program(1.3) [buildInfo = dict({{"coremlc-component-MIL", "3510.2.1"}, {"coremlc-version", "3500.32.1"}, {"coremltools-component-milinternal", ""}, {"coremltools-version", "9.0"}})] { func length_1(tensor inputs_embeds, state> key_cache, tensor position_id, tensor position_index_seed, state> value_cache) { tensor layers_1_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524992))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524416))))[name = string("layers_1_self_attn_v_proj_weight_cast_fp16")]; tensor layers_1_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(525312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13120640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13108288))))[name = string("layers_1_mlp_up_proj_weight_cast_fp16")]; tensor layers_2_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13126848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651776))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651200))))[name = string("layers_2_self_attn_v_proj_weight_cast_fp16")]; tensor layers_2_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13652096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26247424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26235072))))[name = string("layers_2_mlp_up_proj_weight_cast_fp16")]; tensor layers_3_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26253632))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778560))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26777984))))[name = string("layers_3_self_attn_v_proj_weight_cast_fp16")]; tensor layers_3_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30977408))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30973248))))[name = string("layers_3_self_attn_o_proj_weight_cast_fp16")]; tensor layers_3_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30979520))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43566656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43562496))))[name = string("layers_3_mlp_down_proj_weight_cast_fp16")]; tensor layers_4_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43568768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093696))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093120))))[name = string("layers_4_self_attn_v_proj_weight_cast_fp16")]; tensor layers_4_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44094016))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48292544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48288384))))[name = string("layers_4_self_attn_o_proj_weight_cast_fp16")]; tensor layers_4_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48294656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60877632))))[name = string("layers_4_mlp_gate_proj_weight_cast_fp16")]; tensor layers_4_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60896192))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73491520))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73479168))))[name = string("layers_4_mlp_up_proj_weight_cast_fp16")]; tensor layers_4_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73497728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86084864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86080704))))[name = string("layers_4_mlp_down_proj_weight_cast_fp16")]; tensor layers_5_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86086976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611328))))[name = string("layers_5_self_attn_v_proj_weight_cast_fp16")]; tensor layers_5_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86612224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90810752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90806592))))[name = string("layers_5_self_attn_o_proj_weight_cast_fp16")]; tensor layers_5_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90812864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103408192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103395840))))[name = string("layers_5_mlp_up_proj_weight_cast_fp16")]; tensor layers_5_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103414400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116001536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115997376))))[name = string("layers_5_mlp_down_proj_weight_cast_fp16")]; tensor layers_6_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116003648))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528576))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528000))))[name = string("layers_6_self_attn_v_proj_weight_cast_fp16")]; tensor layers_6_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528896))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120727424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120723264))))[name = string("layers_6_self_attn_o_proj_weight_cast_fp16")]; tensor layers_6_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120729536))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133324864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133312512))))[name = string("layers_6_mlp_gate_proj_weight_cast_fp16")]; tensor layers_6_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133331072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145926400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145914048))))[name = string("layers_6_mlp_up_proj_weight_cast_fp16")]; tensor layers_6_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145932608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158519744))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158515584))))[name = string("layers_6_mlp_down_proj_weight_cast_fp16")]; tensor layers_7_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158521856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046208))))[name = string("layers_7_self_attn_v_proj_weight_cast_fp16")]; tensor layers_7_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159047104))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163245632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163241472))))[name = string("layers_7_self_attn_o_proj_weight_cast_fp16")]; tensor layers_7_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163247744))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175843072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175830720))))[name = string("layers_7_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175849280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374208))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176373632))))[name = string("layers_8_self_attn_v_proj_weight_cast_fp16")]; tensor layers_8_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374528))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180573056))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180568896))))[name = string("layers_8_self_attn_o_proj_weight_cast_fp16")]; tensor layers_8_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180575168))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193170496))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193158144))))[name = string("layers_8_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193176704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205772032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205759680))))[name = string("layers_8_mlp_up_proj_weight_cast_fp16")]; tensor layers_8_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205778240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218365376))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218361216))))[name = string("layers_8_mlp_down_proj_weight_cast_fp16")]; tensor layers_9_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218367488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218891840))))[name = string("layers_9_self_attn_v_proj_weight_cast_fp16")]; tensor layers_9_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223091264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223087104))))[name = string("layers_9_self_attn_o_proj_weight_cast_fp16")]; tensor layers_9_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223093376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235688704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235676352))))[name = string("layers_9_mlp_gate_proj_weight_cast_fp16")]; tensor layers_9_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235694912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248290240))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248277888))))[name = string("layers_9_mlp_up_proj_weight_cast_fp16")]; tensor layers_9_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248296448))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260883584))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260879424))))[name = string("layers_9_mlp_down_proj_weight_cast_fp16")]; tensor layers_10_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260885696))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410048))))[name = string("layers_10_self_attn_v_proj_weight_cast_fp16")]; tensor layers_10_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410944))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265609472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265605312))))[name = string("layers_10_self_attn_o_proj_weight_cast_fp16")]; tensor layers_10_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265611584))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278206912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278194560))))[name = string("layers_10_mlp_gate_proj_weight_cast_fp16")]; tensor layers_10_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278213120))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290808448))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290796096))))[name = string("layers_10_mlp_up_proj_weight_cast_fp16")]; tensor layers_10_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290814656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303401792))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303397632))))[name = string("layers_10_mlp_down_proj_weight_cast_fp16")]; tensor layers_11_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303403904))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307602432))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307598272))))[name = string("layers_11_self_attn_q_proj_weight_cast_fp16")]; tensor layers_11_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307604544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308128896))))[name = string("layers_11_self_attn_k_proj_weight_cast_fp16")]; tensor layers_11_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129792))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654720))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654144))))[name = string("layers_11_self_attn_v_proj_weight_cast_fp16")]; tensor layers_11_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308655040))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312853568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312849408))))[name = string("layers_11_self_attn_o_proj_weight_cast_fp16")]; tensor layers_11_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312855680))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325451008))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325438656))))[name = string("layers_11_mlp_gate_proj_weight_cast_fp16")]; tensor layers_11_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325457216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338052544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338040192))))[name = string("layers_11_mlp_up_proj_weight_cast_fp16")]; tensor layers_11_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338058752))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350645888))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350641728))))[name = string("layers_11_mlp_down_proj_weight_cast_fp16")]; tensor layers_12_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350648000))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354846528))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354842368))))[name = string("layers_12_self_attn_q_proj_weight_cast_fp16")]; tensor layers_12_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354848640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355372992))))[name = string("layers_12_self_attn_k_proj_weight_cast_fp16")]; tensor layers_12_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373888))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898816))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898240))))[name = string("layers_12_self_attn_v_proj_weight_cast_fp16")]; tensor layers_12_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355899136))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360097664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360093504))))[name = string("layers_12_self_attn_o_proj_weight_cast_fp16")]; tensor layers_12_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360099776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372695104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372682752))))[name = string("layers_12_mlp_gate_proj_weight_cast_fp16")]; tensor layers_12_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372701312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385296640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385284288))))[name = string("layers_12_mlp_up_proj_weight_cast_fp16")]; tensor layers_12_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385302848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397885824))))[name = string("layers_12_mlp_down_proj_weight_cast_fp16")]; tensor layers_13_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397892096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402090624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402086464))))[name = string("layers_13_self_attn_q_proj_weight_cast_fp16")]; tensor layers_13_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402092736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617088))))[name = string("layers_13_self_attn_k_proj_weight_cast_fp16")]; tensor layers_13_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617984))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142336))))[name = string("layers_13_self_attn_v_proj_weight_cast_fp16")]; tensor layers_13_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403143232))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407341760))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407337600))))[name = string("layers_13_self_attn_o_proj_weight_cast_fp16")]; tensor layers_13_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407343872))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419939200))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419926848))))[name = string("layers_13_mlp_gate_proj_weight_cast_fp16")]; tensor layers_13_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419945408))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432532544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432528384))))[name = string("layers_13_mlp_down_proj_weight_cast_fp16")]; tensor layers_14_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432534656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436733184))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436729024))))[name = string("layers_14_self_attn_q_proj_weight_cast_fp16")]; tensor layers_14_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436735296))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437259648))))[name = string("layers_14_self_attn_v_proj_weight_cast_fp16")]; tensor layers_14_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441459072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441454912))))[name = string("layers_14_self_attn_o_proj_weight_cast_fp16")]; tensor layers_14_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441461184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454056512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454044160))))[name = string("layers_14_mlp_gate_proj_weight_cast_fp16")]; tensor layers_14_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454062720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466658048))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466645696))))[name = string("layers_14_mlp_up_proj_weight_cast_fp16")]; tensor layers_14_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466664256))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479251392))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479247232))))[name = string("layers_14_mlp_down_proj_weight_cast_fp16")]; tensor layers_15_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479253504))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483452032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483447872))))[name = string("layers_15_self_attn_q_proj_weight_cast_fp16")]; tensor layers_15_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483454144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483978496))))[name = string("layers_15_self_attn_k_proj_weight_cast_fp16")]; tensor layers_15_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504320))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484503744))))[name = string("layers_15_self_attn_v_proj_weight_cast_fp16")]; tensor layers_15_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488703168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488699008))))[name = string("layers_15_self_attn_o_proj_weight_cast_fp16")]; tensor layers_15_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488705280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501300608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501288256))))[name = string("layers_15_mlp_gate_proj_weight_cast_fp16")]; tensor layers_15_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501306816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513902144))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513889792))))[name = string("layers_15_mlp_up_proj_weight_cast_fp16")]; tensor layers_15_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513908352))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526495488))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526491328))))[name = string("layers_15_mlp_down_proj_weight_cast_fp16")]; tensor layers_16_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526497600))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530696128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530691968))))[name = string("layers_16_self_attn_q_proj_weight_cast_fp16")]; tensor layers_16_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530698240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531222592))))[name = string("layers_16_self_attn_k_proj_weight_cast_fp16")]; tensor layers_16_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531747840))))[name = string("layers_16_self_attn_v_proj_weight_cast_fp16")]; tensor layers_16_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535947264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535943104))))[name = string("layers_16_self_attn_o_proj_weight_cast_fp16")]; tensor layers_16_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535949376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548536512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548532352))))[name = string("layers_16_mlp_down_proj_weight_cast_fp16")]; tensor layers_17_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548538624))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552737152))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552732992))))[name = string("layers_17_self_attn_q_proj_weight_cast_fp16")]; tensor layers_17_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552739264))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553263616))))[name = string("layers_17_self_attn_k_proj_weight_cast_fp16")]; tensor layers_17_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264512))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553788864))))[name = string("layers_17_self_attn_v_proj_weight_cast_fp16")]; tensor layers_17_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789760))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557988288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557984128))))[name = string("layers_17_self_attn_o_proj_weight_cast_fp16")]; tensor layers_17_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557990400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570585728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570573376))))[name = string("layers_17_mlp_gate_proj_weight_cast_fp16")]; tensor layers_17_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570591936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583187264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583174912))))[name = string("layers_17_mlp_up_proj_weight_cast_fp16")]; tensor layers_17_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583193472))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595780608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595776448))))[name = string("layers_17_mlp_down_proj_weight_cast_fp16")]; tensor layers_18_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595782720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599981248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599977088))))[name = string("layers_18_self_attn_q_proj_weight_cast_fp16")]; tensor layers_18_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599983360))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600507712))))[name = string("layers_18_self_attn_k_proj_weight_cast_fp16")]; tensor layers_18_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601032960))))[name = string("layers_18_self_attn_v_proj_weight_cast_fp16")]; tensor layers_18_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605232384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605228224))))[name = string("layers_18_self_attn_o_proj_weight_cast_fp16")]; tensor layers_18_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605234496))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617829824))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617817472))))[name = string("layers_18_mlp_gate_proj_weight_cast_fp16")]; tensor layers_18_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617836032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630431360))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630419008))))[name = string("layers_18_mlp_up_proj_weight_cast_fp16")]; tensor layers_18_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630437568))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643024704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643020544))))[name = string("layers_18_mlp_down_proj_weight_cast_fp16")]; tensor layers_19_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643026816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647225344))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647221184))))[name = string("layers_19_self_attn_q_proj_weight_cast_fp16")]; tensor layers_19_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647227456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647751808))))[name = string("layers_19_self_attn_k_proj_weight_cast_fp16")]; tensor layers_19_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660348032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660335680))))[name = string("layers_19_mlp_gate_proj_weight_cast_fp16")]; tensor layers_19_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660354240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672949568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672937216))))[name = string("layers_19_mlp_up_proj_weight_cast_fp16")]; tensor layers_19_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672955776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685542912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685538752))))[name = string("layers_19_mlp_down_proj_weight_cast_fp16")]; tensor layers_20_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685545024))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689743552))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689739392))))[name = string("layers_20_self_attn_q_proj_weight_cast_fp16")]; tensor layers_20_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689745664))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270592))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270016))))[name = string("layers_20_self_attn_k_proj_weight_cast_fp16")]; tensor layers_20_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694469440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694465280))))[name = string("layers_20_self_attn_o_proj_weight_cast_fp16")]; tensor layers_20_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694471552))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707066880))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707054528))))[name = string("layers_20_mlp_gate_proj_weight_cast_fp16")]; tensor layers_20_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707073088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719660224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719656064))))[name = string("layers_20_mlp_down_proj_weight_cast_fp16")]; tensor layers_21_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719662336))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723860864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723856704))))[name = string("layers_21_self_attn_q_proj_weight_cast_fp16")]; tensor layers_21_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723862976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387328))))[name = string("layers_21_self_attn_k_proj_weight_cast_fp16")]; tensor layers_21_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724388224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728586752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728582592))))[name = string("layers_21_self_attn_o_proj_weight_cast_fp16")]; tensor layers_21_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728588864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741184192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741171840))))[name = string("layers_21_mlp_gate_proj_weight_cast_fp16")]; tensor layers_21_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741190400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753785728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753773376))))[name = string("layers_21_mlp_up_proj_weight_cast_fp16")]; tensor layers_21_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753791936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766379072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766374912))))[name = string("layers_21_mlp_down_proj_weight_cast_fp16")]; tensor layers_22_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766381184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770579712))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770575552))))[name = string("layers_22_self_attn_q_proj_weight_cast_fp16")]; tensor layers_22_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770581824))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106176))))[name = string("layers_22_self_attn_k_proj_weight_cast_fp16")]; tensor layers_22_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771107072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783702400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783690048))))[name = string("layers_22_mlp_gate_proj_weight_cast_fp16")]; tensor layers_22_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783708608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796303936))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796291584))))[name = string("layers_22_mlp_up_proj_weight_cast_fp16")]; tensor layers_22_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796310144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808897280))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808893120))))[name = string("layers_22_mlp_down_proj_weight_cast_fp16")]; tensor layers_23_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808899392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813097920))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813093760))))[name = string("layers_23_self_attn_q_proj_weight_cast_fp16")]; tensor layers_23_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813100032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624960))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624384))))[name = string("layers_23_self_attn_k_proj_weight_cast_fp16")]; tensor layers_23_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813625280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817823808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817819648))))[name = string("layers_23_self_attn_o_proj_weight_cast_fp16")]; tensor layers_23_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817825920))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830421248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830408896))))[name = string("layers_23_mlp_gate_proj_weight_cast_fp16")]; tensor layers_23_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830427456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843022784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843010432))))[name = string("layers_23_mlp_up_proj_weight_cast_fp16")]; tensor layers_23_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843028992))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855616128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855611968))))[name = string("layers_23_mlp_down_proj_weight_cast_fp16")]; tensor layers_24_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855618240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859816768))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859812608))))[name = string("layers_24_self_attn_q_proj_weight_cast_fp16")]; tensor layers_24_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859818880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343232))))[name = string("layers_24_self_attn_k_proj_weight_cast_fp16")]; tensor layers_24_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860344128))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864542656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864538496))))[name = string("layers_24_self_attn_o_proj_weight_cast_fp16")]; tensor layers_24_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864544768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877140096))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877127744))))[name = string("layers_24_mlp_gate_proj_weight_cast_fp16")]; tensor layers_24_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877146304))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889741632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889729280))))[name = string("layers_24_mlp_up_proj_weight_cast_fp16")]; tensor layers_24_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889747840))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902334976))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902330816))))[name = string("layers_24_mlp_down_proj_weight_cast_fp16")]; tensor layers_25_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902337088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906535616))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906531456))))[name = string("layers_25_self_attn_q_proj_weight_cast_fp16")]; tensor layers_25_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906537728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062080))))[name = string("layers_25_self_attn_k_proj_weight_cast_fp16")]; tensor layers_25_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911261504))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911257344))))[name = string("layers_25_self_attn_o_proj_weight_cast_fp16")]; tensor layers_25_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911263616))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923858944))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923846592))))[name = string("layers_25_mlp_gate_proj_weight_cast_fp16")]; tensor layers_25_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923865152))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936460480))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936448128))))[name = string("layers_25_mlp_up_proj_weight_cast_fp16")]; tensor layers_26_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936466688))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940665216))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940661056))))[name = string("layers_26_self_attn_q_proj_weight_cast_fp16")]; tensor layers_26_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940667328))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192256))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941191680))))[name = string("layers_26_self_attn_k_proj_weight_cast_fp16")]; tensor layers_26_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192576))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945391104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945386944))))[name = string("layers_26_self_attn_o_proj_weight_cast_fp16")]; tensor layers_26_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945393216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957988544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957976192))))[name = string("layers_26_mlp_gate_proj_weight_cast_fp16")]; int32 var_765 = const()[name = string("op_765"), val = int32(0)]; tensor var_766 = mul(x = position_index_seed, y = var_765)[name = string("op_766")]; int32 var_768 = const()[name = string("op_768"), val = int32(1)]; tensor ones = add(x = var_766, y = var_768)[name = string("ones")]; int32 var_770 = const()[name = string("op_770"), val = int32(0)]; bool var_772_exclusive_0 = const()[name = string("op_772_exclusive_0"), val = bool(false)]; bool var_772_reverse_0 = const()[name = string("op_772_reverse_0"), val = bool(false)]; tensor var_772 = cumsum(axis = var_770, exclusive = var_772_exclusive_0, reverse = var_772_reverse_0, x = ones)[name = string("op_772")]; int32 var_774 = const()[name = string("op_774"), val = int32(1)]; tensor position_offsets = sub(x = var_772, y = var_774)[name = string("position_offsets")]; tensor position_ids_1 = add(x = position_offsets, y = position_id)[name = string("position_ids_1")]; bool var_784_keep_dims_0 = const()[name = string("op_784_keep_dims_0"), val = bool(false)]; int32 var_784 = reduce_sum(keep_dims = var_784_keep_dims_0, x = ones)[name = string("op_784")]; int32 var_786 = const()[name = string("op_786"), val = int32(1)]; int32 offset = sub(x = var_784, y = var_786)[name = string("offset")]; tensor var_789 = add(x = position_id, y = offset)[name = string("op_789")]; int32 var_791 = const()[name = string("op_791"), val = int32(1)]; tensor cache_position_end = add(x = var_789, y = var_791)[name = string("cache_position_end")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor greater_equal_0 = greater_equal(x = position_ids_1, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(32768)]; tensor add_0 = add(x = position_ids_1, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = position_ids_1, b = add_0, cond = greater_equal_0)[name = string("select_0")]; tensor rope_emb_cos_cached_to_fp16 = const()[name = string("rope_emb_cos_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957994752)))]; int32 cos_1_batch_dims_0 = const()[name = string("cos_1_batch_dims_0"), val = int32(0)]; bool cos_1_validate_indices_0 = const()[name = string("cos_1_validate_indices_0"), val = bool(false)]; int32 greater_equal_2_y_0 = const()[name = string("greater_equal_2_y_0"), val = int32(0)]; tensor greater_equal_2 = greater_equal(x = select_0, y = greater_equal_2_y_0)[name = string("greater_equal_2")]; int32 slice_by_index_2 = const()[name = string("slice_by_index_2"), val = int32(32768)]; tensor add_2 = add(x = select_0, y = slice_by_index_2)[name = string("add_2")]; tensor select_2 = select(a = select_0, b = add_2, cond = greater_equal_2)[name = string("select_2")]; int32 cos_1_cast_fp16_axis_1 = const()[name = string("cos_1_cast_fp16_axis_1"), val = int32(0)]; tensor cos_1_cast_fp16 = gather(axis = cos_1_cast_fp16_axis_1, batch_dims = cos_1_batch_dims_0, indices = select_2, validate_indices = cos_1_validate_indices_0, x = rope_emb_cos_cached_to_fp16)[name = string("cos_1_cast_fp16")]; tensor rope_emb_sin_cached_to_fp16 = const()[name = string("rope_emb_sin_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(966383424)))]; int32 sin_1_batch_dims_0 = const()[name = string("sin_1_batch_dims_0"), val = int32(0)]; bool sin_1_validate_indices_0 = const()[name = string("sin_1_validate_indices_0"), val = bool(false)]; int32 sin_1_cast_fp16_axis_1 = const()[name = string("sin_1_cast_fp16_axis_1"), val = int32(0)]; tensor sin_1_cast_fp16 = gather(axis = sin_1_cast_fp16_axis_1, batch_dims = sin_1_batch_dims_0, indices = select_2, validate_indices = sin_1_validate_indices_0, x = rope_emb_sin_cached_to_fp16)[name = string("sin_1_cast_fp16")]; tensor var_865_perm_0 = const()[name = string("op_865_perm_0"), val = tensor([-1, -2])]; tensor var_867_axes_0 = const()[name = string("op_867_axes_0"), val = tensor([0])]; tensor var_865_cast_fp16 = transpose(perm = var_865_perm_0, x = cos_1_cast_fp16)[name = string("transpose_171")]; tensor var_867_cast_fp16 = expand_dims(axes = var_867_axes_0, x = var_865_cast_fp16)[name = string("op_867_cast_fp16")]; tensor var_869_axes_0 = const()[name = string("op_869_axes_0"), val = tensor([0])]; tensor var_869_cast_fp16 = expand_dims(axes = var_869_axes_0, x = var_867_cast_fp16)[name = string("op_869_cast_fp16")]; tensor var_874_perm_0 = const()[name = string("op_874_perm_0"), val = tensor([-1, -2])]; tensor var_876_axes_0 = const()[name = string("op_876_axes_0"), val = tensor([0])]; tensor var_874_cast_fp16 = transpose(perm = var_874_perm_0, x = sin_1_cast_fp16)[name = string("transpose_170")]; tensor var_876_cast_fp16 = expand_dims(axes = var_876_axes_0, x = var_874_cast_fp16)[name = string("op_876_cast_fp16")]; tensor var_878_axes_0 = const()[name = string("op_878_axes_0"), val = tensor([0])]; tensor var_878_cast_fp16 = expand_dims(axes = var_878_axes_0, x = var_876_cast_fp16)[name = string("op_878_cast_fp16")]; string position_ids_1_to_uint16_dtype_0 = const()[name = string("position_ids_1_to_uint16_dtype_0"), val = string("uint16")]; tensor causal_mask = const()[name = string("causal_mask"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(974772096)))]; int32 mask_axis_0 = const()[name = string("mask_axis_0"), val = int32(1)]; int32 mask_batch_dims_0 = const()[name = string("mask_batch_dims_0"), val = int32(0)]; bool mask_validate_indices_0 = const()[name = string("mask_validate_indices_0"), val = bool(false)]; tensor position_ids_1_to_uint16 = cast(dtype = position_ids_1_to_uint16_dtype_0, x = position_ids_1)[name = string("cast_3")]; tensor mask_cast_uint16 = gather(axis = mask_axis_0, batch_dims = mask_batch_dims_0, indices = position_ids_1_to_uint16, validate_indices = mask_validate_indices_0, x = causal_mask)[name = string("mask_cast_uint16")]; tensor var_895_axes_0 = const()[name = string("op_895_axes_0"), val = tensor([0])]; tensor var_895 = expand_dims(axes = var_895_axes_0, x = mask_cast_uint16)[name = string("op_895")]; tensor attn_mask_1_axes_0 = const()[name = string("attn_mask_1_axes_0"), val = tensor([0])]; tensor attn_mask_1 = expand_dims(axes = attn_mask_1_axes_0, x = var_895)[name = string("attn_mask_1")]; string inputs_embeds_to_fp16_dtype_0 = const()[name = string("inputs_embeds_to_fp16_dtype_0"), val = string("fp16")]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor inputs_embeds_to_fp16 = cast(dtype = inputs_embeds_to_fp16_dtype_0, x = inputs_embeds)[name = string("cast_2")]; tensor var_906_cast_fp16 = mul(x = inputs_embeds_to_fp16, y = const_0_promoted_to_fp16)[name = string("op_906_cast_fp16")]; int32 var_904 = const()[name = string("op_904"), val = int32(1)]; bool doubled_1_interleave_0 = const()[name = string("doubled_1_interleave_0"), val = bool(false)]; tensor doubled_1_cast_fp16 = concat(axis = var_904, interleave = doubled_1_interleave_0, values = (inputs_embeds_to_fp16, var_906_cast_fp16))[name = string("doubled_1_cast_fp16")]; tensor out_1_axes_0 = const()[name = string("out_1_axes_0"), val = tensor([1])]; tensor out_1_gamma_0_to_fp16 = const()[name = string("out_1_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983160768)))]; fp16 var_916_to_fp16 = const()[name = string("op_916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_1_cast_fp16 = layer_norm(axes = out_1_axes_0, epsilon = var_916_to_fp16, gamma = out_1_gamma_0_to_fp16, x = doubled_1_cast_fp16)[name = string("out_1_cast_fp16")]; tensor var_927_split_sizes_0 = const()[name = string("op_927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_927_axis_0 = const()[name = string("op_927_axis_0"), val = int32(1)]; tensor var_927_cast_fp16_0, tensor var_927_cast_fp16_1 = split(axis = var_927_axis_0, split_sizes = var_927_split_sizes_0, x = out_1_cast_fp16)[name = string("op_927_cast_fp16")]; tensor layers_0_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983169024)))]; tensor query_states_1_strides_0 = const()[name = string("query_states_1_strides_0"), val = tensor([1, 1])]; string query_states_1_pad_type_0 = const()[name = string("query_states_1_pad_type_0"), val = string("valid")]; tensor query_states_1_pad_0 = const()[name = string("query_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_1_dilations_0 = const()[name = string("query_states_1_dilations_0"), val = tensor([1, 1])]; int32 query_states_1_groups_0 = const()[name = string("query_states_1_groups_0"), val = int32(1)]; tensor query_states_1_cast_fp16 = conv(dilations = query_states_1_dilations_0, groups = query_states_1_groups_0, pad = query_states_1_pad_0, pad_type = query_states_1_pad_type_0, strides = query_states_1_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("query_states_1_cast_fp16")]; tensor layers_0_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(991557696)))]; tensor key_states_1_strides_0 = const()[name = string("key_states_1_strides_0"), val = tensor([1, 1])]; string key_states_1_pad_type_0 = const()[name = string("key_states_1_pad_type_0"), val = string("valid")]; tensor key_states_1_pad_0 = const()[name = string("key_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_1_dilations_0 = const()[name = string("key_states_1_dilations_0"), val = tensor([1, 1])]; int32 key_states_1_groups_0 = const()[name = string("key_states_1_groups_0"), val = int32(1)]; tensor key_states_1_cast_fp16 = conv(dilations = key_states_1_dilations_0, groups = key_states_1_groups_0, pad = key_states_1_pad_0, pad_type = key_states_1_pad_type_0, strides = key_states_1_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("key_states_1_cast_fp16")]; tensor layers_0_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(992606336)))]; tensor value_states_1_strides_0 = const()[name = string("value_states_1_strides_0"), val = tensor([1, 1])]; string value_states_1_pad_type_0 = const()[name = string("value_states_1_pad_type_0"), val = string("valid")]; tensor value_states_1_pad_0 = const()[name = string("value_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_1_dilations_0 = const()[name = string("value_states_1_dilations_0"), val = tensor([1, 1])]; int32 value_states_1_groups_0 = const()[name = string("value_states_1_groups_0"), val = int32(1)]; tensor value_states_1_cast_fp16 = conv(dilations = value_states_1_dilations_0, groups = value_states_1_groups_0, pad = value_states_1_pad_0, pad_type = value_states_1_pad_type_0, strides = value_states_1_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("value_states_1_cast_fp16")]; tensor concat_0x = const()[name = string("concat_0x"), val = tensor([1, 16, 128, -1])]; tensor x_1_cast_fp16 = reshape(shape = concat_0x, x = query_states_1_cast_fp16)[name = string("x_1_cast_fp16")]; tensor concat_1x = const()[name = string("concat_1x"), val = tensor([1, 2, 128, -1])]; tensor var_984_cast_fp16 = reshape(shape = concat_1x, x = key_states_1_cast_fp16)[name = string("op_984_cast_fp16")]; tensor concat_2x = const()[name = string("concat_2x"), val = tensor([1, 2, 128, -1])]; tensor var_991_cast_fp16 = reshape(shape = concat_2x, x = value_states_1_cast_fp16)[name = string("op_991_cast_fp16")]; tensor var_995_cast_fp16 = mul(x = x_1_cast_fp16, y = var_869_cast_fp16)[name = string("op_995_cast_fp16")]; tensor var_996_split_sizes_0 = const()[name = string("op_996_split_sizes_0"), val = tensor([64, 64])]; int32 var_996_axis_0 = const()[name = string("op_996_axis_0"), val = int32(-2)]; tensor var_996_cast_fp16_0, tensor var_996_cast_fp16_1 = split(axis = var_996_axis_0, split_sizes = var_996_split_sizes_0, x = x_1_cast_fp16)[name = string("op_996_cast_fp16")]; fp16 const_2_promoted_to_fp16 = const()[name = string("const_2_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_998_cast_fp16 = mul(x = var_996_cast_fp16_1, y = const_2_promoted_to_fp16)[name = string("op_998_cast_fp16")]; int32 var_1000 = const()[name = string("op_1000"), val = int32(-2)]; bool var_1001_interleave_0 = const()[name = string("op_1001_interleave_0"), val = bool(false)]; tensor var_1001_cast_fp16 = concat(axis = var_1000, interleave = var_1001_interleave_0, values = (var_998_cast_fp16, var_996_cast_fp16_0))[name = string("op_1001_cast_fp16")]; tensor var_1002_cast_fp16 = mul(x = var_1001_cast_fp16, y = var_878_cast_fp16)[name = string("op_1002_cast_fp16")]; tensor query_states_3_cast_fp16 = add(x = var_995_cast_fp16, y = var_1002_cast_fp16)[name = string("query_states_3_cast_fp16")]; tensor var_1008_cast_fp16 = mul(x = var_984_cast_fp16, y = var_869_cast_fp16)[name = string("op_1008_cast_fp16")]; tensor var_1009_split_sizes_0 = const()[name = string("op_1009_split_sizes_0"), val = tensor([64, 64])]; int32 var_1009_axis_0 = const()[name = string("op_1009_axis_0"), val = int32(-2)]; tensor var_1009_cast_fp16_0, tensor var_1009_cast_fp16_1 = split(axis = var_1009_axis_0, split_sizes = var_1009_split_sizes_0, x = var_984_cast_fp16)[name = string("op_1009_cast_fp16")]; fp16 const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1011_cast_fp16 = mul(x = var_1009_cast_fp16_1, y = const_3_promoted_to_fp16)[name = string("op_1011_cast_fp16")]; int32 var_1013 = const()[name = string("op_1013"), val = int32(-2)]; bool var_1014_interleave_0 = const()[name = string("op_1014_interleave_0"), val = bool(false)]; tensor var_1014_cast_fp16 = concat(axis = var_1013, interleave = var_1014_interleave_0, values = (var_1011_cast_fp16, var_1009_cast_fp16_0))[name = string("op_1014_cast_fp16")]; tensor var_1015_cast_fp16 = mul(x = var_1014_cast_fp16, y = var_878_cast_fp16)[name = string("op_1015_cast_fp16")]; tensor key_states_5_cast_fp16 = add(x = var_1008_cast_fp16, y = var_1015_cast_fp16)[name = string("key_states_5_cast_fp16")]; tensor read_state_0 = read_state(input = key_cache)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; int32 concat_5_axis_0 = const()[name = string("concat_5_axis_0"), val = int32(0)]; bool concat_5_interleave_0 = const()[name = string("concat_5_interleave_0"), val = bool(false)]; tensor concat_5 = concat(axis = concat_5_axis_0, interleave = concat_5_interleave_0, values = (expand_dims_0, expand_dims_1, position_id, expand_dims_3))[name = string("concat_5")]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; tensor concat_6_values1_0 = const()[name = string("concat_6_values1_0"), val = tensor([0])]; tensor concat_6_values3_0 = const()[name = string("concat_6_values3_0"), val = tensor([0])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_4, concat_6_values1_0, cache_position_end, concat_6_values3_0))[name = string("concat_6")]; tensor key_states_7_perm_0 = const()[name = string("key_states_7_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_1_stride_0 = const()[name = string("key_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_7_cast_fp16 = transpose(perm = key_states_7_perm_0, x = key_states_5_cast_fp16)[name = string("transpose_169")]; tensor key_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = key_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = key_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_1_squeeze_mask_0, stride = key_cache_internal_tensor_assign_1_stride_0, update = key_states_7_cast_fp16, x = read_state_0)[name = string("key_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_1_cast_fp16, input = key_cache)[name = string("coreml_update_state_56_write_state")]; tensor coreml_update_state_56 = read_state(input = key_cache)[name = string("coreml_update_state_56")]; tensor read_state_1 = read_state(input = value_cache)[name = string("read_state_1")]; tensor value_states_3_perm_0 = const()[name = string("value_states_3_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_1_stride_0 = const()[name = string("value_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_3_cast_fp16 = transpose(perm = value_states_3_perm_0, x = var_991_cast_fp16)[name = string("transpose_168")]; tensor value_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = value_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = value_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_1_squeeze_mask_0, stride = value_cache_internal_tensor_assign_1_stride_0, update = value_states_3_cast_fp16, x = read_state_1)[name = string("value_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_1_cast_fp16, input = value_cache)[name = string("coreml_update_state_57_write_state")]; tensor coreml_update_state_57 = read_state(input = value_cache)[name = string("coreml_update_state_57")]; tensor var_1085_begin_0 = const()[name = string("op_1085_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1085_end_0 = const()[name = string("op_1085_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1085_end_mask_0 = const()[name = string("op_1085_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1085_cast_fp16 = slice_by_index(begin = var_1085_begin_0, end = var_1085_end_0, end_mask = var_1085_end_mask_0, x = coreml_update_state_56)[name = string("op_1085_cast_fp16")]; tensor tile_0 = const()[name = string("tile_0"), val = tensor([1, 1])]; int32 var_1088_axis_0 = const()[name = string("op_1088_axis_0"), val = int32(1)]; tensor var_1088_cast_fp16_0, tensor var_1088_cast_fp16_1 = split(axis = var_1088_axis_0, split_sizes = tile_0, x = var_1085_cast_fp16)[name = string("op_1088_cast_fp16")]; tensor var_1095_begin_0 = const()[name = string("op_1095_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1095_end_0 = const()[name = string("op_1095_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1095_end_mask_0 = const()[name = string("op_1095_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1095_cast_fp16 = slice_by_index(begin = var_1095_begin_0, end = var_1095_end_0, end_mask = var_1095_end_mask_0, x = coreml_update_state_57)[name = string("op_1095_cast_fp16")]; tensor tile_1 = const()[name = string("tile_1"), val = tensor([1, 1])]; int32 var_1098_axis_0 = const()[name = string("op_1098_axis_0"), val = int32(1)]; tensor var_1098_cast_fp16_0, tensor var_1098_cast_fp16_1 = split(axis = var_1098_axis_0, split_sizes = tile_1, x = var_1095_cast_fp16)[name = string("op_1098_cast_fp16")]; tensor var_1101_split_sizes_0 = const()[name = string("op_1101_split_sizes_0"), val = tensor([8, 8])]; int32 var_1101_axis_0 = const()[name = string("op_1101_axis_0"), val = int32(1)]; tensor var_1101_0, tensor var_1101_1 = split(axis = var_1101_axis_0, split_sizes = var_1101_split_sizes_0, x = query_states_3_cast_fp16)[name = string("op_1101")]; bool attn_weights_1_transpose_x_0 = const()[name = string("attn_weights_1_transpose_x_0"), val = bool(false)]; bool attn_weights_1_transpose_y_0 = const()[name = string("attn_weights_1_transpose_y_0"), val = bool(false)]; tensor attn_weights_1_cast_fp16 = matmul(transpose_x = attn_weights_1_transpose_x_0, transpose_y = attn_weights_1_transpose_y_0, x = var_1088_cast_fp16_0, y = var_1101_0)[name = string("attn_weights_1_cast_fp16")]; fp16 var_1104_to_fp16 = const()[name = string("op_1104_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_3_cast_fp16 = mul(x = attn_weights_1_cast_fp16, y = var_1104_to_fp16)[name = string("attn_weights_3_cast_fp16")]; tensor attn_weights_5_cast_fp16 = add(x = attn_weights_3_cast_fp16, y = attn_mask_1)[name = string("attn_weights_5_cast_fp16")]; int32 var_1108 = const()[name = string("op_1108"), val = int32(-2)]; tensor attn_weights_7_cast_fp16 = softmax(axis = var_1108, x = attn_weights_5_cast_fp16)[name = string("attn_weights_7_cast_fp16")]; bool var_1114_transpose_x_1 = const()[name = string("op_1114_transpose_x_1"), val = bool(true)]; bool var_1114_transpose_y_1 = const()[name = string("op_1114_transpose_y_1"), val = bool(false)]; tensor var_1114_cast_fp16 = matmul(transpose_x = var_1114_transpose_x_1, transpose_y = var_1114_transpose_y_1, x = attn_weights_7_cast_fp16, y = var_1098_cast_fp16_0)[name = string("op_1114_cast_fp16")]; bool attn_weights_9_transpose_x_0 = const()[name = string("attn_weights_9_transpose_x_0"), val = bool(false)]; bool attn_weights_9_transpose_y_0 = const()[name = string("attn_weights_9_transpose_y_0"), val = bool(false)]; tensor attn_weights_9_cast_fp16 = matmul(transpose_x = attn_weights_9_transpose_x_0, transpose_y = attn_weights_9_transpose_y_0, x = var_1088_cast_fp16_1, y = var_1101_1)[name = string("attn_weights_9_cast_fp16")]; fp16 var_1116_to_fp16 = const()[name = string("op_1116_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_11_cast_fp16 = mul(x = attn_weights_9_cast_fp16, y = var_1116_to_fp16)[name = string("attn_weights_11_cast_fp16")]; tensor attn_weights_13_cast_fp16 = add(x = attn_weights_11_cast_fp16, y = attn_mask_1)[name = string("attn_weights_13_cast_fp16")]; int32 var_1120 = const()[name = string("op_1120"), val = int32(-2)]; tensor attn_weights_15_cast_fp16 = softmax(axis = var_1120, x = attn_weights_13_cast_fp16)[name = string("attn_weights_15_cast_fp16")]; bool attn_output_1_transpose_x_1 = const()[name = string("attn_output_1_transpose_x_1"), val = bool(true)]; bool attn_output_1_transpose_y_1 = const()[name = string("attn_output_1_transpose_y_1"), val = bool(false)]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_1, transpose_y = attn_output_1_transpose_y_1, x = attn_weights_15_cast_fp16, y = var_1098_cast_fp16_1)[name = string("attn_output_1_cast_fp16")]; int32 var_1128 = const()[name = string("op_1128"), val = int32(1)]; bool attn_output_3_interleave_0 = const()[name = string("attn_output_3_interleave_0"), val = bool(false)]; tensor attn_output_3_cast_fp16 = concat(axis = var_1128, interleave = attn_output_3_interleave_0, values = (var_1114_cast_fp16, attn_output_1_cast_fp16))[name = string("attn_output_3_cast_fp16")]; tensor var_1132_perm_0 = const()[name = string("op_1132_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_11x = const()[name = string("concat_11x"), val = tensor([1, 2048, 1, -1])]; tensor var_1132_cast_fp16 = transpose(perm = var_1132_perm_0, x = attn_output_3_cast_fp16)[name = string("transpose_167")]; tensor attn_output_7_cast_fp16 = reshape(shape = concat_11x, x = var_1132_cast_fp16)[name = string("attn_output_7_cast_fp16")]; tensor layers_0_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(993654976)))]; tensor hidden_states_3_strides_0 = const()[name = string("hidden_states_3_strides_0"), val = tensor([1, 1])]; string hidden_states_3_pad_type_0 = const()[name = string("hidden_states_3_pad_type_0"), val = string("valid")]; tensor hidden_states_3_pad_0 = const()[name = string("hidden_states_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_3_dilations_0 = const()[name = string("hidden_states_3_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_3_groups_0 = const()[name = string("hidden_states_3_groups_0"), val = int32(1)]; tensor hidden_states_3_cast_fp16 = conv(dilations = hidden_states_3_dilations_0, groups = hidden_states_3_groups_0, pad = hidden_states_3_pad_0, pad_type = hidden_states_3_pad_type_0, strides = hidden_states_3_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16, x = attn_output_7_cast_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor hidden_states_5_cast_fp16 = add(x = inputs_embeds_to_fp16, y = hidden_states_3_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1165_cast_fp16 = mul(x = hidden_states_5_cast_fp16, y = const_8_promoted_to_fp16)[name = string("op_1165_cast_fp16")]; int32 var_1163 = const()[name = string("op_1163"), val = int32(1)]; bool doubled_5_interleave_0 = const()[name = string("doubled_5_interleave_0"), val = bool(false)]; tensor doubled_5_cast_fp16 = concat(axis = var_1163, interleave = doubled_5_interleave_0, values = (hidden_states_5_cast_fp16, var_1165_cast_fp16))[name = string("doubled_5_cast_fp16")]; tensor out_3_axes_0 = const()[name = string("out_3_axes_0"), val = tensor([1])]; tensor out_3_gamma_0_to_fp16 = const()[name = string("out_3_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002043648)))]; fp16 var_1175_to_fp16 = const()[name = string("op_1175_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_3_cast_fp16 = layer_norm(axes = out_3_axes_0, epsilon = var_1175_to_fp16, gamma = out_3_gamma_0_to_fp16, x = doubled_5_cast_fp16)[name = string("out_3_cast_fp16")]; tensor var_1186_split_sizes_0 = const()[name = string("op_1186_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1186_axis_0 = const()[name = string("op_1186_axis_0"), val = int32(1)]; tensor var_1186_cast_fp16_0, tensor var_1186_cast_fp16_1 = split(axis = var_1186_axis_0, split_sizes = var_1186_split_sizes_0, x = out_3_cast_fp16)[name = string("op_1186_cast_fp16")]; tensor layers_0_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002051904)))]; tensor input_1_strides_0 = const()[name = string("input_1_strides_0"), val = tensor([1, 1])]; string input_1_pad_type_0 = const()[name = string("input_1_pad_type_0"), val = string("valid")]; tensor input_1_pad_0 = const()[name = string("input_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_1_dilations_0 = const()[name = string("input_1_dilations_0"), val = tensor([1, 1])]; int32 input_1_groups_0 = const()[name = string("input_1_groups_0"), val = int32(1)]; tensor input_1_cast_fp16 = conv(dilations = input_1_dilations_0, groups = input_1_groups_0, pad = input_1_pad_0, pad_type = input_1_pad_type_0, strides = input_1_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("input_1_cast_fp16")]; tensor var_1203_cast_fp16 = silu(x = input_1_cast_fp16)[name = string("op_1203_cast_fp16")]; tensor layers_0_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1027217792)))]; tensor var_1209_strides_0 = const()[name = string("op_1209_strides_0"), val = tensor([1, 1])]; string var_1209_pad_type_0 = const()[name = string("op_1209_pad_type_0"), val = string("valid")]; tensor var_1209_pad_0 = const()[name = string("op_1209_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1209_dilations_0 = const()[name = string("op_1209_dilations_0"), val = tensor([1, 1])]; int32 var_1209_groups_0 = const()[name = string("op_1209_groups_0"), val = int32(1)]; tensor var_1209_cast_fp16 = conv(dilations = var_1209_dilations_0, groups = var_1209_groups_0, pad = var_1209_pad_0, pad_type = var_1209_pad_type_0, strides = var_1209_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("op_1209_cast_fp16")]; tensor x_9_cast_fp16 = mul(x = var_1203_cast_fp16, y = var_1209_cast_fp16)[name = string("x_9_cast_fp16")]; tensor layers_0_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1052383680)))]; tensor hidden_states_7_strides_0 = const()[name = string("hidden_states_7_strides_0"), val = tensor([1, 1])]; string hidden_states_7_pad_type_0 = const()[name = string("hidden_states_7_pad_type_0"), val = string("valid")]; tensor hidden_states_7_pad_0 = const()[name = string("hidden_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_7_dilations_0 = const()[name = string("hidden_states_7_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_7_groups_0 = const()[name = string("hidden_states_7_groups_0"), val = int32(1)]; tensor hidden_states_7_cast_fp16 = conv(dilations = hidden_states_7_dilations_0, groups = hidden_states_7_groups_0, pad = hidden_states_7_pad_0, pad_type = hidden_states_7_pad_type_0, strides = hidden_states_7_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16, x = x_9_cast_fp16)[name = string("hidden_states_7_cast_fp16")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = hidden_states_7_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; fp16 const_10_promoted_to_fp16 = const()[name = string("const_10_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1227_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_10_promoted_to_fp16)[name = string("op_1227_cast_fp16")]; int32 var_1225 = const()[name = string("op_1225"), val = int32(1)]; bool doubled_9_interleave_0 = const()[name = string("doubled_9_interleave_0"), val = bool(false)]; tensor doubled_9_cast_fp16 = concat(axis = var_1225, interleave = doubled_9_interleave_0, values = (hidden_states_9_cast_fp16, var_1227_cast_fp16))[name = string("doubled_9_cast_fp16")]; tensor out_5_axes_0 = const()[name = string("out_5_axes_0"), val = tensor([1])]; tensor out_5_gamma_0_to_fp16 = const()[name = string("out_5_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077549568)))]; fp16 var_1237_to_fp16 = const()[name = string("op_1237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_5_cast_fp16 = layer_norm(axes = out_5_axes_0, epsilon = var_1237_to_fp16, gamma = out_5_gamma_0_to_fp16, x = doubled_9_cast_fp16)[name = string("out_5_cast_fp16")]; tensor var_1248_split_sizes_0 = const()[name = string("op_1248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1248_axis_0 = const()[name = string("op_1248_axis_0"), val = int32(1)]; tensor var_1248_cast_fp16_0, tensor var_1248_cast_fp16_1 = split(axis = var_1248_axis_0, split_sizes = var_1248_split_sizes_0, x = out_5_cast_fp16)[name = string("op_1248_cast_fp16")]; tensor layers_1_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077557824)))]; tensor query_states_7_strides_0 = const()[name = string("query_states_7_strides_0"), val = tensor([1, 1])]; string query_states_7_pad_type_0 = const()[name = string("query_states_7_pad_type_0"), val = string("valid")]; tensor query_states_7_pad_0 = const()[name = string("query_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_7_dilations_0 = const()[name = string("query_states_7_dilations_0"), val = tensor([1, 1])]; int32 query_states_7_groups_0 = const()[name = string("query_states_7_groups_0"), val = int32(1)]; tensor query_states_7_cast_fp16 = conv(dilations = query_states_7_dilations_0, groups = query_states_7_groups_0, pad = query_states_7_pad_0, pad_type = query_states_7_pad_type_0, strides = query_states_7_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("query_states_7_cast_fp16")]; tensor layers_1_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1085946496)))]; tensor key_states_11_strides_0 = const()[name = string("key_states_11_strides_0"), val = tensor([1, 1])]; string key_states_11_pad_type_0 = const()[name = string("key_states_11_pad_type_0"), val = string("valid")]; tensor key_states_11_pad_0 = const()[name = string("key_states_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_11_dilations_0 = const()[name = string("key_states_11_dilations_0"), val = tensor([1, 1])]; int32 key_states_11_groups_0 = const()[name = string("key_states_11_groups_0"), val = int32(1)]; tensor key_states_11_cast_fp16 = conv(dilations = key_states_11_dilations_0, groups = key_states_11_groups_0, pad = key_states_11_pad_0, pad_type = key_states_11_pad_type_0, strides = key_states_11_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("key_states_11_cast_fp16")]; tensor value_states_7_strides_0 = const()[name = string("value_states_7_strides_0"), val = tensor([1, 1])]; string value_states_7_pad_type_0 = const()[name = string("value_states_7_pad_type_0"), val = string("valid")]; tensor value_states_7_pad_0 = const()[name = string("value_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_7_dilations_0 = const()[name = string("value_states_7_dilations_0"), val = tensor([1, 1])]; int32 value_states_7_groups_0 = const()[name = string("value_states_7_groups_0"), val = int32(1)]; tensor value_states_7_cast_fp16 = conv(dilations = value_states_7_dilations_0, groups = value_states_7_groups_0, pad = value_states_7_pad_0, pad_type = value_states_7_pad_type_0, strides = value_states_7_strides_0, weight = layers_1_self_attn_v_proj_weight_cast_fp16, x = var_1248_cast_fp16_0)[name = string("value_states_7_cast_fp16")]; tensor concat_12x = const()[name = string("concat_12x"), val = tensor([1, 16, 128, -1])]; tensor x_11_cast_fp16 = reshape(shape = concat_12x, x = query_states_7_cast_fp16)[name = string("x_11_cast_fp16")]; tensor concat_13x = const()[name = string("concat_13x"), val = tensor([1, 2, 128, -1])]; tensor var_1305_cast_fp16 = reshape(shape = concat_13x, x = key_states_11_cast_fp16)[name = string("op_1305_cast_fp16")]; tensor concat_14x = const()[name = string("concat_14x"), val = tensor([1, 2, 128, -1])]; tensor var_1312_cast_fp16 = reshape(shape = concat_14x, x = value_states_7_cast_fp16)[name = string("op_1312_cast_fp16")]; tensor var_1316_cast_fp16 = mul(x = x_11_cast_fp16, y = var_869_cast_fp16)[name = string("op_1316_cast_fp16")]; tensor var_1317_split_sizes_0 = const()[name = string("op_1317_split_sizes_0"), val = tensor([64, 64])]; int32 var_1317_axis_0 = const()[name = string("op_1317_axis_0"), val = int32(-2)]; tensor var_1317_cast_fp16_0, tensor var_1317_cast_fp16_1 = split(axis = var_1317_axis_0, split_sizes = var_1317_split_sizes_0, x = x_11_cast_fp16)[name = string("op_1317_cast_fp16")]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1319_cast_fp16 = mul(x = var_1317_cast_fp16_1, y = const_12_promoted_to_fp16)[name = string("op_1319_cast_fp16")]; int32 var_1321 = const()[name = string("op_1321"), val = int32(-2)]; bool var_1322_interleave_0 = const()[name = string("op_1322_interleave_0"), val = bool(false)]; tensor var_1322_cast_fp16 = concat(axis = var_1321, interleave = var_1322_interleave_0, values = (var_1319_cast_fp16, var_1317_cast_fp16_0))[name = string("op_1322_cast_fp16")]; tensor var_1323_cast_fp16 = mul(x = var_1322_cast_fp16, y = var_878_cast_fp16)[name = string("op_1323_cast_fp16")]; tensor query_states_9_cast_fp16 = add(x = var_1316_cast_fp16, y = var_1323_cast_fp16)[name = string("query_states_9_cast_fp16")]; tensor var_1329_cast_fp16 = mul(x = var_1305_cast_fp16, y = var_869_cast_fp16)[name = string("op_1329_cast_fp16")]; tensor var_1330_split_sizes_0 = const()[name = string("op_1330_split_sizes_0"), val = tensor([64, 64])]; int32 var_1330_axis_0 = const()[name = string("op_1330_axis_0"), val = int32(-2)]; tensor var_1330_cast_fp16_0, tensor var_1330_cast_fp16_1 = split(axis = var_1330_axis_0, split_sizes = var_1330_split_sizes_0, x = var_1305_cast_fp16)[name = string("op_1330_cast_fp16")]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1332_cast_fp16 = mul(x = var_1330_cast_fp16_1, y = const_13_promoted_to_fp16)[name = string("op_1332_cast_fp16")]; int32 var_1334 = const()[name = string("op_1334"), val = int32(-2)]; bool var_1335_interleave_0 = const()[name = string("op_1335_interleave_0"), val = bool(false)]; tensor var_1335_cast_fp16 = concat(axis = var_1334, interleave = var_1335_interleave_0, values = (var_1332_cast_fp16, var_1330_cast_fp16_0))[name = string("op_1335_cast_fp16")]; tensor var_1336_cast_fp16 = mul(x = var_1335_cast_fp16, y = var_878_cast_fp16)[name = string("op_1336_cast_fp16")]; tensor key_states_15_cast_fp16 = add(x = var_1329_cast_fp16, y = var_1336_cast_fp16)[name = string("key_states_15_cast_fp16")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; int32 concat_17_axis_0 = const()[name = string("concat_17_axis_0"), val = int32(0)]; bool concat_17_interleave_0 = const()[name = string("concat_17_interleave_0"), val = bool(false)]; tensor concat_17 = concat(axis = concat_17_axis_0, interleave = concat_17_interleave_0, values = (expand_dims_12, expand_dims_13, position_id, expand_dims_15))[name = string("concat_17")]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; tensor concat_18_values1_0 = const()[name = string("concat_18_values1_0"), val = tensor([0])]; tensor concat_18_values3_0 = const()[name = string("concat_18_values3_0"), val = tensor([0])]; int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_16, concat_18_values1_0, cache_position_end, concat_18_values3_0))[name = string("concat_18")]; tensor key_states_17_perm_0 = const()[name = string("key_states_17_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_2_stride_0 = const()[name = string("key_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_17_cast_fp16 = transpose(perm = key_states_17_perm_0, x = key_states_15_cast_fp16)[name = string("transpose_166")]; tensor key_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = key_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = key_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_2_squeeze_mask_0, stride = key_cache_internal_tensor_assign_2_stride_0, update = key_states_17_cast_fp16, x = coreml_update_state_56)[name = string("key_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_2_cast_fp16, input = key_cache)[name = string("coreml_update_state_58_write_state")]; tensor coreml_update_state_58 = read_state(input = key_cache)[name = string("coreml_update_state_58")]; tensor value_states_9_perm_0 = const()[name = string("value_states_9_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_2_stride_0 = const()[name = string("value_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_9_cast_fp16 = transpose(perm = value_states_9_perm_0, x = var_1312_cast_fp16)[name = string("transpose_165")]; tensor value_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = value_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = value_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_2_squeeze_mask_0, stride = value_cache_internal_tensor_assign_2_stride_0, update = value_states_9_cast_fp16, x = coreml_update_state_57)[name = string("value_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_2_cast_fp16, input = value_cache)[name = string("coreml_update_state_59_write_state")]; tensor coreml_update_state_59 = read_state(input = value_cache)[name = string("coreml_update_state_59")]; tensor var_1406_begin_0 = const()[name = string("op_1406_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1406_end_0 = const()[name = string("op_1406_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1406_end_mask_0 = const()[name = string("op_1406_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1406_cast_fp16 = slice_by_index(begin = var_1406_begin_0, end = var_1406_end_0, end_mask = var_1406_end_mask_0, x = coreml_update_state_58)[name = string("op_1406_cast_fp16")]; tensor tile_2 = const()[name = string("tile_2"), val = tensor([1, 1])]; int32 var_1409_axis_0 = const()[name = string("op_1409_axis_0"), val = int32(1)]; tensor var_1409_cast_fp16_0, tensor var_1409_cast_fp16_1 = split(axis = var_1409_axis_0, split_sizes = tile_2, x = var_1406_cast_fp16)[name = string("op_1409_cast_fp16")]; tensor var_1416_begin_0 = const()[name = string("op_1416_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1416_end_0 = const()[name = string("op_1416_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1416_end_mask_0 = const()[name = string("op_1416_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1416_cast_fp16 = slice_by_index(begin = var_1416_begin_0, end = var_1416_end_0, end_mask = var_1416_end_mask_0, x = coreml_update_state_59)[name = string("op_1416_cast_fp16")]; tensor tile_3 = const()[name = string("tile_3"), val = tensor([1, 1])]; int32 var_1419_axis_0 = const()[name = string("op_1419_axis_0"), val = int32(1)]; tensor var_1419_cast_fp16_0, tensor var_1419_cast_fp16_1 = split(axis = var_1419_axis_0, split_sizes = tile_3, x = var_1416_cast_fp16)[name = string("op_1419_cast_fp16")]; tensor var_1422_split_sizes_0 = const()[name = string("op_1422_split_sizes_0"), val = tensor([8, 8])]; int32 var_1422_axis_0 = const()[name = string("op_1422_axis_0"), val = int32(1)]; tensor var_1422_0, tensor var_1422_1 = split(axis = var_1422_axis_0, split_sizes = var_1422_split_sizes_0, x = query_states_9_cast_fp16)[name = string("op_1422")]; bool attn_weights_17_transpose_x_0 = const()[name = string("attn_weights_17_transpose_x_0"), val = bool(false)]; bool attn_weights_17_transpose_y_0 = const()[name = string("attn_weights_17_transpose_y_0"), val = bool(false)]; tensor attn_weights_17_cast_fp16 = matmul(transpose_x = attn_weights_17_transpose_x_0, transpose_y = attn_weights_17_transpose_y_0, x = var_1409_cast_fp16_0, y = var_1422_0)[name = string("attn_weights_17_cast_fp16")]; fp16 var_1425_to_fp16 = const()[name = string("op_1425_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_19_cast_fp16 = mul(x = attn_weights_17_cast_fp16, y = var_1425_to_fp16)[name = string("attn_weights_19_cast_fp16")]; tensor attn_weights_21_cast_fp16 = add(x = attn_weights_19_cast_fp16, y = attn_mask_1)[name = string("attn_weights_21_cast_fp16")]; int32 var_1429 = const()[name = string("op_1429"), val = int32(-2)]; tensor attn_weights_23_cast_fp16 = softmax(axis = var_1429, x = attn_weights_21_cast_fp16)[name = string("attn_weights_23_cast_fp16")]; bool var_1435_transpose_x_1 = const()[name = string("op_1435_transpose_x_1"), val = bool(true)]; bool var_1435_transpose_y_1 = const()[name = string("op_1435_transpose_y_1"), val = bool(false)]; tensor var_1435_cast_fp16 = matmul(transpose_x = var_1435_transpose_x_1, transpose_y = var_1435_transpose_y_1, x = attn_weights_23_cast_fp16, y = var_1419_cast_fp16_0)[name = string("op_1435_cast_fp16")]; bool attn_weights_25_transpose_x_0 = const()[name = string("attn_weights_25_transpose_x_0"), val = bool(false)]; bool attn_weights_25_transpose_y_0 = const()[name = string("attn_weights_25_transpose_y_0"), val = bool(false)]; tensor attn_weights_25_cast_fp16 = matmul(transpose_x = attn_weights_25_transpose_x_0, transpose_y = attn_weights_25_transpose_y_0, x = var_1409_cast_fp16_1, y = var_1422_1)[name = string("attn_weights_25_cast_fp16")]; fp16 var_1437_to_fp16 = const()[name = string("op_1437_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_27_cast_fp16 = mul(x = attn_weights_25_cast_fp16, y = var_1437_to_fp16)[name = string("attn_weights_27_cast_fp16")]; tensor attn_weights_29_cast_fp16 = add(x = attn_weights_27_cast_fp16, y = attn_mask_1)[name = string("attn_weights_29_cast_fp16")]; int32 var_1441 = const()[name = string("op_1441"), val = int32(-2)]; tensor attn_weights_31_cast_fp16 = softmax(axis = var_1441, x = attn_weights_29_cast_fp16)[name = string("attn_weights_31_cast_fp16")]; bool attn_output_9_transpose_x_1 = const()[name = string("attn_output_9_transpose_x_1"), val = bool(true)]; bool attn_output_9_transpose_y_1 = const()[name = string("attn_output_9_transpose_y_1"), val = bool(false)]; tensor attn_output_9_cast_fp16 = matmul(transpose_x = attn_output_9_transpose_x_1, transpose_y = attn_output_9_transpose_y_1, x = attn_weights_31_cast_fp16, y = var_1419_cast_fp16_1)[name = string("attn_output_9_cast_fp16")]; int32 var_1449 = const()[name = string("op_1449"), val = int32(1)]; bool attn_output_11_interleave_0 = const()[name = string("attn_output_11_interleave_0"), val = bool(false)]; tensor attn_output_11_cast_fp16 = concat(axis = var_1449, interleave = attn_output_11_interleave_0, values = (var_1435_cast_fp16, attn_output_9_cast_fp16))[name = string("attn_output_11_cast_fp16")]; tensor var_1453_perm_0 = const()[name = string("op_1453_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_23x = const()[name = string("concat_23x"), val = tensor([1, 2048, 1, -1])]; tensor var_1453_cast_fp16 = transpose(perm = var_1453_perm_0, x = attn_output_11_cast_fp16)[name = string("transpose_164")]; tensor attn_output_15_cast_fp16 = reshape(shape = concat_23x, x = var_1453_cast_fp16)[name = string("attn_output_15_cast_fp16")]; tensor layers_1_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086995136)))]; tensor hidden_states_13_strides_0 = const()[name = string("hidden_states_13_strides_0"), val = tensor([1, 1])]; string hidden_states_13_pad_type_0 = const()[name = string("hidden_states_13_pad_type_0"), val = string("valid")]; tensor hidden_states_13_pad_0 = const()[name = string("hidden_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_13_dilations_0 = const()[name = string("hidden_states_13_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_13_groups_0 = const()[name = string("hidden_states_13_groups_0"), val = int32(1)]; tensor hidden_states_13_cast_fp16 = conv(dilations = hidden_states_13_dilations_0, groups = hidden_states_13_groups_0, pad = hidden_states_13_pad_0, pad_type = hidden_states_13_pad_type_0, strides = hidden_states_13_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16, x = attn_output_15_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; tensor hidden_states_15_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = hidden_states_13_cast_fp16)[name = string("hidden_states_15_cast_fp16")]; fp16 const_18_promoted_to_fp16 = const()[name = string("const_18_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1486_cast_fp16 = mul(x = hidden_states_15_cast_fp16, y = const_18_promoted_to_fp16)[name = string("op_1486_cast_fp16")]; int32 var_1484 = const()[name = string("op_1484"), val = int32(1)]; bool doubled_13_interleave_0 = const()[name = string("doubled_13_interleave_0"), val = bool(false)]; tensor doubled_13_cast_fp16 = concat(axis = var_1484, interleave = doubled_13_interleave_0, values = (hidden_states_15_cast_fp16, var_1486_cast_fp16))[name = string("doubled_13_cast_fp16")]; tensor out_7_axes_0 = const()[name = string("out_7_axes_0"), val = tensor([1])]; tensor out_7_gamma_0_to_fp16 = const()[name = string("out_7_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095383808)))]; fp16 var_1496_to_fp16 = const()[name = string("op_1496_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_7_cast_fp16 = layer_norm(axes = out_7_axes_0, epsilon = var_1496_to_fp16, gamma = out_7_gamma_0_to_fp16, x = doubled_13_cast_fp16)[name = string("out_7_cast_fp16")]; tensor var_1507_split_sizes_0 = const()[name = string("op_1507_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1507_axis_0 = const()[name = string("op_1507_axis_0"), val = int32(1)]; tensor var_1507_cast_fp16_0, tensor var_1507_cast_fp16_1 = split(axis = var_1507_axis_0, split_sizes = var_1507_split_sizes_0, x = out_7_cast_fp16)[name = string("op_1507_cast_fp16")]; tensor layers_1_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095392064)))]; tensor input_3_strides_0 = const()[name = string("input_3_strides_0"), val = tensor([1, 1])]; string input_3_pad_type_0 = const()[name = string("input_3_pad_type_0"), val = string("valid")]; tensor input_3_pad_0 = const()[name = string("input_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_3_dilations_0 = const()[name = string("input_3_dilations_0"), val = tensor([1, 1])]; int32 input_3_groups_0 = const()[name = string("input_3_groups_0"), val = int32(1)]; tensor input_3_cast_fp16 = conv(dilations = input_3_dilations_0, groups = input_3_groups_0, pad = input_3_pad_0, pad_type = input_3_pad_type_0, strides = input_3_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16, x = var_1507_cast_fp16_0)[name = string("input_3_cast_fp16")]; tensor var_1524_cast_fp16 = silu(x = input_3_cast_fp16)[name = string("op_1524_cast_fp16")]; tensor var_1530_strides_0 = const()[name = string("op_1530_strides_0"), val = tensor([1, 1])]; string var_1530_pad_type_0 = const()[name = string("op_1530_pad_type_0"), val = string("valid")]; tensor var_1530_pad_0 = const()[name = string("op_1530_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1530_dilations_0 = const()[name = string("op_1530_dilations_0"), val = tensor([1, 1])]; int32 var_1530_groups_0 = const()[name = string("op_1530_groups_0"), val = int32(1)]; tensor var_1530_cast_fp16 = conv(dilations = var_1530_dilations_0, groups = var_1530_groups_0, pad = var_1530_pad_0, pad_type = var_1530_pad_type_0, strides = var_1530_strides_0, weight = layers_1_mlp_up_proj_weight_cast_fp16, x = var_1507_cast_fp16_0)[name = string("op_1530_cast_fp16")]; tensor x_19_cast_fp16 = mul(x = var_1524_cast_fp16, y = var_1530_cast_fp16)[name = string("x_19_cast_fp16")]; tensor layers_1_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1120557952)))]; tensor hidden_states_17_strides_0 = const()[name = string("hidden_states_17_strides_0"), val = tensor([1, 1])]; string hidden_states_17_pad_type_0 = const()[name = string("hidden_states_17_pad_type_0"), val = string("valid")]; tensor hidden_states_17_pad_0 = const()[name = string("hidden_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_17_dilations_0 = const()[name = string("hidden_states_17_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_17_groups_0 = const()[name = string("hidden_states_17_groups_0"), val = int32(1)]; tensor hidden_states_17_cast_fp16 = conv(dilations = hidden_states_17_dilations_0, groups = hidden_states_17_groups_0, pad = hidden_states_17_pad_0, pad_type = hidden_states_17_pad_type_0, strides = hidden_states_17_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16, x = x_19_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; tensor hidden_states_19_cast_fp16 = add(x = hidden_states_15_cast_fp16, y = hidden_states_17_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1548_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_1548_cast_fp16")]; int32 var_1546 = const()[name = string("op_1546"), val = int32(1)]; bool doubled_17_interleave_0 = const()[name = string("doubled_17_interleave_0"), val = bool(false)]; tensor doubled_17_cast_fp16 = concat(axis = var_1546, interleave = doubled_17_interleave_0, values = (hidden_states_19_cast_fp16, var_1548_cast_fp16))[name = string("doubled_17_cast_fp16")]; tensor out_9_axes_0 = const()[name = string("out_9_axes_0"), val = tensor([1])]; tensor out_9_gamma_0_to_fp16 = const()[name = string("out_9_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145723840)))]; fp16 var_1558_to_fp16 = const()[name = string("op_1558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_9_cast_fp16 = layer_norm(axes = out_9_axes_0, epsilon = var_1558_to_fp16, gamma = out_9_gamma_0_to_fp16, x = doubled_17_cast_fp16)[name = string("out_9_cast_fp16")]; tensor var_1569_split_sizes_0 = const()[name = string("op_1569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1569_axis_0 = const()[name = string("op_1569_axis_0"), val = int32(1)]; tensor var_1569_cast_fp16_0, tensor var_1569_cast_fp16_1 = split(axis = var_1569_axis_0, split_sizes = var_1569_split_sizes_0, x = out_9_cast_fp16)[name = string("op_1569_cast_fp16")]; tensor layers_2_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145732096)))]; tensor query_states_13_strides_0 = const()[name = string("query_states_13_strides_0"), val = tensor([1, 1])]; string query_states_13_pad_type_0 = const()[name = string("query_states_13_pad_type_0"), val = string("valid")]; tensor query_states_13_pad_0 = const()[name = string("query_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_13_dilations_0 = const()[name = string("query_states_13_dilations_0"), val = tensor([1, 1])]; int32 query_states_13_groups_0 = const()[name = string("query_states_13_groups_0"), val = int32(1)]; tensor query_states_13_cast_fp16 = conv(dilations = query_states_13_dilations_0, groups = query_states_13_groups_0, pad = query_states_13_pad_0, pad_type = query_states_13_pad_type_0, strides = query_states_13_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("query_states_13_cast_fp16")]; tensor layers_2_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1154120768)))]; tensor key_states_21_strides_0 = const()[name = string("key_states_21_strides_0"), val = tensor([1, 1])]; string key_states_21_pad_type_0 = const()[name = string("key_states_21_pad_type_0"), val = string("valid")]; tensor key_states_21_pad_0 = const()[name = string("key_states_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_21_dilations_0 = const()[name = string("key_states_21_dilations_0"), val = tensor([1, 1])]; int32 key_states_21_groups_0 = const()[name = string("key_states_21_groups_0"), val = int32(1)]; tensor key_states_21_cast_fp16 = conv(dilations = key_states_21_dilations_0, groups = key_states_21_groups_0, pad = key_states_21_pad_0, pad_type = key_states_21_pad_type_0, strides = key_states_21_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("key_states_21_cast_fp16")]; tensor value_states_13_strides_0 = const()[name = string("value_states_13_strides_0"), val = tensor([1, 1])]; string value_states_13_pad_type_0 = const()[name = string("value_states_13_pad_type_0"), val = string("valid")]; tensor value_states_13_pad_0 = const()[name = string("value_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_13_dilations_0 = const()[name = string("value_states_13_dilations_0"), val = tensor([1, 1])]; int32 value_states_13_groups_0 = const()[name = string("value_states_13_groups_0"), val = int32(1)]; tensor value_states_13_cast_fp16 = conv(dilations = value_states_13_dilations_0, groups = value_states_13_groups_0, pad = value_states_13_pad_0, pad_type = value_states_13_pad_type_0, strides = value_states_13_strides_0, weight = layers_2_self_attn_v_proj_weight_cast_fp16, x = var_1569_cast_fp16_0)[name = string("value_states_13_cast_fp16")]; tensor concat_24x = const()[name = string("concat_24x"), val = tensor([1, 16, 128, -1])]; tensor x_21_cast_fp16 = reshape(shape = concat_24x, x = query_states_13_cast_fp16)[name = string("x_21_cast_fp16")]; tensor concat_25x = const()[name = string("concat_25x"), val = tensor([1, 2, 128, -1])]; tensor var_1626_cast_fp16 = reshape(shape = concat_25x, x = key_states_21_cast_fp16)[name = string("op_1626_cast_fp16")]; tensor concat_26x = const()[name = string("concat_26x"), val = tensor([1, 2, 128, -1])]; tensor var_1633_cast_fp16 = reshape(shape = concat_26x, x = value_states_13_cast_fp16)[name = string("op_1633_cast_fp16")]; tensor var_1637_cast_fp16 = mul(x = x_21_cast_fp16, y = var_869_cast_fp16)[name = string("op_1637_cast_fp16")]; tensor var_1638_split_sizes_0 = const()[name = string("op_1638_split_sizes_0"), val = tensor([64, 64])]; int32 var_1638_axis_0 = const()[name = string("op_1638_axis_0"), val = int32(-2)]; tensor var_1638_cast_fp16_0, tensor var_1638_cast_fp16_1 = split(axis = var_1638_axis_0, split_sizes = var_1638_split_sizes_0, x = x_21_cast_fp16)[name = string("op_1638_cast_fp16")]; fp16 const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1640_cast_fp16 = mul(x = var_1638_cast_fp16_1, y = const_22_promoted_to_fp16)[name = string("op_1640_cast_fp16")]; int32 var_1642 = const()[name = string("op_1642"), val = int32(-2)]; bool var_1643_interleave_0 = const()[name = string("op_1643_interleave_0"), val = bool(false)]; tensor var_1643_cast_fp16 = concat(axis = var_1642, interleave = var_1643_interleave_0, values = (var_1640_cast_fp16, var_1638_cast_fp16_0))[name = string("op_1643_cast_fp16")]; tensor var_1644_cast_fp16 = mul(x = var_1643_cast_fp16, y = var_878_cast_fp16)[name = string("op_1644_cast_fp16")]; tensor query_states_15_cast_fp16 = add(x = var_1637_cast_fp16, y = var_1644_cast_fp16)[name = string("query_states_15_cast_fp16")]; tensor var_1650_cast_fp16 = mul(x = var_1626_cast_fp16, y = var_869_cast_fp16)[name = string("op_1650_cast_fp16")]; tensor var_1651_split_sizes_0 = const()[name = string("op_1651_split_sizes_0"), val = tensor([64, 64])]; int32 var_1651_axis_0 = const()[name = string("op_1651_axis_0"), val = int32(-2)]; tensor var_1651_cast_fp16_0, tensor var_1651_cast_fp16_1 = split(axis = var_1651_axis_0, split_sizes = var_1651_split_sizes_0, x = var_1626_cast_fp16)[name = string("op_1651_cast_fp16")]; fp16 const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1653_cast_fp16 = mul(x = var_1651_cast_fp16_1, y = const_23_promoted_to_fp16)[name = string("op_1653_cast_fp16")]; int32 var_1655 = const()[name = string("op_1655"), val = int32(-2)]; bool var_1656_interleave_0 = const()[name = string("op_1656_interleave_0"), val = bool(false)]; tensor var_1656_cast_fp16 = concat(axis = var_1655, interleave = var_1656_interleave_0, values = (var_1653_cast_fp16, var_1651_cast_fp16_0))[name = string("op_1656_cast_fp16")]; tensor var_1657_cast_fp16 = mul(x = var_1656_cast_fp16, y = var_878_cast_fp16)[name = string("op_1657_cast_fp16")]; tensor key_states_25_cast_fp16 = add(x = var_1650_cast_fp16, y = var_1657_cast_fp16)[name = string("key_states_25_cast_fp16")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; int32 concat_29_axis_0 = const()[name = string("concat_29_axis_0"), val = int32(0)]; bool concat_29_interleave_0 = const()[name = string("concat_29_interleave_0"), val = bool(false)]; tensor concat_29 = concat(axis = concat_29_axis_0, interleave = concat_29_interleave_0, values = (expand_dims_24, expand_dims_25, position_id, expand_dims_27))[name = string("concat_29")]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; tensor concat_30_values1_0 = const()[name = string("concat_30_values1_0"), val = tensor([0])]; tensor concat_30_values3_0 = const()[name = string("concat_30_values3_0"), val = tensor([0])]; int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_28, concat_30_values1_0, cache_position_end, concat_30_values3_0))[name = string("concat_30")]; tensor key_states_27_perm_0 = const()[name = string("key_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_3_stride_0 = const()[name = string("key_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_27_cast_fp16 = transpose(perm = key_states_27_perm_0, x = key_states_25_cast_fp16)[name = string("transpose_163")]; tensor key_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = key_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = key_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_3_squeeze_mask_0, stride = key_cache_internal_tensor_assign_3_stride_0, update = key_states_27_cast_fp16, x = coreml_update_state_58)[name = string("key_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_3_cast_fp16, input = key_cache)[name = string("coreml_update_state_60_write_state")]; tensor coreml_update_state_60 = read_state(input = key_cache)[name = string("coreml_update_state_60")]; tensor value_states_15_perm_0 = const()[name = string("value_states_15_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_3_stride_0 = const()[name = string("value_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_15_cast_fp16 = transpose(perm = value_states_15_perm_0, x = var_1633_cast_fp16)[name = string("transpose_162")]; tensor value_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = value_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = value_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_3_squeeze_mask_0, stride = value_cache_internal_tensor_assign_3_stride_0, update = value_states_15_cast_fp16, x = coreml_update_state_59)[name = string("value_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_3_cast_fp16, input = value_cache)[name = string("coreml_update_state_61_write_state")]; tensor coreml_update_state_61 = read_state(input = value_cache)[name = string("coreml_update_state_61")]; tensor var_1727_begin_0 = const()[name = string("op_1727_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1727_end_0 = const()[name = string("op_1727_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1727_end_mask_0 = const()[name = string("op_1727_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1727_cast_fp16 = slice_by_index(begin = var_1727_begin_0, end = var_1727_end_0, end_mask = var_1727_end_mask_0, x = coreml_update_state_60)[name = string("op_1727_cast_fp16")]; tensor tile_4 = const()[name = string("tile_4"), val = tensor([1, 1])]; int32 var_1730_axis_0 = const()[name = string("op_1730_axis_0"), val = int32(1)]; tensor var_1730_cast_fp16_0, tensor var_1730_cast_fp16_1 = split(axis = var_1730_axis_0, split_sizes = tile_4, x = var_1727_cast_fp16)[name = string("op_1730_cast_fp16")]; tensor var_1737_begin_0 = const()[name = string("op_1737_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1737_end_0 = const()[name = string("op_1737_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1737_end_mask_0 = const()[name = string("op_1737_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1737_cast_fp16 = slice_by_index(begin = var_1737_begin_0, end = var_1737_end_0, end_mask = var_1737_end_mask_0, x = coreml_update_state_61)[name = string("op_1737_cast_fp16")]; tensor tile_5 = const()[name = string("tile_5"), val = tensor([1, 1])]; int32 var_1740_axis_0 = const()[name = string("op_1740_axis_0"), val = int32(1)]; tensor var_1740_cast_fp16_0, tensor var_1740_cast_fp16_1 = split(axis = var_1740_axis_0, split_sizes = tile_5, x = var_1737_cast_fp16)[name = string("op_1740_cast_fp16")]; tensor var_1743_split_sizes_0 = const()[name = string("op_1743_split_sizes_0"), val = tensor([8, 8])]; int32 var_1743_axis_0 = const()[name = string("op_1743_axis_0"), val = int32(1)]; tensor var_1743_0, tensor var_1743_1 = split(axis = var_1743_axis_0, split_sizes = var_1743_split_sizes_0, x = query_states_15_cast_fp16)[name = string("op_1743")]; bool attn_weights_33_transpose_x_0 = const()[name = string("attn_weights_33_transpose_x_0"), val = bool(false)]; bool attn_weights_33_transpose_y_0 = const()[name = string("attn_weights_33_transpose_y_0"), val = bool(false)]; tensor attn_weights_33_cast_fp16 = matmul(transpose_x = attn_weights_33_transpose_x_0, transpose_y = attn_weights_33_transpose_y_0, x = var_1730_cast_fp16_0, y = var_1743_0)[name = string("attn_weights_33_cast_fp16")]; fp16 var_1746_to_fp16 = const()[name = string("op_1746_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_35_cast_fp16 = mul(x = attn_weights_33_cast_fp16, y = var_1746_to_fp16)[name = string("attn_weights_35_cast_fp16")]; tensor attn_weights_37_cast_fp16 = add(x = attn_weights_35_cast_fp16, y = attn_mask_1)[name = string("attn_weights_37_cast_fp16")]; int32 var_1750 = const()[name = string("op_1750"), val = int32(-2)]; tensor attn_weights_39_cast_fp16 = softmax(axis = var_1750, x = attn_weights_37_cast_fp16)[name = string("attn_weights_39_cast_fp16")]; bool var_1756_transpose_x_1 = const()[name = string("op_1756_transpose_x_1"), val = bool(true)]; bool var_1756_transpose_y_1 = const()[name = string("op_1756_transpose_y_1"), val = bool(false)]; tensor var_1756_cast_fp16 = matmul(transpose_x = var_1756_transpose_x_1, transpose_y = var_1756_transpose_y_1, x = attn_weights_39_cast_fp16, y = var_1740_cast_fp16_0)[name = string("op_1756_cast_fp16")]; bool attn_weights_41_transpose_x_0 = const()[name = string("attn_weights_41_transpose_x_0"), val = bool(false)]; bool attn_weights_41_transpose_y_0 = const()[name = string("attn_weights_41_transpose_y_0"), val = bool(false)]; tensor attn_weights_41_cast_fp16 = matmul(transpose_x = attn_weights_41_transpose_x_0, transpose_y = attn_weights_41_transpose_y_0, x = var_1730_cast_fp16_1, y = var_1743_1)[name = string("attn_weights_41_cast_fp16")]; fp16 var_1758_to_fp16 = const()[name = string("op_1758_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_43_cast_fp16 = mul(x = attn_weights_41_cast_fp16, y = var_1758_to_fp16)[name = string("attn_weights_43_cast_fp16")]; tensor attn_weights_45_cast_fp16 = add(x = attn_weights_43_cast_fp16, y = attn_mask_1)[name = string("attn_weights_45_cast_fp16")]; int32 var_1762 = const()[name = string("op_1762"), val = int32(-2)]; tensor attn_weights_47_cast_fp16 = softmax(axis = var_1762, x = attn_weights_45_cast_fp16)[name = string("attn_weights_47_cast_fp16")]; bool attn_output_17_transpose_x_1 = const()[name = string("attn_output_17_transpose_x_1"), val = bool(true)]; bool attn_output_17_transpose_y_1 = const()[name = string("attn_output_17_transpose_y_1"), val = bool(false)]; tensor attn_output_17_cast_fp16 = matmul(transpose_x = attn_output_17_transpose_x_1, transpose_y = attn_output_17_transpose_y_1, x = attn_weights_47_cast_fp16, y = var_1740_cast_fp16_1)[name = string("attn_output_17_cast_fp16")]; int32 var_1770 = const()[name = string("op_1770"), val = int32(1)]; bool attn_output_19_interleave_0 = const()[name = string("attn_output_19_interleave_0"), val = bool(false)]; tensor attn_output_19_cast_fp16 = concat(axis = var_1770, interleave = attn_output_19_interleave_0, values = (var_1756_cast_fp16, attn_output_17_cast_fp16))[name = string("attn_output_19_cast_fp16")]; tensor var_1774_perm_0 = const()[name = string("op_1774_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_35x = const()[name = string("concat_35x"), val = tensor([1, 2048, 1, -1])]; tensor var_1774_cast_fp16 = transpose(perm = var_1774_perm_0, x = attn_output_19_cast_fp16)[name = string("transpose_161")]; tensor attn_output_23_cast_fp16 = reshape(shape = concat_35x, x = var_1774_cast_fp16)[name = string("attn_output_23_cast_fp16")]; tensor layers_2_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1155169408)))]; tensor hidden_states_23_strides_0 = const()[name = string("hidden_states_23_strides_0"), val = tensor([1, 1])]; string hidden_states_23_pad_type_0 = const()[name = string("hidden_states_23_pad_type_0"), val = string("valid")]; tensor hidden_states_23_pad_0 = const()[name = string("hidden_states_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_23_dilations_0 = const()[name = string("hidden_states_23_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_23_groups_0 = const()[name = string("hidden_states_23_groups_0"), val = int32(1)]; tensor hidden_states_23_cast_fp16 = conv(dilations = hidden_states_23_dilations_0, groups = hidden_states_23_groups_0, pad = hidden_states_23_pad_0, pad_type = hidden_states_23_pad_type_0, strides = hidden_states_23_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16, x = attn_output_23_cast_fp16)[name = string("hidden_states_23_cast_fp16")]; tensor hidden_states_25_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = hidden_states_23_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1807_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_1807_cast_fp16")]; int32 var_1805 = const()[name = string("op_1805"), val = int32(1)]; bool doubled_21_interleave_0 = const()[name = string("doubled_21_interleave_0"), val = bool(false)]; tensor doubled_21_cast_fp16 = concat(axis = var_1805, interleave = doubled_21_interleave_0, values = (hidden_states_25_cast_fp16, var_1807_cast_fp16))[name = string("doubled_21_cast_fp16")]; tensor out_11_axes_0 = const()[name = string("out_11_axes_0"), val = tensor([1])]; tensor out_11_gamma_0_to_fp16 = const()[name = string("out_11_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163558080)))]; fp16 var_1817_to_fp16 = const()[name = string("op_1817_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_11_cast_fp16 = layer_norm(axes = out_11_axes_0, epsilon = var_1817_to_fp16, gamma = out_11_gamma_0_to_fp16, x = doubled_21_cast_fp16)[name = string("out_11_cast_fp16")]; tensor var_1828_split_sizes_0 = const()[name = string("op_1828_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1828_axis_0 = const()[name = string("op_1828_axis_0"), val = int32(1)]; tensor var_1828_cast_fp16_0, tensor var_1828_cast_fp16_1 = split(axis = var_1828_axis_0, split_sizes = var_1828_split_sizes_0, x = out_11_cast_fp16)[name = string("op_1828_cast_fp16")]; tensor layers_2_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163566336)))]; tensor input_5_strides_0 = const()[name = string("input_5_strides_0"), val = tensor([1, 1])]; string input_5_pad_type_0 = const()[name = string("input_5_pad_type_0"), val = string("valid")]; tensor input_5_pad_0 = const()[name = string("input_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_5_dilations_0 = const()[name = string("input_5_dilations_0"), val = tensor([1, 1])]; int32 input_5_groups_0 = const()[name = string("input_5_groups_0"), val = int32(1)]; tensor input_5_cast_fp16 = conv(dilations = input_5_dilations_0, groups = input_5_groups_0, pad = input_5_pad_0, pad_type = input_5_pad_type_0, strides = input_5_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16, x = var_1828_cast_fp16_0)[name = string("input_5_cast_fp16")]; tensor var_1845_cast_fp16 = silu(x = input_5_cast_fp16)[name = string("op_1845_cast_fp16")]; tensor var_1851_strides_0 = const()[name = string("op_1851_strides_0"), val = tensor([1, 1])]; string var_1851_pad_type_0 = const()[name = string("op_1851_pad_type_0"), val = string("valid")]; tensor var_1851_pad_0 = const()[name = string("op_1851_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1851_dilations_0 = const()[name = string("op_1851_dilations_0"), val = tensor([1, 1])]; int32 var_1851_groups_0 = const()[name = string("op_1851_groups_0"), val = int32(1)]; tensor var_1851_cast_fp16 = conv(dilations = var_1851_dilations_0, groups = var_1851_groups_0, pad = var_1851_pad_0, pad_type = var_1851_pad_type_0, strides = var_1851_strides_0, weight = layers_2_mlp_up_proj_weight_cast_fp16, x = var_1828_cast_fp16_0)[name = string("op_1851_cast_fp16")]; tensor x_29_cast_fp16 = mul(x = var_1845_cast_fp16, y = var_1851_cast_fp16)[name = string("x_29_cast_fp16")]; tensor layers_2_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1188732224)))]; tensor hidden_states_27_strides_0 = const()[name = string("hidden_states_27_strides_0"), val = tensor([1, 1])]; string hidden_states_27_pad_type_0 = const()[name = string("hidden_states_27_pad_type_0"), val = string("valid")]; tensor hidden_states_27_pad_0 = const()[name = string("hidden_states_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_27_dilations_0 = const()[name = string("hidden_states_27_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_27_groups_0 = const()[name = string("hidden_states_27_groups_0"), val = int32(1)]; tensor hidden_states_27_cast_fp16 = conv(dilations = hidden_states_27_dilations_0, groups = hidden_states_27_groups_0, pad = hidden_states_27_pad_0, pad_type = hidden_states_27_pad_type_0, strides = hidden_states_27_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16, x = x_29_cast_fp16)[name = string("hidden_states_27_cast_fp16")]; tensor hidden_states_29_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = hidden_states_27_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; fp16 const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1869_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_30_promoted_to_fp16)[name = string("op_1869_cast_fp16")]; int32 var_1867 = const()[name = string("op_1867"), val = int32(1)]; bool doubled_25_interleave_0 = const()[name = string("doubled_25_interleave_0"), val = bool(false)]; tensor doubled_25_cast_fp16 = concat(axis = var_1867, interleave = doubled_25_interleave_0, values = (hidden_states_29_cast_fp16, var_1869_cast_fp16))[name = string("doubled_25_cast_fp16")]; tensor out_13_axes_0 = const()[name = string("out_13_axes_0"), val = tensor([1])]; tensor out_13_gamma_0_to_fp16 = const()[name = string("out_13_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213898112)))]; fp16 var_1879_to_fp16 = const()[name = string("op_1879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_13_cast_fp16 = layer_norm(axes = out_13_axes_0, epsilon = var_1879_to_fp16, gamma = out_13_gamma_0_to_fp16, x = doubled_25_cast_fp16)[name = string("out_13_cast_fp16")]; tensor var_1890_split_sizes_0 = const()[name = string("op_1890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1890_axis_0 = const()[name = string("op_1890_axis_0"), val = int32(1)]; tensor var_1890_cast_fp16_0, tensor var_1890_cast_fp16_1 = split(axis = var_1890_axis_0, split_sizes = var_1890_split_sizes_0, x = out_13_cast_fp16)[name = string("op_1890_cast_fp16")]; tensor layers_3_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213906368)))]; tensor query_states_19_strides_0 = const()[name = string("query_states_19_strides_0"), val = tensor([1, 1])]; string query_states_19_pad_type_0 = const()[name = string("query_states_19_pad_type_0"), val = string("valid")]; tensor query_states_19_pad_0 = const()[name = string("query_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_19_dilations_0 = const()[name = string("query_states_19_dilations_0"), val = tensor([1, 1])]; int32 query_states_19_groups_0 = const()[name = string("query_states_19_groups_0"), val = int32(1)]; tensor query_states_19_cast_fp16 = conv(dilations = query_states_19_dilations_0, groups = query_states_19_groups_0, pad = query_states_19_pad_0, pad_type = query_states_19_pad_type_0, strides = query_states_19_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("query_states_19_cast_fp16")]; tensor layers_3_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1222295040)))]; tensor key_states_31_strides_0 = const()[name = string("key_states_31_strides_0"), val = tensor([1, 1])]; string key_states_31_pad_type_0 = const()[name = string("key_states_31_pad_type_0"), val = string("valid")]; tensor key_states_31_pad_0 = const()[name = string("key_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_31_dilations_0 = const()[name = string("key_states_31_dilations_0"), val = tensor([1, 1])]; int32 key_states_31_groups_0 = const()[name = string("key_states_31_groups_0"), val = int32(1)]; tensor key_states_31_cast_fp16 = conv(dilations = key_states_31_dilations_0, groups = key_states_31_groups_0, pad = key_states_31_pad_0, pad_type = key_states_31_pad_type_0, strides = key_states_31_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("key_states_31_cast_fp16")]; tensor value_states_19_strides_0 = const()[name = string("value_states_19_strides_0"), val = tensor([1, 1])]; string value_states_19_pad_type_0 = const()[name = string("value_states_19_pad_type_0"), val = string("valid")]; tensor value_states_19_pad_0 = const()[name = string("value_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_19_dilations_0 = const()[name = string("value_states_19_dilations_0"), val = tensor([1, 1])]; int32 value_states_19_groups_0 = const()[name = string("value_states_19_groups_0"), val = int32(1)]; tensor value_states_19_cast_fp16 = conv(dilations = value_states_19_dilations_0, groups = value_states_19_groups_0, pad = value_states_19_pad_0, pad_type = value_states_19_pad_type_0, strides = value_states_19_strides_0, weight = layers_3_self_attn_v_proj_weight_cast_fp16, x = var_1890_cast_fp16_0)[name = string("value_states_19_cast_fp16")]; tensor concat_36x = const()[name = string("concat_36x"), val = tensor([1, 16, 128, -1])]; tensor x_31_cast_fp16 = reshape(shape = concat_36x, x = query_states_19_cast_fp16)[name = string("x_31_cast_fp16")]; tensor concat_37x = const()[name = string("concat_37x"), val = tensor([1, 2, 128, -1])]; tensor var_1947_cast_fp16 = reshape(shape = concat_37x, x = key_states_31_cast_fp16)[name = string("op_1947_cast_fp16")]; tensor concat_38x = const()[name = string("concat_38x"), val = tensor([1, 2, 128, -1])]; tensor var_1954_cast_fp16 = reshape(shape = concat_38x, x = value_states_19_cast_fp16)[name = string("op_1954_cast_fp16")]; tensor var_1958_cast_fp16 = mul(x = x_31_cast_fp16, y = var_869_cast_fp16)[name = string("op_1958_cast_fp16")]; tensor var_1959_split_sizes_0 = const()[name = string("op_1959_split_sizes_0"), val = tensor([64, 64])]; int32 var_1959_axis_0 = const()[name = string("op_1959_axis_0"), val = int32(-2)]; tensor var_1959_cast_fp16_0, tensor var_1959_cast_fp16_1 = split(axis = var_1959_axis_0, split_sizes = var_1959_split_sizes_0, x = x_31_cast_fp16)[name = string("op_1959_cast_fp16")]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1961_cast_fp16 = mul(x = var_1959_cast_fp16_1, y = const_32_promoted_to_fp16)[name = string("op_1961_cast_fp16")]; int32 var_1963 = const()[name = string("op_1963"), val = int32(-2)]; bool var_1964_interleave_0 = const()[name = string("op_1964_interleave_0"), val = bool(false)]; tensor var_1964_cast_fp16 = concat(axis = var_1963, interleave = var_1964_interleave_0, values = (var_1961_cast_fp16, var_1959_cast_fp16_0))[name = string("op_1964_cast_fp16")]; tensor var_1965_cast_fp16 = mul(x = var_1964_cast_fp16, y = var_878_cast_fp16)[name = string("op_1965_cast_fp16")]; tensor query_states_21_cast_fp16 = add(x = var_1958_cast_fp16, y = var_1965_cast_fp16)[name = string("query_states_21_cast_fp16")]; tensor var_1971_cast_fp16 = mul(x = var_1947_cast_fp16, y = var_869_cast_fp16)[name = string("op_1971_cast_fp16")]; tensor var_1972_split_sizes_0 = const()[name = string("op_1972_split_sizes_0"), val = tensor([64, 64])]; int32 var_1972_axis_0 = const()[name = string("op_1972_axis_0"), val = int32(-2)]; tensor var_1972_cast_fp16_0, tensor var_1972_cast_fp16_1 = split(axis = var_1972_axis_0, split_sizes = var_1972_split_sizes_0, x = var_1947_cast_fp16)[name = string("op_1972_cast_fp16")]; fp16 const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1974_cast_fp16 = mul(x = var_1972_cast_fp16_1, y = const_33_promoted_to_fp16)[name = string("op_1974_cast_fp16")]; int32 var_1976 = const()[name = string("op_1976"), val = int32(-2)]; bool var_1977_interleave_0 = const()[name = string("op_1977_interleave_0"), val = bool(false)]; tensor var_1977_cast_fp16 = concat(axis = var_1976, interleave = var_1977_interleave_0, values = (var_1974_cast_fp16, var_1972_cast_fp16_0))[name = string("op_1977_cast_fp16")]; tensor var_1978_cast_fp16 = mul(x = var_1977_cast_fp16, y = var_878_cast_fp16)[name = string("op_1978_cast_fp16")]; tensor key_states_35_cast_fp16 = add(x = var_1971_cast_fp16, y = var_1978_cast_fp16)[name = string("key_states_35_cast_fp16")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; int32 concat_41_axis_0 = const()[name = string("concat_41_axis_0"), val = int32(0)]; bool concat_41_interleave_0 = const()[name = string("concat_41_interleave_0"), val = bool(false)]; tensor concat_41 = concat(axis = concat_41_axis_0, interleave = concat_41_interleave_0, values = (expand_dims_36, expand_dims_37, position_id, expand_dims_39))[name = string("concat_41")]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; tensor concat_42_values1_0 = const()[name = string("concat_42_values1_0"), val = tensor([0])]; tensor concat_42_values3_0 = const()[name = string("concat_42_values3_0"), val = tensor([0])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_40, concat_42_values1_0, cache_position_end, concat_42_values3_0))[name = string("concat_42")]; tensor key_states_37_perm_0 = const()[name = string("key_states_37_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_4_stride_0 = const()[name = string("key_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_37_cast_fp16 = transpose(perm = key_states_37_perm_0, x = key_states_35_cast_fp16)[name = string("transpose_160")]; tensor key_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = key_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = key_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_4_squeeze_mask_0, stride = key_cache_internal_tensor_assign_4_stride_0, update = key_states_37_cast_fp16, x = coreml_update_state_60)[name = string("key_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_4_cast_fp16, input = key_cache)[name = string("coreml_update_state_62_write_state")]; tensor coreml_update_state_62 = read_state(input = key_cache)[name = string("coreml_update_state_62")]; tensor value_states_21_perm_0 = const()[name = string("value_states_21_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_4_stride_0 = const()[name = string("value_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_21_cast_fp16 = transpose(perm = value_states_21_perm_0, x = var_1954_cast_fp16)[name = string("transpose_159")]; tensor value_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = value_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = value_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_4_squeeze_mask_0, stride = value_cache_internal_tensor_assign_4_stride_0, update = value_states_21_cast_fp16, x = coreml_update_state_61)[name = string("value_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_4_cast_fp16, input = value_cache)[name = string("coreml_update_state_63_write_state")]; tensor coreml_update_state_63 = read_state(input = value_cache)[name = string("coreml_update_state_63")]; tensor var_2048_begin_0 = const()[name = string("op_2048_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2048_end_0 = const()[name = string("op_2048_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2048_end_mask_0 = const()[name = string("op_2048_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2048_cast_fp16 = slice_by_index(begin = var_2048_begin_0, end = var_2048_end_0, end_mask = var_2048_end_mask_0, x = coreml_update_state_62)[name = string("op_2048_cast_fp16")]; tensor tile_6 = const()[name = string("tile_6"), val = tensor([1, 1])]; int32 var_2051_axis_0 = const()[name = string("op_2051_axis_0"), val = int32(1)]; tensor var_2051_cast_fp16_0, tensor var_2051_cast_fp16_1 = split(axis = var_2051_axis_0, split_sizes = tile_6, x = var_2048_cast_fp16)[name = string("op_2051_cast_fp16")]; tensor var_2058_begin_0 = const()[name = string("op_2058_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2058_end_0 = const()[name = string("op_2058_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2058_end_mask_0 = const()[name = string("op_2058_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2058_cast_fp16 = slice_by_index(begin = var_2058_begin_0, end = var_2058_end_0, end_mask = var_2058_end_mask_0, x = coreml_update_state_63)[name = string("op_2058_cast_fp16")]; tensor tile_7 = const()[name = string("tile_7"), val = tensor([1, 1])]; int32 var_2061_axis_0 = const()[name = string("op_2061_axis_0"), val = int32(1)]; tensor var_2061_cast_fp16_0, tensor var_2061_cast_fp16_1 = split(axis = var_2061_axis_0, split_sizes = tile_7, x = var_2058_cast_fp16)[name = string("op_2061_cast_fp16")]; tensor var_2064_split_sizes_0 = const()[name = string("op_2064_split_sizes_0"), val = tensor([8, 8])]; int32 var_2064_axis_0 = const()[name = string("op_2064_axis_0"), val = int32(1)]; tensor var_2064_0, tensor var_2064_1 = split(axis = var_2064_axis_0, split_sizes = var_2064_split_sizes_0, x = query_states_21_cast_fp16)[name = string("op_2064")]; bool attn_weights_49_transpose_x_0 = const()[name = string("attn_weights_49_transpose_x_0"), val = bool(false)]; bool attn_weights_49_transpose_y_0 = const()[name = string("attn_weights_49_transpose_y_0"), val = bool(false)]; tensor attn_weights_49_cast_fp16 = matmul(transpose_x = attn_weights_49_transpose_x_0, transpose_y = attn_weights_49_transpose_y_0, x = var_2051_cast_fp16_0, y = var_2064_0)[name = string("attn_weights_49_cast_fp16")]; fp16 var_2067_to_fp16 = const()[name = string("op_2067_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_51_cast_fp16 = mul(x = attn_weights_49_cast_fp16, y = var_2067_to_fp16)[name = string("attn_weights_51_cast_fp16")]; tensor attn_weights_53_cast_fp16 = add(x = attn_weights_51_cast_fp16, y = attn_mask_1)[name = string("attn_weights_53_cast_fp16")]; int32 var_2071 = const()[name = string("op_2071"), val = int32(-2)]; tensor attn_weights_55_cast_fp16 = softmax(axis = var_2071, x = attn_weights_53_cast_fp16)[name = string("attn_weights_55_cast_fp16")]; bool var_2077_transpose_x_1 = const()[name = string("op_2077_transpose_x_1"), val = bool(true)]; bool var_2077_transpose_y_1 = const()[name = string("op_2077_transpose_y_1"), val = bool(false)]; tensor var_2077_cast_fp16 = matmul(transpose_x = var_2077_transpose_x_1, transpose_y = var_2077_transpose_y_1, x = attn_weights_55_cast_fp16, y = var_2061_cast_fp16_0)[name = string("op_2077_cast_fp16")]; bool attn_weights_57_transpose_x_0 = const()[name = string("attn_weights_57_transpose_x_0"), val = bool(false)]; bool attn_weights_57_transpose_y_0 = const()[name = string("attn_weights_57_transpose_y_0"), val = bool(false)]; tensor attn_weights_57_cast_fp16 = matmul(transpose_x = attn_weights_57_transpose_x_0, transpose_y = attn_weights_57_transpose_y_0, x = var_2051_cast_fp16_1, y = var_2064_1)[name = string("attn_weights_57_cast_fp16")]; fp16 var_2079_to_fp16 = const()[name = string("op_2079_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_59_cast_fp16 = mul(x = attn_weights_57_cast_fp16, y = var_2079_to_fp16)[name = string("attn_weights_59_cast_fp16")]; tensor attn_weights_61_cast_fp16 = add(x = attn_weights_59_cast_fp16, y = attn_mask_1)[name = string("attn_weights_61_cast_fp16")]; int32 var_2083 = const()[name = string("op_2083"), val = int32(-2)]; tensor attn_weights_63_cast_fp16 = softmax(axis = var_2083, x = attn_weights_61_cast_fp16)[name = string("attn_weights_63_cast_fp16")]; bool attn_output_25_transpose_x_1 = const()[name = string("attn_output_25_transpose_x_1"), val = bool(true)]; bool attn_output_25_transpose_y_1 = const()[name = string("attn_output_25_transpose_y_1"), val = bool(false)]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_1, transpose_y = attn_output_25_transpose_y_1, x = attn_weights_63_cast_fp16, y = var_2061_cast_fp16_1)[name = string("attn_output_25_cast_fp16")]; int32 var_2091 = const()[name = string("op_2091"), val = int32(1)]; bool attn_output_27_interleave_0 = const()[name = string("attn_output_27_interleave_0"), val = bool(false)]; tensor attn_output_27_cast_fp16 = concat(axis = var_2091, interleave = attn_output_27_interleave_0, values = (var_2077_cast_fp16, attn_output_25_cast_fp16))[name = string("attn_output_27_cast_fp16")]; tensor var_2095_perm_0 = const()[name = string("op_2095_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_47x = const()[name = string("concat_47x"), val = tensor([1, 2048, 1, -1])]; tensor var_2095_cast_fp16 = transpose(perm = var_2095_perm_0, x = attn_output_27_cast_fp16)[name = string("transpose_158")]; tensor attn_output_31_cast_fp16 = reshape(shape = concat_47x, x = var_2095_cast_fp16)[name = string("attn_output_31_cast_fp16")]; tensor hidden_states_33_strides_0 = const()[name = string("hidden_states_33_strides_0"), val = tensor([1, 1])]; string hidden_states_33_pad_type_0 = const()[name = string("hidden_states_33_pad_type_0"), val = string("valid")]; tensor hidden_states_33_pad_0 = const()[name = string("hidden_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_33_dilations_0 = const()[name = string("hidden_states_33_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_33_groups_0 = const()[name = string("hidden_states_33_groups_0"), val = int32(1)]; tensor hidden_states_33_cast_fp16 = conv(dilations = hidden_states_33_dilations_0, groups = hidden_states_33_groups_0, pad = hidden_states_33_pad_0, pad_type = hidden_states_33_pad_type_0, strides = hidden_states_33_strides_0, weight = layers_3_self_attn_o_proj_weight_cast_fp16, x = attn_output_31_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; tensor hidden_states_35_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = hidden_states_33_cast_fp16)[name = string("hidden_states_35_cast_fp16")]; fp16 const_38_promoted_to_fp16 = const()[name = string("const_38_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2128_cast_fp16 = mul(x = hidden_states_35_cast_fp16, y = const_38_promoted_to_fp16)[name = string("op_2128_cast_fp16")]; int32 var_2126 = const()[name = string("op_2126"), val = int32(1)]; bool doubled_29_interleave_0 = const()[name = string("doubled_29_interleave_0"), val = bool(false)]; tensor doubled_29_cast_fp16 = concat(axis = var_2126, interleave = doubled_29_interleave_0, values = (hidden_states_35_cast_fp16, var_2128_cast_fp16))[name = string("doubled_29_cast_fp16")]; tensor out_15_axes_0 = const()[name = string("out_15_axes_0"), val = tensor([1])]; tensor out_15_gamma_0_to_fp16 = const()[name = string("out_15_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223343680)))]; fp16 var_2138_to_fp16 = const()[name = string("op_2138_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_15_cast_fp16 = layer_norm(axes = out_15_axes_0, epsilon = var_2138_to_fp16, gamma = out_15_gamma_0_to_fp16, x = doubled_29_cast_fp16)[name = string("out_15_cast_fp16")]; tensor var_2149_split_sizes_0 = const()[name = string("op_2149_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2149_axis_0 = const()[name = string("op_2149_axis_0"), val = int32(1)]; tensor var_2149_cast_fp16_0, tensor var_2149_cast_fp16_1 = split(axis = var_2149_axis_0, split_sizes = var_2149_split_sizes_0, x = out_15_cast_fp16)[name = string("op_2149_cast_fp16")]; tensor layers_3_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223351936)))]; tensor input_7_strides_0 = const()[name = string("input_7_strides_0"), val = tensor([1, 1])]; string input_7_pad_type_0 = const()[name = string("input_7_pad_type_0"), val = string("valid")]; tensor input_7_pad_0 = const()[name = string("input_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_7_dilations_0 = const()[name = string("input_7_dilations_0"), val = tensor([1, 1])]; int32 input_7_groups_0 = const()[name = string("input_7_groups_0"), val = int32(1)]; tensor input_7_cast_fp16 = conv(dilations = input_7_dilations_0, groups = input_7_groups_0, pad = input_7_pad_0, pad_type = input_7_pad_type_0, strides = input_7_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("input_7_cast_fp16")]; tensor var_2166_cast_fp16 = silu(x = input_7_cast_fp16)[name = string("op_2166_cast_fp16")]; tensor layers_3_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1248517824)))]; tensor var_2172_strides_0 = const()[name = string("op_2172_strides_0"), val = tensor([1, 1])]; string var_2172_pad_type_0 = const()[name = string("op_2172_pad_type_0"), val = string("valid")]; tensor var_2172_pad_0 = const()[name = string("op_2172_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2172_dilations_0 = const()[name = string("op_2172_dilations_0"), val = tensor([1, 1])]; int32 var_2172_groups_0 = const()[name = string("op_2172_groups_0"), val = int32(1)]; tensor var_2172_cast_fp16 = conv(dilations = var_2172_dilations_0, groups = var_2172_groups_0, pad = var_2172_pad_0, pad_type = var_2172_pad_type_0, strides = var_2172_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("op_2172_cast_fp16")]; tensor x_39_cast_fp16 = mul(x = var_2166_cast_fp16, y = var_2172_cast_fp16)[name = string("x_39_cast_fp16")]; tensor hidden_states_37_strides_0 = const()[name = string("hidden_states_37_strides_0"), val = tensor([1, 1])]; string hidden_states_37_pad_type_0 = const()[name = string("hidden_states_37_pad_type_0"), val = string("valid")]; tensor hidden_states_37_pad_0 = const()[name = string("hidden_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_37_dilations_0 = const()[name = string("hidden_states_37_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_37_groups_0 = const()[name = string("hidden_states_37_groups_0"), val = int32(1)]; tensor hidden_states_37_cast_fp16 = conv(dilations = hidden_states_37_dilations_0, groups = hidden_states_37_groups_0, pad = hidden_states_37_pad_0, pad_type = hidden_states_37_pad_type_0, strides = hidden_states_37_strides_0, weight = layers_3_mlp_down_proj_weight_cast_fp16, x = x_39_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; tensor hidden_states_39_cast_fp16 = add(x = hidden_states_35_cast_fp16, y = hidden_states_37_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; fp16 const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2190_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_40_promoted_to_fp16)[name = string("op_2190_cast_fp16")]; int32 var_2188 = const()[name = string("op_2188"), val = int32(1)]; bool doubled_33_interleave_0 = const()[name = string("doubled_33_interleave_0"), val = bool(false)]; tensor doubled_33_cast_fp16 = concat(axis = var_2188, interleave = doubled_33_interleave_0, values = (hidden_states_39_cast_fp16, var_2190_cast_fp16))[name = string("doubled_33_cast_fp16")]; tensor out_17_axes_0 = const()[name = string("out_17_axes_0"), val = tensor([1])]; tensor out_17_gamma_0_to_fp16 = const()[name = string("out_17_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273683712)))]; fp16 var_2200_to_fp16 = const()[name = string("op_2200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_17_cast_fp16 = layer_norm(axes = out_17_axes_0, epsilon = var_2200_to_fp16, gamma = out_17_gamma_0_to_fp16, x = doubled_33_cast_fp16)[name = string("out_17_cast_fp16")]; tensor var_2211_split_sizes_0 = const()[name = string("op_2211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2211_axis_0 = const()[name = string("op_2211_axis_0"), val = int32(1)]; tensor var_2211_cast_fp16_0, tensor var_2211_cast_fp16_1 = split(axis = var_2211_axis_0, split_sizes = var_2211_split_sizes_0, x = out_17_cast_fp16)[name = string("op_2211_cast_fp16")]; tensor layers_4_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273691968)))]; tensor query_states_25_strides_0 = const()[name = string("query_states_25_strides_0"), val = tensor([1, 1])]; string query_states_25_pad_type_0 = const()[name = string("query_states_25_pad_type_0"), val = string("valid")]; tensor query_states_25_pad_0 = const()[name = string("query_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_25_dilations_0 = const()[name = string("query_states_25_dilations_0"), val = tensor([1, 1])]; int32 query_states_25_groups_0 = const()[name = string("query_states_25_groups_0"), val = int32(1)]; tensor query_states_25_cast_fp16 = conv(dilations = query_states_25_dilations_0, groups = query_states_25_groups_0, pad = query_states_25_pad_0, pad_type = query_states_25_pad_type_0, strides = query_states_25_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("query_states_25_cast_fp16")]; tensor layers_4_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1282080640)))]; tensor key_states_41_strides_0 = const()[name = string("key_states_41_strides_0"), val = tensor([1, 1])]; string key_states_41_pad_type_0 = const()[name = string("key_states_41_pad_type_0"), val = string("valid")]; tensor key_states_41_pad_0 = const()[name = string("key_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_41_dilations_0 = const()[name = string("key_states_41_dilations_0"), val = tensor([1, 1])]; int32 key_states_41_groups_0 = const()[name = string("key_states_41_groups_0"), val = int32(1)]; tensor key_states_41_cast_fp16 = conv(dilations = key_states_41_dilations_0, groups = key_states_41_groups_0, pad = key_states_41_pad_0, pad_type = key_states_41_pad_type_0, strides = key_states_41_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("key_states_41_cast_fp16")]; tensor value_states_25_strides_0 = const()[name = string("value_states_25_strides_0"), val = tensor([1, 1])]; string value_states_25_pad_type_0 = const()[name = string("value_states_25_pad_type_0"), val = string("valid")]; tensor value_states_25_pad_0 = const()[name = string("value_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_25_dilations_0 = const()[name = string("value_states_25_dilations_0"), val = tensor([1, 1])]; int32 value_states_25_groups_0 = const()[name = string("value_states_25_groups_0"), val = int32(1)]; tensor value_states_25_cast_fp16 = conv(dilations = value_states_25_dilations_0, groups = value_states_25_groups_0, pad = value_states_25_pad_0, pad_type = value_states_25_pad_type_0, strides = value_states_25_strides_0, weight = layers_4_self_attn_v_proj_weight_cast_fp16, x = var_2211_cast_fp16_0)[name = string("value_states_25_cast_fp16")]; tensor concat_48x = const()[name = string("concat_48x"), val = tensor([1, 16, 128, -1])]; tensor x_41_cast_fp16 = reshape(shape = concat_48x, x = query_states_25_cast_fp16)[name = string("x_41_cast_fp16")]; tensor concat_49x = const()[name = string("concat_49x"), val = tensor([1, 2, 128, -1])]; tensor var_2268_cast_fp16 = reshape(shape = concat_49x, x = key_states_41_cast_fp16)[name = string("op_2268_cast_fp16")]; tensor concat_50x = const()[name = string("concat_50x"), val = tensor([1, 2, 128, -1])]; tensor var_2275_cast_fp16 = reshape(shape = concat_50x, x = value_states_25_cast_fp16)[name = string("op_2275_cast_fp16")]; tensor var_2279_cast_fp16 = mul(x = x_41_cast_fp16, y = var_869_cast_fp16)[name = string("op_2279_cast_fp16")]; tensor var_2280_split_sizes_0 = const()[name = string("op_2280_split_sizes_0"), val = tensor([64, 64])]; int32 var_2280_axis_0 = const()[name = string("op_2280_axis_0"), val = int32(-2)]; tensor var_2280_cast_fp16_0, tensor var_2280_cast_fp16_1 = split(axis = var_2280_axis_0, split_sizes = var_2280_split_sizes_0, x = x_41_cast_fp16)[name = string("op_2280_cast_fp16")]; fp16 const_42_promoted_to_fp16 = const()[name = string("const_42_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2282_cast_fp16 = mul(x = var_2280_cast_fp16_1, y = const_42_promoted_to_fp16)[name = string("op_2282_cast_fp16")]; int32 var_2284 = const()[name = string("op_2284"), val = int32(-2)]; bool var_2285_interleave_0 = const()[name = string("op_2285_interleave_0"), val = bool(false)]; tensor var_2285_cast_fp16 = concat(axis = var_2284, interleave = var_2285_interleave_0, values = (var_2282_cast_fp16, var_2280_cast_fp16_0))[name = string("op_2285_cast_fp16")]; tensor var_2286_cast_fp16 = mul(x = var_2285_cast_fp16, y = var_878_cast_fp16)[name = string("op_2286_cast_fp16")]; tensor query_states_27_cast_fp16 = add(x = var_2279_cast_fp16, y = var_2286_cast_fp16)[name = string("query_states_27_cast_fp16")]; tensor var_2292_cast_fp16 = mul(x = var_2268_cast_fp16, y = var_869_cast_fp16)[name = string("op_2292_cast_fp16")]; tensor var_2293_split_sizes_0 = const()[name = string("op_2293_split_sizes_0"), val = tensor([64, 64])]; int32 var_2293_axis_0 = const()[name = string("op_2293_axis_0"), val = int32(-2)]; tensor var_2293_cast_fp16_0, tensor var_2293_cast_fp16_1 = split(axis = var_2293_axis_0, split_sizes = var_2293_split_sizes_0, x = var_2268_cast_fp16)[name = string("op_2293_cast_fp16")]; fp16 const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2295_cast_fp16 = mul(x = var_2293_cast_fp16_1, y = const_43_promoted_to_fp16)[name = string("op_2295_cast_fp16")]; int32 var_2297 = const()[name = string("op_2297"), val = int32(-2)]; bool var_2298_interleave_0 = const()[name = string("op_2298_interleave_0"), val = bool(false)]; tensor var_2298_cast_fp16 = concat(axis = var_2297, interleave = var_2298_interleave_0, values = (var_2295_cast_fp16, var_2293_cast_fp16_0))[name = string("op_2298_cast_fp16")]; tensor var_2299_cast_fp16 = mul(x = var_2298_cast_fp16, y = var_878_cast_fp16)[name = string("op_2299_cast_fp16")]; tensor key_states_45_cast_fp16 = add(x = var_2292_cast_fp16, y = var_2299_cast_fp16)[name = string("key_states_45_cast_fp16")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; int32 concat_53_axis_0 = const()[name = string("concat_53_axis_0"), val = int32(0)]; bool concat_53_interleave_0 = const()[name = string("concat_53_interleave_0"), val = bool(false)]; tensor concat_53 = concat(axis = concat_53_axis_0, interleave = concat_53_interleave_0, values = (expand_dims_48, expand_dims_49, position_id, expand_dims_51))[name = string("concat_53")]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; tensor concat_54_values1_0 = const()[name = string("concat_54_values1_0"), val = tensor([0])]; tensor concat_54_values3_0 = const()[name = string("concat_54_values3_0"), val = tensor([0])]; int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_52, concat_54_values1_0, cache_position_end, concat_54_values3_0))[name = string("concat_54")]; tensor key_states_47_perm_0 = const()[name = string("key_states_47_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_5_stride_0 = const()[name = string("key_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_47_cast_fp16 = transpose(perm = key_states_47_perm_0, x = key_states_45_cast_fp16)[name = string("transpose_157")]; tensor key_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = key_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = key_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_5_squeeze_mask_0, stride = key_cache_internal_tensor_assign_5_stride_0, update = key_states_47_cast_fp16, x = coreml_update_state_62)[name = string("key_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_5_cast_fp16, input = key_cache)[name = string("coreml_update_state_64_write_state")]; tensor coreml_update_state_64 = read_state(input = key_cache)[name = string("coreml_update_state_64")]; tensor value_states_27_perm_0 = const()[name = string("value_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_5_stride_0 = const()[name = string("value_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_27_cast_fp16 = transpose(perm = value_states_27_perm_0, x = var_2275_cast_fp16)[name = string("transpose_156")]; tensor value_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = value_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = value_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_5_squeeze_mask_0, stride = value_cache_internal_tensor_assign_5_stride_0, update = value_states_27_cast_fp16, x = coreml_update_state_63)[name = string("value_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_5_cast_fp16, input = value_cache)[name = string("coreml_update_state_65_write_state")]; tensor coreml_update_state_65 = read_state(input = value_cache)[name = string("coreml_update_state_65")]; tensor var_2369_begin_0 = const()[name = string("op_2369_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2369_end_0 = const()[name = string("op_2369_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2369_end_mask_0 = const()[name = string("op_2369_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2369_cast_fp16 = slice_by_index(begin = var_2369_begin_0, end = var_2369_end_0, end_mask = var_2369_end_mask_0, x = coreml_update_state_64)[name = string("op_2369_cast_fp16")]; tensor tile_8 = const()[name = string("tile_8"), val = tensor([1, 1])]; int32 var_2372_axis_0 = const()[name = string("op_2372_axis_0"), val = int32(1)]; tensor var_2372_cast_fp16_0, tensor var_2372_cast_fp16_1 = split(axis = var_2372_axis_0, split_sizes = tile_8, x = var_2369_cast_fp16)[name = string("op_2372_cast_fp16")]; tensor var_2379_begin_0 = const()[name = string("op_2379_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2379_end_0 = const()[name = string("op_2379_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2379_end_mask_0 = const()[name = string("op_2379_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2379_cast_fp16 = slice_by_index(begin = var_2379_begin_0, end = var_2379_end_0, end_mask = var_2379_end_mask_0, x = coreml_update_state_65)[name = string("op_2379_cast_fp16")]; tensor tile_9 = const()[name = string("tile_9"), val = tensor([1, 1])]; int32 var_2382_axis_0 = const()[name = string("op_2382_axis_0"), val = int32(1)]; tensor var_2382_cast_fp16_0, tensor var_2382_cast_fp16_1 = split(axis = var_2382_axis_0, split_sizes = tile_9, x = var_2379_cast_fp16)[name = string("op_2382_cast_fp16")]; tensor var_2385_split_sizes_0 = const()[name = string("op_2385_split_sizes_0"), val = tensor([8, 8])]; int32 var_2385_axis_0 = const()[name = string("op_2385_axis_0"), val = int32(1)]; tensor var_2385_0, tensor var_2385_1 = split(axis = var_2385_axis_0, split_sizes = var_2385_split_sizes_0, x = query_states_27_cast_fp16)[name = string("op_2385")]; bool attn_weights_65_transpose_x_0 = const()[name = string("attn_weights_65_transpose_x_0"), val = bool(false)]; bool attn_weights_65_transpose_y_0 = const()[name = string("attn_weights_65_transpose_y_0"), val = bool(false)]; tensor attn_weights_65_cast_fp16 = matmul(transpose_x = attn_weights_65_transpose_x_0, transpose_y = attn_weights_65_transpose_y_0, x = var_2372_cast_fp16_0, y = var_2385_0)[name = string("attn_weights_65_cast_fp16")]; fp16 var_2388_to_fp16 = const()[name = string("op_2388_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_67_cast_fp16 = mul(x = attn_weights_65_cast_fp16, y = var_2388_to_fp16)[name = string("attn_weights_67_cast_fp16")]; tensor attn_weights_69_cast_fp16 = add(x = attn_weights_67_cast_fp16, y = attn_mask_1)[name = string("attn_weights_69_cast_fp16")]; int32 var_2392 = const()[name = string("op_2392"), val = int32(-2)]; tensor attn_weights_71_cast_fp16 = softmax(axis = var_2392, x = attn_weights_69_cast_fp16)[name = string("attn_weights_71_cast_fp16")]; bool var_2398_transpose_x_1 = const()[name = string("op_2398_transpose_x_1"), val = bool(true)]; bool var_2398_transpose_y_1 = const()[name = string("op_2398_transpose_y_1"), val = bool(false)]; tensor var_2398_cast_fp16 = matmul(transpose_x = var_2398_transpose_x_1, transpose_y = var_2398_transpose_y_1, x = attn_weights_71_cast_fp16, y = var_2382_cast_fp16_0)[name = string("op_2398_cast_fp16")]; bool attn_weights_73_transpose_x_0 = const()[name = string("attn_weights_73_transpose_x_0"), val = bool(false)]; bool attn_weights_73_transpose_y_0 = const()[name = string("attn_weights_73_transpose_y_0"), val = bool(false)]; tensor attn_weights_73_cast_fp16 = matmul(transpose_x = attn_weights_73_transpose_x_0, transpose_y = attn_weights_73_transpose_y_0, x = var_2372_cast_fp16_1, y = var_2385_1)[name = string("attn_weights_73_cast_fp16")]; fp16 var_2400_to_fp16 = const()[name = string("op_2400_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_75_cast_fp16 = mul(x = attn_weights_73_cast_fp16, y = var_2400_to_fp16)[name = string("attn_weights_75_cast_fp16")]; tensor attn_weights_77_cast_fp16 = add(x = attn_weights_75_cast_fp16, y = attn_mask_1)[name = string("attn_weights_77_cast_fp16")]; int32 var_2404 = const()[name = string("op_2404"), val = int32(-2)]; tensor attn_weights_79_cast_fp16 = softmax(axis = var_2404, x = attn_weights_77_cast_fp16)[name = string("attn_weights_79_cast_fp16")]; bool attn_output_33_transpose_x_1 = const()[name = string("attn_output_33_transpose_x_1"), val = bool(true)]; bool attn_output_33_transpose_y_1 = const()[name = string("attn_output_33_transpose_y_1"), val = bool(false)]; tensor attn_output_33_cast_fp16 = matmul(transpose_x = attn_output_33_transpose_x_1, transpose_y = attn_output_33_transpose_y_1, x = attn_weights_79_cast_fp16, y = var_2382_cast_fp16_1)[name = string("attn_output_33_cast_fp16")]; int32 var_2412 = const()[name = string("op_2412"), val = int32(1)]; bool attn_output_35_interleave_0 = const()[name = string("attn_output_35_interleave_0"), val = bool(false)]; tensor attn_output_35_cast_fp16 = concat(axis = var_2412, interleave = attn_output_35_interleave_0, values = (var_2398_cast_fp16, attn_output_33_cast_fp16))[name = string("attn_output_35_cast_fp16")]; tensor var_2416_perm_0 = const()[name = string("op_2416_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_59x = const()[name = string("concat_59x"), val = tensor([1, 2048, 1, -1])]; tensor var_2416_cast_fp16 = transpose(perm = var_2416_perm_0, x = attn_output_35_cast_fp16)[name = string("transpose_155")]; tensor attn_output_39_cast_fp16 = reshape(shape = concat_59x, x = var_2416_cast_fp16)[name = string("attn_output_39_cast_fp16")]; tensor hidden_states_43_strides_0 = const()[name = string("hidden_states_43_strides_0"), val = tensor([1, 1])]; string hidden_states_43_pad_type_0 = const()[name = string("hidden_states_43_pad_type_0"), val = string("valid")]; tensor hidden_states_43_pad_0 = const()[name = string("hidden_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_43_dilations_0 = const()[name = string("hidden_states_43_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_43_groups_0 = const()[name = string("hidden_states_43_groups_0"), val = int32(1)]; tensor hidden_states_43_cast_fp16 = conv(dilations = hidden_states_43_dilations_0, groups = hidden_states_43_groups_0, pad = hidden_states_43_pad_0, pad_type = hidden_states_43_pad_type_0, strides = hidden_states_43_strides_0, weight = layers_4_self_attn_o_proj_weight_cast_fp16, x = attn_output_39_cast_fp16)[name = string("hidden_states_43_cast_fp16")]; tensor hidden_states_45_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = hidden_states_43_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; fp16 const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2449_cast_fp16 = mul(x = hidden_states_45_cast_fp16, y = const_48_promoted_to_fp16)[name = string("op_2449_cast_fp16")]; int32 var_2447 = const()[name = string("op_2447"), val = int32(1)]; bool doubled_37_interleave_0 = const()[name = string("doubled_37_interleave_0"), val = bool(false)]; tensor doubled_37_cast_fp16 = concat(axis = var_2447, interleave = doubled_37_interleave_0, values = (hidden_states_45_cast_fp16, var_2449_cast_fp16))[name = string("doubled_37_cast_fp16")]; tensor out_19_axes_0 = const()[name = string("out_19_axes_0"), val = tensor([1])]; tensor out_19_gamma_0_to_fp16 = const()[name = string("out_19_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283129280)))]; fp16 var_2459_to_fp16 = const()[name = string("op_2459_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_19_cast_fp16 = layer_norm(axes = out_19_axes_0, epsilon = var_2459_to_fp16, gamma = out_19_gamma_0_to_fp16, x = doubled_37_cast_fp16)[name = string("out_19_cast_fp16")]; tensor var_2470_split_sizes_0 = const()[name = string("op_2470_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2470_axis_0 = const()[name = string("op_2470_axis_0"), val = int32(1)]; tensor var_2470_cast_fp16_0, tensor var_2470_cast_fp16_1 = split(axis = var_2470_axis_0, split_sizes = var_2470_split_sizes_0, x = out_19_cast_fp16)[name = string("op_2470_cast_fp16")]; tensor input_9_strides_0 = const()[name = string("input_9_strides_0"), val = tensor([1, 1])]; string input_9_pad_type_0 = const()[name = string("input_9_pad_type_0"), val = string("valid")]; tensor input_9_pad_0 = const()[name = string("input_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_9_dilations_0 = const()[name = string("input_9_dilations_0"), val = tensor([1, 1])]; int32 input_9_groups_0 = const()[name = string("input_9_groups_0"), val = int32(1)]; tensor input_9_cast_fp16 = conv(dilations = input_9_dilations_0, groups = input_9_groups_0, pad = input_9_pad_0, pad_type = input_9_pad_type_0, strides = input_9_strides_0, weight = layers_4_mlp_gate_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("input_9_cast_fp16")]; tensor var_2487_cast_fp16 = silu(x = input_9_cast_fp16)[name = string("op_2487_cast_fp16")]; tensor var_2493_strides_0 = const()[name = string("op_2493_strides_0"), val = tensor([1, 1])]; string var_2493_pad_type_0 = const()[name = string("op_2493_pad_type_0"), val = string("valid")]; tensor var_2493_pad_0 = const()[name = string("op_2493_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2493_dilations_0 = const()[name = string("op_2493_dilations_0"), val = tensor([1, 1])]; int32 var_2493_groups_0 = const()[name = string("op_2493_groups_0"), val = int32(1)]; tensor var_2493_cast_fp16 = conv(dilations = var_2493_dilations_0, groups = var_2493_groups_0, pad = var_2493_pad_0, pad_type = var_2493_pad_type_0, strides = var_2493_strides_0, weight = layers_4_mlp_up_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("op_2493_cast_fp16")]; tensor x_49_cast_fp16 = mul(x = var_2487_cast_fp16, y = var_2493_cast_fp16)[name = string("x_49_cast_fp16")]; tensor hidden_states_47_strides_0 = const()[name = string("hidden_states_47_strides_0"), val = tensor([1, 1])]; string hidden_states_47_pad_type_0 = const()[name = string("hidden_states_47_pad_type_0"), val = string("valid")]; tensor hidden_states_47_pad_0 = const()[name = string("hidden_states_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_47_dilations_0 = const()[name = string("hidden_states_47_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_47_groups_0 = const()[name = string("hidden_states_47_groups_0"), val = int32(1)]; tensor hidden_states_47_cast_fp16 = conv(dilations = hidden_states_47_dilations_0, groups = hidden_states_47_groups_0, pad = hidden_states_47_pad_0, pad_type = hidden_states_47_pad_type_0, strides = hidden_states_47_strides_0, weight = layers_4_mlp_down_proj_weight_cast_fp16, x = x_49_cast_fp16)[name = string("hidden_states_47_cast_fp16")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = hidden_states_47_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; fp16 const_50_promoted_to_fp16 = const()[name = string("const_50_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2511_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_50_promoted_to_fp16)[name = string("op_2511_cast_fp16")]; int32 var_2509 = const()[name = string("op_2509"), val = int32(1)]; bool doubled_41_interleave_0 = const()[name = string("doubled_41_interleave_0"), val = bool(false)]; tensor doubled_41_cast_fp16 = concat(axis = var_2509, interleave = doubled_41_interleave_0, values = (hidden_states_49_cast_fp16, var_2511_cast_fp16))[name = string("doubled_41_cast_fp16")]; tensor out_21_axes_0 = const()[name = string("out_21_axes_0"), val = tensor([1])]; tensor out_21_gamma_0_to_fp16 = const()[name = string("out_21_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283137536)))]; fp16 var_2521_to_fp16 = const()[name = string("op_2521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_21_cast_fp16 = layer_norm(axes = out_21_axes_0, epsilon = var_2521_to_fp16, gamma = out_21_gamma_0_to_fp16, x = doubled_41_cast_fp16)[name = string("out_21_cast_fp16")]; tensor var_2532_split_sizes_0 = const()[name = string("op_2532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2532_axis_0 = const()[name = string("op_2532_axis_0"), val = int32(1)]; tensor var_2532_cast_fp16_0, tensor var_2532_cast_fp16_1 = split(axis = var_2532_axis_0, split_sizes = var_2532_split_sizes_0, x = out_21_cast_fp16)[name = string("op_2532_cast_fp16")]; tensor layers_5_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283145792)))]; tensor query_states_31_strides_0 = const()[name = string("query_states_31_strides_0"), val = tensor([1, 1])]; string query_states_31_pad_type_0 = const()[name = string("query_states_31_pad_type_0"), val = string("valid")]; tensor query_states_31_pad_0 = const()[name = string("query_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_31_dilations_0 = const()[name = string("query_states_31_dilations_0"), val = tensor([1, 1])]; int32 query_states_31_groups_0 = const()[name = string("query_states_31_groups_0"), val = int32(1)]; tensor query_states_31_cast_fp16 = conv(dilations = query_states_31_dilations_0, groups = query_states_31_groups_0, pad = query_states_31_pad_0, pad_type = query_states_31_pad_type_0, strides = query_states_31_strides_0, weight = layers_5_self_attn_q_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("query_states_31_cast_fp16")]; tensor layers_5_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1291534464)))]; tensor key_states_51_strides_0 = const()[name = string("key_states_51_strides_0"), val = tensor([1, 1])]; string key_states_51_pad_type_0 = const()[name = string("key_states_51_pad_type_0"), val = string("valid")]; tensor key_states_51_pad_0 = const()[name = string("key_states_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_51_dilations_0 = const()[name = string("key_states_51_dilations_0"), val = tensor([1, 1])]; int32 key_states_51_groups_0 = const()[name = string("key_states_51_groups_0"), val = int32(1)]; tensor key_states_51_cast_fp16 = conv(dilations = key_states_51_dilations_0, groups = key_states_51_groups_0, pad = key_states_51_pad_0, pad_type = key_states_51_pad_type_0, strides = key_states_51_strides_0, weight = layers_5_self_attn_k_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("key_states_51_cast_fp16")]; tensor value_states_31_strides_0 = const()[name = string("value_states_31_strides_0"), val = tensor([1, 1])]; string value_states_31_pad_type_0 = const()[name = string("value_states_31_pad_type_0"), val = string("valid")]; tensor value_states_31_pad_0 = const()[name = string("value_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_31_dilations_0 = const()[name = string("value_states_31_dilations_0"), val = tensor([1, 1])]; int32 value_states_31_groups_0 = const()[name = string("value_states_31_groups_0"), val = int32(1)]; tensor value_states_31_cast_fp16 = conv(dilations = value_states_31_dilations_0, groups = value_states_31_groups_0, pad = value_states_31_pad_0, pad_type = value_states_31_pad_type_0, strides = value_states_31_strides_0, weight = layers_5_self_attn_v_proj_weight_cast_fp16, x = var_2532_cast_fp16_0)[name = string("value_states_31_cast_fp16")]; tensor concat_60x = const()[name = string("concat_60x"), val = tensor([1, 16, 128, -1])]; tensor x_51_cast_fp16 = reshape(shape = concat_60x, x = query_states_31_cast_fp16)[name = string("x_51_cast_fp16")]; tensor concat_61x = const()[name = string("concat_61x"), val = tensor([1, 2, 128, -1])]; tensor var_2589_cast_fp16 = reshape(shape = concat_61x, x = key_states_51_cast_fp16)[name = string("op_2589_cast_fp16")]; tensor concat_62x = const()[name = string("concat_62x"), val = tensor([1, 2, 128, -1])]; tensor var_2596_cast_fp16 = reshape(shape = concat_62x, x = value_states_31_cast_fp16)[name = string("op_2596_cast_fp16")]; tensor var_2600_cast_fp16 = mul(x = x_51_cast_fp16, y = var_869_cast_fp16)[name = string("op_2600_cast_fp16")]; tensor var_2601_split_sizes_0 = const()[name = string("op_2601_split_sizes_0"), val = tensor([64, 64])]; int32 var_2601_axis_0 = const()[name = string("op_2601_axis_0"), val = int32(-2)]; tensor var_2601_cast_fp16_0, tensor var_2601_cast_fp16_1 = split(axis = var_2601_axis_0, split_sizes = var_2601_split_sizes_0, x = x_51_cast_fp16)[name = string("op_2601_cast_fp16")]; fp16 const_52_promoted_to_fp16 = const()[name = string("const_52_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2603_cast_fp16 = mul(x = var_2601_cast_fp16_1, y = const_52_promoted_to_fp16)[name = string("op_2603_cast_fp16")]; int32 var_2605 = const()[name = string("op_2605"), val = int32(-2)]; bool var_2606_interleave_0 = const()[name = string("op_2606_interleave_0"), val = bool(false)]; tensor var_2606_cast_fp16 = concat(axis = var_2605, interleave = var_2606_interleave_0, values = (var_2603_cast_fp16, var_2601_cast_fp16_0))[name = string("op_2606_cast_fp16")]; tensor var_2607_cast_fp16 = mul(x = var_2606_cast_fp16, y = var_878_cast_fp16)[name = string("op_2607_cast_fp16")]; tensor query_states_33_cast_fp16 = add(x = var_2600_cast_fp16, y = var_2607_cast_fp16)[name = string("query_states_33_cast_fp16")]; tensor var_2613_cast_fp16 = mul(x = var_2589_cast_fp16, y = var_869_cast_fp16)[name = string("op_2613_cast_fp16")]; tensor var_2614_split_sizes_0 = const()[name = string("op_2614_split_sizes_0"), val = tensor([64, 64])]; int32 var_2614_axis_0 = const()[name = string("op_2614_axis_0"), val = int32(-2)]; tensor var_2614_cast_fp16_0, tensor var_2614_cast_fp16_1 = split(axis = var_2614_axis_0, split_sizes = var_2614_split_sizes_0, x = var_2589_cast_fp16)[name = string("op_2614_cast_fp16")]; fp16 const_53_promoted_to_fp16 = const()[name = string("const_53_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2616_cast_fp16 = mul(x = var_2614_cast_fp16_1, y = const_53_promoted_to_fp16)[name = string("op_2616_cast_fp16")]; int32 var_2618 = const()[name = string("op_2618"), val = int32(-2)]; bool var_2619_interleave_0 = const()[name = string("op_2619_interleave_0"), val = bool(false)]; tensor var_2619_cast_fp16 = concat(axis = var_2618, interleave = var_2619_interleave_0, values = (var_2616_cast_fp16, var_2614_cast_fp16_0))[name = string("op_2619_cast_fp16")]; tensor var_2620_cast_fp16 = mul(x = var_2619_cast_fp16, y = var_878_cast_fp16)[name = string("op_2620_cast_fp16")]; tensor key_states_55_cast_fp16 = add(x = var_2613_cast_fp16, y = var_2620_cast_fp16)[name = string("key_states_55_cast_fp16")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; int32 concat_65_axis_0 = const()[name = string("concat_65_axis_0"), val = int32(0)]; bool concat_65_interleave_0 = const()[name = string("concat_65_interleave_0"), val = bool(false)]; tensor concat_65 = concat(axis = concat_65_axis_0, interleave = concat_65_interleave_0, values = (expand_dims_60, expand_dims_61, position_id, expand_dims_63))[name = string("concat_65")]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; tensor concat_66_values1_0 = const()[name = string("concat_66_values1_0"), val = tensor([0])]; tensor concat_66_values3_0 = const()[name = string("concat_66_values3_0"), val = tensor([0])]; int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_64, concat_66_values1_0, cache_position_end, concat_66_values3_0))[name = string("concat_66")]; tensor key_states_57_perm_0 = const()[name = string("key_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_6_stride_0 = const()[name = string("key_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_57_cast_fp16 = transpose(perm = key_states_57_perm_0, x = key_states_55_cast_fp16)[name = string("transpose_154")]; tensor key_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = key_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = key_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_6_squeeze_mask_0, stride = key_cache_internal_tensor_assign_6_stride_0, update = key_states_57_cast_fp16, x = coreml_update_state_64)[name = string("key_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_6_cast_fp16, input = key_cache)[name = string("coreml_update_state_66_write_state")]; tensor coreml_update_state_66 = read_state(input = key_cache)[name = string("coreml_update_state_66")]; tensor value_states_33_perm_0 = const()[name = string("value_states_33_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_6_stride_0 = const()[name = string("value_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_33_cast_fp16 = transpose(perm = value_states_33_perm_0, x = var_2596_cast_fp16)[name = string("transpose_153")]; tensor value_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = value_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = value_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_6_squeeze_mask_0, stride = value_cache_internal_tensor_assign_6_stride_0, update = value_states_33_cast_fp16, x = coreml_update_state_65)[name = string("value_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_6_cast_fp16, input = value_cache)[name = string("coreml_update_state_67_write_state")]; tensor coreml_update_state_67 = read_state(input = value_cache)[name = string("coreml_update_state_67")]; tensor var_2690_begin_0 = const()[name = string("op_2690_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2690_end_0 = const()[name = string("op_2690_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2690_end_mask_0 = const()[name = string("op_2690_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2690_cast_fp16 = slice_by_index(begin = var_2690_begin_0, end = var_2690_end_0, end_mask = var_2690_end_mask_0, x = coreml_update_state_66)[name = string("op_2690_cast_fp16")]; tensor tile_10 = const()[name = string("tile_10"), val = tensor([1, 1])]; int32 var_2693_axis_0 = const()[name = string("op_2693_axis_0"), val = int32(1)]; tensor var_2693_cast_fp16_0, tensor var_2693_cast_fp16_1 = split(axis = var_2693_axis_0, split_sizes = tile_10, x = var_2690_cast_fp16)[name = string("op_2693_cast_fp16")]; tensor var_2700_begin_0 = const()[name = string("op_2700_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2700_end_0 = const()[name = string("op_2700_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2700_end_mask_0 = const()[name = string("op_2700_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2700_cast_fp16 = slice_by_index(begin = var_2700_begin_0, end = var_2700_end_0, end_mask = var_2700_end_mask_0, x = coreml_update_state_67)[name = string("op_2700_cast_fp16")]; tensor tile_11 = const()[name = string("tile_11"), val = tensor([1, 1])]; int32 var_2703_axis_0 = const()[name = string("op_2703_axis_0"), val = int32(1)]; tensor var_2703_cast_fp16_0, tensor var_2703_cast_fp16_1 = split(axis = var_2703_axis_0, split_sizes = tile_11, x = var_2700_cast_fp16)[name = string("op_2703_cast_fp16")]; tensor var_2706_split_sizes_0 = const()[name = string("op_2706_split_sizes_0"), val = tensor([8, 8])]; int32 var_2706_axis_0 = const()[name = string("op_2706_axis_0"), val = int32(1)]; tensor var_2706_0, tensor var_2706_1 = split(axis = var_2706_axis_0, split_sizes = var_2706_split_sizes_0, x = query_states_33_cast_fp16)[name = string("op_2706")]; bool attn_weights_81_transpose_x_0 = const()[name = string("attn_weights_81_transpose_x_0"), val = bool(false)]; bool attn_weights_81_transpose_y_0 = const()[name = string("attn_weights_81_transpose_y_0"), val = bool(false)]; tensor attn_weights_81_cast_fp16 = matmul(transpose_x = attn_weights_81_transpose_x_0, transpose_y = attn_weights_81_transpose_y_0, x = var_2693_cast_fp16_0, y = var_2706_0)[name = string("attn_weights_81_cast_fp16")]; fp16 var_2709_to_fp16 = const()[name = string("op_2709_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_83_cast_fp16 = mul(x = attn_weights_81_cast_fp16, y = var_2709_to_fp16)[name = string("attn_weights_83_cast_fp16")]; tensor attn_weights_85_cast_fp16 = add(x = attn_weights_83_cast_fp16, y = attn_mask_1)[name = string("attn_weights_85_cast_fp16")]; int32 var_2713 = const()[name = string("op_2713"), val = int32(-2)]; tensor attn_weights_87_cast_fp16 = softmax(axis = var_2713, x = attn_weights_85_cast_fp16)[name = string("attn_weights_87_cast_fp16")]; bool var_2719_transpose_x_1 = const()[name = string("op_2719_transpose_x_1"), val = bool(true)]; bool var_2719_transpose_y_1 = const()[name = string("op_2719_transpose_y_1"), val = bool(false)]; tensor var_2719_cast_fp16 = matmul(transpose_x = var_2719_transpose_x_1, transpose_y = var_2719_transpose_y_1, x = attn_weights_87_cast_fp16, y = var_2703_cast_fp16_0)[name = string("op_2719_cast_fp16")]; bool attn_weights_89_transpose_x_0 = const()[name = string("attn_weights_89_transpose_x_0"), val = bool(false)]; bool attn_weights_89_transpose_y_0 = const()[name = string("attn_weights_89_transpose_y_0"), val = bool(false)]; tensor attn_weights_89_cast_fp16 = matmul(transpose_x = attn_weights_89_transpose_x_0, transpose_y = attn_weights_89_transpose_y_0, x = var_2693_cast_fp16_1, y = var_2706_1)[name = string("attn_weights_89_cast_fp16")]; fp16 var_2721_to_fp16 = const()[name = string("op_2721_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_91_cast_fp16 = mul(x = attn_weights_89_cast_fp16, y = var_2721_to_fp16)[name = string("attn_weights_91_cast_fp16")]; tensor attn_weights_93_cast_fp16 = add(x = attn_weights_91_cast_fp16, y = attn_mask_1)[name = string("attn_weights_93_cast_fp16")]; int32 var_2725 = const()[name = string("op_2725"), val = int32(-2)]; tensor attn_weights_95_cast_fp16 = softmax(axis = var_2725, x = attn_weights_93_cast_fp16)[name = string("attn_weights_95_cast_fp16")]; bool attn_output_41_transpose_x_1 = const()[name = string("attn_output_41_transpose_x_1"), val = bool(true)]; bool attn_output_41_transpose_y_1 = const()[name = string("attn_output_41_transpose_y_1"), val = bool(false)]; tensor attn_output_41_cast_fp16 = matmul(transpose_x = attn_output_41_transpose_x_1, transpose_y = attn_output_41_transpose_y_1, x = attn_weights_95_cast_fp16, y = var_2703_cast_fp16_1)[name = string("attn_output_41_cast_fp16")]; int32 var_2733 = const()[name = string("op_2733"), val = int32(1)]; bool attn_output_43_interleave_0 = const()[name = string("attn_output_43_interleave_0"), val = bool(false)]; tensor attn_output_43_cast_fp16 = concat(axis = var_2733, interleave = attn_output_43_interleave_0, values = (var_2719_cast_fp16, attn_output_41_cast_fp16))[name = string("attn_output_43_cast_fp16")]; tensor var_2737_perm_0 = const()[name = string("op_2737_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_71x = const()[name = string("concat_71x"), val = tensor([1, 2048, 1, -1])]; tensor var_2737_cast_fp16 = transpose(perm = var_2737_perm_0, x = attn_output_43_cast_fp16)[name = string("transpose_152")]; tensor attn_output_47_cast_fp16 = reshape(shape = concat_71x, x = var_2737_cast_fp16)[name = string("attn_output_47_cast_fp16")]; tensor hidden_states_53_strides_0 = const()[name = string("hidden_states_53_strides_0"), val = tensor([1, 1])]; string hidden_states_53_pad_type_0 = const()[name = string("hidden_states_53_pad_type_0"), val = string("valid")]; tensor hidden_states_53_pad_0 = const()[name = string("hidden_states_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_53_dilations_0 = const()[name = string("hidden_states_53_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_53_groups_0 = const()[name = string("hidden_states_53_groups_0"), val = int32(1)]; tensor hidden_states_53_cast_fp16 = conv(dilations = hidden_states_53_dilations_0, groups = hidden_states_53_groups_0, pad = hidden_states_53_pad_0, pad_type = hidden_states_53_pad_type_0, strides = hidden_states_53_strides_0, weight = layers_5_self_attn_o_proj_weight_cast_fp16, x = attn_output_47_cast_fp16)[name = string("hidden_states_53_cast_fp16")]; tensor hidden_states_55_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = hidden_states_53_cast_fp16)[name = string("hidden_states_55_cast_fp16")]; fp16 const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2770_cast_fp16 = mul(x = hidden_states_55_cast_fp16, y = const_58_promoted_to_fp16)[name = string("op_2770_cast_fp16")]; int32 var_2768 = const()[name = string("op_2768"), val = int32(1)]; bool doubled_45_interleave_0 = const()[name = string("doubled_45_interleave_0"), val = bool(false)]; tensor doubled_45_cast_fp16 = concat(axis = var_2768, interleave = doubled_45_interleave_0, values = (hidden_states_55_cast_fp16, var_2770_cast_fp16))[name = string("doubled_45_cast_fp16")]; tensor out_23_axes_0 = const()[name = string("out_23_axes_0"), val = tensor([1])]; tensor out_23_gamma_0_to_fp16 = const()[name = string("out_23_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292583104)))]; fp16 var_2780_to_fp16 = const()[name = string("op_2780_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_23_cast_fp16 = layer_norm(axes = out_23_axes_0, epsilon = var_2780_to_fp16, gamma = out_23_gamma_0_to_fp16, x = doubled_45_cast_fp16)[name = string("out_23_cast_fp16")]; tensor var_2791_split_sizes_0 = const()[name = string("op_2791_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2791_axis_0 = const()[name = string("op_2791_axis_0"), val = int32(1)]; tensor var_2791_cast_fp16_0, tensor var_2791_cast_fp16_1 = split(axis = var_2791_axis_0, split_sizes = var_2791_split_sizes_0, x = out_23_cast_fp16)[name = string("op_2791_cast_fp16")]; tensor layers_5_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_5_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292591360)))]; tensor input_11_strides_0 = const()[name = string("input_11_strides_0"), val = tensor([1, 1])]; string input_11_pad_type_0 = const()[name = string("input_11_pad_type_0"), val = string("valid")]; tensor input_11_pad_0 = const()[name = string("input_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_11_dilations_0 = const()[name = string("input_11_dilations_0"), val = tensor([1, 1])]; int32 input_11_groups_0 = const()[name = string("input_11_groups_0"), val = int32(1)]; tensor input_11_cast_fp16 = conv(dilations = input_11_dilations_0, groups = input_11_groups_0, pad = input_11_pad_0, pad_type = input_11_pad_type_0, strides = input_11_strides_0, weight = layers_5_mlp_gate_proj_weight_to_fp16, x = var_2791_cast_fp16_0)[name = string("input_11_cast_fp16")]; tensor var_2808_cast_fp16 = silu(x = input_11_cast_fp16)[name = string("op_2808_cast_fp16")]; tensor var_2814_strides_0 = const()[name = string("op_2814_strides_0"), val = tensor([1, 1])]; string var_2814_pad_type_0 = const()[name = string("op_2814_pad_type_0"), val = string("valid")]; tensor var_2814_pad_0 = const()[name = string("op_2814_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2814_dilations_0 = const()[name = string("op_2814_dilations_0"), val = tensor([1, 1])]; int32 var_2814_groups_0 = const()[name = string("op_2814_groups_0"), val = int32(1)]; tensor var_2814_cast_fp16 = conv(dilations = var_2814_dilations_0, groups = var_2814_groups_0, pad = var_2814_pad_0, pad_type = var_2814_pad_type_0, strides = var_2814_strides_0, weight = layers_5_mlp_up_proj_weight_cast_fp16, x = var_2791_cast_fp16_0)[name = string("op_2814_cast_fp16")]; tensor x_59_cast_fp16 = mul(x = var_2808_cast_fp16, y = var_2814_cast_fp16)[name = string("x_59_cast_fp16")]; tensor hidden_states_57_strides_0 = const()[name = string("hidden_states_57_strides_0"), val = tensor([1, 1])]; string hidden_states_57_pad_type_0 = const()[name = string("hidden_states_57_pad_type_0"), val = string("valid")]; tensor hidden_states_57_pad_0 = const()[name = string("hidden_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_57_dilations_0 = const()[name = string("hidden_states_57_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_57_groups_0 = const()[name = string("hidden_states_57_groups_0"), val = int32(1)]; tensor hidden_states_57_cast_fp16 = conv(dilations = hidden_states_57_dilations_0, groups = hidden_states_57_groups_0, pad = hidden_states_57_pad_0, pad_type = hidden_states_57_pad_type_0, strides = hidden_states_57_strides_0, weight = layers_5_mlp_down_proj_weight_cast_fp16, x = x_59_cast_fp16)[name = string("hidden_states_57_cast_fp16")]; tensor hidden_states_59_cast_fp16 = add(x = hidden_states_55_cast_fp16, y = hidden_states_57_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2832_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_2832_cast_fp16")]; int32 var_2830 = const()[name = string("op_2830"), val = int32(1)]; bool doubled_49_interleave_0 = const()[name = string("doubled_49_interleave_0"), val = bool(false)]; tensor doubled_49_cast_fp16 = concat(axis = var_2830, interleave = doubled_49_interleave_0, values = (hidden_states_59_cast_fp16, var_2832_cast_fp16))[name = string("doubled_49_cast_fp16")]; tensor out_25_axes_0 = const()[name = string("out_25_axes_0"), val = tensor([1])]; tensor out_25_gamma_0_to_fp16 = const()[name = string("out_25_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317757248)))]; fp16 var_2842_to_fp16 = const()[name = string("op_2842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_25_cast_fp16 = layer_norm(axes = out_25_axes_0, epsilon = var_2842_to_fp16, gamma = out_25_gamma_0_to_fp16, x = doubled_49_cast_fp16)[name = string("out_25_cast_fp16")]; tensor var_2853_split_sizes_0 = const()[name = string("op_2853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2853_axis_0 = const()[name = string("op_2853_axis_0"), val = int32(1)]; tensor var_2853_cast_fp16_0, tensor var_2853_cast_fp16_1 = split(axis = var_2853_axis_0, split_sizes = var_2853_split_sizes_0, x = out_25_cast_fp16)[name = string("op_2853_cast_fp16")]; tensor layers_6_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317765504)))]; tensor query_states_37_strides_0 = const()[name = string("query_states_37_strides_0"), val = tensor([1, 1])]; string query_states_37_pad_type_0 = const()[name = string("query_states_37_pad_type_0"), val = string("valid")]; tensor query_states_37_pad_0 = const()[name = string("query_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_37_dilations_0 = const()[name = string("query_states_37_dilations_0"), val = tensor([1, 1])]; int32 query_states_37_groups_0 = const()[name = string("query_states_37_groups_0"), val = int32(1)]; tensor query_states_37_cast_fp16 = conv(dilations = query_states_37_dilations_0, groups = query_states_37_groups_0, pad = query_states_37_pad_0, pad_type = query_states_37_pad_type_0, strides = query_states_37_strides_0, weight = layers_6_self_attn_q_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("query_states_37_cast_fp16")]; tensor layers_6_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1326154176)))]; tensor key_states_61_strides_0 = const()[name = string("key_states_61_strides_0"), val = tensor([1, 1])]; string key_states_61_pad_type_0 = const()[name = string("key_states_61_pad_type_0"), val = string("valid")]; tensor key_states_61_pad_0 = const()[name = string("key_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_61_dilations_0 = const()[name = string("key_states_61_dilations_0"), val = tensor([1, 1])]; int32 key_states_61_groups_0 = const()[name = string("key_states_61_groups_0"), val = int32(1)]; tensor key_states_61_cast_fp16 = conv(dilations = key_states_61_dilations_0, groups = key_states_61_groups_0, pad = key_states_61_pad_0, pad_type = key_states_61_pad_type_0, strides = key_states_61_strides_0, weight = layers_6_self_attn_k_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("key_states_61_cast_fp16")]; tensor value_states_37_strides_0 = const()[name = string("value_states_37_strides_0"), val = tensor([1, 1])]; string value_states_37_pad_type_0 = const()[name = string("value_states_37_pad_type_0"), val = string("valid")]; tensor value_states_37_pad_0 = const()[name = string("value_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_37_dilations_0 = const()[name = string("value_states_37_dilations_0"), val = tensor([1, 1])]; int32 value_states_37_groups_0 = const()[name = string("value_states_37_groups_0"), val = int32(1)]; tensor value_states_37_cast_fp16 = conv(dilations = value_states_37_dilations_0, groups = value_states_37_groups_0, pad = value_states_37_pad_0, pad_type = value_states_37_pad_type_0, strides = value_states_37_strides_0, weight = layers_6_self_attn_v_proj_weight_cast_fp16, x = var_2853_cast_fp16_0)[name = string("value_states_37_cast_fp16")]; tensor concat_72x = const()[name = string("concat_72x"), val = tensor([1, 16, 128, -1])]; tensor x_61_cast_fp16 = reshape(shape = concat_72x, x = query_states_37_cast_fp16)[name = string("x_61_cast_fp16")]; tensor concat_73x = const()[name = string("concat_73x"), val = tensor([1, 2, 128, -1])]; tensor var_2910_cast_fp16 = reshape(shape = concat_73x, x = key_states_61_cast_fp16)[name = string("op_2910_cast_fp16")]; tensor concat_74x = const()[name = string("concat_74x"), val = tensor([1, 2, 128, -1])]; tensor var_2917_cast_fp16 = reshape(shape = concat_74x, x = value_states_37_cast_fp16)[name = string("op_2917_cast_fp16")]; tensor var_2921_cast_fp16 = mul(x = x_61_cast_fp16, y = var_869_cast_fp16)[name = string("op_2921_cast_fp16")]; tensor var_2922_split_sizes_0 = const()[name = string("op_2922_split_sizes_0"), val = tensor([64, 64])]; int32 var_2922_axis_0 = const()[name = string("op_2922_axis_0"), val = int32(-2)]; tensor var_2922_cast_fp16_0, tensor var_2922_cast_fp16_1 = split(axis = var_2922_axis_0, split_sizes = var_2922_split_sizes_0, x = x_61_cast_fp16)[name = string("op_2922_cast_fp16")]; fp16 const_62_promoted_to_fp16 = const()[name = string("const_62_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2924_cast_fp16 = mul(x = var_2922_cast_fp16_1, y = const_62_promoted_to_fp16)[name = string("op_2924_cast_fp16")]; int32 var_2926 = const()[name = string("op_2926"), val = int32(-2)]; bool var_2927_interleave_0 = const()[name = string("op_2927_interleave_0"), val = bool(false)]; tensor var_2927_cast_fp16 = concat(axis = var_2926, interleave = var_2927_interleave_0, values = (var_2924_cast_fp16, var_2922_cast_fp16_0))[name = string("op_2927_cast_fp16")]; tensor var_2928_cast_fp16 = mul(x = var_2927_cast_fp16, y = var_878_cast_fp16)[name = string("op_2928_cast_fp16")]; tensor query_states_39_cast_fp16 = add(x = var_2921_cast_fp16, y = var_2928_cast_fp16)[name = string("query_states_39_cast_fp16")]; tensor var_2934_cast_fp16 = mul(x = var_2910_cast_fp16, y = var_869_cast_fp16)[name = string("op_2934_cast_fp16")]; tensor var_2935_split_sizes_0 = const()[name = string("op_2935_split_sizes_0"), val = tensor([64, 64])]; int32 var_2935_axis_0 = const()[name = string("op_2935_axis_0"), val = int32(-2)]; tensor var_2935_cast_fp16_0, tensor var_2935_cast_fp16_1 = split(axis = var_2935_axis_0, split_sizes = var_2935_split_sizes_0, x = var_2910_cast_fp16)[name = string("op_2935_cast_fp16")]; fp16 const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2937_cast_fp16 = mul(x = var_2935_cast_fp16_1, y = const_63_promoted_to_fp16)[name = string("op_2937_cast_fp16")]; int32 var_2939 = const()[name = string("op_2939"), val = int32(-2)]; bool var_2940_interleave_0 = const()[name = string("op_2940_interleave_0"), val = bool(false)]; tensor var_2940_cast_fp16 = concat(axis = var_2939, interleave = var_2940_interleave_0, values = (var_2937_cast_fp16, var_2935_cast_fp16_0))[name = string("op_2940_cast_fp16")]; tensor var_2941_cast_fp16 = mul(x = var_2940_cast_fp16, y = var_878_cast_fp16)[name = string("op_2941_cast_fp16")]; tensor key_states_65_cast_fp16 = add(x = var_2934_cast_fp16, y = var_2941_cast_fp16)[name = string("key_states_65_cast_fp16")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; int32 concat_77_axis_0 = const()[name = string("concat_77_axis_0"), val = int32(0)]; bool concat_77_interleave_0 = const()[name = string("concat_77_interleave_0"), val = bool(false)]; tensor concat_77 = concat(axis = concat_77_axis_0, interleave = concat_77_interleave_0, values = (expand_dims_72, expand_dims_73, position_id, expand_dims_75))[name = string("concat_77")]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; tensor concat_78_values1_0 = const()[name = string("concat_78_values1_0"), val = tensor([0])]; tensor concat_78_values3_0 = const()[name = string("concat_78_values3_0"), val = tensor([0])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_76, concat_78_values1_0, cache_position_end, concat_78_values3_0))[name = string("concat_78")]; tensor key_states_67_perm_0 = const()[name = string("key_states_67_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_7_stride_0 = const()[name = string("key_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_67_cast_fp16 = transpose(perm = key_states_67_perm_0, x = key_states_65_cast_fp16)[name = string("transpose_151")]; tensor key_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = key_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = key_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_7_squeeze_mask_0, stride = key_cache_internal_tensor_assign_7_stride_0, update = key_states_67_cast_fp16, x = coreml_update_state_66)[name = string("key_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_7_cast_fp16, input = key_cache)[name = string("coreml_update_state_68_write_state")]; tensor coreml_update_state_68 = read_state(input = key_cache)[name = string("coreml_update_state_68")]; tensor value_states_39_perm_0 = const()[name = string("value_states_39_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_7_stride_0 = const()[name = string("value_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_39_cast_fp16 = transpose(perm = value_states_39_perm_0, x = var_2917_cast_fp16)[name = string("transpose_150")]; tensor value_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = value_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = value_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_7_squeeze_mask_0, stride = value_cache_internal_tensor_assign_7_stride_0, update = value_states_39_cast_fp16, x = coreml_update_state_67)[name = string("value_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_7_cast_fp16, input = value_cache)[name = string("coreml_update_state_69_write_state")]; tensor coreml_update_state_69 = read_state(input = value_cache)[name = string("coreml_update_state_69")]; tensor var_3011_begin_0 = const()[name = string("op_3011_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3011_end_0 = const()[name = string("op_3011_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3011_end_mask_0 = const()[name = string("op_3011_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3011_cast_fp16 = slice_by_index(begin = var_3011_begin_0, end = var_3011_end_0, end_mask = var_3011_end_mask_0, x = coreml_update_state_68)[name = string("op_3011_cast_fp16")]; tensor tile_12 = const()[name = string("tile_12"), val = tensor([1, 1])]; int32 var_3014_axis_0 = const()[name = string("op_3014_axis_0"), val = int32(1)]; tensor var_3014_cast_fp16_0, tensor var_3014_cast_fp16_1 = split(axis = var_3014_axis_0, split_sizes = tile_12, x = var_3011_cast_fp16)[name = string("op_3014_cast_fp16")]; tensor var_3021_begin_0 = const()[name = string("op_3021_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3021_end_0 = const()[name = string("op_3021_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3021_end_mask_0 = const()[name = string("op_3021_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3021_cast_fp16 = slice_by_index(begin = var_3021_begin_0, end = var_3021_end_0, end_mask = var_3021_end_mask_0, x = coreml_update_state_69)[name = string("op_3021_cast_fp16")]; tensor tile_13 = const()[name = string("tile_13"), val = tensor([1, 1])]; int32 var_3024_axis_0 = const()[name = string("op_3024_axis_0"), val = int32(1)]; tensor var_3024_cast_fp16_0, tensor var_3024_cast_fp16_1 = split(axis = var_3024_axis_0, split_sizes = tile_13, x = var_3021_cast_fp16)[name = string("op_3024_cast_fp16")]; tensor var_3027_split_sizes_0 = const()[name = string("op_3027_split_sizes_0"), val = tensor([8, 8])]; int32 var_3027_axis_0 = const()[name = string("op_3027_axis_0"), val = int32(1)]; tensor var_3027_0, tensor var_3027_1 = split(axis = var_3027_axis_0, split_sizes = var_3027_split_sizes_0, x = query_states_39_cast_fp16)[name = string("op_3027")]; bool attn_weights_97_transpose_x_0 = const()[name = string("attn_weights_97_transpose_x_0"), val = bool(false)]; bool attn_weights_97_transpose_y_0 = const()[name = string("attn_weights_97_transpose_y_0"), val = bool(false)]; tensor attn_weights_97_cast_fp16 = matmul(transpose_x = attn_weights_97_transpose_x_0, transpose_y = attn_weights_97_transpose_y_0, x = var_3014_cast_fp16_0, y = var_3027_0)[name = string("attn_weights_97_cast_fp16")]; fp16 var_3030_to_fp16 = const()[name = string("op_3030_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_99_cast_fp16 = mul(x = attn_weights_97_cast_fp16, y = var_3030_to_fp16)[name = string("attn_weights_99_cast_fp16")]; tensor attn_weights_101_cast_fp16 = add(x = attn_weights_99_cast_fp16, y = attn_mask_1)[name = string("attn_weights_101_cast_fp16")]; int32 var_3034 = const()[name = string("op_3034"), val = int32(-2)]; tensor attn_weights_103_cast_fp16 = softmax(axis = var_3034, x = attn_weights_101_cast_fp16)[name = string("attn_weights_103_cast_fp16")]; bool var_3040_transpose_x_1 = const()[name = string("op_3040_transpose_x_1"), val = bool(true)]; bool var_3040_transpose_y_1 = const()[name = string("op_3040_transpose_y_1"), val = bool(false)]; tensor var_3040_cast_fp16 = matmul(transpose_x = var_3040_transpose_x_1, transpose_y = var_3040_transpose_y_1, x = attn_weights_103_cast_fp16, y = var_3024_cast_fp16_0)[name = string("op_3040_cast_fp16")]; bool attn_weights_105_transpose_x_0 = const()[name = string("attn_weights_105_transpose_x_0"), val = bool(false)]; bool attn_weights_105_transpose_y_0 = const()[name = string("attn_weights_105_transpose_y_0"), val = bool(false)]; tensor attn_weights_105_cast_fp16 = matmul(transpose_x = attn_weights_105_transpose_x_0, transpose_y = attn_weights_105_transpose_y_0, x = var_3014_cast_fp16_1, y = var_3027_1)[name = string("attn_weights_105_cast_fp16")]; fp16 var_3042_to_fp16 = const()[name = string("op_3042_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_107_cast_fp16 = mul(x = attn_weights_105_cast_fp16, y = var_3042_to_fp16)[name = string("attn_weights_107_cast_fp16")]; tensor attn_weights_109_cast_fp16 = add(x = attn_weights_107_cast_fp16, y = attn_mask_1)[name = string("attn_weights_109_cast_fp16")]; int32 var_3046 = const()[name = string("op_3046"), val = int32(-2)]; tensor attn_weights_111_cast_fp16 = softmax(axis = var_3046, x = attn_weights_109_cast_fp16)[name = string("attn_weights_111_cast_fp16")]; bool attn_output_49_transpose_x_1 = const()[name = string("attn_output_49_transpose_x_1"), val = bool(true)]; bool attn_output_49_transpose_y_1 = const()[name = string("attn_output_49_transpose_y_1"), val = bool(false)]; tensor attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_1, transpose_y = attn_output_49_transpose_y_1, x = attn_weights_111_cast_fp16, y = var_3024_cast_fp16_1)[name = string("attn_output_49_cast_fp16")]; int32 var_3054 = const()[name = string("op_3054"), val = int32(1)]; bool attn_output_51_interleave_0 = const()[name = string("attn_output_51_interleave_0"), val = bool(false)]; tensor attn_output_51_cast_fp16 = concat(axis = var_3054, interleave = attn_output_51_interleave_0, values = (var_3040_cast_fp16, attn_output_49_cast_fp16))[name = string("attn_output_51_cast_fp16")]; tensor var_3058_perm_0 = const()[name = string("op_3058_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_83x = const()[name = string("concat_83x"), val = tensor([1, 2048, 1, -1])]; tensor var_3058_cast_fp16 = transpose(perm = var_3058_perm_0, x = attn_output_51_cast_fp16)[name = string("transpose_149")]; tensor attn_output_55_cast_fp16 = reshape(shape = concat_83x, x = var_3058_cast_fp16)[name = string("attn_output_55_cast_fp16")]; tensor hidden_states_63_strides_0 = const()[name = string("hidden_states_63_strides_0"), val = tensor([1, 1])]; string hidden_states_63_pad_type_0 = const()[name = string("hidden_states_63_pad_type_0"), val = string("valid")]; tensor hidden_states_63_pad_0 = const()[name = string("hidden_states_63_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_63_dilations_0 = const()[name = string("hidden_states_63_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_63_groups_0 = const()[name = string("hidden_states_63_groups_0"), val = int32(1)]; tensor hidden_states_63_cast_fp16 = conv(dilations = hidden_states_63_dilations_0, groups = hidden_states_63_groups_0, pad = hidden_states_63_pad_0, pad_type = hidden_states_63_pad_type_0, strides = hidden_states_63_strides_0, weight = layers_6_self_attn_o_proj_weight_cast_fp16, x = attn_output_55_cast_fp16)[name = string("hidden_states_63_cast_fp16")]; tensor hidden_states_65_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = hidden_states_63_cast_fp16)[name = string("hidden_states_65_cast_fp16")]; fp16 const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3091_cast_fp16 = mul(x = hidden_states_65_cast_fp16, y = const_68_promoted_to_fp16)[name = string("op_3091_cast_fp16")]; int32 var_3089 = const()[name = string("op_3089"), val = int32(1)]; bool doubled_53_interleave_0 = const()[name = string("doubled_53_interleave_0"), val = bool(false)]; tensor doubled_53_cast_fp16 = concat(axis = var_3089, interleave = doubled_53_interleave_0, values = (hidden_states_65_cast_fp16, var_3091_cast_fp16))[name = string("doubled_53_cast_fp16")]; tensor out_27_axes_0 = const()[name = string("out_27_axes_0"), val = tensor([1])]; tensor out_27_gamma_0_to_fp16 = const()[name = string("out_27_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327202816)))]; fp16 var_3101_to_fp16 = const()[name = string("op_3101_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_27_cast_fp16 = layer_norm(axes = out_27_axes_0, epsilon = var_3101_to_fp16, gamma = out_27_gamma_0_to_fp16, x = doubled_53_cast_fp16)[name = string("out_27_cast_fp16")]; tensor var_3112_split_sizes_0 = const()[name = string("op_3112_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3112_axis_0 = const()[name = string("op_3112_axis_0"), val = int32(1)]; tensor var_3112_cast_fp16_0, tensor var_3112_cast_fp16_1 = split(axis = var_3112_axis_0, split_sizes = var_3112_split_sizes_0, x = out_27_cast_fp16)[name = string("op_3112_cast_fp16")]; tensor input_13_strides_0 = const()[name = string("input_13_strides_0"), val = tensor([1, 1])]; string input_13_pad_type_0 = const()[name = string("input_13_pad_type_0"), val = string("valid")]; tensor input_13_pad_0 = const()[name = string("input_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_13_dilations_0 = const()[name = string("input_13_dilations_0"), val = tensor([1, 1])]; int32 input_13_groups_0 = const()[name = string("input_13_groups_0"), val = int32(1)]; tensor input_13_cast_fp16 = conv(dilations = input_13_dilations_0, groups = input_13_groups_0, pad = input_13_pad_0, pad_type = input_13_pad_type_0, strides = input_13_strides_0, weight = layers_6_mlp_gate_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("input_13_cast_fp16")]; tensor var_3129_cast_fp16 = silu(x = input_13_cast_fp16)[name = string("op_3129_cast_fp16")]; tensor var_3135_strides_0 = const()[name = string("op_3135_strides_0"), val = tensor([1, 1])]; string var_3135_pad_type_0 = const()[name = string("op_3135_pad_type_0"), val = string("valid")]; tensor var_3135_pad_0 = const()[name = string("op_3135_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3135_dilations_0 = const()[name = string("op_3135_dilations_0"), val = tensor([1, 1])]; int32 var_3135_groups_0 = const()[name = string("op_3135_groups_0"), val = int32(1)]; tensor var_3135_cast_fp16 = conv(dilations = var_3135_dilations_0, groups = var_3135_groups_0, pad = var_3135_pad_0, pad_type = var_3135_pad_type_0, strides = var_3135_strides_0, weight = layers_6_mlp_up_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("op_3135_cast_fp16")]; tensor x_69_cast_fp16 = mul(x = var_3129_cast_fp16, y = var_3135_cast_fp16)[name = string("x_69_cast_fp16")]; tensor hidden_states_67_strides_0 = const()[name = string("hidden_states_67_strides_0"), val = tensor([1, 1])]; string hidden_states_67_pad_type_0 = const()[name = string("hidden_states_67_pad_type_0"), val = string("valid")]; tensor hidden_states_67_pad_0 = const()[name = string("hidden_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_67_dilations_0 = const()[name = string("hidden_states_67_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_67_groups_0 = const()[name = string("hidden_states_67_groups_0"), val = int32(1)]; tensor hidden_states_67_cast_fp16 = conv(dilations = hidden_states_67_dilations_0, groups = hidden_states_67_groups_0, pad = hidden_states_67_pad_0, pad_type = hidden_states_67_pad_type_0, strides = hidden_states_67_strides_0, weight = layers_6_mlp_down_proj_weight_cast_fp16, x = x_69_cast_fp16)[name = string("hidden_states_67_cast_fp16")]; tensor hidden_states_69_cast_fp16 = add(x = hidden_states_65_cast_fp16, y = hidden_states_67_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3153_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_70_promoted_to_fp16)[name = string("op_3153_cast_fp16")]; int32 var_3151 = const()[name = string("op_3151"), val = int32(1)]; bool doubled_57_interleave_0 = const()[name = string("doubled_57_interleave_0"), val = bool(false)]; tensor doubled_57_cast_fp16 = concat(axis = var_3151, interleave = doubled_57_interleave_0, values = (hidden_states_69_cast_fp16, var_3153_cast_fp16))[name = string("doubled_57_cast_fp16")]; tensor out_29_axes_0 = const()[name = string("out_29_axes_0"), val = tensor([1])]; tensor out_29_gamma_0_to_fp16 = const()[name = string("out_29_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327211072)))]; fp16 var_3163_to_fp16 = const()[name = string("op_3163_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_29_cast_fp16 = layer_norm(axes = out_29_axes_0, epsilon = var_3163_to_fp16, gamma = out_29_gamma_0_to_fp16, x = doubled_57_cast_fp16)[name = string("out_29_cast_fp16")]; tensor var_3174_split_sizes_0 = const()[name = string("op_3174_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3174_axis_0 = const()[name = string("op_3174_axis_0"), val = int32(1)]; tensor var_3174_cast_fp16_0, tensor var_3174_cast_fp16_1 = split(axis = var_3174_axis_0, split_sizes = var_3174_split_sizes_0, x = out_29_cast_fp16)[name = string("op_3174_cast_fp16")]; tensor layers_7_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327219328)))]; tensor query_states_43_strides_0 = const()[name = string("query_states_43_strides_0"), val = tensor([1, 1])]; string query_states_43_pad_type_0 = const()[name = string("query_states_43_pad_type_0"), val = string("valid")]; tensor query_states_43_pad_0 = const()[name = string("query_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_43_dilations_0 = const()[name = string("query_states_43_dilations_0"), val = tensor([1, 1])]; int32 query_states_43_groups_0 = const()[name = string("query_states_43_groups_0"), val = int32(1)]; tensor query_states_43_cast_fp16 = conv(dilations = query_states_43_dilations_0, groups = query_states_43_groups_0, pad = query_states_43_pad_0, pad_type = query_states_43_pad_type_0, strides = query_states_43_strides_0, weight = layers_7_self_attn_q_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("query_states_43_cast_fp16")]; tensor layers_7_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1335608000)))]; tensor key_states_71_strides_0 = const()[name = string("key_states_71_strides_0"), val = tensor([1, 1])]; string key_states_71_pad_type_0 = const()[name = string("key_states_71_pad_type_0"), val = string("valid")]; tensor key_states_71_pad_0 = const()[name = string("key_states_71_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_71_dilations_0 = const()[name = string("key_states_71_dilations_0"), val = tensor([1, 1])]; int32 key_states_71_groups_0 = const()[name = string("key_states_71_groups_0"), val = int32(1)]; tensor key_states_71_cast_fp16 = conv(dilations = key_states_71_dilations_0, groups = key_states_71_groups_0, pad = key_states_71_pad_0, pad_type = key_states_71_pad_type_0, strides = key_states_71_strides_0, weight = layers_7_self_attn_k_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("key_states_71_cast_fp16")]; tensor value_states_43_strides_0 = const()[name = string("value_states_43_strides_0"), val = tensor([1, 1])]; string value_states_43_pad_type_0 = const()[name = string("value_states_43_pad_type_0"), val = string("valid")]; tensor value_states_43_pad_0 = const()[name = string("value_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_43_dilations_0 = const()[name = string("value_states_43_dilations_0"), val = tensor([1, 1])]; int32 value_states_43_groups_0 = const()[name = string("value_states_43_groups_0"), val = int32(1)]; tensor value_states_43_cast_fp16 = conv(dilations = value_states_43_dilations_0, groups = value_states_43_groups_0, pad = value_states_43_pad_0, pad_type = value_states_43_pad_type_0, strides = value_states_43_strides_0, weight = layers_7_self_attn_v_proj_weight_cast_fp16, x = var_3174_cast_fp16_0)[name = string("value_states_43_cast_fp16")]; tensor concat_84x = const()[name = string("concat_84x"), val = tensor([1, 16, 128, -1])]; tensor x_71_cast_fp16 = reshape(shape = concat_84x, x = query_states_43_cast_fp16)[name = string("x_71_cast_fp16")]; tensor concat_85x = const()[name = string("concat_85x"), val = tensor([1, 2, 128, -1])]; tensor var_3231_cast_fp16 = reshape(shape = concat_85x, x = key_states_71_cast_fp16)[name = string("op_3231_cast_fp16")]; tensor concat_86x = const()[name = string("concat_86x"), val = tensor([1, 2, 128, -1])]; tensor var_3238_cast_fp16 = reshape(shape = concat_86x, x = value_states_43_cast_fp16)[name = string("op_3238_cast_fp16")]; tensor var_3242_cast_fp16 = mul(x = x_71_cast_fp16, y = var_869_cast_fp16)[name = string("op_3242_cast_fp16")]; tensor var_3243_split_sizes_0 = const()[name = string("op_3243_split_sizes_0"), val = tensor([64, 64])]; int32 var_3243_axis_0 = const()[name = string("op_3243_axis_0"), val = int32(-2)]; tensor var_3243_cast_fp16_0, tensor var_3243_cast_fp16_1 = split(axis = var_3243_axis_0, split_sizes = var_3243_split_sizes_0, x = x_71_cast_fp16)[name = string("op_3243_cast_fp16")]; fp16 const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3245_cast_fp16 = mul(x = var_3243_cast_fp16_1, y = const_72_promoted_to_fp16)[name = string("op_3245_cast_fp16")]; int32 var_3247 = const()[name = string("op_3247"), val = int32(-2)]; bool var_3248_interleave_0 = const()[name = string("op_3248_interleave_0"), val = bool(false)]; tensor var_3248_cast_fp16 = concat(axis = var_3247, interleave = var_3248_interleave_0, values = (var_3245_cast_fp16, var_3243_cast_fp16_0))[name = string("op_3248_cast_fp16")]; tensor var_3249_cast_fp16 = mul(x = var_3248_cast_fp16, y = var_878_cast_fp16)[name = string("op_3249_cast_fp16")]; tensor query_states_45_cast_fp16 = add(x = var_3242_cast_fp16, y = var_3249_cast_fp16)[name = string("query_states_45_cast_fp16")]; tensor var_3255_cast_fp16 = mul(x = var_3231_cast_fp16, y = var_869_cast_fp16)[name = string("op_3255_cast_fp16")]; tensor var_3256_split_sizes_0 = const()[name = string("op_3256_split_sizes_0"), val = tensor([64, 64])]; int32 var_3256_axis_0 = const()[name = string("op_3256_axis_0"), val = int32(-2)]; tensor var_3256_cast_fp16_0, tensor var_3256_cast_fp16_1 = split(axis = var_3256_axis_0, split_sizes = var_3256_split_sizes_0, x = var_3231_cast_fp16)[name = string("op_3256_cast_fp16")]; fp16 const_73_promoted_to_fp16 = const()[name = string("const_73_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3258_cast_fp16 = mul(x = var_3256_cast_fp16_1, y = const_73_promoted_to_fp16)[name = string("op_3258_cast_fp16")]; int32 var_3260 = const()[name = string("op_3260"), val = int32(-2)]; bool var_3261_interleave_0 = const()[name = string("op_3261_interleave_0"), val = bool(false)]; tensor var_3261_cast_fp16 = concat(axis = var_3260, interleave = var_3261_interleave_0, values = (var_3258_cast_fp16, var_3256_cast_fp16_0))[name = string("op_3261_cast_fp16")]; tensor var_3262_cast_fp16 = mul(x = var_3261_cast_fp16, y = var_878_cast_fp16)[name = string("op_3262_cast_fp16")]; tensor key_states_75_cast_fp16 = add(x = var_3255_cast_fp16, y = var_3262_cast_fp16)[name = string("key_states_75_cast_fp16")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; int32 concat_89_axis_0 = const()[name = string("concat_89_axis_0"), val = int32(0)]; bool concat_89_interleave_0 = const()[name = string("concat_89_interleave_0"), val = bool(false)]; tensor concat_89 = concat(axis = concat_89_axis_0, interleave = concat_89_interleave_0, values = (expand_dims_84, expand_dims_85, position_id, expand_dims_87))[name = string("concat_89")]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; tensor concat_90_values1_0 = const()[name = string("concat_90_values1_0"), val = tensor([0])]; tensor concat_90_values3_0 = const()[name = string("concat_90_values3_0"), val = tensor([0])]; int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_88, concat_90_values1_0, cache_position_end, concat_90_values3_0))[name = string("concat_90")]; tensor key_states_77_perm_0 = const()[name = string("key_states_77_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_8_stride_0 = const()[name = string("key_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_77_cast_fp16 = transpose(perm = key_states_77_perm_0, x = key_states_75_cast_fp16)[name = string("transpose_148")]; tensor key_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = key_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = key_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_8_squeeze_mask_0, stride = key_cache_internal_tensor_assign_8_stride_0, update = key_states_77_cast_fp16, x = coreml_update_state_68)[name = string("key_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_8_cast_fp16, input = key_cache)[name = string("coreml_update_state_70_write_state")]; tensor coreml_update_state_70 = read_state(input = key_cache)[name = string("coreml_update_state_70")]; tensor value_states_45_perm_0 = const()[name = string("value_states_45_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_8_stride_0 = const()[name = string("value_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_45_cast_fp16 = transpose(perm = value_states_45_perm_0, x = var_3238_cast_fp16)[name = string("transpose_147")]; tensor value_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = value_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = value_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_8_squeeze_mask_0, stride = value_cache_internal_tensor_assign_8_stride_0, update = value_states_45_cast_fp16, x = coreml_update_state_69)[name = string("value_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_8_cast_fp16, input = value_cache)[name = string("coreml_update_state_71_write_state")]; tensor coreml_update_state_71 = read_state(input = value_cache)[name = string("coreml_update_state_71")]; tensor var_3332_begin_0 = const()[name = string("op_3332_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3332_end_0 = const()[name = string("op_3332_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3332_end_mask_0 = const()[name = string("op_3332_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3332_cast_fp16 = slice_by_index(begin = var_3332_begin_0, end = var_3332_end_0, end_mask = var_3332_end_mask_0, x = coreml_update_state_70)[name = string("op_3332_cast_fp16")]; tensor tile_14 = const()[name = string("tile_14"), val = tensor([1, 1])]; int32 var_3335_axis_0 = const()[name = string("op_3335_axis_0"), val = int32(1)]; tensor var_3335_cast_fp16_0, tensor var_3335_cast_fp16_1 = split(axis = var_3335_axis_0, split_sizes = tile_14, x = var_3332_cast_fp16)[name = string("op_3335_cast_fp16")]; tensor var_3342_begin_0 = const()[name = string("op_3342_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3342_end_0 = const()[name = string("op_3342_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3342_end_mask_0 = const()[name = string("op_3342_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3342_cast_fp16 = slice_by_index(begin = var_3342_begin_0, end = var_3342_end_0, end_mask = var_3342_end_mask_0, x = coreml_update_state_71)[name = string("op_3342_cast_fp16")]; tensor tile_15 = const()[name = string("tile_15"), val = tensor([1, 1])]; int32 var_3345_axis_0 = const()[name = string("op_3345_axis_0"), val = int32(1)]; tensor var_3345_cast_fp16_0, tensor var_3345_cast_fp16_1 = split(axis = var_3345_axis_0, split_sizes = tile_15, x = var_3342_cast_fp16)[name = string("op_3345_cast_fp16")]; tensor var_3348_split_sizes_0 = const()[name = string("op_3348_split_sizes_0"), val = tensor([8, 8])]; int32 var_3348_axis_0 = const()[name = string("op_3348_axis_0"), val = int32(1)]; tensor var_3348_0, tensor var_3348_1 = split(axis = var_3348_axis_0, split_sizes = var_3348_split_sizes_0, x = query_states_45_cast_fp16)[name = string("op_3348")]; bool attn_weights_113_transpose_x_0 = const()[name = string("attn_weights_113_transpose_x_0"), val = bool(false)]; bool attn_weights_113_transpose_y_0 = const()[name = string("attn_weights_113_transpose_y_0"), val = bool(false)]; tensor attn_weights_113_cast_fp16 = matmul(transpose_x = attn_weights_113_transpose_x_0, transpose_y = attn_weights_113_transpose_y_0, x = var_3335_cast_fp16_0, y = var_3348_0)[name = string("attn_weights_113_cast_fp16")]; fp16 var_3351_to_fp16 = const()[name = string("op_3351_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_115_cast_fp16 = mul(x = attn_weights_113_cast_fp16, y = var_3351_to_fp16)[name = string("attn_weights_115_cast_fp16")]; tensor attn_weights_117_cast_fp16 = add(x = attn_weights_115_cast_fp16, y = attn_mask_1)[name = string("attn_weights_117_cast_fp16")]; int32 var_3355 = const()[name = string("op_3355"), val = int32(-2)]; tensor attn_weights_119_cast_fp16 = softmax(axis = var_3355, x = attn_weights_117_cast_fp16)[name = string("attn_weights_119_cast_fp16")]; bool var_3361_transpose_x_1 = const()[name = string("op_3361_transpose_x_1"), val = bool(true)]; bool var_3361_transpose_y_1 = const()[name = string("op_3361_transpose_y_1"), val = bool(false)]; tensor var_3361_cast_fp16 = matmul(transpose_x = var_3361_transpose_x_1, transpose_y = var_3361_transpose_y_1, x = attn_weights_119_cast_fp16, y = var_3345_cast_fp16_0)[name = string("op_3361_cast_fp16")]; bool attn_weights_121_transpose_x_0 = const()[name = string("attn_weights_121_transpose_x_0"), val = bool(false)]; bool attn_weights_121_transpose_y_0 = const()[name = string("attn_weights_121_transpose_y_0"), val = bool(false)]; tensor attn_weights_121_cast_fp16 = matmul(transpose_x = attn_weights_121_transpose_x_0, transpose_y = attn_weights_121_transpose_y_0, x = var_3335_cast_fp16_1, y = var_3348_1)[name = string("attn_weights_121_cast_fp16")]; fp16 var_3363_to_fp16 = const()[name = string("op_3363_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_123_cast_fp16 = mul(x = attn_weights_121_cast_fp16, y = var_3363_to_fp16)[name = string("attn_weights_123_cast_fp16")]; tensor attn_weights_125_cast_fp16 = add(x = attn_weights_123_cast_fp16, y = attn_mask_1)[name = string("attn_weights_125_cast_fp16")]; int32 var_3367 = const()[name = string("op_3367"), val = int32(-2)]; tensor attn_weights_127_cast_fp16 = softmax(axis = var_3367, x = attn_weights_125_cast_fp16)[name = string("attn_weights_127_cast_fp16")]; bool attn_output_57_transpose_x_1 = const()[name = string("attn_output_57_transpose_x_1"), val = bool(true)]; bool attn_output_57_transpose_y_1 = const()[name = string("attn_output_57_transpose_y_1"), val = bool(false)]; tensor attn_output_57_cast_fp16 = matmul(transpose_x = attn_output_57_transpose_x_1, transpose_y = attn_output_57_transpose_y_1, x = attn_weights_127_cast_fp16, y = var_3345_cast_fp16_1)[name = string("attn_output_57_cast_fp16")]; int32 var_3375 = const()[name = string("op_3375"), val = int32(1)]; bool attn_output_59_interleave_0 = const()[name = string("attn_output_59_interleave_0"), val = bool(false)]; tensor attn_output_59_cast_fp16 = concat(axis = var_3375, interleave = attn_output_59_interleave_0, values = (var_3361_cast_fp16, attn_output_57_cast_fp16))[name = string("attn_output_59_cast_fp16")]; tensor var_3379_perm_0 = const()[name = string("op_3379_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_95x = const()[name = string("concat_95x"), val = tensor([1, 2048, 1, -1])]; tensor var_3379_cast_fp16 = transpose(perm = var_3379_perm_0, x = attn_output_59_cast_fp16)[name = string("transpose_146")]; tensor attn_output_63_cast_fp16 = reshape(shape = concat_95x, x = var_3379_cast_fp16)[name = string("attn_output_63_cast_fp16")]; tensor hidden_states_73_strides_0 = const()[name = string("hidden_states_73_strides_0"), val = tensor([1, 1])]; string hidden_states_73_pad_type_0 = const()[name = string("hidden_states_73_pad_type_0"), val = string("valid")]; tensor hidden_states_73_pad_0 = const()[name = string("hidden_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_73_dilations_0 = const()[name = string("hidden_states_73_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_73_groups_0 = const()[name = string("hidden_states_73_groups_0"), val = int32(1)]; tensor hidden_states_73_cast_fp16 = conv(dilations = hidden_states_73_dilations_0, groups = hidden_states_73_groups_0, pad = hidden_states_73_pad_0, pad_type = hidden_states_73_pad_type_0, strides = hidden_states_73_strides_0, weight = layers_7_self_attn_o_proj_weight_cast_fp16, x = attn_output_63_cast_fp16)[name = string("hidden_states_73_cast_fp16")]; tensor hidden_states_75_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = hidden_states_73_cast_fp16)[name = string("hidden_states_75_cast_fp16")]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3412_cast_fp16 = mul(x = hidden_states_75_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_3412_cast_fp16")]; int32 var_3410 = const()[name = string("op_3410"), val = int32(1)]; bool doubled_61_interleave_0 = const()[name = string("doubled_61_interleave_0"), val = bool(false)]; tensor doubled_61_cast_fp16 = concat(axis = var_3410, interleave = doubled_61_interleave_0, values = (hidden_states_75_cast_fp16, var_3412_cast_fp16))[name = string("doubled_61_cast_fp16")]; tensor out_31_axes_0 = const()[name = string("out_31_axes_0"), val = tensor([1])]; tensor out_31_gamma_0_to_fp16 = const()[name = string("out_31_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336656640)))]; fp16 var_3422_to_fp16 = const()[name = string("op_3422_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_31_cast_fp16 = layer_norm(axes = out_31_axes_0, epsilon = var_3422_to_fp16, gamma = out_31_gamma_0_to_fp16, x = doubled_61_cast_fp16)[name = string("out_31_cast_fp16")]; tensor var_3433_split_sizes_0 = const()[name = string("op_3433_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3433_axis_0 = const()[name = string("op_3433_axis_0"), val = int32(1)]; tensor var_3433_cast_fp16_0, tensor var_3433_cast_fp16_1 = split(axis = var_3433_axis_0, split_sizes = var_3433_split_sizes_0, x = out_31_cast_fp16)[name = string("op_3433_cast_fp16")]; tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; tensor input_15_cast_fp16 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = layers_7_mlp_gate_proj_weight_cast_fp16, x = var_3433_cast_fp16_0)[name = string("input_15_cast_fp16")]; tensor var_3450_cast_fp16 = silu(x = input_15_cast_fp16)[name = string("op_3450_cast_fp16")]; tensor layers_7_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336664896)))]; tensor var_3456_strides_0 = const()[name = string("op_3456_strides_0"), val = tensor([1, 1])]; string var_3456_pad_type_0 = const()[name = string("op_3456_pad_type_0"), val = string("valid")]; tensor var_3456_pad_0 = const()[name = string("op_3456_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3456_dilations_0 = const()[name = string("op_3456_dilations_0"), val = tensor([1, 1])]; int32 var_3456_groups_0 = const()[name = string("op_3456_groups_0"), val = int32(1)]; tensor var_3456_cast_fp16 = conv(dilations = var_3456_dilations_0, groups = var_3456_groups_0, pad = var_3456_pad_0, pad_type = var_3456_pad_type_0, strides = var_3456_strides_0, weight = layers_7_mlp_up_proj_weight_to_fp16, x = var_3433_cast_fp16_0)[name = string("op_3456_cast_fp16")]; tensor x_79_cast_fp16 = mul(x = var_3450_cast_fp16, y = var_3456_cast_fp16)[name = string("x_79_cast_fp16")]; tensor layers_7_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1361830784)))]; tensor hidden_states_77_strides_0 = const()[name = string("hidden_states_77_strides_0"), val = tensor([1, 1])]; string hidden_states_77_pad_type_0 = const()[name = string("hidden_states_77_pad_type_0"), val = string("valid")]; tensor hidden_states_77_pad_0 = const()[name = string("hidden_states_77_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_77_dilations_0 = const()[name = string("hidden_states_77_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_77_groups_0 = const()[name = string("hidden_states_77_groups_0"), val = int32(1)]; tensor hidden_states_77_cast_fp16 = conv(dilations = hidden_states_77_dilations_0, groups = hidden_states_77_groups_0, pad = hidden_states_77_pad_0, pad_type = hidden_states_77_pad_type_0, strides = hidden_states_77_strides_0, weight = layers_7_mlp_down_proj_weight_to_fp16, x = x_79_cast_fp16)[name = string("hidden_states_77_cast_fp16")]; tensor hidden_states_79_cast_fp16 = add(x = hidden_states_75_cast_fp16, y = hidden_states_77_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3474_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_80_promoted_to_fp16)[name = string("op_3474_cast_fp16")]; int32 var_3472 = const()[name = string("op_3472"), val = int32(1)]; bool doubled_65_interleave_0 = const()[name = string("doubled_65_interleave_0"), val = bool(false)]; tensor doubled_65_cast_fp16 = concat(axis = var_3472, interleave = doubled_65_interleave_0, values = (hidden_states_79_cast_fp16, var_3474_cast_fp16))[name = string("doubled_65_cast_fp16")]; tensor out_33_axes_0 = const()[name = string("out_33_axes_0"), val = tensor([1])]; tensor out_33_gamma_0_to_fp16 = const()[name = string("out_33_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1386996672)))]; fp16 var_3484_to_fp16 = const()[name = string("op_3484_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_33_cast_fp16 = layer_norm(axes = out_33_axes_0, epsilon = var_3484_to_fp16, gamma = out_33_gamma_0_to_fp16, x = doubled_65_cast_fp16)[name = string("out_33_cast_fp16")]; tensor var_3495_split_sizes_0 = const()[name = string("op_3495_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3495_axis_0 = const()[name = string("op_3495_axis_0"), val = int32(1)]; tensor var_3495_cast_fp16_0, tensor var_3495_cast_fp16_1 = split(axis = var_3495_axis_0, split_sizes = var_3495_split_sizes_0, x = out_33_cast_fp16)[name = string("op_3495_cast_fp16")]; tensor layers_8_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1387004928)))]; tensor query_states_49_strides_0 = const()[name = string("query_states_49_strides_0"), val = tensor([1, 1])]; string query_states_49_pad_type_0 = const()[name = string("query_states_49_pad_type_0"), val = string("valid")]; tensor query_states_49_pad_0 = const()[name = string("query_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_49_dilations_0 = const()[name = string("query_states_49_dilations_0"), val = tensor([1, 1])]; int32 query_states_49_groups_0 = const()[name = string("query_states_49_groups_0"), val = int32(1)]; tensor query_states_49_cast_fp16 = conv(dilations = query_states_49_dilations_0, groups = query_states_49_groups_0, pad = query_states_49_pad_0, pad_type = query_states_49_pad_type_0, strides = query_states_49_strides_0, weight = layers_8_self_attn_q_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("query_states_49_cast_fp16")]; tensor layers_8_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1395393600)))]; tensor key_states_81_strides_0 = const()[name = string("key_states_81_strides_0"), val = tensor([1, 1])]; string key_states_81_pad_type_0 = const()[name = string("key_states_81_pad_type_0"), val = string("valid")]; tensor key_states_81_pad_0 = const()[name = string("key_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_81_dilations_0 = const()[name = string("key_states_81_dilations_0"), val = tensor([1, 1])]; int32 key_states_81_groups_0 = const()[name = string("key_states_81_groups_0"), val = int32(1)]; tensor key_states_81_cast_fp16 = conv(dilations = key_states_81_dilations_0, groups = key_states_81_groups_0, pad = key_states_81_pad_0, pad_type = key_states_81_pad_type_0, strides = key_states_81_strides_0, weight = layers_8_self_attn_k_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("key_states_81_cast_fp16")]; tensor value_states_49_strides_0 = const()[name = string("value_states_49_strides_0"), val = tensor([1, 1])]; string value_states_49_pad_type_0 = const()[name = string("value_states_49_pad_type_0"), val = string("valid")]; tensor value_states_49_pad_0 = const()[name = string("value_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_49_dilations_0 = const()[name = string("value_states_49_dilations_0"), val = tensor([1, 1])]; int32 value_states_49_groups_0 = const()[name = string("value_states_49_groups_0"), val = int32(1)]; tensor value_states_49_cast_fp16 = conv(dilations = value_states_49_dilations_0, groups = value_states_49_groups_0, pad = value_states_49_pad_0, pad_type = value_states_49_pad_type_0, strides = value_states_49_strides_0, weight = layers_8_self_attn_v_proj_weight_cast_fp16, x = var_3495_cast_fp16_0)[name = string("value_states_49_cast_fp16")]; tensor concat_96x = const()[name = string("concat_96x"), val = tensor([1, 16, 128, -1])]; tensor x_81_cast_fp16 = reshape(shape = concat_96x, x = query_states_49_cast_fp16)[name = string("x_81_cast_fp16")]; tensor concat_97x = const()[name = string("concat_97x"), val = tensor([1, 2, 128, -1])]; tensor var_3552_cast_fp16 = reshape(shape = concat_97x, x = key_states_81_cast_fp16)[name = string("op_3552_cast_fp16")]; tensor concat_98x = const()[name = string("concat_98x"), val = tensor([1, 2, 128, -1])]; tensor var_3559_cast_fp16 = reshape(shape = concat_98x, x = value_states_49_cast_fp16)[name = string("op_3559_cast_fp16")]; tensor var_3563_cast_fp16 = mul(x = x_81_cast_fp16, y = var_869_cast_fp16)[name = string("op_3563_cast_fp16")]; tensor var_3564_split_sizes_0 = const()[name = string("op_3564_split_sizes_0"), val = tensor([64, 64])]; int32 var_3564_axis_0 = const()[name = string("op_3564_axis_0"), val = int32(-2)]; tensor var_3564_cast_fp16_0, tensor var_3564_cast_fp16_1 = split(axis = var_3564_axis_0, split_sizes = var_3564_split_sizes_0, x = x_81_cast_fp16)[name = string("op_3564_cast_fp16")]; fp16 const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3566_cast_fp16 = mul(x = var_3564_cast_fp16_1, y = const_82_promoted_to_fp16)[name = string("op_3566_cast_fp16")]; int32 var_3568 = const()[name = string("op_3568"), val = int32(-2)]; bool var_3569_interleave_0 = const()[name = string("op_3569_interleave_0"), val = bool(false)]; tensor var_3569_cast_fp16 = concat(axis = var_3568, interleave = var_3569_interleave_0, values = (var_3566_cast_fp16, var_3564_cast_fp16_0))[name = string("op_3569_cast_fp16")]; tensor var_3570_cast_fp16 = mul(x = var_3569_cast_fp16, y = var_878_cast_fp16)[name = string("op_3570_cast_fp16")]; tensor query_states_51_cast_fp16 = add(x = var_3563_cast_fp16, y = var_3570_cast_fp16)[name = string("query_states_51_cast_fp16")]; tensor var_3576_cast_fp16 = mul(x = var_3552_cast_fp16, y = var_869_cast_fp16)[name = string("op_3576_cast_fp16")]; tensor var_3577_split_sizes_0 = const()[name = string("op_3577_split_sizes_0"), val = tensor([64, 64])]; int32 var_3577_axis_0 = const()[name = string("op_3577_axis_0"), val = int32(-2)]; tensor var_3577_cast_fp16_0, tensor var_3577_cast_fp16_1 = split(axis = var_3577_axis_0, split_sizes = var_3577_split_sizes_0, x = var_3552_cast_fp16)[name = string("op_3577_cast_fp16")]; fp16 const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3579_cast_fp16 = mul(x = var_3577_cast_fp16_1, y = const_83_promoted_to_fp16)[name = string("op_3579_cast_fp16")]; int32 var_3581 = const()[name = string("op_3581"), val = int32(-2)]; bool var_3582_interleave_0 = const()[name = string("op_3582_interleave_0"), val = bool(false)]; tensor var_3582_cast_fp16 = concat(axis = var_3581, interleave = var_3582_interleave_0, values = (var_3579_cast_fp16, var_3577_cast_fp16_0))[name = string("op_3582_cast_fp16")]; tensor var_3583_cast_fp16 = mul(x = var_3582_cast_fp16, y = var_878_cast_fp16)[name = string("op_3583_cast_fp16")]; tensor key_states_85_cast_fp16 = add(x = var_3576_cast_fp16, y = var_3583_cast_fp16)[name = string("key_states_85_cast_fp16")]; tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; int32 concat_101_axis_0 = const()[name = string("concat_101_axis_0"), val = int32(0)]; bool concat_101_interleave_0 = const()[name = string("concat_101_interleave_0"), val = bool(false)]; tensor concat_101 = concat(axis = concat_101_axis_0, interleave = concat_101_interleave_0, values = (expand_dims_96, expand_dims_97, position_id, expand_dims_99))[name = string("concat_101")]; tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; tensor concat_102_values1_0 = const()[name = string("concat_102_values1_0"), val = tensor([0])]; tensor concat_102_values3_0 = const()[name = string("concat_102_values3_0"), val = tensor([0])]; int32 concat_102_axis_0 = const()[name = string("concat_102_axis_0"), val = int32(0)]; bool concat_102_interleave_0 = const()[name = string("concat_102_interleave_0"), val = bool(false)]; tensor concat_102 = concat(axis = concat_102_axis_0, interleave = concat_102_interleave_0, values = (expand_dims_100, concat_102_values1_0, cache_position_end, concat_102_values3_0))[name = string("concat_102")]; tensor key_states_87_perm_0 = const()[name = string("key_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_9_stride_0 = const()[name = string("key_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_87_cast_fp16 = transpose(perm = key_states_87_perm_0, x = key_states_85_cast_fp16)[name = string("transpose_145")]; tensor key_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = key_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = key_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_9_squeeze_mask_0, stride = key_cache_internal_tensor_assign_9_stride_0, update = key_states_87_cast_fp16, x = coreml_update_state_70)[name = string("key_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_9_cast_fp16, input = key_cache)[name = string("coreml_update_state_72_write_state")]; tensor coreml_update_state_72 = read_state(input = key_cache)[name = string("coreml_update_state_72")]; tensor value_states_51_perm_0 = const()[name = string("value_states_51_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_9_stride_0 = const()[name = string("value_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_51_cast_fp16 = transpose(perm = value_states_51_perm_0, x = var_3559_cast_fp16)[name = string("transpose_144")]; tensor value_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = value_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = value_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_9_squeeze_mask_0, stride = value_cache_internal_tensor_assign_9_stride_0, update = value_states_51_cast_fp16, x = coreml_update_state_71)[name = string("value_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_9_cast_fp16, input = value_cache)[name = string("coreml_update_state_73_write_state")]; tensor coreml_update_state_73 = read_state(input = value_cache)[name = string("coreml_update_state_73")]; tensor var_3653_begin_0 = const()[name = string("op_3653_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3653_end_0 = const()[name = string("op_3653_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3653_end_mask_0 = const()[name = string("op_3653_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3653_cast_fp16 = slice_by_index(begin = var_3653_begin_0, end = var_3653_end_0, end_mask = var_3653_end_mask_0, x = coreml_update_state_72)[name = string("op_3653_cast_fp16")]; tensor tile_16 = const()[name = string("tile_16"), val = tensor([1, 1])]; int32 var_3656_axis_0 = const()[name = string("op_3656_axis_0"), val = int32(1)]; tensor var_3656_cast_fp16_0, tensor var_3656_cast_fp16_1 = split(axis = var_3656_axis_0, split_sizes = tile_16, x = var_3653_cast_fp16)[name = string("op_3656_cast_fp16")]; tensor var_3663_begin_0 = const()[name = string("op_3663_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3663_end_0 = const()[name = string("op_3663_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3663_end_mask_0 = const()[name = string("op_3663_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3663_cast_fp16 = slice_by_index(begin = var_3663_begin_0, end = var_3663_end_0, end_mask = var_3663_end_mask_0, x = coreml_update_state_73)[name = string("op_3663_cast_fp16")]; tensor tile_17 = const()[name = string("tile_17"), val = tensor([1, 1])]; int32 var_3666_axis_0 = const()[name = string("op_3666_axis_0"), val = int32(1)]; tensor var_3666_cast_fp16_0, tensor var_3666_cast_fp16_1 = split(axis = var_3666_axis_0, split_sizes = tile_17, x = var_3663_cast_fp16)[name = string("op_3666_cast_fp16")]; tensor var_3669_split_sizes_0 = const()[name = string("op_3669_split_sizes_0"), val = tensor([8, 8])]; int32 var_3669_axis_0 = const()[name = string("op_3669_axis_0"), val = int32(1)]; tensor var_3669_0, tensor var_3669_1 = split(axis = var_3669_axis_0, split_sizes = var_3669_split_sizes_0, x = query_states_51_cast_fp16)[name = string("op_3669")]; bool attn_weights_129_transpose_x_0 = const()[name = string("attn_weights_129_transpose_x_0"), val = bool(false)]; bool attn_weights_129_transpose_y_0 = const()[name = string("attn_weights_129_transpose_y_0"), val = bool(false)]; tensor attn_weights_129_cast_fp16 = matmul(transpose_x = attn_weights_129_transpose_x_0, transpose_y = attn_weights_129_transpose_y_0, x = var_3656_cast_fp16_0, y = var_3669_0)[name = string("attn_weights_129_cast_fp16")]; fp16 var_3672_to_fp16 = const()[name = string("op_3672_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_131_cast_fp16 = mul(x = attn_weights_129_cast_fp16, y = var_3672_to_fp16)[name = string("attn_weights_131_cast_fp16")]; tensor attn_weights_133_cast_fp16 = add(x = attn_weights_131_cast_fp16, y = attn_mask_1)[name = string("attn_weights_133_cast_fp16")]; int32 var_3676 = const()[name = string("op_3676"), val = int32(-2)]; tensor attn_weights_135_cast_fp16 = softmax(axis = var_3676, x = attn_weights_133_cast_fp16)[name = string("attn_weights_135_cast_fp16")]; bool var_3682_transpose_x_1 = const()[name = string("op_3682_transpose_x_1"), val = bool(true)]; bool var_3682_transpose_y_1 = const()[name = string("op_3682_transpose_y_1"), val = bool(false)]; tensor var_3682_cast_fp16 = matmul(transpose_x = var_3682_transpose_x_1, transpose_y = var_3682_transpose_y_1, x = attn_weights_135_cast_fp16, y = var_3666_cast_fp16_0)[name = string("op_3682_cast_fp16")]; bool attn_weights_137_transpose_x_0 = const()[name = string("attn_weights_137_transpose_x_0"), val = bool(false)]; bool attn_weights_137_transpose_y_0 = const()[name = string("attn_weights_137_transpose_y_0"), val = bool(false)]; tensor attn_weights_137_cast_fp16 = matmul(transpose_x = attn_weights_137_transpose_x_0, transpose_y = attn_weights_137_transpose_y_0, x = var_3656_cast_fp16_1, y = var_3669_1)[name = string("attn_weights_137_cast_fp16")]; fp16 var_3684_to_fp16 = const()[name = string("op_3684_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_139_cast_fp16 = mul(x = attn_weights_137_cast_fp16, y = var_3684_to_fp16)[name = string("attn_weights_139_cast_fp16")]; tensor attn_weights_141_cast_fp16 = add(x = attn_weights_139_cast_fp16, y = attn_mask_1)[name = string("attn_weights_141_cast_fp16")]; int32 var_3688 = const()[name = string("op_3688"), val = int32(-2)]; tensor attn_weights_143_cast_fp16 = softmax(axis = var_3688, x = attn_weights_141_cast_fp16)[name = string("attn_weights_143_cast_fp16")]; bool attn_output_65_transpose_x_1 = const()[name = string("attn_output_65_transpose_x_1"), val = bool(true)]; bool attn_output_65_transpose_y_1 = const()[name = string("attn_output_65_transpose_y_1"), val = bool(false)]; tensor attn_output_65_cast_fp16 = matmul(transpose_x = attn_output_65_transpose_x_1, transpose_y = attn_output_65_transpose_y_1, x = attn_weights_143_cast_fp16, y = var_3666_cast_fp16_1)[name = string("attn_output_65_cast_fp16")]; int32 var_3696 = const()[name = string("op_3696"), val = int32(1)]; bool attn_output_67_interleave_0 = const()[name = string("attn_output_67_interleave_0"), val = bool(false)]; tensor attn_output_67_cast_fp16 = concat(axis = var_3696, interleave = attn_output_67_interleave_0, values = (var_3682_cast_fp16, attn_output_65_cast_fp16))[name = string("attn_output_67_cast_fp16")]; tensor var_3700_perm_0 = const()[name = string("op_3700_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_107x = const()[name = string("concat_107x"), val = tensor([1, 2048, 1, -1])]; tensor var_3700_cast_fp16 = transpose(perm = var_3700_perm_0, x = attn_output_67_cast_fp16)[name = string("transpose_143")]; tensor attn_output_71_cast_fp16 = reshape(shape = concat_107x, x = var_3700_cast_fp16)[name = string("attn_output_71_cast_fp16")]; tensor hidden_states_83_strides_0 = const()[name = string("hidden_states_83_strides_0"), val = tensor([1, 1])]; string hidden_states_83_pad_type_0 = const()[name = string("hidden_states_83_pad_type_0"), val = string("valid")]; tensor hidden_states_83_pad_0 = const()[name = string("hidden_states_83_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_83_dilations_0 = const()[name = string("hidden_states_83_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_83_groups_0 = const()[name = string("hidden_states_83_groups_0"), val = int32(1)]; tensor hidden_states_83_cast_fp16 = conv(dilations = hidden_states_83_dilations_0, groups = hidden_states_83_groups_0, pad = hidden_states_83_pad_0, pad_type = hidden_states_83_pad_type_0, strides = hidden_states_83_strides_0, weight = layers_8_self_attn_o_proj_weight_cast_fp16, x = attn_output_71_cast_fp16)[name = string("hidden_states_83_cast_fp16")]; tensor hidden_states_85_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = hidden_states_83_cast_fp16)[name = string("hidden_states_85_cast_fp16")]; fp16 const_88_promoted_to_fp16 = const()[name = string("const_88_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3733_cast_fp16 = mul(x = hidden_states_85_cast_fp16, y = const_88_promoted_to_fp16)[name = string("op_3733_cast_fp16")]; int32 var_3731 = const()[name = string("op_3731"), val = int32(1)]; bool doubled_69_interleave_0 = const()[name = string("doubled_69_interleave_0"), val = bool(false)]; tensor doubled_69_cast_fp16 = concat(axis = var_3731, interleave = doubled_69_interleave_0, values = (hidden_states_85_cast_fp16, var_3733_cast_fp16))[name = string("doubled_69_cast_fp16")]; tensor out_35_axes_0 = const()[name = string("out_35_axes_0"), val = tensor([1])]; tensor out_35_gamma_0_to_fp16 = const()[name = string("out_35_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396442240)))]; fp16 var_3743_to_fp16 = const()[name = string("op_3743_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_35_cast_fp16 = layer_norm(axes = out_35_axes_0, epsilon = var_3743_to_fp16, gamma = out_35_gamma_0_to_fp16, x = doubled_69_cast_fp16)[name = string("out_35_cast_fp16")]; tensor var_3754_split_sizes_0 = const()[name = string("op_3754_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3754_axis_0 = const()[name = string("op_3754_axis_0"), val = int32(1)]; tensor var_3754_cast_fp16_0, tensor var_3754_cast_fp16_1 = split(axis = var_3754_axis_0, split_sizes = var_3754_split_sizes_0, x = out_35_cast_fp16)[name = string("op_3754_cast_fp16")]; tensor input_17_strides_0 = const()[name = string("input_17_strides_0"), val = tensor([1, 1])]; string input_17_pad_type_0 = const()[name = string("input_17_pad_type_0"), val = string("valid")]; tensor input_17_pad_0 = const()[name = string("input_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_17_dilations_0 = const()[name = string("input_17_dilations_0"), val = tensor([1, 1])]; int32 input_17_groups_0 = const()[name = string("input_17_groups_0"), val = int32(1)]; tensor input_17_cast_fp16 = conv(dilations = input_17_dilations_0, groups = input_17_groups_0, pad = input_17_pad_0, pad_type = input_17_pad_type_0, strides = input_17_strides_0, weight = layers_8_mlp_gate_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("input_17_cast_fp16")]; tensor var_3771_cast_fp16 = silu(x = input_17_cast_fp16)[name = string("op_3771_cast_fp16")]; tensor var_3777_strides_0 = const()[name = string("op_3777_strides_0"), val = tensor([1, 1])]; string var_3777_pad_type_0 = const()[name = string("op_3777_pad_type_0"), val = string("valid")]; tensor var_3777_pad_0 = const()[name = string("op_3777_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3777_dilations_0 = const()[name = string("op_3777_dilations_0"), val = tensor([1, 1])]; int32 var_3777_groups_0 = const()[name = string("op_3777_groups_0"), val = int32(1)]; tensor var_3777_cast_fp16 = conv(dilations = var_3777_dilations_0, groups = var_3777_groups_0, pad = var_3777_pad_0, pad_type = var_3777_pad_type_0, strides = var_3777_strides_0, weight = layers_8_mlp_up_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("op_3777_cast_fp16")]; tensor x_89_cast_fp16 = mul(x = var_3771_cast_fp16, y = var_3777_cast_fp16)[name = string("x_89_cast_fp16")]; tensor hidden_states_87_strides_0 = const()[name = string("hidden_states_87_strides_0"), val = tensor([1, 1])]; string hidden_states_87_pad_type_0 = const()[name = string("hidden_states_87_pad_type_0"), val = string("valid")]; tensor hidden_states_87_pad_0 = const()[name = string("hidden_states_87_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_87_dilations_0 = const()[name = string("hidden_states_87_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_87_groups_0 = const()[name = string("hidden_states_87_groups_0"), val = int32(1)]; tensor hidden_states_87_cast_fp16 = conv(dilations = hidden_states_87_dilations_0, groups = hidden_states_87_groups_0, pad = hidden_states_87_pad_0, pad_type = hidden_states_87_pad_type_0, strides = hidden_states_87_strides_0, weight = layers_8_mlp_down_proj_weight_cast_fp16, x = x_89_cast_fp16)[name = string("hidden_states_87_cast_fp16")]; tensor hidden_states_89_cast_fp16 = add(x = hidden_states_85_cast_fp16, y = hidden_states_87_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3795_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_3795_cast_fp16")]; int32 var_3793 = const()[name = string("op_3793"), val = int32(1)]; bool doubled_73_interleave_0 = const()[name = string("doubled_73_interleave_0"), val = bool(false)]; tensor doubled_73_cast_fp16 = concat(axis = var_3793, interleave = doubled_73_interleave_0, values = (hidden_states_89_cast_fp16, var_3795_cast_fp16))[name = string("doubled_73_cast_fp16")]; tensor out_37_axes_0 = const()[name = string("out_37_axes_0"), val = tensor([1])]; tensor out_37_gamma_0_to_fp16 = const()[name = string("out_37_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396450496)))]; fp16 var_3805_to_fp16 = const()[name = string("op_3805_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_37_cast_fp16 = layer_norm(axes = out_37_axes_0, epsilon = var_3805_to_fp16, gamma = out_37_gamma_0_to_fp16, x = doubled_73_cast_fp16)[name = string("out_37_cast_fp16")]; tensor var_3816_split_sizes_0 = const()[name = string("op_3816_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3816_axis_0 = const()[name = string("op_3816_axis_0"), val = int32(1)]; tensor var_3816_cast_fp16_0, tensor var_3816_cast_fp16_1 = split(axis = var_3816_axis_0, split_sizes = var_3816_split_sizes_0, x = out_37_cast_fp16)[name = string("op_3816_cast_fp16")]; tensor layers_9_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396458752)))]; tensor query_states_55_strides_0 = const()[name = string("query_states_55_strides_0"), val = tensor([1, 1])]; string query_states_55_pad_type_0 = const()[name = string("query_states_55_pad_type_0"), val = string("valid")]; tensor query_states_55_pad_0 = const()[name = string("query_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_55_dilations_0 = const()[name = string("query_states_55_dilations_0"), val = tensor([1, 1])]; int32 query_states_55_groups_0 = const()[name = string("query_states_55_groups_0"), val = int32(1)]; tensor query_states_55_cast_fp16 = conv(dilations = query_states_55_dilations_0, groups = query_states_55_groups_0, pad = query_states_55_pad_0, pad_type = query_states_55_pad_type_0, strides = query_states_55_strides_0, weight = layers_9_self_attn_q_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("query_states_55_cast_fp16")]; tensor layers_9_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1404847424)))]; tensor key_states_91_strides_0 = const()[name = string("key_states_91_strides_0"), val = tensor([1, 1])]; string key_states_91_pad_type_0 = const()[name = string("key_states_91_pad_type_0"), val = string("valid")]; tensor key_states_91_pad_0 = const()[name = string("key_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_91_dilations_0 = const()[name = string("key_states_91_dilations_0"), val = tensor([1, 1])]; int32 key_states_91_groups_0 = const()[name = string("key_states_91_groups_0"), val = int32(1)]; tensor key_states_91_cast_fp16 = conv(dilations = key_states_91_dilations_0, groups = key_states_91_groups_0, pad = key_states_91_pad_0, pad_type = key_states_91_pad_type_0, strides = key_states_91_strides_0, weight = layers_9_self_attn_k_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("key_states_91_cast_fp16")]; tensor value_states_55_strides_0 = const()[name = string("value_states_55_strides_0"), val = tensor([1, 1])]; string value_states_55_pad_type_0 = const()[name = string("value_states_55_pad_type_0"), val = string("valid")]; tensor value_states_55_pad_0 = const()[name = string("value_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_55_dilations_0 = const()[name = string("value_states_55_dilations_0"), val = tensor([1, 1])]; int32 value_states_55_groups_0 = const()[name = string("value_states_55_groups_0"), val = int32(1)]; tensor value_states_55_cast_fp16 = conv(dilations = value_states_55_dilations_0, groups = value_states_55_groups_0, pad = value_states_55_pad_0, pad_type = value_states_55_pad_type_0, strides = value_states_55_strides_0, weight = layers_9_self_attn_v_proj_weight_cast_fp16, x = var_3816_cast_fp16_0)[name = string("value_states_55_cast_fp16")]; tensor concat_108x = const()[name = string("concat_108x"), val = tensor([1, 16, 128, -1])]; tensor x_91_cast_fp16 = reshape(shape = concat_108x, x = query_states_55_cast_fp16)[name = string("x_91_cast_fp16")]; tensor concat_109x = const()[name = string("concat_109x"), val = tensor([1, 2, 128, -1])]; tensor var_3873_cast_fp16 = reshape(shape = concat_109x, x = key_states_91_cast_fp16)[name = string("op_3873_cast_fp16")]; tensor concat_110x = const()[name = string("concat_110x"), val = tensor([1, 2, 128, -1])]; tensor var_3880_cast_fp16 = reshape(shape = concat_110x, x = value_states_55_cast_fp16)[name = string("op_3880_cast_fp16")]; tensor var_3884_cast_fp16 = mul(x = x_91_cast_fp16, y = var_869_cast_fp16)[name = string("op_3884_cast_fp16")]; tensor var_3885_split_sizes_0 = const()[name = string("op_3885_split_sizes_0"), val = tensor([64, 64])]; int32 var_3885_axis_0 = const()[name = string("op_3885_axis_0"), val = int32(-2)]; tensor var_3885_cast_fp16_0, tensor var_3885_cast_fp16_1 = split(axis = var_3885_axis_0, split_sizes = var_3885_split_sizes_0, x = x_91_cast_fp16)[name = string("op_3885_cast_fp16")]; fp16 const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3887_cast_fp16 = mul(x = var_3885_cast_fp16_1, y = const_92_promoted_to_fp16)[name = string("op_3887_cast_fp16")]; int32 var_3889 = const()[name = string("op_3889"), val = int32(-2)]; bool var_3890_interleave_0 = const()[name = string("op_3890_interleave_0"), val = bool(false)]; tensor var_3890_cast_fp16 = concat(axis = var_3889, interleave = var_3890_interleave_0, values = (var_3887_cast_fp16, var_3885_cast_fp16_0))[name = string("op_3890_cast_fp16")]; tensor var_3891_cast_fp16 = mul(x = var_3890_cast_fp16, y = var_878_cast_fp16)[name = string("op_3891_cast_fp16")]; tensor query_states_57_cast_fp16 = add(x = var_3884_cast_fp16, y = var_3891_cast_fp16)[name = string("query_states_57_cast_fp16")]; tensor var_3897_cast_fp16 = mul(x = var_3873_cast_fp16, y = var_869_cast_fp16)[name = string("op_3897_cast_fp16")]; tensor var_3898_split_sizes_0 = const()[name = string("op_3898_split_sizes_0"), val = tensor([64, 64])]; int32 var_3898_axis_0 = const()[name = string("op_3898_axis_0"), val = int32(-2)]; tensor var_3898_cast_fp16_0, tensor var_3898_cast_fp16_1 = split(axis = var_3898_axis_0, split_sizes = var_3898_split_sizes_0, x = var_3873_cast_fp16)[name = string("op_3898_cast_fp16")]; fp16 const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3900_cast_fp16 = mul(x = var_3898_cast_fp16_1, y = const_93_promoted_to_fp16)[name = string("op_3900_cast_fp16")]; int32 var_3902 = const()[name = string("op_3902"), val = int32(-2)]; bool var_3903_interleave_0 = const()[name = string("op_3903_interleave_0"), val = bool(false)]; tensor var_3903_cast_fp16 = concat(axis = var_3902, interleave = var_3903_interleave_0, values = (var_3900_cast_fp16, var_3898_cast_fp16_0))[name = string("op_3903_cast_fp16")]; tensor var_3904_cast_fp16 = mul(x = var_3903_cast_fp16, y = var_878_cast_fp16)[name = string("op_3904_cast_fp16")]; tensor key_states_95_cast_fp16 = add(x = var_3897_cast_fp16, y = var_3904_cast_fp16)[name = string("key_states_95_cast_fp16")]; tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; int32 concat_113_axis_0 = const()[name = string("concat_113_axis_0"), val = int32(0)]; bool concat_113_interleave_0 = const()[name = string("concat_113_interleave_0"), val = bool(false)]; tensor concat_113 = concat(axis = concat_113_axis_0, interleave = concat_113_interleave_0, values = (expand_dims_108, expand_dims_109, position_id, expand_dims_111))[name = string("concat_113")]; tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; tensor concat_114_values1_0 = const()[name = string("concat_114_values1_0"), val = tensor([0])]; tensor concat_114_values3_0 = const()[name = string("concat_114_values3_0"), val = tensor([0])]; int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_112, concat_114_values1_0, cache_position_end, concat_114_values3_0))[name = string("concat_114")]; tensor key_states_97_perm_0 = const()[name = string("key_states_97_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_10_stride_0 = const()[name = string("key_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_97_cast_fp16 = transpose(perm = key_states_97_perm_0, x = key_states_95_cast_fp16)[name = string("transpose_142")]; tensor key_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = key_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = key_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_10_squeeze_mask_0, stride = key_cache_internal_tensor_assign_10_stride_0, update = key_states_97_cast_fp16, x = coreml_update_state_72)[name = string("key_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_10_cast_fp16, input = key_cache)[name = string("coreml_update_state_74_write_state")]; tensor coreml_update_state_74 = read_state(input = key_cache)[name = string("coreml_update_state_74")]; tensor value_states_57_perm_0 = const()[name = string("value_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_10_stride_0 = const()[name = string("value_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_57_cast_fp16 = transpose(perm = value_states_57_perm_0, x = var_3880_cast_fp16)[name = string("transpose_141")]; tensor value_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = value_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = value_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_10_squeeze_mask_0, stride = value_cache_internal_tensor_assign_10_stride_0, update = value_states_57_cast_fp16, x = coreml_update_state_73)[name = string("value_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_10_cast_fp16, input = value_cache)[name = string("coreml_update_state_75_write_state")]; tensor coreml_update_state_75 = read_state(input = value_cache)[name = string("coreml_update_state_75")]; tensor var_3974_begin_0 = const()[name = string("op_3974_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3974_end_0 = const()[name = string("op_3974_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3974_end_mask_0 = const()[name = string("op_3974_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3974_cast_fp16 = slice_by_index(begin = var_3974_begin_0, end = var_3974_end_0, end_mask = var_3974_end_mask_0, x = coreml_update_state_74)[name = string("op_3974_cast_fp16")]; tensor tile_18 = const()[name = string("tile_18"), val = tensor([1, 1])]; int32 var_3977_axis_0 = const()[name = string("op_3977_axis_0"), val = int32(1)]; tensor var_3977_cast_fp16_0, tensor var_3977_cast_fp16_1 = split(axis = var_3977_axis_0, split_sizes = tile_18, x = var_3974_cast_fp16)[name = string("op_3977_cast_fp16")]; tensor var_3984_begin_0 = const()[name = string("op_3984_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3984_end_0 = const()[name = string("op_3984_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3984_end_mask_0 = const()[name = string("op_3984_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3984_cast_fp16 = slice_by_index(begin = var_3984_begin_0, end = var_3984_end_0, end_mask = var_3984_end_mask_0, x = coreml_update_state_75)[name = string("op_3984_cast_fp16")]; tensor tile_19 = const()[name = string("tile_19"), val = tensor([1, 1])]; int32 var_3987_axis_0 = const()[name = string("op_3987_axis_0"), val = int32(1)]; tensor var_3987_cast_fp16_0, tensor var_3987_cast_fp16_1 = split(axis = var_3987_axis_0, split_sizes = tile_19, x = var_3984_cast_fp16)[name = string("op_3987_cast_fp16")]; tensor var_3990_split_sizes_0 = const()[name = string("op_3990_split_sizes_0"), val = tensor([8, 8])]; int32 var_3990_axis_0 = const()[name = string("op_3990_axis_0"), val = int32(1)]; tensor var_3990_0, tensor var_3990_1 = split(axis = var_3990_axis_0, split_sizes = var_3990_split_sizes_0, x = query_states_57_cast_fp16)[name = string("op_3990")]; bool attn_weights_145_transpose_x_0 = const()[name = string("attn_weights_145_transpose_x_0"), val = bool(false)]; bool attn_weights_145_transpose_y_0 = const()[name = string("attn_weights_145_transpose_y_0"), val = bool(false)]; tensor attn_weights_145_cast_fp16 = matmul(transpose_x = attn_weights_145_transpose_x_0, transpose_y = attn_weights_145_transpose_y_0, x = var_3977_cast_fp16_0, y = var_3990_0)[name = string("attn_weights_145_cast_fp16")]; fp16 var_3993_to_fp16 = const()[name = string("op_3993_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_147_cast_fp16 = mul(x = attn_weights_145_cast_fp16, y = var_3993_to_fp16)[name = string("attn_weights_147_cast_fp16")]; tensor attn_weights_149_cast_fp16 = add(x = attn_weights_147_cast_fp16, y = attn_mask_1)[name = string("attn_weights_149_cast_fp16")]; int32 var_3997 = const()[name = string("op_3997"), val = int32(-2)]; tensor attn_weights_151_cast_fp16 = softmax(axis = var_3997, x = attn_weights_149_cast_fp16)[name = string("attn_weights_151_cast_fp16")]; bool var_4003_transpose_x_1 = const()[name = string("op_4003_transpose_x_1"), val = bool(true)]; bool var_4003_transpose_y_1 = const()[name = string("op_4003_transpose_y_1"), val = bool(false)]; tensor var_4003_cast_fp16 = matmul(transpose_x = var_4003_transpose_x_1, transpose_y = var_4003_transpose_y_1, x = attn_weights_151_cast_fp16, y = var_3987_cast_fp16_0)[name = string("op_4003_cast_fp16")]; bool attn_weights_153_transpose_x_0 = const()[name = string("attn_weights_153_transpose_x_0"), val = bool(false)]; bool attn_weights_153_transpose_y_0 = const()[name = string("attn_weights_153_transpose_y_0"), val = bool(false)]; tensor attn_weights_153_cast_fp16 = matmul(transpose_x = attn_weights_153_transpose_x_0, transpose_y = attn_weights_153_transpose_y_0, x = var_3977_cast_fp16_1, y = var_3990_1)[name = string("attn_weights_153_cast_fp16")]; fp16 var_4005_to_fp16 = const()[name = string("op_4005_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_155_cast_fp16 = mul(x = attn_weights_153_cast_fp16, y = var_4005_to_fp16)[name = string("attn_weights_155_cast_fp16")]; tensor attn_weights_157_cast_fp16 = add(x = attn_weights_155_cast_fp16, y = attn_mask_1)[name = string("attn_weights_157_cast_fp16")]; int32 var_4009 = const()[name = string("op_4009"), val = int32(-2)]; tensor attn_weights_159_cast_fp16 = softmax(axis = var_4009, x = attn_weights_157_cast_fp16)[name = string("attn_weights_159_cast_fp16")]; bool attn_output_73_transpose_x_1 = const()[name = string("attn_output_73_transpose_x_1"), val = bool(true)]; bool attn_output_73_transpose_y_1 = const()[name = string("attn_output_73_transpose_y_1"), val = bool(false)]; tensor attn_output_73_cast_fp16 = matmul(transpose_x = attn_output_73_transpose_x_1, transpose_y = attn_output_73_transpose_y_1, x = attn_weights_159_cast_fp16, y = var_3987_cast_fp16_1)[name = string("attn_output_73_cast_fp16")]; int32 var_4017 = const()[name = string("op_4017"), val = int32(1)]; bool attn_output_75_interleave_0 = const()[name = string("attn_output_75_interleave_0"), val = bool(false)]; tensor attn_output_75_cast_fp16 = concat(axis = var_4017, interleave = attn_output_75_interleave_0, values = (var_4003_cast_fp16, attn_output_73_cast_fp16))[name = string("attn_output_75_cast_fp16")]; tensor var_4021_perm_0 = const()[name = string("op_4021_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_119x = const()[name = string("concat_119x"), val = tensor([1, 2048, 1, -1])]; tensor var_4021_cast_fp16 = transpose(perm = var_4021_perm_0, x = attn_output_75_cast_fp16)[name = string("transpose_140")]; tensor attn_output_79_cast_fp16 = reshape(shape = concat_119x, x = var_4021_cast_fp16)[name = string("attn_output_79_cast_fp16")]; tensor hidden_states_93_strides_0 = const()[name = string("hidden_states_93_strides_0"), val = tensor([1, 1])]; string hidden_states_93_pad_type_0 = const()[name = string("hidden_states_93_pad_type_0"), val = string("valid")]; tensor hidden_states_93_pad_0 = const()[name = string("hidden_states_93_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_93_dilations_0 = const()[name = string("hidden_states_93_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_93_groups_0 = const()[name = string("hidden_states_93_groups_0"), val = int32(1)]; tensor hidden_states_93_cast_fp16 = conv(dilations = hidden_states_93_dilations_0, groups = hidden_states_93_groups_0, pad = hidden_states_93_pad_0, pad_type = hidden_states_93_pad_type_0, strides = hidden_states_93_strides_0, weight = layers_9_self_attn_o_proj_weight_cast_fp16, x = attn_output_79_cast_fp16)[name = string("hidden_states_93_cast_fp16")]; tensor hidden_states_95_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = hidden_states_93_cast_fp16)[name = string("hidden_states_95_cast_fp16")]; fp16 const_98_promoted_to_fp16 = const()[name = string("const_98_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4054_cast_fp16 = mul(x = hidden_states_95_cast_fp16, y = const_98_promoted_to_fp16)[name = string("op_4054_cast_fp16")]; int32 var_4052 = const()[name = string("op_4052"), val = int32(1)]; bool doubled_77_interleave_0 = const()[name = string("doubled_77_interleave_0"), val = bool(false)]; tensor doubled_77_cast_fp16 = concat(axis = var_4052, interleave = doubled_77_interleave_0, values = (hidden_states_95_cast_fp16, var_4054_cast_fp16))[name = string("doubled_77_cast_fp16")]; tensor out_39_axes_0 = const()[name = string("out_39_axes_0"), val = tensor([1])]; tensor out_39_gamma_0_to_fp16 = const()[name = string("out_39_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405896064)))]; fp16 var_4064_to_fp16 = const()[name = string("op_4064_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_39_cast_fp16 = layer_norm(axes = out_39_axes_0, epsilon = var_4064_to_fp16, gamma = out_39_gamma_0_to_fp16, x = doubled_77_cast_fp16)[name = string("out_39_cast_fp16")]; tensor var_4075_split_sizes_0 = const()[name = string("op_4075_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4075_axis_0 = const()[name = string("op_4075_axis_0"), val = int32(1)]; tensor var_4075_cast_fp16_0, tensor var_4075_cast_fp16_1 = split(axis = var_4075_axis_0, split_sizes = var_4075_split_sizes_0, x = out_39_cast_fp16)[name = string("op_4075_cast_fp16")]; tensor input_19_strides_0 = const()[name = string("input_19_strides_0"), val = tensor([1, 1])]; string input_19_pad_type_0 = const()[name = string("input_19_pad_type_0"), val = string("valid")]; tensor input_19_pad_0 = const()[name = string("input_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_19_dilations_0 = const()[name = string("input_19_dilations_0"), val = tensor([1, 1])]; int32 input_19_groups_0 = const()[name = string("input_19_groups_0"), val = int32(1)]; tensor input_19_cast_fp16 = conv(dilations = input_19_dilations_0, groups = input_19_groups_0, pad = input_19_pad_0, pad_type = input_19_pad_type_0, strides = input_19_strides_0, weight = layers_9_mlp_gate_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("input_19_cast_fp16")]; tensor var_4092_cast_fp16 = silu(x = input_19_cast_fp16)[name = string("op_4092_cast_fp16")]; tensor var_4098_strides_0 = const()[name = string("op_4098_strides_0"), val = tensor([1, 1])]; string var_4098_pad_type_0 = const()[name = string("op_4098_pad_type_0"), val = string("valid")]; tensor var_4098_pad_0 = const()[name = string("op_4098_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4098_dilations_0 = const()[name = string("op_4098_dilations_0"), val = tensor([1, 1])]; int32 var_4098_groups_0 = const()[name = string("op_4098_groups_0"), val = int32(1)]; tensor var_4098_cast_fp16 = conv(dilations = var_4098_dilations_0, groups = var_4098_groups_0, pad = var_4098_pad_0, pad_type = var_4098_pad_type_0, strides = var_4098_strides_0, weight = layers_9_mlp_up_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("op_4098_cast_fp16")]; tensor x_99_cast_fp16 = mul(x = var_4092_cast_fp16, y = var_4098_cast_fp16)[name = string("x_99_cast_fp16")]; tensor hidden_states_97_strides_0 = const()[name = string("hidden_states_97_strides_0"), val = tensor([1, 1])]; string hidden_states_97_pad_type_0 = const()[name = string("hidden_states_97_pad_type_0"), val = string("valid")]; tensor hidden_states_97_pad_0 = const()[name = string("hidden_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_97_dilations_0 = const()[name = string("hidden_states_97_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_97_groups_0 = const()[name = string("hidden_states_97_groups_0"), val = int32(1)]; tensor hidden_states_97_cast_fp16 = conv(dilations = hidden_states_97_dilations_0, groups = hidden_states_97_groups_0, pad = hidden_states_97_pad_0, pad_type = hidden_states_97_pad_type_0, strides = hidden_states_97_strides_0, weight = layers_9_mlp_down_proj_weight_cast_fp16, x = x_99_cast_fp16)[name = string("hidden_states_97_cast_fp16")]; tensor hidden_states_99_cast_fp16 = add(x = hidden_states_95_cast_fp16, y = hidden_states_97_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; fp16 const_100_promoted_to_fp16 = const()[name = string("const_100_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4116_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_100_promoted_to_fp16)[name = string("op_4116_cast_fp16")]; int32 var_4114 = const()[name = string("op_4114"), val = int32(1)]; bool doubled_81_interleave_0 = const()[name = string("doubled_81_interleave_0"), val = bool(false)]; tensor doubled_81_cast_fp16 = concat(axis = var_4114, interleave = doubled_81_interleave_0, values = (hidden_states_99_cast_fp16, var_4116_cast_fp16))[name = string("doubled_81_cast_fp16")]; tensor out_41_axes_0 = const()[name = string("out_41_axes_0"), val = tensor([1])]; tensor out_41_gamma_0_to_fp16 = const()[name = string("out_41_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405904320)))]; fp16 var_4126_to_fp16 = const()[name = string("op_4126_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_41_cast_fp16 = layer_norm(axes = out_41_axes_0, epsilon = var_4126_to_fp16, gamma = out_41_gamma_0_to_fp16, x = doubled_81_cast_fp16)[name = string("out_41_cast_fp16")]; tensor var_4137_split_sizes_0 = const()[name = string("op_4137_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4137_axis_0 = const()[name = string("op_4137_axis_0"), val = int32(1)]; tensor var_4137_cast_fp16_0, tensor var_4137_cast_fp16_1 = split(axis = var_4137_axis_0, split_sizes = var_4137_split_sizes_0, x = out_41_cast_fp16)[name = string("op_4137_cast_fp16")]; tensor layers_10_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405912576)))]; tensor query_states_61_strides_0 = const()[name = string("query_states_61_strides_0"), val = tensor([1, 1])]; string query_states_61_pad_type_0 = const()[name = string("query_states_61_pad_type_0"), val = string("valid")]; tensor query_states_61_pad_0 = const()[name = string("query_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_61_dilations_0 = const()[name = string("query_states_61_dilations_0"), val = tensor([1, 1])]; int32 query_states_61_groups_0 = const()[name = string("query_states_61_groups_0"), val = int32(1)]; tensor query_states_61_cast_fp16 = conv(dilations = query_states_61_dilations_0, groups = query_states_61_groups_0, pad = query_states_61_pad_0, pad_type = query_states_61_pad_type_0, strides = query_states_61_strides_0, weight = layers_10_self_attn_q_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("query_states_61_cast_fp16")]; tensor layers_10_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1414301248)))]; tensor key_states_101_strides_0 = const()[name = string("key_states_101_strides_0"), val = tensor([1, 1])]; string key_states_101_pad_type_0 = const()[name = string("key_states_101_pad_type_0"), val = string("valid")]; tensor key_states_101_pad_0 = const()[name = string("key_states_101_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_101_dilations_0 = const()[name = string("key_states_101_dilations_0"), val = tensor([1, 1])]; int32 key_states_101_groups_0 = const()[name = string("key_states_101_groups_0"), val = int32(1)]; tensor key_states_101_cast_fp16 = conv(dilations = key_states_101_dilations_0, groups = key_states_101_groups_0, pad = key_states_101_pad_0, pad_type = key_states_101_pad_type_0, strides = key_states_101_strides_0, weight = layers_10_self_attn_k_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("key_states_101_cast_fp16")]; tensor value_states_61_strides_0 = const()[name = string("value_states_61_strides_0"), val = tensor([1, 1])]; string value_states_61_pad_type_0 = const()[name = string("value_states_61_pad_type_0"), val = string("valid")]; tensor value_states_61_pad_0 = const()[name = string("value_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_61_dilations_0 = const()[name = string("value_states_61_dilations_0"), val = tensor([1, 1])]; int32 value_states_61_groups_0 = const()[name = string("value_states_61_groups_0"), val = int32(1)]; tensor value_states_61_cast_fp16 = conv(dilations = value_states_61_dilations_0, groups = value_states_61_groups_0, pad = value_states_61_pad_0, pad_type = value_states_61_pad_type_0, strides = value_states_61_strides_0, weight = layers_10_self_attn_v_proj_weight_cast_fp16, x = var_4137_cast_fp16_0)[name = string("value_states_61_cast_fp16")]; tensor concat_120x = const()[name = string("concat_120x"), val = tensor([1, 16, 128, -1])]; tensor x_101_cast_fp16 = reshape(shape = concat_120x, x = query_states_61_cast_fp16)[name = string("x_101_cast_fp16")]; tensor concat_121x = const()[name = string("concat_121x"), val = tensor([1, 2, 128, -1])]; tensor var_4194_cast_fp16 = reshape(shape = concat_121x, x = key_states_101_cast_fp16)[name = string("op_4194_cast_fp16")]; tensor concat_122x = const()[name = string("concat_122x"), val = tensor([1, 2, 128, -1])]; tensor var_4201_cast_fp16 = reshape(shape = concat_122x, x = value_states_61_cast_fp16)[name = string("op_4201_cast_fp16")]; tensor var_4205_cast_fp16 = mul(x = x_101_cast_fp16, y = var_869_cast_fp16)[name = string("op_4205_cast_fp16")]; tensor var_4206_split_sizes_0 = const()[name = string("op_4206_split_sizes_0"), val = tensor([64, 64])]; int32 var_4206_axis_0 = const()[name = string("op_4206_axis_0"), val = int32(-2)]; tensor var_4206_cast_fp16_0, tensor var_4206_cast_fp16_1 = split(axis = var_4206_axis_0, split_sizes = var_4206_split_sizes_0, x = x_101_cast_fp16)[name = string("op_4206_cast_fp16")]; fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4208_cast_fp16 = mul(x = var_4206_cast_fp16_1, y = const_102_promoted_to_fp16)[name = string("op_4208_cast_fp16")]; int32 var_4210 = const()[name = string("op_4210"), val = int32(-2)]; bool var_4211_interleave_0 = const()[name = string("op_4211_interleave_0"), val = bool(false)]; tensor var_4211_cast_fp16 = concat(axis = var_4210, interleave = var_4211_interleave_0, values = (var_4208_cast_fp16, var_4206_cast_fp16_0))[name = string("op_4211_cast_fp16")]; tensor var_4212_cast_fp16 = mul(x = var_4211_cast_fp16, y = var_878_cast_fp16)[name = string("op_4212_cast_fp16")]; tensor query_states_63_cast_fp16 = add(x = var_4205_cast_fp16, y = var_4212_cast_fp16)[name = string("query_states_63_cast_fp16")]; tensor var_4218_cast_fp16 = mul(x = var_4194_cast_fp16, y = var_869_cast_fp16)[name = string("op_4218_cast_fp16")]; tensor var_4219_split_sizes_0 = const()[name = string("op_4219_split_sizes_0"), val = tensor([64, 64])]; int32 var_4219_axis_0 = const()[name = string("op_4219_axis_0"), val = int32(-2)]; tensor var_4219_cast_fp16_0, tensor var_4219_cast_fp16_1 = split(axis = var_4219_axis_0, split_sizes = var_4219_split_sizes_0, x = var_4194_cast_fp16)[name = string("op_4219_cast_fp16")]; fp16 const_103_promoted_to_fp16 = const()[name = string("const_103_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4221_cast_fp16 = mul(x = var_4219_cast_fp16_1, y = const_103_promoted_to_fp16)[name = string("op_4221_cast_fp16")]; int32 var_4223 = const()[name = string("op_4223"), val = int32(-2)]; bool var_4224_interleave_0 = const()[name = string("op_4224_interleave_0"), val = bool(false)]; tensor var_4224_cast_fp16 = concat(axis = var_4223, interleave = var_4224_interleave_0, values = (var_4221_cast_fp16, var_4219_cast_fp16_0))[name = string("op_4224_cast_fp16")]; tensor var_4225_cast_fp16 = mul(x = var_4224_cast_fp16, y = var_878_cast_fp16)[name = string("op_4225_cast_fp16")]; tensor key_states_105_cast_fp16 = add(x = var_4218_cast_fp16, y = var_4225_cast_fp16)[name = string("key_states_105_cast_fp16")]; tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; int32 concat_125_axis_0 = const()[name = string("concat_125_axis_0"), val = int32(0)]; bool concat_125_interleave_0 = const()[name = string("concat_125_interleave_0"), val = bool(false)]; tensor concat_125 = concat(axis = concat_125_axis_0, interleave = concat_125_interleave_0, values = (expand_dims_120, expand_dims_121, position_id, expand_dims_123))[name = string("concat_125")]; tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; tensor concat_126_values1_0 = const()[name = string("concat_126_values1_0"), val = tensor([0])]; tensor concat_126_values3_0 = const()[name = string("concat_126_values3_0"), val = tensor([0])]; int32 concat_126_axis_0 = const()[name = string("concat_126_axis_0"), val = int32(0)]; bool concat_126_interleave_0 = const()[name = string("concat_126_interleave_0"), val = bool(false)]; tensor concat_126 = concat(axis = concat_126_axis_0, interleave = concat_126_interleave_0, values = (expand_dims_124, concat_126_values1_0, cache_position_end, concat_126_values3_0))[name = string("concat_126")]; tensor key_states_107_perm_0 = const()[name = string("key_states_107_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_11_stride_0 = const()[name = string("key_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_107_cast_fp16 = transpose(perm = key_states_107_perm_0, x = key_states_105_cast_fp16)[name = string("transpose_139")]; tensor key_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = key_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = key_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_11_squeeze_mask_0, stride = key_cache_internal_tensor_assign_11_stride_0, update = key_states_107_cast_fp16, x = coreml_update_state_74)[name = string("key_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_11_cast_fp16, input = key_cache)[name = string("coreml_update_state_76_write_state")]; tensor coreml_update_state_76 = read_state(input = key_cache)[name = string("coreml_update_state_76")]; tensor value_states_63_perm_0 = const()[name = string("value_states_63_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_11_stride_0 = const()[name = string("value_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_63_cast_fp16 = transpose(perm = value_states_63_perm_0, x = var_4201_cast_fp16)[name = string("transpose_138")]; tensor value_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = value_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = value_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_11_squeeze_mask_0, stride = value_cache_internal_tensor_assign_11_stride_0, update = value_states_63_cast_fp16, x = coreml_update_state_75)[name = string("value_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_11_cast_fp16, input = value_cache)[name = string("coreml_update_state_77_write_state")]; tensor coreml_update_state_77 = read_state(input = value_cache)[name = string("coreml_update_state_77")]; tensor var_4295_begin_0 = const()[name = string("op_4295_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4295_end_0 = const()[name = string("op_4295_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4295_end_mask_0 = const()[name = string("op_4295_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4295_cast_fp16 = slice_by_index(begin = var_4295_begin_0, end = var_4295_end_0, end_mask = var_4295_end_mask_0, x = coreml_update_state_76)[name = string("op_4295_cast_fp16")]; tensor tile_20 = const()[name = string("tile_20"), val = tensor([1, 1])]; int32 var_4298_axis_0 = const()[name = string("op_4298_axis_0"), val = int32(1)]; tensor var_4298_cast_fp16_0, tensor var_4298_cast_fp16_1 = split(axis = var_4298_axis_0, split_sizes = tile_20, x = var_4295_cast_fp16)[name = string("op_4298_cast_fp16")]; tensor var_4305_begin_0 = const()[name = string("op_4305_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4305_end_0 = const()[name = string("op_4305_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4305_end_mask_0 = const()[name = string("op_4305_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4305_cast_fp16 = slice_by_index(begin = var_4305_begin_0, end = var_4305_end_0, end_mask = var_4305_end_mask_0, x = coreml_update_state_77)[name = string("op_4305_cast_fp16")]; tensor tile_21 = const()[name = string("tile_21"), val = tensor([1, 1])]; int32 var_4308_axis_0 = const()[name = string("op_4308_axis_0"), val = int32(1)]; tensor var_4308_cast_fp16_0, tensor var_4308_cast_fp16_1 = split(axis = var_4308_axis_0, split_sizes = tile_21, x = var_4305_cast_fp16)[name = string("op_4308_cast_fp16")]; tensor var_4311_split_sizes_0 = const()[name = string("op_4311_split_sizes_0"), val = tensor([8, 8])]; int32 var_4311_axis_0 = const()[name = string("op_4311_axis_0"), val = int32(1)]; tensor var_4311_0, tensor var_4311_1 = split(axis = var_4311_axis_0, split_sizes = var_4311_split_sizes_0, x = query_states_63_cast_fp16)[name = string("op_4311")]; bool attn_weights_161_transpose_x_0 = const()[name = string("attn_weights_161_transpose_x_0"), val = bool(false)]; bool attn_weights_161_transpose_y_0 = const()[name = string("attn_weights_161_transpose_y_0"), val = bool(false)]; tensor attn_weights_161_cast_fp16 = matmul(transpose_x = attn_weights_161_transpose_x_0, transpose_y = attn_weights_161_transpose_y_0, x = var_4298_cast_fp16_0, y = var_4311_0)[name = string("attn_weights_161_cast_fp16")]; fp16 var_4314_to_fp16 = const()[name = string("op_4314_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_163_cast_fp16 = mul(x = attn_weights_161_cast_fp16, y = var_4314_to_fp16)[name = string("attn_weights_163_cast_fp16")]; tensor attn_weights_165_cast_fp16 = add(x = attn_weights_163_cast_fp16, y = attn_mask_1)[name = string("attn_weights_165_cast_fp16")]; int32 var_4318 = const()[name = string("op_4318"), val = int32(-2)]; tensor attn_weights_167_cast_fp16 = softmax(axis = var_4318, x = attn_weights_165_cast_fp16)[name = string("attn_weights_167_cast_fp16")]; bool var_4324_transpose_x_1 = const()[name = string("op_4324_transpose_x_1"), val = bool(true)]; bool var_4324_transpose_y_1 = const()[name = string("op_4324_transpose_y_1"), val = bool(false)]; tensor var_4324_cast_fp16 = matmul(transpose_x = var_4324_transpose_x_1, transpose_y = var_4324_transpose_y_1, x = attn_weights_167_cast_fp16, y = var_4308_cast_fp16_0)[name = string("op_4324_cast_fp16")]; bool attn_weights_169_transpose_x_0 = const()[name = string("attn_weights_169_transpose_x_0"), val = bool(false)]; bool attn_weights_169_transpose_y_0 = const()[name = string("attn_weights_169_transpose_y_0"), val = bool(false)]; tensor attn_weights_169_cast_fp16 = matmul(transpose_x = attn_weights_169_transpose_x_0, transpose_y = attn_weights_169_transpose_y_0, x = var_4298_cast_fp16_1, y = var_4311_1)[name = string("attn_weights_169_cast_fp16")]; fp16 var_4326_to_fp16 = const()[name = string("op_4326_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_171_cast_fp16 = mul(x = attn_weights_169_cast_fp16, y = var_4326_to_fp16)[name = string("attn_weights_171_cast_fp16")]; tensor attn_weights_173_cast_fp16 = add(x = attn_weights_171_cast_fp16, y = attn_mask_1)[name = string("attn_weights_173_cast_fp16")]; int32 var_4330 = const()[name = string("op_4330"), val = int32(-2)]; tensor attn_weights_175_cast_fp16 = softmax(axis = var_4330, x = attn_weights_173_cast_fp16)[name = string("attn_weights_175_cast_fp16")]; bool attn_output_81_transpose_x_1 = const()[name = string("attn_output_81_transpose_x_1"), val = bool(true)]; bool attn_output_81_transpose_y_1 = const()[name = string("attn_output_81_transpose_y_1"), val = bool(false)]; tensor attn_output_81_cast_fp16 = matmul(transpose_x = attn_output_81_transpose_x_1, transpose_y = attn_output_81_transpose_y_1, x = attn_weights_175_cast_fp16, y = var_4308_cast_fp16_1)[name = string("attn_output_81_cast_fp16")]; int32 var_4338 = const()[name = string("op_4338"), val = int32(1)]; bool attn_output_83_interleave_0 = const()[name = string("attn_output_83_interleave_0"), val = bool(false)]; tensor attn_output_83_cast_fp16 = concat(axis = var_4338, interleave = attn_output_83_interleave_0, values = (var_4324_cast_fp16, attn_output_81_cast_fp16))[name = string("attn_output_83_cast_fp16")]; tensor var_4342_perm_0 = const()[name = string("op_4342_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_131x = const()[name = string("concat_131x"), val = tensor([1, 2048, 1, -1])]; tensor var_4342_cast_fp16 = transpose(perm = var_4342_perm_0, x = attn_output_83_cast_fp16)[name = string("transpose_137")]; tensor attn_output_87_cast_fp16 = reshape(shape = concat_131x, x = var_4342_cast_fp16)[name = string("attn_output_87_cast_fp16")]; tensor hidden_states_103_strides_0 = const()[name = string("hidden_states_103_strides_0"), val = tensor([1, 1])]; string hidden_states_103_pad_type_0 = const()[name = string("hidden_states_103_pad_type_0"), val = string("valid")]; tensor hidden_states_103_pad_0 = const()[name = string("hidden_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_103_dilations_0 = const()[name = string("hidden_states_103_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_103_groups_0 = const()[name = string("hidden_states_103_groups_0"), val = int32(1)]; tensor hidden_states_103_cast_fp16 = conv(dilations = hidden_states_103_dilations_0, groups = hidden_states_103_groups_0, pad = hidden_states_103_pad_0, pad_type = hidden_states_103_pad_type_0, strides = hidden_states_103_strides_0, weight = layers_10_self_attn_o_proj_weight_cast_fp16, x = attn_output_87_cast_fp16)[name = string("hidden_states_103_cast_fp16")]; tensor hidden_states_105_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = hidden_states_103_cast_fp16)[name = string("hidden_states_105_cast_fp16")]; fp16 const_108_promoted_to_fp16 = const()[name = string("const_108_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4375_cast_fp16 = mul(x = hidden_states_105_cast_fp16, y = const_108_promoted_to_fp16)[name = string("op_4375_cast_fp16")]; int32 var_4373 = const()[name = string("op_4373"), val = int32(1)]; bool doubled_85_interleave_0 = const()[name = string("doubled_85_interleave_0"), val = bool(false)]; tensor doubled_85_cast_fp16 = concat(axis = var_4373, interleave = doubled_85_interleave_0, values = (hidden_states_105_cast_fp16, var_4375_cast_fp16))[name = string("doubled_85_cast_fp16")]; tensor out_43_axes_0 = const()[name = string("out_43_axes_0"), val = tensor([1])]; tensor out_43_gamma_0_to_fp16 = const()[name = string("out_43_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415349888)))]; fp16 var_4385_to_fp16 = const()[name = string("op_4385_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_43_cast_fp16 = layer_norm(axes = out_43_axes_0, epsilon = var_4385_to_fp16, gamma = out_43_gamma_0_to_fp16, x = doubled_85_cast_fp16)[name = string("out_43_cast_fp16")]; tensor var_4396_split_sizes_0 = const()[name = string("op_4396_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4396_axis_0 = const()[name = string("op_4396_axis_0"), val = int32(1)]; tensor var_4396_cast_fp16_0, tensor var_4396_cast_fp16_1 = split(axis = var_4396_axis_0, split_sizes = var_4396_split_sizes_0, x = out_43_cast_fp16)[name = string("op_4396_cast_fp16")]; tensor input_21_strides_0 = const()[name = string("input_21_strides_0"), val = tensor([1, 1])]; string input_21_pad_type_0 = const()[name = string("input_21_pad_type_0"), val = string("valid")]; tensor input_21_pad_0 = const()[name = string("input_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_21_dilations_0 = const()[name = string("input_21_dilations_0"), val = tensor([1, 1])]; int32 input_21_groups_0 = const()[name = string("input_21_groups_0"), val = int32(1)]; tensor input_21_cast_fp16 = conv(dilations = input_21_dilations_0, groups = input_21_groups_0, pad = input_21_pad_0, pad_type = input_21_pad_type_0, strides = input_21_strides_0, weight = layers_10_mlp_gate_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("input_21_cast_fp16")]; tensor var_4413_cast_fp16 = silu(x = input_21_cast_fp16)[name = string("op_4413_cast_fp16")]; tensor var_4419_strides_0 = const()[name = string("op_4419_strides_0"), val = tensor([1, 1])]; string var_4419_pad_type_0 = const()[name = string("op_4419_pad_type_0"), val = string("valid")]; tensor var_4419_pad_0 = const()[name = string("op_4419_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4419_dilations_0 = const()[name = string("op_4419_dilations_0"), val = tensor([1, 1])]; int32 var_4419_groups_0 = const()[name = string("op_4419_groups_0"), val = int32(1)]; tensor var_4419_cast_fp16 = conv(dilations = var_4419_dilations_0, groups = var_4419_groups_0, pad = var_4419_pad_0, pad_type = var_4419_pad_type_0, strides = var_4419_strides_0, weight = layers_10_mlp_up_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("op_4419_cast_fp16")]; tensor x_109_cast_fp16 = mul(x = var_4413_cast_fp16, y = var_4419_cast_fp16)[name = string("x_109_cast_fp16")]; tensor hidden_states_107_strides_0 = const()[name = string("hidden_states_107_strides_0"), val = tensor([1, 1])]; string hidden_states_107_pad_type_0 = const()[name = string("hidden_states_107_pad_type_0"), val = string("valid")]; tensor hidden_states_107_pad_0 = const()[name = string("hidden_states_107_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_107_dilations_0 = const()[name = string("hidden_states_107_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_107_groups_0 = const()[name = string("hidden_states_107_groups_0"), val = int32(1)]; tensor hidden_states_107_cast_fp16 = conv(dilations = hidden_states_107_dilations_0, groups = hidden_states_107_groups_0, pad = hidden_states_107_pad_0, pad_type = hidden_states_107_pad_type_0, strides = hidden_states_107_strides_0, weight = layers_10_mlp_down_proj_weight_cast_fp16, x = x_109_cast_fp16)[name = string("hidden_states_107_cast_fp16")]; tensor hidden_states_109_cast_fp16 = add(x = hidden_states_105_cast_fp16, y = hidden_states_107_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; fp16 const_110_promoted_to_fp16 = const()[name = string("const_110_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4437_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_110_promoted_to_fp16)[name = string("op_4437_cast_fp16")]; int32 var_4435 = const()[name = string("op_4435"), val = int32(1)]; bool doubled_89_interleave_0 = const()[name = string("doubled_89_interleave_0"), val = bool(false)]; tensor doubled_89_cast_fp16 = concat(axis = var_4435, interleave = doubled_89_interleave_0, values = (hidden_states_109_cast_fp16, var_4437_cast_fp16))[name = string("doubled_89_cast_fp16")]; tensor out_45_axes_0 = const()[name = string("out_45_axes_0"), val = tensor([1])]; tensor out_45_gamma_0_to_fp16 = const()[name = string("out_45_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415358144)))]; fp16 var_4447_to_fp16 = const()[name = string("op_4447_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_45_cast_fp16 = layer_norm(axes = out_45_axes_0, epsilon = var_4447_to_fp16, gamma = out_45_gamma_0_to_fp16, x = doubled_89_cast_fp16)[name = string("out_45_cast_fp16")]; tensor var_4458_split_sizes_0 = const()[name = string("op_4458_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4458_axis_0 = const()[name = string("op_4458_axis_0"), val = int32(1)]; tensor var_4458_cast_fp16_0, tensor var_4458_cast_fp16_1 = split(axis = var_4458_axis_0, split_sizes = var_4458_split_sizes_0, x = out_45_cast_fp16)[name = string("op_4458_cast_fp16")]; tensor query_states_67_strides_0 = const()[name = string("query_states_67_strides_0"), val = tensor([1, 1])]; string query_states_67_pad_type_0 = const()[name = string("query_states_67_pad_type_0"), val = string("valid")]; tensor query_states_67_pad_0 = const()[name = string("query_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_67_dilations_0 = const()[name = string("query_states_67_dilations_0"), val = tensor([1, 1])]; int32 query_states_67_groups_0 = const()[name = string("query_states_67_groups_0"), val = int32(1)]; tensor query_states_67_cast_fp16 = conv(dilations = query_states_67_dilations_0, groups = query_states_67_groups_0, pad = query_states_67_pad_0, pad_type = query_states_67_pad_type_0, strides = query_states_67_strides_0, weight = layers_11_self_attn_q_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("query_states_67_cast_fp16")]; tensor key_states_111_strides_0 = const()[name = string("key_states_111_strides_0"), val = tensor([1, 1])]; string key_states_111_pad_type_0 = const()[name = string("key_states_111_pad_type_0"), val = string("valid")]; tensor key_states_111_pad_0 = const()[name = string("key_states_111_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_111_dilations_0 = const()[name = string("key_states_111_dilations_0"), val = tensor([1, 1])]; int32 key_states_111_groups_0 = const()[name = string("key_states_111_groups_0"), val = int32(1)]; tensor key_states_111_cast_fp16 = conv(dilations = key_states_111_dilations_0, groups = key_states_111_groups_0, pad = key_states_111_pad_0, pad_type = key_states_111_pad_type_0, strides = key_states_111_strides_0, weight = layers_11_self_attn_k_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("key_states_111_cast_fp16")]; tensor value_states_67_strides_0 = const()[name = string("value_states_67_strides_0"), val = tensor([1, 1])]; string value_states_67_pad_type_0 = const()[name = string("value_states_67_pad_type_0"), val = string("valid")]; tensor value_states_67_pad_0 = const()[name = string("value_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_67_dilations_0 = const()[name = string("value_states_67_dilations_0"), val = tensor([1, 1])]; int32 value_states_67_groups_0 = const()[name = string("value_states_67_groups_0"), val = int32(1)]; tensor value_states_67_cast_fp16 = conv(dilations = value_states_67_dilations_0, groups = value_states_67_groups_0, pad = value_states_67_pad_0, pad_type = value_states_67_pad_type_0, strides = value_states_67_strides_0, weight = layers_11_self_attn_v_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("value_states_67_cast_fp16")]; tensor concat_132x = const()[name = string("concat_132x"), val = tensor([1, 16, 128, -1])]; tensor x_111_cast_fp16 = reshape(shape = concat_132x, x = query_states_67_cast_fp16)[name = string("x_111_cast_fp16")]; tensor concat_133x = const()[name = string("concat_133x"), val = tensor([1, 2, 128, -1])]; tensor var_4515_cast_fp16 = reshape(shape = concat_133x, x = key_states_111_cast_fp16)[name = string("op_4515_cast_fp16")]; tensor concat_134x = const()[name = string("concat_134x"), val = tensor([1, 2, 128, -1])]; tensor var_4522_cast_fp16 = reshape(shape = concat_134x, x = value_states_67_cast_fp16)[name = string("op_4522_cast_fp16")]; tensor var_4526_cast_fp16 = mul(x = x_111_cast_fp16, y = var_869_cast_fp16)[name = string("op_4526_cast_fp16")]; tensor var_4527_split_sizes_0 = const()[name = string("op_4527_split_sizes_0"), val = tensor([64, 64])]; int32 var_4527_axis_0 = const()[name = string("op_4527_axis_0"), val = int32(-2)]; tensor var_4527_cast_fp16_0, tensor var_4527_cast_fp16_1 = split(axis = var_4527_axis_0, split_sizes = var_4527_split_sizes_0, x = x_111_cast_fp16)[name = string("op_4527_cast_fp16")]; fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4529_cast_fp16 = mul(x = var_4527_cast_fp16_1, y = const_112_promoted_to_fp16)[name = string("op_4529_cast_fp16")]; int32 var_4531 = const()[name = string("op_4531"), val = int32(-2)]; bool var_4532_interleave_0 = const()[name = string("op_4532_interleave_0"), val = bool(false)]; tensor var_4532_cast_fp16 = concat(axis = var_4531, interleave = var_4532_interleave_0, values = (var_4529_cast_fp16, var_4527_cast_fp16_0))[name = string("op_4532_cast_fp16")]; tensor var_4533_cast_fp16 = mul(x = var_4532_cast_fp16, y = var_878_cast_fp16)[name = string("op_4533_cast_fp16")]; tensor query_states_69_cast_fp16 = add(x = var_4526_cast_fp16, y = var_4533_cast_fp16)[name = string("query_states_69_cast_fp16")]; tensor var_4539_cast_fp16 = mul(x = var_4515_cast_fp16, y = var_869_cast_fp16)[name = string("op_4539_cast_fp16")]; tensor var_4540_split_sizes_0 = const()[name = string("op_4540_split_sizes_0"), val = tensor([64, 64])]; int32 var_4540_axis_0 = const()[name = string("op_4540_axis_0"), val = int32(-2)]; tensor var_4540_cast_fp16_0, tensor var_4540_cast_fp16_1 = split(axis = var_4540_axis_0, split_sizes = var_4540_split_sizes_0, x = var_4515_cast_fp16)[name = string("op_4540_cast_fp16")]; fp16 const_113_promoted_to_fp16 = const()[name = string("const_113_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4542_cast_fp16 = mul(x = var_4540_cast_fp16_1, y = const_113_promoted_to_fp16)[name = string("op_4542_cast_fp16")]; int32 var_4544 = const()[name = string("op_4544"), val = int32(-2)]; bool var_4545_interleave_0 = const()[name = string("op_4545_interleave_0"), val = bool(false)]; tensor var_4545_cast_fp16 = concat(axis = var_4544, interleave = var_4545_interleave_0, values = (var_4542_cast_fp16, var_4540_cast_fp16_0))[name = string("op_4545_cast_fp16")]; tensor var_4546_cast_fp16 = mul(x = var_4545_cast_fp16, y = var_878_cast_fp16)[name = string("op_4546_cast_fp16")]; tensor key_states_115_cast_fp16 = add(x = var_4539_cast_fp16, y = var_4546_cast_fp16)[name = string("key_states_115_cast_fp16")]; tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; int32 concat_137_axis_0 = const()[name = string("concat_137_axis_0"), val = int32(0)]; bool concat_137_interleave_0 = const()[name = string("concat_137_interleave_0"), val = bool(false)]; tensor concat_137 = concat(axis = concat_137_axis_0, interleave = concat_137_interleave_0, values = (expand_dims_132, expand_dims_133, position_id, expand_dims_135))[name = string("concat_137")]; tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; tensor concat_138_values1_0 = const()[name = string("concat_138_values1_0"), val = tensor([0])]; tensor concat_138_values3_0 = const()[name = string("concat_138_values3_0"), val = tensor([0])]; int32 concat_138_axis_0 = const()[name = string("concat_138_axis_0"), val = int32(0)]; bool concat_138_interleave_0 = const()[name = string("concat_138_interleave_0"), val = bool(false)]; tensor concat_138 = concat(axis = concat_138_axis_0, interleave = concat_138_interleave_0, values = (expand_dims_136, concat_138_values1_0, cache_position_end, concat_138_values3_0))[name = string("concat_138")]; tensor key_states_117_perm_0 = const()[name = string("key_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_12_stride_0 = const()[name = string("key_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_117_cast_fp16 = transpose(perm = key_states_117_perm_0, x = key_states_115_cast_fp16)[name = string("transpose_136")]; tensor key_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = key_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = key_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_12_squeeze_mask_0, stride = key_cache_internal_tensor_assign_12_stride_0, update = key_states_117_cast_fp16, x = coreml_update_state_76)[name = string("key_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_12_cast_fp16, input = key_cache)[name = string("coreml_update_state_78_write_state")]; tensor coreml_update_state_78 = read_state(input = key_cache)[name = string("coreml_update_state_78")]; tensor value_states_69_perm_0 = const()[name = string("value_states_69_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_12_stride_0 = const()[name = string("value_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_69_cast_fp16 = transpose(perm = value_states_69_perm_0, x = var_4522_cast_fp16)[name = string("transpose_135")]; tensor value_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = value_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = value_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_12_squeeze_mask_0, stride = value_cache_internal_tensor_assign_12_stride_0, update = value_states_69_cast_fp16, x = coreml_update_state_77)[name = string("value_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_12_cast_fp16, input = value_cache)[name = string("coreml_update_state_79_write_state")]; tensor coreml_update_state_79 = read_state(input = value_cache)[name = string("coreml_update_state_79")]; tensor var_4616_begin_0 = const()[name = string("op_4616_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4616_end_0 = const()[name = string("op_4616_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4616_end_mask_0 = const()[name = string("op_4616_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4616_cast_fp16 = slice_by_index(begin = var_4616_begin_0, end = var_4616_end_0, end_mask = var_4616_end_mask_0, x = coreml_update_state_78)[name = string("op_4616_cast_fp16")]; tensor tile_22 = const()[name = string("tile_22"), val = tensor([1, 1])]; int32 var_4619_axis_0 = const()[name = string("op_4619_axis_0"), val = int32(1)]; tensor var_4619_cast_fp16_0, tensor var_4619_cast_fp16_1 = split(axis = var_4619_axis_0, split_sizes = tile_22, x = var_4616_cast_fp16)[name = string("op_4619_cast_fp16")]; tensor var_4626_begin_0 = const()[name = string("op_4626_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4626_end_0 = const()[name = string("op_4626_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4626_end_mask_0 = const()[name = string("op_4626_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4626_cast_fp16 = slice_by_index(begin = var_4626_begin_0, end = var_4626_end_0, end_mask = var_4626_end_mask_0, x = coreml_update_state_79)[name = string("op_4626_cast_fp16")]; tensor tile_23 = const()[name = string("tile_23"), val = tensor([1, 1])]; int32 var_4629_axis_0 = const()[name = string("op_4629_axis_0"), val = int32(1)]; tensor var_4629_cast_fp16_0, tensor var_4629_cast_fp16_1 = split(axis = var_4629_axis_0, split_sizes = tile_23, x = var_4626_cast_fp16)[name = string("op_4629_cast_fp16")]; tensor var_4632_split_sizes_0 = const()[name = string("op_4632_split_sizes_0"), val = tensor([8, 8])]; int32 var_4632_axis_0 = const()[name = string("op_4632_axis_0"), val = int32(1)]; tensor var_4632_0, tensor var_4632_1 = split(axis = var_4632_axis_0, split_sizes = var_4632_split_sizes_0, x = query_states_69_cast_fp16)[name = string("op_4632")]; bool attn_weights_177_transpose_x_0 = const()[name = string("attn_weights_177_transpose_x_0"), val = bool(false)]; bool attn_weights_177_transpose_y_0 = const()[name = string("attn_weights_177_transpose_y_0"), val = bool(false)]; tensor attn_weights_177_cast_fp16 = matmul(transpose_x = attn_weights_177_transpose_x_0, transpose_y = attn_weights_177_transpose_y_0, x = var_4619_cast_fp16_0, y = var_4632_0)[name = string("attn_weights_177_cast_fp16")]; fp16 var_4635_to_fp16 = const()[name = string("op_4635_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_179_cast_fp16 = mul(x = attn_weights_177_cast_fp16, y = var_4635_to_fp16)[name = string("attn_weights_179_cast_fp16")]; tensor attn_weights_181_cast_fp16 = add(x = attn_weights_179_cast_fp16, y = attn_mask_1)[name = string("attn_weights_181_cast_fp16")]; int32 var_4639 = const()[name = string("op_4639"), val = int32(-2)]; tensor attn_weights_183_cast_fp16 = softmax(axis = var_4639, x = attn_weights_181_cast_fp16)[name = string("attn_weights_183_cast_fp16")]; bool var_4645_transpose_x_1 = const()[name = string("op_4645_transpose_x_1"), val = bool(true)]; bool var_4645_transpose_y_1 = const()[name = string("op_4645_transpose_y_1"), val = bool(false)]; tensor var_4645_cast_fp16 = matmul(transpose_x = var_4645_transpose_x_1, transpose_y = var_4645_transpose_y_1, x = attn_weights_183_cast_fp16, y = var_4629_cast_fp16_0)[name = string("op_4645_cast_fp16")]; bool attn_weights_185_transpose_x_0 = const()[name = string("attn_weights_185_transpose_x_0"), val = bool(false)]; bool attn_weights_185_transpose_y_0 = const()[name = string("attn_weights_185_transpose_y_0"), val = bool(false)]; tensor attn_weights_185_cast_fp16 = matmul(transpose_x = attn_weights_185_transpose_x_0, transpose_y = attn_weights_185_transpose_y_0, x = var_4619_cast_fp16_1, y = var_4632_1)[name = string("attn_weights_185_cast_fp16")]; fp16 var_4647_to_fp16 = const()[name = string("op_4647_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_187_cast_fp16 = mul(x = attn_weights_185_cast_fp16, y = var_4647_to_fp16)[name = string("attn_weights_187_cast_fp16")]; tensor attn_weights_189_cast_fp16 = add(x = attn_weights_187_cast_fp16, y = attn_mask_1)[name = string("attn_weights_189_cast_fp16")]; int32 var_4651 = const()[name = string("op_4651"), val = int32(-2)]; tensor attn_weights_191_cast_fp16 = softmax(axis = var_4651, x = attn_weights_189_cast_fp16)[name = string("attn_weights_191_cast_fp16")]; bool attn_output_89_transpose_x_1 = const()[name = string("attn_output_89_transpose_x_1"), val = bool(true)]; bool attn_output_89_transpose_y_1 = const()[name = string("attn_output_89_transpose_y_1"), val = bool(false)]; tensor attn_output_89_cast_fp16 = matmul(transpose_x = attn_output_89_transpose_x_1, transpose_y = attn_output_89_transpose_y_1, x = attn_weights_191_cast_fp16, y = var_4629_cast_fp16_1)[name = string("attn_output_89_cast_fp16")]; int32 var_4659 = const()[name = string("op_4659"), val = int32(1)]; bool attn_output_91_interleave_0 = const()[name = string("attn_output_91_interleave_0"), val = bool(false)]; tensor attn_output_91_cast_fp16 = concat(axis = var_4659, interleave = attn_output_91_interleave_0, values = (var_4645_cast_fp16, attn_output_89_cast_fp16))[name = string("attn_output_91_cast_fp16")]; tensor var_4663_perm_0 = const()[name = string("op_4663_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_143x = const()[name = string("concat_143x"), val = tensor([1, 2048, 1, -1])]; tensor var_4663_cast_fp16 = transpose(perm = var_4663_perm_0, x = attn_output_91_cast_fp16)[name = string("transpose_134")]; tensor attn_output_95_cast_fp16 = reshape(shape = concat_143x, x = var_4663_cast_fp16)[name = string("attn_output_95_cast_fp16")]; tensor hidden_states_113_strides_0 = const()[name = string("hidden_states_113_strides_0"), val = tensor([1, 1])]; string hidden_states_113_pad_type_0 = const()[name = string("hidden_states_113_pad_type_0"), val = string("valid")]; tensor hidden_states_113_pad_0 = const()[name = string("hidden_states_113_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_113_dilations_0 = const()[name = string("hidden_states_113_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_113_groups_0 = const()[name = string("hidden_states_113_groups_0"), val = int32(1)]; tensor hidden_states_113_cast_fp16 = conv(dilations = hidden_states_113_dilations_0, groups = hidden_states_113_groups_0, pad = hidden_states_113_pad_0, pad_type = hidden_states_113_pad_type_0, strides = hidden_states_113_strides_0, weight = layers_11_self_attn_o_proj_weight_cast_fp16, x = attn_output_95_cast_fp16)[name = string("hidden_states_113_cast_fp16")]; tensor hidden_states_115_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = hidden_states_113_cast_fp16)[name = string("hidden_states_115_cast_fp16")]; fp16 const_118_promoted_to_fp16 = const()[name = string("const_118_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4696_cast_fp16 = mul(x = hidden_states_115_cast_fp16, y = const_118_promoted_to_fp16)[name = string("op_4696_cast_fp16")]; int32 var_4694 = const()[name = string("op_4694"), val = int32(1)]; bool doubled_93_interleave_0 = const()[name = string("doubled_93_interleave_0"), val = bool(false)]; tensor doubled_93_cast_fp16 = concat(axis = var_4694, interleave = doubled_93_interleave_0, values = (hidden_states_115_cast_fp16, var_4696_cast_fp16))[name = string("doubled_93_cast_fp16")]; tensor out_47_axes_0 = const()[name = string("out_47_axes_0"), val = tensor([1])]; tensor out_47_gamma_0_to_fp16 = const()[name = string("out_47_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415366400)))]; fp16 var_4706_to_fp16 = const()[name = string("op_4706_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_47_cast_fp16 = layer_norm(axes = out_47_axes_0, epsilon = var_4706_to_fp16, gamma = out_47_gamma_0_to_fp16, x = doubled_93_cast_fp16)[name = string("out_47_cast_fp16")]; tensor var_4717_split_sizes_0 = const()[name = string("op_4717_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4717_axis_0 = const()[name = string("op_4717_axis_0"), val = int32(1)]; tensor var_4717_cast_fp16_0, tensor var_4717_cast_fp16_1 = split(axis = var_4717_axis_0, split_sizes = var_4717_split_sizes_0, x = out_47_cast_fp16)[name = string("op_4717_cast_fp16")]; tensor input_23_strides_0 = const()[name = string("input_23_strides_0"), val = tensor([1, 1])]; string input_23_pad_type_0 = const()[name = string("input_23_pad_type_0"), val = string("valid")]; tensor input_23_pad_0 = const()[name = string("input_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_23_dilations_0 = const()[name = string("input_23_dilations_0"), val = tensor([1, 1])]; int32 input_23_groups_0 = const()[name = string("input_23_groups_0"), val = int32(1)]; tensor input_23_cast_fp16 = conv(dilations = input_23_dilations_0, groups = input_23_groups_0, pad = input_23_pad_0, pad_type = input_23_pad_type_0, strides = input_23_strides_0, weight = layers_11_mlp_gate_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("input_23_cast_fp16")]; tensor var_4734_cast_fp16 = silu(x = input_23_cast_fp16)[name = string("op_4734_cast_fp16")]; tensor var_4740_strides_0 = const()[name = string("op_4740_strides_0"), val = tensor([1, 1])]; string var_4740_pad_type_0 = const()[name = string("op_4740_pad_type_0"), val = string("valid")]; tensor var_4740_pad_0 = const()[name = string("op_4740_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4740_dilations_0 = const()[name = string("op_4740_dilations_0"), val = tensor([1, 1])]; int32 var_4740_groups_0 = const()[name = string("op_4740_groups_0"), val = int32(1)]; tensor var_4740_cast_fp16 = conv(dilations = var_4740_dilations_0, groups = var_4740_groups_0, pad = var_4740_pad_0, pad_type = var_4740_pad_type_0, strides = var_4740_strides_0, weight = layers_11_mlp_up_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("op_4740_cast_fp16")]; tensor x_119_cast_fp16 = mul(x = var_4734_cast_fp16, y = var_4740_cast_fp16)[name = string("x_119_cast_fp16")]; tensor hidden_states_117_strides_0 = const()[name = string("hidden_states_117_strides_0"), val = tensor([1, 1])]; string hidden_states_117_pad_type_0 = const()[name = string("hidden_states_117_pad_type_0"), val = string("valid")]; tensor hidden_states_117_pad_0 = const()[name = string("hidden_states_117_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_117_dilations_0 = const()[name = string("hidden_states_117_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_117_groups_0 = const()[name = string("hidden_states_117_groups_0"), val = int32(1)]; tensor hidden_states_117_cast_fp16 = conv(dilations = hidden_states_117_dilations_0, groups = hidden_states_117_groups_0, pad = hidden_states_117_pad_0, pad_type = hidden_states_117_pad_type_0, strides = hidden_states_117_strides_0, weight = layers_11_mlp_down_proj_weight_cast_fp16, x = x_119_cast_fp16)[name = string("hidden_states_117_cast_fp16")]; tensor hidden_states_119_cast_fp16 = add(x = hidden_states_115_cast_fp16, y = hidden_states_117_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; fp16 const_120_promoted_to_fp16 = const()[name = string("const_120_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4758_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_120_promoted_to_fp16)[name = string("op_4758_cast_fp16")]; int32 var_4756 = const()[name = string("op_4756"), val = int32(1)]; bool doubled_97_interleave_0 = const()[name = string("doubled_97_interleave_0"), val = bool(false)]; tensor doubled_97_cast_fp16 = concat(axis = var_4756, interleave = doubled_97_interleave_0, values = (hidden_states_119_cast_fp16, var_4758_cast_fp16))[name = string("doubled_97_cast_fp16")]; tensor out_49_axes_0 = const()[name = string("out_49_axes_0"), val = tensor([1])]; tensor out_49_gamma_0_to_fp16 = const()[name = string("out_49_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415374656)))]; fp16 var_4768_to_fp16 = const()[name = string("op_4768_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_49_cast_fp16 = layer_norm(axes = out_49_axes_0, epsilon = var_4768_to_fp16, gamma = out_49_gamma_0_to_fp16, x = doubled_97_cast_fp16)[name = string("out_49_cast_fp16")]; tensor var_4779_split_sizes_0 = const()[name = string("op_4779_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4779_axis_0 = const()[name = string("op_4779_axis_0"), val = int32(1)]; tensor var_4779_cast_fp16_0, tensor var_4779_cast_fp16_1 = split(axis = var_4779_axis_0, split_sizes = var_4779_split_sizes_0, x = out_49_cast_fp16)[name = string("op_4779_cast_fp16")]; tensor query_states_73_strides_0 = const()[name = string("query_states_73_strides_0"), val = tensor([1, 1])]; string query_states_73_pad_type_0 = const()[name = string("query_states_73_pad_type_0"), val = string("valid")]; tensor query_states_73_pad_0 = const()[name = string("query_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_73_dilations_0 = const()[name = string("query_states_73_dilations_0"), val = tensor([1, 1])]; int32 query_states_73_groups_0 = const()[name = string("query_states_73_groups_0"), val = int32(1)]; tensor query_states_73_cast_fp16 = conv(dilations = query_states_73_dilations_0, groups = query_states_73_groups_0, pad = query_states_73_pad_0, pad_type = query_states_73_pad_type_0, strides = query_states_73_strides_0, weight = layers_12_self_attn_q_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("query_states_73_cast_fp16")]; tensor key_states_121_strides_0 = const()[name = string("key_states_121_strides_0"), val = tensor([1, 1])]; string key_states_121_pad_type_0 = const()[name = string("key_states_121_pad_type_0"), val = string("valid")]; tensor key_states_121_pad_0 = const()[name = string("key_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_121_dilations_0 = const()[name = string("key_states_121_dilations_0"), val = tensor([1, 1])]; int32 key_states_121_groups_0 = const()[name = string("key_states_121_groups_0"), val = int32(1)]; tensor key_states_121_cast_fp16 = conv(dilations = key_states_121_dilations_0, groups = key_states_121_groups_0, pad = key_states_121_pad_0, pad_type = key_states_121_pad_type_0, strides = key_states_121_strides_0, weight = layers_12_self_attn_k_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("key_states_121_cast_fp16")]; tensor value_states_73_strides_0 = const()[name = string("value_states_73_strides_0"), val = tensor([1, 1])]; string value_states_73_pad_type_0 = const()[name = string("value_states_73_pad_type_0"), val = string("valid")]; tensor value_states_73_pad_0 = const()[name = string("value_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_73_dilations_0 = const()[name = string("value_states_73_dilations_0"), val = tensor([1, 1])]; int32 value_states_73_groups_0 = const()[name = string("value_states_73_groups_0"), val = int32(1)]; tensor value_states_73_cast_fp16 = conv(dilations = value_states_73_dilations_0, groups = value_states_73_groups_0, pad = value_states_73_pad_0, pad_type = value_states_73_pad_type_0, strides = value_states_73_strides_0, weight = layers_12_self_attn_v_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("value_states_73_cast_fp16")]; tensor concat_144x = const()[name = string("concat_144x"), val = tensor([1, 16, 128, -1])]; tensor x_121_cast_fp16 = reshape(shape = concat_144x, x = query_states_73_cast_fp16)[name = string("x_121_cast_fp16")]; tensor concat_145x = const()[name = string("concat_145x"), val = tensor([1, 2, 128, -1])]; tensor var_4836_cast_fp16 = reshape(shape = concat_145x, x = key_states_121_cast_fp16)[name = string("op_4836_cast_fp16")]; tensor concat_146x = const()[name = string("concat_146x"), val = tensor([1, 2, 128, -1])]; tensor var_4843_cast_fp16 = reshape(shape = concat_146x, x = value_states_73_cast_fp16)[name = string("op_4843_cast_fp16")]; tensor var_4847_cast_fp16 = mul(x = x_121_cast_fp16, y = var_869_cast_fp16)[name = string("op_4847_cast_fp16")]; tensor var_4848_split_sizes_0 = const()[name = string("op_4848_split_sizes_0"), val = tensor([64, 64])]; int32 var_4848_axis_0 = const()[name = string("op_4848_axis_0"), val = int32(-2)]; tensor var_4848_cast_fp16_0, tensor var_4848_cast_fp16_1 = split(axis = var_4848_axis_0, split_sizes = var_4848_split_sizes_0, x = x_121_cast_fp16)[name = string("op_4848_cast_fp16")]; fp16 const_122_promoted_to_fp16 = const()[name = string("const_122_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4850_cast_fp16 = mul(x = var_4848_cast_fp16_1, y = const_122_promoted_to_fp16)[name = string("op_4850_cast_fp16")]; int32 var_4852 = const()[name = string("op_4852"), val = int32(-2)]; bool var_4853_interleave_0 = const()[name = string("op_4853_interleave_0"), val = bool(false)]; tensor var_4853_cast_fp16 = concat(axis = var_4852, interleave = var_4853_interleave_0, values = (var_4850_cast_fp16, var_4848_cast_fp16_0))[name = string("op_4853_cast_fp16")]; tensor var_4854_cast_fp16 = mul(x = var_4853_cast_fp16, y = var_878_cast_fp16)[name = string("op_4854_cast_fp16")]; tensor query_states_75_cast_fp16 = add(x = var_4847_cast_fp16, y = var_4854_cast_fp16)[name = string("query_states_75_cast_fp16")]; tensor var_4860_cast_fp16 = mul(x = var_4836_cast_fp16, y = var_869_cast_fp16)[name = string("op_4860_cast_fp16")]; tensor var_4861_split_sizes_0 = const()[name = string("op_4861_split_sizes_0"), val = tensor([64, 64])]; int32 var_4861_axis_0 = const()[name = string("op_4861_axis_0"), val = int32(-2)]; tensor var_4861_cast_fp16_0, tensor var_4861_cast_fp16_1 = split(axis = var_4861_axis_0, split_sizes = var_4861_split_sizes_0, x = var_4836_cast_fp16)[name = string("op_4861_cast_fp16")]; fp16 const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4863_cast_fp16 = mul(x = var_4861_cast_fp16_1, y = const_123_promoted_to_fp16)[name = string("op_4863_cast_fp16")]; int32 var_4865 = const()[name = string("op_4865"), val = int32(-2)]; bool var_4866_interleave_0 = const()[name = string("op_4866_interleave_0"), val = bool(false)]; tensor var_4866_cast_fp16 = concat(axis = var_4865, interleave = var_4866_interleave_0, values = (var_4863_cast_fp16, var_4861_cast_fp16_0))[name = string("op_4866_cast_fp16")]; tensor var_4867_cast_fp16 = mul(x = var_4866_cast_fp16, y = var_878_cast_fp16)[name = string("op_4867_cast_fp16")]; tensor key_states_125_cast_fp16 = add(x = var_4860_cast_fp16, y = var_4867_cast_fp16)[name = string("key_states_125_cast_fp16")]; tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; int32 concat_149_axis_0 = const()[name = string("concat_149_axis_0"), val = int32(0)]; bool concat_149_interleave_0 = const()[name = string("concat_149_interleave_0"), val = bool(false)]; tensor concat_149 = concat(axis = concat_149_axis_0, interleave = concat_149_interleave_0, values = (expand_dims_144, expand_dims_145, position_id, expand_dims_147))[name = string("concat_149")]; tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; tensor concat_150_values1_0 = const()[name = string("concat_150_values1_0"), val = tensor([0])]; tensor concat_150_values3_0 = const()[name = string("concat_150_values3_0"), val = tensor([0])]; int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_148, concat_150_values1_0, cache_position_end, concat_150_values3_0))[name = string("concat_150")]; tensor key_states_127_perm_0 = const()[name = string("key_states_127_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_13_stride_0 = const()[name = string("key_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_127_cast_fp16 = transpose(perm = key_states_127_perm_0, x = key_states_125_cast_fp16)[name = string("transpose_133")]; tensor key_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = key_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = key_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_13_squeeze_mask_0, stride = key_cache_internal_tensor_assign_13_stride_0, update = key_states_127_cast_fp16, x = coreml_update_state_78)[name = string("key_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_13_cast_fp16, input = key_cache)[name = string("coreml_update_state_80_write_state")]; tensor coreml_update_state_80 = read_state(input = key_cache)[name = string("coreml_update_state_80")]; tensor value_states_75_perm_0 = const()[name = string("value_states_75_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_13_stride_0 = const()[name = string("value_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_75_cast_fp16 = transpose(perm = value_states_75_perm_0, x = var_4843_cast_fp16)[name = string("transpose_132")]; tensor value_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = value_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = value_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_13_squeeze_mask_0, stride = value_cache_internal_tensor_assign_13_stride_0, update = value_states_75_cast_fp16, x = coreml_update_state_79)[name = string("value_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_13_cast_fp16, input = value_cache)[name = string("coreml_update_state_81_write_state")]; tensor coreml_update_state_81 = read_state(input = value_cache)[name = string("coreml_update_state_81")]; tensor var_4937_begin_0 = const()[name = string("op_4937_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4937_end_0 = const()[name = string("op_4937_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4937_end_mask_0 = const()[name = string("op_4937_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4937_cast_fp16 = slice_by_index(begin = var_4937_begin_0, end = var_4937_end_0, end_mask = var_4937_end_mask_0, x = coreml_update_state_80)[name = string("op_4937_cast_fp16")]; tensor tile_24 = const()[name = string("tile_24"), val = tensor([1, 1])]; int32 var_4940_axis_0 = const()[name = string("op_4940_axis_0"), val = int32(1)]; tensor var_4940_cast_fp16_0, tensor var_4940_cast_fp16_1 = split(axis = var_4940_axis_0, split_sizes = tile_24, x = var_4937_cast_fp16)[name = string("op_4940_cast_fp16")]; tensor var_4947_begin_0 = const()[name = string("op_4947_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4947_end_0 = const()[name = string("op_4947_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4947_end_mask_0 = const()[name = string("op_4947_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4947_cast_fp16 = slice_by_index(begin = var_4947_begin_0, end = var_4947_end_0, end_mask = var_4947_end_mask_0, x = coreml_update_state_81)[name = string("op_4947_cast_fp16")]; tensor tile_25 = const()[name = string("tile_25"), val = tensor([1, 1])]; int32 var_4950_axis_0 = const()[name = string("op_4950_axis_0"), val = int32(1)]; tensor var_4950_cast_fp16_0, tensor var_4950_cast_fp16_1 = split(axis = var_4950_axis_0, split_sizes = tile_25, x = var_4947_cast_fp16)[name = string("op_4950_cast_fp16")]; tensor var_4953_split_sizes_0 = const()[name = string("op_4953_split_sizes_0"), val = tensor([8, 8])]; int32 var_4953_axis_0 = const()[name = string("op_4953_axis_0"), val = int32(1)]; tensor var_4953_0, tensor var_4953_1 = split(axis = var_4953_axis_0, split_sizes = var_4953_split_sizes_0, x = query_states_75_cast_fp16)[name = string("op_4953")]; bool attn_weights_193_transpose_x_0 = const()[name = string("attn_weights_193_transpose_x_0"), val = bool(false)]; bool attn_weights_193_transpose_y_0 = const()[name = string("attn_weights_193_transpose_y_0"), val = bool(false)]; tensor attn_weights_193_cast_fp16 = matmul(transpose_x = attn_weights_193_transpose_x_0, transpose_y = attn_weights_193_transpose_y_0, x = var_4940_cast_fp16_0, y = var_4953_0)[name = string("attn_weights_193_cast_fp16")]; fp16 var_4956_to_fp16 = const()[name = string("op_4956_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_195_cast_fp16 = mul(x = attn_weights_193_cast_fp16, y = var_4956_to_fp16)[name = string("attn_weights_195_cast_fp16")]; tensor attn_weights_197_cast_fp16 = add(x = attn_weights_195_cast_fp16, y = attn_mask_1)[name = string("attn_weights_197_cast_fp16")]; int32 var_4960 = const()[name = string("op_4960"), val = int32(-2)]; tensor attn_weights_199_cast_fp16 = softmax(axis = var_4960, x = attn_weights_197_cast_fp16)[name = string("attn_weights_199_cast_fp16")]; bool var_4966_transpose_x_1 = const()[name = string("op_4966_transpose_x_1"), val = bool(true)]; bool var_4966_transpose_y_1 = const()[name = string("op_4966_transpose_y_1"), val = bool(false)]; tensor var_4966_cast_fp16 = matmul(transpose_x = var_4966_transpose_x_1, transpose_y = var_4966_transpose_y_1, x = attn_weights_199_cast_fp16, y = var_4950_cast_fp16_0)[name = string("op_4966_cast_fp16")]; bool attn_weights_201_transpose_x_0 = const()[name = string("attn_weights_201_transpose_x_0"), val = bool(false)]; bool attn_weights_201_transpose_y_0 = const()[name = string("attn_weights_201_transpose_y_0"), val = bool(false)]; tensor attn_weights_201_cast_fp16 = matmul(transpose_x = attn_weights_201_transpose_x_0, transpose_y = attn_weights_201_transpose_y_0, x = var_4940_cast_fp16_1, y = var_4953_1)[name = string("attn_weights_201_cast_fp16")]; fp16 var_4968_to_fp16 = const()[name = string("op_4968_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_203_cast_fp16 = mul(x = attn_weights_201_cast_fp16, y = var_4968_to_fp16)[name = string("attn_weights_203_cast_fp16")]; tensor attn_weights_205_cast_fp16 = add(x = attn_weights_203_cast_fp16, y = attn_mask_1)[name = string("attn_weights_205_cast_fp16")]; int32 var_4972 = const()[name = string("op_4972"), val = int32(-2)]; tensor attn_weights_207_cast_fp16 = softmax(axis = var_4972, x = attn_weights_205_cast_fp16)[name = string("attn_weights_207_cast_fp16")]; bool attn_output_97_transpose_x_1 = const()[name = string("attn_output_97_transpose_x_1"), val = bool(true)]; bool attn_output_97_transpose_y_1 = const()[name = string("attn_output_97_transpose_y_1"), val = bool(false)]; tensor attn_output_97_cast_fp16 = matmul(transpose_x = attn_output_97_transpose_x_1, transpose_y = attn_output_97_transpose_y_1, x = attn_weights_207_cast_fp16, y = var_4950_cast_fp16_1)[name = string("attn_output_97_cast_fp16")]; int32 var_4980 = const()[name = string("op_4980"), val = int32(1)]; bool attn_output_99_interleave_0 = const()[name = string("attn_output_99_interleave_0"), val = bool(false)]; tensor attn_output_99_cast_fp16 = concat(axis = var_4980, interleave = attn_output_99_interleave_0, values = (var_4966_cast_fp16, attn_output_97_cast_fp16))[name = string("attn_output_99_cast_fp16")]; tensor var_4984_perm_0 = const()[name = string("op_4984_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_155x = const()[name = string("concat_155x"), val = tensor([1, 2048, 1, -1])]; tensor var_4984_cast_fp16 = transpose(perm = var_4984_perm_0, x = attn_output_99_cast_fp16)[name = string("transpose_131")]; tensor attn_output_103_cast_fp16 = reshape(shape = concat_155x, x = var_4984_cast_fp16)[name = string("attn_output_103_cast_fp16")]; tensor hidden_states_123_strides_0 = const()[name = string("hidden_states_123_strides_0"), val = tensor([1, 1])]; string hidden_states_123_pad_type_0 = const()[name = string("hidden_states_123_pad_type_0"), val = string("valid")]; tensor hidden_states_123_pad_0 = const()[name = string("hidden_states_123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_123_dilations_0 = const()[name = string("hidden_states_123_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_123_groups_0 = const()[name = string("hidden_states_123_groups_0"), val = int32(1)]; tensor hidden_states_123_cast_fp16 = conv(dilations = hidden_states_123_dilations_0, groups = hidden_states_123_groups_0, pad = hidden_states_123_pad_0, pad_type = hidden_states_123_pad_type_0, strides = hidden_states_123_strides_0, weight = layers_12_self_attn_o_proj_weight_cast_fp16, x = attn_output_103_cast_fp16)[name = string("hidden_states_123_cast_fp16")]; tensor hidden_states_125_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = hidden_states_123_cast_fp16)[name = string("hidden_states_125_cast_fp16")]; fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5017_cast_fp16 = mul(x = hidden_states_125_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_5017_cast_fp16")]; int32 var_5015 = const()[name = string("op_5015"), val = int32(1)]; bool doubled_101_interleave_0 = const()[name = string("doubled_101_interleave_0"), val = bool(false)]; tensor doubled_101_cast_fp16 = concat(axis = var_5015, interleave = doubled_101_interleave_0, values = (hidden_states_125_cast_fp16, var_5017_cast_fp16))[name = string("doubled_101_cast_fp16")]; tensor out_51_axes_0 = const()[name = string("out_51_axes_0"), val = tensor([1])]; tensor out_51_gamma_0_to_fp16 = const()[name = string("out_51_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415382912)))]; fp16 var_5027_to_fp16 = const()[name = string("op_5027_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_51_cast_fp16 = layer_norm(axes = out_51_axes_0, epsilon = var_5027_to_fp16, gamma = out_51_gamma_0_to_fp16, x = doubled_101_cast_fp16)[name = string("out_51_cast_fp16")]; tensor var_5038_split_sizes_0 = const()[name = string("op_5038_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5038_axis_0 = const()[name = string("op_5038_axis_0"), val = int32(1)]; tensor var_5038_cast_fp16_0, tensor var_5038_cast_fp16_1 = split(axis = var_5038_axis_0, split_sizes = var_5038_split_sizes_0, x = out_51_cast_fp16)[name = string("op_5038_cast_fp16")]; tensor input_25_strides_0 = const()[name = string("input_25_strides_0"), val = tensor([1, 1])]; string input_25_pad_type_0 = const()[name = string("input_25_pad_type_0"), val = string("valid")]; tensor input_25_pad_0 = const()[name = string("input_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_25_dilations_0 = const()[name = string("input_25_dilations_0"), val = tensor([1, 1])]; int32 input_25_groups_0 = const()[name = string("input_25_groups_0"), val = int32(1)]; tensor input_25_cast_fp16 = conv(dilations = input_25_dilations_0, groups = input_25_groups_0, pad = input_25_pad_0, pad_type = input_25_pad_type_0, strides = input_25_strides_0, weight = layers_12_mlp_gate_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("input_25_cast_fp16")]; tensor var_5055_cast_fp16 = silu(x = input_25_cast_fp16)[name = string("op_5055_cast_fp16")]; tensor var_5061_strides_0 = const()[name = string("op_5061_strides_0"), val = tensor([1, 1])]; string var_5061_pad_type_0 = const()[name = string("op_5061_pad_type_0"), val = string("valid")]; tensor var_5061_pad_0 = const()[name = string("op_5061_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5061_dilations_0 = const()[name = string("op_5061_dilations_0"), val = tensor([1, 1])]; int32 var_5061_groups_0 = const()[name = string("op_5061_groups_0"), val = int32(1)]; tensor var_5061_cast_fp16 = conv(dilations = var_5061_dilations_0, groups = var_5061_groups_0, pad = var_5061_pad_0, pad_type = var_5061_pad_type_0, strides = var_5061_strides_0, weight = layers_12_mlp_up_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("op_5061_cast_fp16")]; tensor x_129_cast_fp16 = mul(x = var_5055_cast_fp16, y = var_5061_cast_fp16)[name = string("x_129_cast_fp16")]; tensor hidden_states_127_strides_0 = const()[name = string("hidden_states_127_strides_0"), val = tensor([1, 1])]; string hidden_states_127_pad_type_0 = const()[name = string("hidden_states_127_pad_type_0"), val = string("valid")]; tensor hidden_states_127_pad_0 = const()[name = string("hidden_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_127_dilations_0 = const()[name = string("hidden_states_127_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_127_groups_0 = const()[name = string("hidden_states_127_groups_0"), val = int32(1)]; tensor hidden_states_127_cast_fp16 = conv(dilations = hidden_states_127_dilations_0, groups = hidden_states_127_groups_0, pad = hidden_states_127_pad_0, pad_type = hidden_states_127_pad_type_0, strides = hidden_states_127_strides_0, weight = layers_12_mlp_down_proj_weight_cast_fp16, x = x_129_cast_fp16)[name = string("hidden_states_127_cast_fp16")]; tensor hidden_states_129_cast_fp16 = add(x = hidden_states_125_cast_fp16, y = hidden_states_127_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5079_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_130_promoted_to_fp16)[name = string("op_5079_cast_fp16")]; int32 var_5077 = const()[name = string("op_5077"), val = int32(1)]; bool doubled_105_interleave_0 = const()[name = string("doubled_105_interleave_0"), val = bool(false)]; tensor doubled_105_cast_fp16 = concat(axis = var_5077, interleave = doubled_105_interleave_0, values = (hidden_states_129_cast_fp16, var_5079_cast_fp16))[name = string("doubled_105_cast_fp16")]; tensor out_53_axes_0 = const()[name = string("out_53_axes_0"), val = tensor([1])]; tensor out_53_gamma_0_to_fp16 = const()[name = string("out_53_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415391168)))]; fp16 var_5089_to_fp16 = const()[name = string("op_5089_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_53_cast_fp16 = layer_norm(axes = out_53_axes_0, epsilon = var_5089_to_fp16, gamma = out_53_gamma_0_to_fp16, x = doubled_105_cast_fp16)[name = string("out_53_cast_fp16")]; tensor var_5100_split_sizes_0 = const()[name = string("op_5100_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5100_axis_0 = const()[name = string("op_5100_axis_0"), val = int32(1)]; tensor var_5100_cast_fp16_0, tensor var_5100_cast_fp16_1 = split(axis = var_5100_axis_0, split_sizes = var_5100_split_sizes_0, x = out_53_cast_fp16)[name = string("op_5100_cast_fp16")]; tensor query_states_79_strides_0 = const()[name = string("query_states_79_strides_0"), val = tensor([1, 1])]; string query_states_79_pad_type_0 = const()[name = string("query_states_79_pad_type_0"), val = string("valid")]; tensor query_states_79_pad_0 = const()[name = string("query_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_79_dilations_0 = const()[name = string("query_states_79_dilations_0"), val = tensor([1, 1])]; int32 query_states_79_groups_0 = const()[name = string("query_states_79_groups_0"), val = int32(1)]; tensor query_states_79_cast_fp16 = conv(dilations = query_states_79_dilations_0, groups = query_states_79_groups_0, pad = query_states_79_pad_0, pad_type = query_states_79_pad_type_0, strides = query_states_79_strides_0, weight = layers_13_self_attn_q_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("query_states_79_cast_fp16")]; tensor key_states_131_strides_0 = const()[name = string("key_states_131_strides_0"), val = tensor([1, 1])]; string key_states_131_pad_type_0 = const()[name = string("key_states_131_pad_type_0"), val = string("valid")]; tensor key_states_131_pad_0 = const()[name = string("key_states_131_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_131_dilations_0 = const()[name = string("key_states_131_dilations_0"), val = tensor([1, 1])]; int32 key_states_131_groups_0 = const()[name = string("key_states_131_groups_0"), val = int32(1)]; tensor key_states_131_cast_fp16 = conv(dilations = key_states_131_dilations_0, groups = key_states_131_groups_0, pad = key_states_131_pad_0, pad_type = key_states_131_pad_type_0, strides = key_states_131_strides_0, weight = layers_13_self_attn_k_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("key_states_131_cast_fp16")]; tensor value_states_79_strides_0 = const()[name = string("value_states_79_strides_0"), val = tensor([1, 1])]; string value_states_79_pad_type_0 = const()[name = string("value_states_79_pad_type_0"), val = string("valid")]; tensor value_states_79_pad_0 = const()[name = string("value_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_79_dilations_0 = const()[name = string("value_states_79_dilations_0"), val = tensor([1, 1])]; int32 value_states_79_groups_0 = const()[name = string("value_states_79_groups_0"), val = int32(1)]; tensor value_states_79_cast_fp16 = conv(dilations = value_states_79_dilations_0, groups = value_states_79_groups_0, pad = value_states_79_pad_0, pad_type = value_states_79_pad_type_0, strides = value_states_79_strides_0, weight = layers_13_self_attn_v_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("value_states_79_cast_fp16")]; tensor concat_156x = const()[name = string("concat_156x"), val = tensor([1, 16, 128, -1])]; tensor x_131_cast_fp16 = reshape(shape = concat_156x, x = query_states_79_cast_fp16)[name = string("x_131_cast_fp16")]; tensor concat_157x = const()[name = string("concat_157x"), val = tensor([1, 2, 128, -1])]; tensor var_5157_cast_fp16 = reshape(shape = concat_157x, x = key_states_131_cast_fp16)[name = string("op_5157_cast_fp16")]; tensor concat_158x = const()[name = string("concat_158x"), val = tensor([1, 2, 128, -1])]; tensor var_5164_cast_fp16 = reshape(shape = concat_158x, x = value_states_79_cast_fp16)[name = string("op_5164_cast_fp16")]; tensor var_5168_cast_fp16 = mul(x = x_131_cast_fp16, y = var_869_cast_fp16)[name = string("op_5168_cast_fp16")]; tensor var_5169_split_sizes_0 = const()[name = string("op_5169_split_sizes_0"), val = tensor([64, 64])]; int32 var_5169_axis_0 = const()[name = string("op_5169_axis_0"), val = int32(-2)]; tensor var_5169_cast_fp16_0, tensor var_5169_cast_fp16_1 = split(axis = var_5169_axis_0, split_sizes = var_5169_split_sizes_0, x = x_131_cast_fp16)[name = string("op_5169_cast_fp16")]; fp16 const_132_promoted_to_fp16 = const()[name = string("const_132_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5171_cast_fp16 = mul(x = var_5169_cast_fp16_1, y = const_132_promoted_to_fp16)[name = string("op_5171_cast_fp16")]; int32 var_5173 = const()[name = string("op_5173"), val = int32(-2)]; bool var_5174_interleave_0 = const()[name = string("op_5174_interleave_0"), val = bool(false)]; tensor var_5174_cast_fp16 = concat(axis = var_5173, interleave = var_5174_interleave_0, values = (var_5171_cast_fp16, var_5169_cast_fp16_0))[name = string("op_5174_cast_fp16")]; tensor var_5175_cast_fp16 = mul(x = var_5174_cast_fp16, y = var_878_cast_fp16)[name = string("op_5175_cast_fp16")]; tensor query_states_81_cast_fp16 = add(x = var_5168_cast_fp16, y = var_5175_cast_fp16)[name = string("query_states_81_cast_fp16")]; tensor var_5181_cast_fp16 = mul(x = var_5157_cast_fp16, y = var_869_cast_fp16)[name = string("op_5181_cast_fp16")]; tensor var_5182_split_sizes_0 = const()[name = string("op_5182_split_sizes_0"), val = tensor([64, 64])]; int32 var_5182_axis_0 = const()[name = string("op_5182_axis_0"), val = int32(-2)]; tensor var_5182_cast_fp16_0, tensor var_5182_cast_fp16_1 = split(axis = var_5182_axis_0, split_sizes = var_5182_split_sizes_0, x = var_5157_cast_fp16)[name = string("op_5182_cast_fp16")]; fp16 const_133_promoted_to_fp16 = const()[name = string("const_133_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5184_cast_fp16 = mul(x = var_5182_cast_fp16_1, y = const_133_promoted_to_fp16)[name = string("op_5184_cast_fp16")]; int32 var_5186 = const()[name = string("op_5186"), val = int32(-2)]; bool var_5187_interleave_0 = const()[name = string("op_5187_interleave_0"), val = bool(false)]; tensor var_5187_cast_fp16 = concat(axis = var_5186, interleave = var_5187_interleave_0, values = (var_5184_cast_fp16, var_5182_cast_fp16_0))[name = string("op_5187_cast_fp16")]; tensor var_5188_cast_fp16 = mul(x = var_5187_cast_fp16, y = var_878_cast_fp16)[name = string("op_5188_cast_fp16")]; tensor key_states_135_cast_fp16 = add(x = var_5181_cast_fp16, y = var_5188_cast_fp16)[name = string("key_states_135_cast_fp16")]; tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; int32 concat_161_axis_0 = const()[name = string("concat_161_axis_0"), val = int32(0)]; bool concat_161_interleave_0 = const()[name = string("concat_161_interleave_0"), val = bool(false)]; tensor concat_161 = concat(axis = concat_161_axis_0, interleave = concat_161_interleave_0, values = (expand_dims_156, expand_dims_157, position_id, expand_dims_159))[name = string("concat_161")]; tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; tensor concat_162_values1_0 = const()[name = string("concat_162_values1_0"), val = tensor([0])]; tensor concat_162_values3_0 = const()[name = string("concat_162_values3_0"), val = tensor([0])]; int32 concat_162_axis_0 = const()[name = string("concat_162_axis_0"), val = int32(0)]; bool concat_162_interleave_0 = const()[name = string("concat_162_interleave_0"), val = bool(false)]; tensor concat_162 = concat(axis = concat_162_axis_0, interleave = concat_162_interleave_0, values = (expand_dims_160, concat_162_values1_0, cache_position_end, concat_162_values3_0))[name = string("concat_162")]; tensor key_states_137_perm_0 = const()[name = string("key_states_137_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_14_stride_0 = const()[name = string("key_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_137_cast_fp16 = transpose(perm = key_states_137_perm_0, x = key_states_135_cast_fp16)[name = string("transpose_130")]; tensor key_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = key_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = key_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_14_squeeze_mask_0, stride = key_cache_internal_tensor_assign_14_stride_0, update = key_states_137_cast_fp16, x = coreml_update_state_80)[name = string("key_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_14_cast_fp16, input = key_cache)[name = string("coreml_update_state_82_write_state")]; tensor coreml_update_state_82 = read_state(input = key_cache)[name = string("coreml_update_state_82")]; tensor value_states_81_perm_0 = const()[name = string("value_states_81_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_14_stride_0 = const()[name = string("value_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_81_cast_fp16 = transpose(perm = value_states_81_perm_0, x = var_5164_cast_fp16)[name = string("transpose_129")]; tensor value_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = value_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = value_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_14_squeeze_mask_0, stride = value_cache_internal_tensor_assign_14_stride_0, update = value_states_81_cast_fp16, x = coreml_update_state_81)[name = string("value_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_14_cast_fp16, input = value_cache)[name = string("coreml_update_state_83_write_state")]; tensor coreml_update_state_83 = read_state(input = value_cache)[name = string("coreml_update_state_83")]; tensor var_5258_begin_0 = const()[name = string("op_5258_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5258_end_0 = const()[name = string("op_5258_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5258_end_mask_0 = const()[name = string("op_5258_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5258_cast_fp16 = slice_by_index(begin = var_5258_begin_0, end = var_5258_end_0, end_mask = var_5258_end_mask_0, x = coreml_update_state_82)[name = string("op_5258_cast_fp16")]; tensor tile_26 = const()[name = string("tile_26"), val = tensor([1, 1])]; int32 var_5261_axis_0 = const()[name = string("op_5261_axis_0"), val = int32(1)]; tensor var_5261_cast_fp16_0, tensor var_5261_cast_fp16_1 = split(axis = var_5261_axis_0, split_sizes = tile_26, x = var_5258_cast_fp16)[name = string("op_5261_cast_fp16")]; tensor var_5268_begin_0 = const()[name = string("op_5268_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5268_end_0 = const()[name = string("op_5268_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5268_end_mask_0 = const()[name = string("op_5268_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5268_cast_fp16 = slice_by_index(begin = var_5268_begin_0, end = var_5268_end_0, end_mask = var_5268_end_mask_0, x = coreml_update_state_83)[name = string("op_5268_cast_fp16")]; tensor tile_27 = const()[name = string("tile_27"), val = tensor([1, 1])]; int32 var_5271_axis_0 = const()[name = string("op_5271_axis_0"), val = int32(1)]; tensor var_5271_cast_fp16_0, tensor var_5271_cast_fp16_1 = split(axis = var_5271_axis_0, split_sizes = tile_27, x = var_5268_cast_fp16)[name = string("op_5271_cast_fp16")]; tensor var_5274_split_sizes_0 = const()[name = string("op_5274_split_sizes_0"), val = tensor([8, 8])]; int32 var_5274_axis_0 = const()[name = string("op_5274_axis_0"), val = int32(1)]; tensor var_5274_0, tensor var_5274_1 = split(axis = var_5274_axis_0, split_sizes = var_5274_split_sizes_0, x = query_states_81_cast_fp16)[name = string("op_5274")]; bool attn_weights_209_transpose_x_0 = const()[name = string("attn_weights_209_transpose_x_0"), val = bool(false)]; bool attn_weights_209_transpose_y_0 = const()[name = string("attn_weights_209_transpose_y_0"), val = bool(false)]; tensor attn_weights_209_cast_fp16 = matmul(transpose_x = attn_weights_209_transpose_x_0, transpose_y = attn_weights_209_transpose_y_0, x = var_5261_cast_fp16_0, y = var_5274_0)[name = string("attn_weights_209_cast_fp16")]; fp16 var_5277_to_fp16 = const()[name = string("op_5277_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_211_cast_fp16 = mul(x = attn_weights_209_cast_fp16, y = var_5277_to_fp16)[name = string("attn_weights_211_cast_fp16")]; tensor attn_weights_213_cast_fp16 = add(x = attn_weights_211_cast_fp16, y = attn_mask_1)[name = string("attn_weights_213_cast_fp16")]; int32 var_5281 = const()[name = string("op_5281"), val = int32(-2)]; tensor attn_weights_215_cast_fp16 = softmax(axis = var_5281, x = attn_weights_213_cast_fp16)[name = string("attn_weights_215_cast_fp16")]; bool var_5287_transpose_x_1 = const()[name = string("op_5287_transpose_x_1"), val = bool(true)]; bool var_5287_transpose_y_1 = const()[name = string("op_5287_transpose_y_1"), val = bool(false)]; tensor var_5287_cast_fp16 = matmul(transpose_x = var_5287_transpose_x_1, transpose_y = var_5287_transpose_y_1, x = attn_weights_215_cast_fp16, y = var_5271_cast_fp16_0)[name = string("op_5287_cast_fp16")]; bool attn_weights_217_transpose_x_0 = const()[name = string("attn_weights_217_transpose_x_0"), val = bool(false)]; bool attn_weights_217_transpose_y_0 = const()[name = string("attn_weights_217_transpose_y_0"), val = bool(false)]; tensor attn_weights_217_cast_fp16 = matmul(transpose_x = attn_weights_217_transpose_x_0, transpose_y = attn_weights_217_transpose_y_0, x = var_5261_cast_fp16_1, y = var_5274_1)[name = string("attn_weights_217_cast_fp16")]; fp16 var_5289_to_fp16 = const()[name = string("op_5289_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_219_cast_fp16 = mul(x = attn_weights_217_cast_fp16, y = var_5289_to_fp16)[name = string("attn_weights_219_cast_fp16")]; tensor attn_weights_221_cast_fp16 = add(x = attn_weights_219_cast_fp16, y = attn_mask_1)[name = string("attn_weights_221_cast_fp16")]; int32 var_5293 = const()[name = string("op_5293"), val = int32(-2)]; tensor attn_weights_223_cast_fp16 = softmax(axis = var_5293, x = attn_weights_221_cast_fp16)[name = string("attn_weights_223_cast_fp16")]; bool attn_output_105_transpose_x_1 = const()[name = string("attn_output_105_transpose_x_1"), val = bool(true)]; bool attn_output_105_transpose_y_1 = const()[name = string("attn_output_105_transpose_y_1"), val = bool(false)]; tensor attn_output_105_cast_fp16 = matmul(transpose_x = attn_output_105_transpose_x_1, transpose_y = attn_output_105_transpose_y_1, x = attn_weights_223_cast_fp16, y = var_5271_cast_fp16_1)[name = string("attn_output_105_cast_fp16")]; int32 var_5301 = const()[name = string("op_5301"), val = int32(1)]; bool attn_output_107_interleave_0 = const()[name = string("attn_output_107_interleave_0"), val = bool(false)]; tensor attn_output_107_cast_fp16 = concat(axis = var_5301, interleave = attn_output_107_interleave_0, values = (var_5287_cast_fp16, attn_output_105_cast_fp16))[name = string("attn_output_107_cast_fp16")]; tensor var_5305_perm_0 = const()[name = string("op_5305_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_167x = const()[name = string("concat_167x"), val = tensor([1, 2048, 1, -1])]; tensor var_5305_cast_fp16 = transpose(perm = var_5305_perm_0, x = attn_output_107_cast_fp16)[name = string("transpose_128")]; tensor attn_output_111_cast_fp16 = reshape(shape = concat_167x, x = var_5305_cast_fp16)[name = string("attn_output_111_cast_fp16")]; tensor hidden_states_133_strides_0 = const()[name = string("hidden_states_133_strides_0"), val = tensor([1, 1])]; string hidden_states_133_pad_type_0 = const()[name = string("hidden_states_133_pad_type_0"), val = string("valid")]; tensor hidden_states_133_pad_0 = const()[name = string("hidden_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_133_dilations_0 = const()[name = string("hidden_states_133_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_133_groups_0 = const()[name = string("hidden_states_133_groups_0"), val = int32(1)]; tensor hidden_states_133_cast_fp16 = conv(dilations = hidden_states_133_dilations_0, groups = hidden_states_133_groups_0, pad = hidden_states_133_pad_0, pad_type = hidden_states_133_pad_type_0, strides = hidden_states_133_strides_0, weight = layers_13_self_attn_o_proj_weight_cast_fp16, x = attn_output_111_cast_fp16)[name = string("hidden_states_133_cast_fp16")]; tensor hidden_states_135_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = hidden_states_133_cast_fp16)[name = string("hidden_states_135_cast_fp16")]; fp16 const_138_promoted_to_fp16 = const()[name = string("const_138_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5338_cast_fp16 = mul(x = hidden_states_135_cast_fp16, y = const_138_promoted_to_fp16)[name = string("op_5338_cast_fp16")]; int32 var_5336 = const()[name = string("op_5336"), val = int32(1)]; bool doubled_109_interleave_0 = const()[name = string("doubled_109_interleave_0"), val = bool(false)]; tensor doubled_109_cast_fp16 = concat(axis = var_5336, interleave = doubled_109_interleave_0, values = (hidden_states_135_cast_fp16, var_5338_cast_fp16))[name = string("doubled_109_cast_fp16")]; tensor out_55_axes_0 = const()[name = string("out_55_axes_0"), val = tensor([1])]; tensor out_55_gamma_0_to_fp16 = const()[name = string("out_55_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415399424)))]; fp16 var_5348_to_fp16 = const()[name = string("op_5348_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_55_cast_fp16 = layer_norm(axes = out_55_axes_0, epsilon = var_5348_to_fp16, gamma = out_55_gamma_0_to_fp16, x = doubled_109_cast_fp16)[name = string("out_55_cast_fp16")]; tensor var_5359_split_sizes_0 = const()[name = string("op_5359_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5359_axis_0 = const()[name = string("op_5359_axis_0"), val = int32(1)]; tensor var_5359_cast_fp16_0, tensor var_5359_cast_fp16_1 = split(axis = var_5359_axis_0, split_sizes = var_5359_split_sizes_0, x = out_55_cast_fp16)[name = string("op_5359_cast_fp16")]; tensor input_27_strides_0 = const()[name = string("input_27_strides_0"), val = tensor([1, 1])]; string input_27_pad_type_0 = const()[name = string("input_27_pad_type_0"), val = string("valid")]; tensor input_27_pad_0 = const()[name = string("input_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_27_dilations_0 = const()[name = string("input_27_dilations_0"), val = tensor([1, 1])]; int32 input_27_groups_0 = const()[name = string("input_27_groups_0"), val = int32(1)]; tensor input_27_cast_fp16 = conv(dilations = input_27_dilations_0, groups = input_27_groups_0, pad = input_27_pad_0, pad_type = input_27_pad_type_0, strides = input_27_strides_0, weight = layers_13_mlp_gate_proj_weight_cast_fp16, x = var_5359_cast_fp16_0)[name = string("input_27_cast_fp16")]; tensor var_5376_cast_fp16 = silu(x = input_27_cast_fp16)[name = string("op_5376_cast_fp16")]; tensor layers_13_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_13_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415407680)))]; tensor var_5382_strides_0 = const()[name = string("op_5382_strides_0"), val = tensor([1, 1])]; string var_5382_pad_type_0 = const()[name = string("op_5382_pad_type_0"), val = string("valid")]; tensor var_5382_pad_0 = const()[name = string("op_5382_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5382_dilations_0 = const()[name = string("op_5382_dilations_0"), val = tensor([1, 1])]; int32 var_5382_groups_0 = const()[name = string("op_5382_groups_0"), val = int32(1)]; tensor var_5382_cast_fp16 = conv(dilations = var_5382_dilations_0, groups = var_5382_groups_0, pad = var_5382_pad_0, pad_type = var_5382_pad_type_0, strides = var_5382_strides_0, weight = layers_13_mlp_up_proj_weight_to_fp16, x = var_5359_cast_fp16_0)[name = string("op_5382_cast_fp16")]; tensor x_139_cast_fp16 = mul(x = var_5376_cast_fp16, y = var_5382_cast_fp16)[name = string("x_139_cast_fp16")]; tensor hidden_states_137_strides_0 = const()[name = string("hidden_states_137_strides_0"), val = tensor([1, 1])]; string hidden_states_137_pad_type_0 = const()[name = string("hidden_states_137_pad_type_0"), val = string("valid")]; tensor hidden_states_137_pad_0 = const()[name = string("hidden_states_137_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_137_dilations_0 = const()[name = string("hidden_states_137_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_137_groups_0 = const()[name = string("hidden_states_137_groups_0"), val = int32(1)]; tensor hidden_states_137_cast_fp16 = conv(dilations = hidden_states_137_dilations_0, groups = hidden_states_137_groups_0, pad = hidden_states_137_pad_0, pad_type = hidden_states_137_pad_type_0, strides = hidden_states_137_strides_0, weight = layers_13_mlp_down_proj_weight_cast_fp16, x = x_139_cast_fp16)[name = string("hidden_states_137_cast_fp16")]; tensor hidden_states_139_cast_fp16 = add(x = hidden_states_135_cast_fp16, y = hidden_states_137_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; fp16 const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5400_cast_fp16 = mul(x = hidden_states_139_cast_fp16, y = const_140_promoted_to_fp16)[name = string("op_5400_cast_fp16")]; int32 var_5398 = const()[name = string("op_5398"), val = int32(1)]; bool doubled_113_interleave_0 = const()[name = string("doubled_113_interleave_0"), val = bool(false)]; tensor doubled_113_cast_fp16 = concat(axis = var_5398, interleave = doubled_113_interleave_0, values = (hidden_states_139_cast_fp16, var_5400_cast_fp16))[name = string("doubled_113_cast_fp16")]; tensor out_57_axes_0 = const()[name = string("out_57_axes_0"), val = tensor([1])]; tensor out_57_gamma_0_to_fp16 = const()[name = string("out_57_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440573568)))]; fp16 var_5410_to_fp16 = const()[name = string("op_5410_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_57_cast_fp16 = layer_norm(axes = out_57_axes_0, epsilon = var_5410_to_fp16, gamma = out_57_gamma_0_to_fp16, x = doubled_113_cast_fp16)[name = string("out_57_cast_fp16")]; tensor var_5421_split_sizes_0 = const()[name = string("op_5421_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5421_axis_0 = const()[name = string("op_5421_axis_0"), val = int32(1)]; tensor var_5421_cast_fp16_0, tensor var_5421_cast_fp16_1 = split(axis = var_5421_axis_0, split_sizes = var_5421_split_sizes_0, x = out_57_cast_fp16)[name = string("op_5421_cast_fp16")]; tensor query_states_85_strides_0 = const()[name = string("query_states_85_strides_0"), val = tensor([1, 1])]; string query_states_85_pad_type_0 = const()[name = string("query_states_85_pad_type_0"), val = string("valid")]; tensor query_states_85_pad_0 = const()[name = string("query_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_85_dilations_0 = const()[name = string("query_states_85_dilations_0"), val = tensor([1, 1])]; int32 query_states_85_groups_0 = const()[name = string("query_states_85_groups_0"), val = int32(1)]; tensor query_states_85_cast_fp16 = conv(dilations = query_states_85_dilations_0, groups = query_states_85_groups_0, pad = query_states_85_pad_0, pad_type = query_states_85_pad_type_0, strides = query_states_85_strides_0, weight = layers_14_self_attn_q_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("query_states_85_cast_fp16")]; tensor layers_14_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_14_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440581824)))]; tensor key_states_141_strides_0 = const()[name = string("key_states_141_strides_0"), val = tensor([1, 1])]; string key_states_141_pad_type_0 = const()[name = string("key_states_141_pad_type_0"), val = string("valid")]; tensor key_states_141_pad_0 = const()[name = string("key_states_141_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_141_dilations_0 = const()[name = string("key_states_141_dilations_0"), val = tensor([1, 1])]; int32 key_states_141_groups_0 = const()[name = string("key_states_141_groups_0"), val = int32(1)]; tensor key_states_141_cast_fp16 = conv(dilations = key_states_141_dilations_0, groups = key_states_141_groups_0, pad = key_states_141_pad_0, pad_type = key_states_141_pad_type_0, strides = key_states_141_strides_0, weight = layers_14_self_attn_k_proj_weight_to_fp16, x = var_5421_cast_fp16_0)[name = string("key_states_141_cast_fp16")]; tensor value_states_85_strides_0 = const()[name = string("value_states_85_strides_0"), val = tensor([1, 1])]; string value_states_85_pad_type_0 = const()[name = string("value_states_85_pad_type_0"), val = string("valid")]; tensor value_states_85_pad_0 = const()[name = string("value_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_85_dilations_0 = const()[name = string("value_states_85_dilations_0"), val = tensor([1, 1])]; int32 value_states_85_groups_0 = const()[name = string("value_states_85_groups_0"), val = int32(1)]; tensor value_states_85_cast_fp16 = conv(dilations = value_states_85_dilations_0, groups = value_states_85_groups_0, pad = value_states_85_pad_0, pad_type = value_states_85_pad_type_0, strides = value_states_85_strides_0, weight = layers_14_self_attn_v_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("value_states_85_cast_fp16")]; tensor concat_168x = const()[name = string("concat_168x"), val = tensor([1, 16, 128, -1])]; tensor x_141_cast_fp16 = reshape(shape = concat_168x, x = query_states_85_cast_fp16)[name = string("x_141_cast_fp16")]; tensor concat_169x = const()[name = string("concat_169x"), val = tensor([1, 2, 128, -1])]; tensor var_5478_cast_fp16 = reshape(shape = concat_169x, x = key_states_141_cast_fp16)[name = string("op_5478_cast_fp16")]; tensor concat_170x = const()[name = string("concat_170x"), val = tensor([1, 2, 128, -1])]; tensor var_5485_cast_fp16 = reshape(shape = concat_170x, x = value_states_85_cast_fp16)[name = string("op_5485_cast_fp16")]; tensor var_5489_cast_fp16 = mul(x = x_141_cast_fp16, y = var_869_cast_fp16)[name = string("op_5489_cast_fp16")]; tensor var_5490_split_sizes_0 = const()[name = string("op_5490_split_sizes_0"), val = tensor([64, 64])]; int32 var_5490_axis_0 = const()[name = string("op_5490_axis_0"), val = int32(-2)]; tensor var_5490_cast_fp16_0, tensor var_5490_cast_fp16_1 = split(axis = var_5490_axis_0, split_sizes = var_5490_split_sizes_0, x = x_141_cast_fp16)[name = string("op_5490_cast_fp16")]; fp16 const_142_promoted_to_fp16 = const()[name = string("const_142_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5492_cast_fp16 = mul(x = var_5490_cast_fp16_1, y = const_142_promoted_to_fp16)[name = string("op_5492_cast_fp16")]; int32 var_5494 = const()[name = string("op_5494"), val = int32(-2)]; bool var_5495_interleave_0 = const()[name = string("op_5495_interleave_0"), val = bool(false)]; tensor var_5495_cast_fp16 = concat(axis = var_5494, interleave = var_5495_interleave_0, values = (var_5492_cast_fp16, var_5490_cast_fp16_0))[name = string("op_5495_cast_fp16")]; tensor var_5496_cast_fp16 = mul(x = var_5495_cast_fp16, y = var_878_cast_fp16)[name = string("op_5496_cast_fp16")]; tensor query_states_87_cast_fp16 = add(x = var_5489_cast_fp16, y = var_5496_cast_fp16)[name = string("query_states_87_cast_fp16")]; tensor var_5502_cast_fp16 = mul(x = var_5478_cast_fp16, y = var_869_cast_fp16)[name = string("op_5502_cast_fp16")]; tensor var_5503_split_sizes_0 = const()[name = string("op_5503_split_sizes_0"), val = tensor([64, 64])]; int32 var_5503_axis_0 = const()[name = string("op_5503_axis_0"), val = int32(-2)]; tensor var_5503_cast_fp16_0, tensor var_5503_cast_fp16_1 = split(axis = var_5503_axis_0, split_sizes = var_5503_split_sizes_0, x = var_5478_cast_fp16)[name = string("op_5503_cast_fp16")]; fp16 const_143_promoted_to_fp16 = const()[name = string("const_143_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5505_cast_fp16 = mul(x = var_5503_cast_fp16_1, y = const_143_promoted_to_fp16)[name = string("op_5505_cast_fp16")]; int32 var_5507 = const()[name = string("op_5507"), val = int32(-2)]; bool var_5508_interleave_0 = const()[name = string("op_5508_interleave_0"), val = bool(false)]; tensor var_5508_cast_fp16 = concat(axis = var_5507, interleave = var_5508_interleave_0, values = (var_5505_cast_fp16, var_5503_cast_fp16_0))[name = string("op_5508_cast_fp16")]; tensor var_5509_cast_fp16 = mul(x = var_5508_cast_fp16, y = var_878_cast_fp16)[name = string("op_5509_cast_fp16")]; tensor key_states_145_cast_fp16 = add(x = var_5502_cast_fp16, y = var_5509_cast_fp16)[name = string("key_states_145_cast_fp16")]; tensor expand_dims_168 = const()[name = string("expand_dims_168"), val = tensor([14])]; tensor expand_dims_169 = const()[name = string("expand_dims_169"), val = tensor([0])]; tensor expand_dims_171 = const()[name = string("expand_dims_171"), val = tensor([0])]; int32 concat_173_axis_0 = const()[name = string("concat_173_axis_0"), val = int32(0)]; bool concat_173_interleave_0 = const()[name = string("concat_173_interleave_0"), val = bool(false)]; tensor concat_173 = concat(axis = concat_173_axis_0, interleave = concat_173_interleave_0, values = (expand_dims_168, expand_dims_169, position_id, expand_dims_171))[name = string("concat_173")]; tensor expand_dims_172 = const()[name = string("expand_dims_172"), val = tensor([15])]; tensor concat_174_values1_0 = const()[name = string("concat_174_values1_0"), val = tensor([0])]; tensor concat_174_values3_0 = const()[name = string("concat_174_values3_0"), val = tensor([0])]; int32 concat_174_axis_0 = const()[name = string("concat_174_axis_0"), val = int32(0)]; bool concat_174_interleave_0 = const()[name = string("concat_174_interleave_0"), val = bool(false)]; tensor concat_174 = concat(axis = concat_174_axis_0, interleave = concat_174_interleave_0, values = (expand_dims_172, concat_174_values1_0, cache_position_end, concat_174_values3_0))[name = string("concat_174")]; tensor key_states_147_perm_0 = const()[name = string("key_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_15_stride_0 = const()[name = string("key_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_147_cast_fp16 = transpose(perm = key_states_147_perm_0, x = key_states_145_cast_fp16)[name = string("transpose_127")]; tensor key_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = key_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = key_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_15_squeeze_mask_0, stride = key_cache_internal_tensor_assign_15_stride_0, update = key_states_147_cast_fp16, x = coreml_update_state_82)[name = string("key_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_15_cast_fp16, input = key_cache)[name = string("coreml_update_state_84_write_state")]; tensor coreml_update_state_84 = read_state(input = key_cache)[name = string("coreml_update_state_84")]; tensor value_states_87_perm_0 = const()[name = string("value_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_15_stride_0 = const()[name = string("value_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_87_cast_fp16 = transpose(perm = value_states_87_perm_0, x = var_5485_cast_fp16)[name = string("transpose_126")]; tensor value_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = value_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = value_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_15_squeeze_mask_0, stride = value_cache_internal_tensor_assign_15_stride_0, update = value_states_87_cast_fp16, x = coreml_update_state_83)[name = string("value_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_15_cast_fp16, input = value_cache)[name = string("coreml_update_state_85_write_state")]; tensor coreml_update_state_85 = read_state(input = value_cache)[name = string("coreml_update_state_85")]; tensor var_5579_begin_0 = const()[name = string("op_5579_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5579_end_0 = const()[name = string("op_5579_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5579_end_mask_0 = const()[name = string("op_5579_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5579_cast_fp16 = slice_by_index(begin = var_5579_begin_0, end = var_5579_end_0, end_mask = var_5579_end_mask_0, x = coreml_update_state_84)[name = string("op_5579_cast_fp16")]; tensor tile_28 = const()[name = string("tile_28"), val = tensor([1, 1])]; int32 var_5582_axis_0 = const()[name = string("op_5582_axis_0"), val = int32(1)]; tensor var_5582_cast_fp16_0, tensor var_5582_cast_fp16_1 = split(axis = var_5582_axis_0, split_sizes = tile_28, x = var_5579_cast_fp16)[name = string("op_5582_cast_fp16")]; tensor var_5589_begin_0 = const()[name = string("op_5589_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5589_end_0 = const()[name = string("op_5589_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5589_end_mask_0 = const()[name = string("op_5589_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5589_cast_fp16 = slice_by_index(begin = var_5589_begin_0, end = var_5589_end_0, end_mask = var_5589_end_mask_0, x = coreml_update_state_85)[name = string("op_5589_cast_fp16")]; tensor tile_29 = const()[name = string("tile_29"), val = tensor([1, 1])]; int32 var_5592_axis_0 = const()[name = string("op_5592_axis_0"), val = int32(1)]; tensor var_5592_cast_fp16_0, tensor var_5592_cast_fp16_1 = split(axis = var_5592_axis_0, split_sizes = tile_29, x = var_5589_cast_fp16)[name = string("op_5592_cast_fp16")]; tensor var_5595_split_sizes_0 = const()[name = string("op_5595_split_sizes_0"), val = tensor([8, 8])]; int32 var_5595_axis_0 = const()[name = string("op_5595_axis_0"), val = int32(1)]; tensor var_5595_0, tensor var_5595_1 = split(axis = var_5595_axis_0, split_sizes = var_5595_split_sizes_0, x = query_states_87_cast_fp16)[name = string("op_5595")]; bool attn_weights_225_transpose_x_0 = const()[name = string("attn_weights_225_transpose_x_0"), val = bool(false)]; bool attn_weights_225_transpose_y_0 = const()[name = string("attn_weights_225_transpose_y_0"), val = bool(false)]; tensor attn_weights_225_cast_fp16 = matmul(transpose_x = attn_weights_225_transpose_x_0, transpose_y = attn_weights_225_transpose_y_0, x = var_5582_cast_fp16_0, y = var_5595_0)[name = string("attn_weights_225_cast_fp16")]; fp16 var_5598_to_fp16 = const()[name = string("op_5598_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_227_cast_fp16 = mul(x = attn_weights_225_cast_fp16, y = var_5598_to_fp16)[name = string("attn_weights_227_cast_fp16")]; tensor attn_weights_229_cast_fp16 = add(x = attn_weights_227_cast_fp16, y = attn_mask_1)[name = string("attn_weights_229_cast_fp16")]; int32 var_5602 = const()[name = string("op_5602"), val = int32(-2)]; tensor attn_weights_231_cast_fp16 = softmax(axis = var_5602, x = attn_weights_229_cast_fp16)[name = string("attn_weights_231_cast_fp16")]; bool var_5608_transpose_x_1 = const()[name = string("op_5608_transpose_x_1"), val = bool(true)]; bool var_5608_transpose_y_1 = const()[name = string("op_5608_transpose_y_1"), val = bool(false)]; tensor var_5608_cast_fp16 = matmul(transpose_x = var_5608_transpose_x_1, transpose_y = var_5608_transpose_y_1, x = attn_weights_231_cast_fp16, y = var_5592_cast_fp16_0)[name = string("op_5608_cast_fp16")]; bool attn_weights_233_transpose_x_0 = const()[name = string("attn_weights_233_transpose_x_0"), val = bool(false)]; bool attn_weights_233_transpose_y_0 = const()[name = string("attn_weights_233_transpose_y_0"), val = bool(false)]; tensor attn_weights_233_cast_fp16 = matmul(transpose_x = attn_weights_233_transpose_x_0, transpose_y = attn_weights_233_transpose_y_0, x = var_5582_cast_fp16_1, y = var_5595_1)[name = string("attn_weights_233_cast_fp16")]; fp16 var_5610_to_fp16 = const()[name = string("op_5610_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_235_cast_fp16 = mul(x = attn_weights_233_cast_fp16, y = var_5610_to_fp16)[name = string("attn_weights_235_cast_fp16")]; tensor attn_weights_237_cast_fp16 = add(x = attn_weights_235_cast_fp16, y = attn_mask_1)[name = string("attn_weights_237_cast_fp16")]; int32 var_5614 = const()[name = string("op_5614"), val = int32(-2)]; tensor attn_weights_239_cast_fp16 = softmax(axis = var_5614, x = attn_weights_237_cast_fp16)[name = string("attn_weights_239_cast_fp16")]; bool attn_output_113_transpose_x_1 = const()[name = string("attn_output_113_transpose_x_1"), val = bool(true)]; bool attn_output_113_transpose_y_1 = const()[name = string("attn_output_113_transpose_y_1"), val = bool(false)]; tensor attn_output_113_cast_fp16 = matmul(transpose_x = attn_output_113_transpose_x_1, transpose_y = attn_output_113_transpose_y_1, x = attn_weights_239_cast_fp16, y = var_5592_cast_fp16_1)[name = string("attn_output_113_cast_fp16")]; int32 var_5622 = const()[name = string("op_5622"), val = int32(1)]; bool attn_output_115_interleave_0 = const()[name = string("attn_output_115_interleave_0"), val = bool(false)]; tensor attn_output_115_cast_fp16 = concat(axis = var_5622, interleave = attn_output_115_interleave_0, values = (var_5608_cast_fp16, attn_output_113_cast_fp16))[name = string("attn_output_115_cast_fp16")]; tensor var_5626_perm_0 = const()[name = string("op_5626_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_179x = const()[name = string("concat_179x"), val = tensor([1, 2048, 1, -1])]; tensor var_5626_cast_fp16 = transpose(perm = var_5626_perm_0, x = attn_output_115_cast_fp16)[name = string("transpose_125")]; tensor attn_output_119_cast_fp16 = reshape(shape = concat_179x, x = var_5626_cast_fp16)[name = string("attn_output_119_cast_fp16")]; tensor hidden_states_143_strides_0 = const()[name = string("hidden_states_143_strides_0"), val = tensor([1, 1])]; string hidden_states_143_pad_type_0 = const()[name = string("hidden_states_143_pad_type_0"), val = string("valid")]; tensor hidden_states_143_pad_0 = const()[name = string("hidden_states_143_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_143_dilations_0 = const()[name = string("hidden_states_143_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_143_groups_0 = const()[name = string("hidden_states_143_groups_0"), val = int32(1)]; tensor hidden_states_143_cast_fp16 = conv(dilations = hidden_states_143_dilations_0, groups = hidden_states_143_groups_0, pad = hidden_states_143_pad_0, pad_type = hidden_states_143_pad_type_0, strides = hidden_states_143_strides_0, weight = layers_14_self_attn_o_proj_weight_cast_fp16, x = attn_output_119_cast_fp16)[name = string("hidden_states_143_cast_fp16")]; tensor hidden_states_145_cast_fp16 = add(x = hidden_states_139_cast_fp16, y = hidden_states_143_cast_fp16)[name = string("hidden_states_145_cast_fp16")]; fp16 const_148_promoted_to_fp16 = const()[name = string("const_148_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5659_cast_fp16 = mul(x = hidden_states_145_cast_fp16, y = const_148_promoted_to_fp16)[name = string("op_5659_cast_fp16")]; int32 var_5657 = const()[name = string("op_5657"), val = int32(1)]; bool doubled_117_interleave_0 = const()[name = string("doubled_117_interleave_0"), val = bool(false)]; tensor doubled_117_cast_fp16 = concat(axis = var_5657, interleave = doubled_117_interleave_0, values = (hidden_states_145_cast_fp16, var_5659_cast_fp16))[name = string("doubled_117_cast_fp16")]; tensor out_59_axes_0 = const()[name = string("out_59_axes_0"), val = tensor([1])]; tensor out_59_gamma_0_to_fp16 = const()[name = string("out_59_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441630464)))]; fp16 var_5669_to_fp16 = const()[name = string("op_5669_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_59_cast_fp16 = layer_norm(axes = out_59_axes_0, epsilon = var_5669_to_fp16, gamma = out_59_gamma_0_to_fp16, x = doubled_117_cast_fp16)[name = string("out_59_cast_fp16")]; tensor var_5680_split_sizes_0 = const()[name = string("op_5680_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5680_axis_0 = const()[name = string("op_5680_axis_0"), val = int32(1)]; tensor var_5680_cast_fp16_0, tensor var_5680_cast_fp16_1 = split(axis = var_5680_axis_0, split_sizes = var_5680_split_sizes_0, x = out_59_cast_fp16)[name = string("op_5680_cast_fp16")]; tensor input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor([1, 1])]; string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; tensor input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor([1, 1])]; int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; tensor input_29_cast_fp16 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_14_mlp_gate_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("input_29_cast_fp16")]; tensor var_5697_cast_fp16 = silu(x = input_29_cast_fp16)[name = string("op_5697_cast_fp16")]; tensor var_5703_strides_0 = const()[name = string("op_5703_strides_0"), val = tensor([1, 1])]; string var_5703_pad_type_0 = const()[name = string("op_5703_pad_type_0"), val = string("valid")]; tensor var_5703_pad_0 = const()[name = string("op_5703_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5703_dilations_0 = const()[name = string("op_5703_dilations_0"), val = tensor([1, 1])]; int32 var_5703_groups_0 = const()[name = string("op_5703_groups_0"), val = int32(1)]; tensor var_5703_cast_fp16 = conv(dilations = var_5703_dilations_0, groups = var_5703_groups_0, pad = var_5703_pad_0, pad_type = var_5703_pad_type_0, strides = var_5703_strides_0, weight = layers_14_mlp_up_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("op_5703_cast_fp16")]; tensor x_149_cast_fp16 = mul(x = var_5697_cast_fp16, y = var_5703_cast_fp16)[name = string("x_149_cast_fp16")]; tensor hidden_states_147_strides_0 = const()[name = string("hidden_states_147_strides_0"), val = tensor([1, 1])]; string hidden_states_147_pad_type_0 = const()[name = string("hidden_states_147_pad_type_0"), val = string("valid")]; tensor hidden_states_147_pad_0 = const()[name = string("hidden_states_147_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_147_dilations_0 = const()[name = string("hidden_states_147_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_147_groups_0 = const()[name = string("hidden_states_147_groups_0"), val = int32(1)]; tensor hidden_states_147_cast_fp16 = conv(dilations = hidden_states_147_dilations_0, groups = hidden_states_147_groups_0, pad = hidden_states_147_pad_0, pad_type = hidden_states_147_pad_type_0, strides = hidden_states_147_strides_0, weight = layers_14_mlp_down_proj_weight_cast_fp16, x = x_149_cast_fp16)[name = string("hidden_states_147_cast_fp16")]; tensor hidden_states_149_cast_fp16 = add(x = hidden_states_145_cast_fp16, y = hidden_states_147_cast_fp16)[name = string("hidden_states_149_cast_fp16")]; fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5721_cast_fp16 = mul(x = hidden_states_149_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_5721_cast_fp16")]; int32 var_5719 = const()[name = string("op_5719"), val = int32(1)]; bool doubled_121_interleave_0 = const()[name = string("doubled_121_interleave_0"), val = bool(false)]; tensor doubled_121_cast_fp16 = concat(axis = var_5719, interleave = doubled_121_interleave_0, values = (hidden_states_149_cast_fp16, var_5721_cast_fp16))[name = string("doubled_121_cast_fp16")]; tensor out_61_axes_0 = const()[name = string("out_61_axes_0"), val = tensor([1])]; tensor out_61_gamma_0_to_fp16 = const()[name = string("out_61_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441638720)))]; fp16 var_5731_to_fp16 = const()[name = string("op_5731_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_61_cast_fp16 = layer_norm(axes = out_61_axes_0, epsilon = var_5731_to_fp16, gamma = out_61_gamma_0_to_fp16, x = doubled_121_cast_fp16)[name = string("out_61_cast_fp16")]; tensor var_5742_split_sizes_0 = const()[name = string("op_5742_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5742_axis_0 = const()[name = string("op_5742_axis_0"), val = int32(1)]; tensor var_5742_cast_fp16_0, tensor var_5742_cast_fp16_1 = split(axis = var_5742_axis_0, split_sizes = var_5742_split_sizes_0, x = out_61_cast_fp16)[name = string("op_5742_cast_fp16")]; tensor query_states_91_strides_0 = const()[name = string("query_states_91_strides_0"), val = tensor([1, 1])]; string query_states_91_pad_type_0 = const()[name = string("query_states_91_pad_type_0"), val = string("valid")]; tensor query_states_91_pad_0 = const()[name = string("query_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_91_dilations_0 = const()[name = string("query_states_91_dilations_0"), val = tensor([1, 1])]; int32 query_states_91_groups_0 = const()[name = string("query_states_91_groups_0"), val = int32(1)]; tensor query_states_91_cast_fp16 = conv(dilations = query_states_91_dilations_0, groups = query_states_91_groups_0, pad = query_states_91_pad_0, pad_type = query_states_91_pad_type_0, strides = query_states_91_strides_0, weight = layers_15_self_attn_q_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("query_states_91_cast_fp16")]; tensor key_states_151_strides_0 = const()[name = string("key_states_151_strides_0"), val = tensor([1, 1])]; string key_states_151_pad_type_0 = const()[name = string("key_states_151_pad_type_0"), val = string("valid")]; tensor key_states_151_pad_0 = const()[name = string("key_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_151_dilations_0 = const()[name = string("key_states_151_dilations_0"), val = tensor([1, 1])]; int32 key_states_151_groups_0 = const()[name = string("key_states_151_groups_0"), val = int32(1)]; tensor key_states_151_cast_fp16 = conv(dilations = key_states_151_dilations_0, groups = key_states_151_groups_0, pad = key_states_151_pad_0, pad_type = key_states_151_pad_type_0, strides = key_states_151_strides_0, weight = layers_15_self_attn_k_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("key_states_151_cast_fp16")]; tensor value_states_91_strides_0 = const()[name = string("value_states_91_strides_0"), val = tensor([1, 1])]; string value_states_91_pad_type_0 = const()[name = string("value_states_91_pad_type_0"), val = string("valid")]; tensor value_states_91_pad_0 = const()[name = string("value_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_91_dilations_0 = const()[name = string("value_states_91_dilations_0"), val = tensor([1, 1])]; int32 value_states_91_groups_0 = const()[name = string("value_states_91_groups_0"), val = int32(1)]; tensor value_states_91_cast_fp16 = conv(dilations = value_states_91_dilations_0, groups = value_states_91_groups_0, pad = value_states_91_pad_0, pad_type = value_states_91_pad_type_0, strides = value_states_91_strides_0, weight = layers_15_self_attn_v_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("value_states_91_cast_fp16")]; tensor concat_180x = const()[name = string("concat_180x"), val = tensor([1, 16, 128, -1])]; tensor x_151_cast_fp16 = reshape(shape = concat_180x, x = query_states_91_cast_fp16)[name = string("x_151_cast_fp16")]; tensor concat_181x = const()[name = string("concat_181x"), val = tensor([1, 2, 128, -1])]; tensor var_5799_cast_fp16 = reshape(shape = concat_181x, x = key_states_151_cast_fp16)[name = string("op_5799_cast_fp16")]; tensor concat_182x = const()[name = string("concat_182x"), val = tensor([1, 2, 128, -1])]; tensor var_5806_cast_fp16 = reshape(shape = concat_182x, x = value_states_91_cast_fp16)[name = string("op_5806_cast_fp16")]; tensor var_5810_cast_fp16 = mul(x = x_151_cast_fp16, y = var_869_cast_fp16)[name = string("op_5810_cast_fp16")]; tensor var_5811_split_sizes_0 = const()[name = string("op_5811_split_sizes_0"), val = tensor([64, 64])]; int32 var_5811_axis_0 = const()[name = string("op_5811_axis_0"), val = int32(-2)]; tensor var_5811_cast_fp16_0, tensor var_5811_cast_fp16_1 = split(axis = var_5811_axis_0, split_sizes = var_5811_split_sizes_0, x = x_151_cast_fp16)[name = string("op_5811_cast_fp16")]; fp16 const_152_promoted_to_fp16 = const()[name = string("const_152_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5813_cast_fp16 = mul(x = var_5811_cast_fp16_1, y = const_152_promoted_to_fp16)[name = string("op_5813_cast_fp16")]; int32 var_5815 = const()[name = string("op_5815"), val = int32(-2)]; bool var_5816_interleave_0 = const()[name = string("op_5816_interleave_0"), val = bool(false)]; tensor var_5816_cast_fp16 = concat(axis = var_5815, interleave = var_5816_interleave_0, values = (var_5813_cast_fp16, var_5811_cast_fp16_0))[name = string("op_5816_cast_fp16")]; tensor var_5817_cast_fp16 = mul(x = var_5816_cast_fp16, y = var_878_cast_fp16)[name = string("op_5817_cast_fp16")]; tensor query_states_93_cast_fp16 = add(x = var_5810_cast_fp16, y = var_5817_cast_fp16)[name = string("query_states_93_cast_fp16")]; tensor var_5823_cast_fp16 = mul(x = var_5799_cast_fp16, y = var_869_cast_fp16)[name = string("op_5823_cast_fp16")]; tensor var_5824_split_sizes_0 = const()[name = string("op_5824_split_sizes_0"), val = tensor([64, 64])]; int32 var_5824_axis_0 = const()[name = string("op_5824_axis_0"), val = int32(-2)]; tensor var_5824_cast_fp16_0, tensor var_5824_cast_fp16_1 = split(axis = var_5824_axis_0, split_sizes = var_5824_split_sizes_0, x = var_5799_cast_fp16)[name = string("op_5824_cast_fp16")]; fp16 const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5826_cast_fp16 = mul(x = var_5824_cast_fp16_1, y = const_153_promoted_to_fp16)[name = string("op_5826_cast_fp16")]; int32 var_5828 = const()[name = string("op_5828"), val = int32(-2)]; bool var_5829_interleave_0 = const()[name = string("op_5829_interleave_0"), val = bool(false)]; tensor var_5829_cast_fp16 = concat(axis = var_5828, interleave = var_5829_interleave_0, values = (var_5826_cast_fp16, var_5824_cast_fp16_0))[name = string("op_5829_cast_fp16")]; tensor var_5830_cast_fp16 = mul(x = var_5829_cast_fp16, y = var_878_cast_fp16)[name = string("op_5830_cast_fp16")]; tensor key_states_155_cast_fp16 = add(x = var_5823_cast_fp16, y = var_5830_cast_fp16)[name = string("key_states_155_cast_fp16")]; tensor expand_dims_180 = const()[name = string("expand_dims_180"), val = tensor([15])]; tensor expand_dims_181 = const()[name = string("expand_dims_181"), val = tensor([0])]; tensor expand_dims_183 = const()[name = string("expand_dims_183"), val = tensor([0])]; int32 concat_185_axis_0 = const()[name = string("concat_185_axis_0"), val = int32(0)]; bool concat_185_interleave_0 = const()[name = string("concat_185_interleave_0"), val = bool(false)]; tensor concat_185 = concat(axis = concat_185_axis_0, interleave = concat_185_interleave_0, values = (expand_dims_180, expand_dims_181, position_id, expand_dims_183))[name = string("concat_185")]; tensor expand_dims_184 = const()[name = string("expand_dims_184"), val = tensor([16])]; tensor concat_186_values1_0 = const()[name = string("concat_186_values1_0"), val = tensor([0])]; tensor concat_186_values3_0 = const()[name = string("concat_186_values3_0"), val = tensor([0])]; int32 concat_186_axis_0 = const()[name = string("concat_186_axis_0"), val = int32(0)]; bool concat_186_interleave_0 = const()[name = string("concat_186_interleave_0"), val = bool(false)]; tensor concat_186 = concat(axis = concat_186_axis_0, interleave = concat_186_interleave_0, values = (expand_dims_184, concat_186_values1_0, cache_position_end, concat_186_values3_0))[name = string("concat_186")]; tensor key_states_157_perm_0 = const()[name = string("key_states_157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_16_stride_0 = const()[name = string("key_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_157_cast_fp16 = transpose(perm = key_states_157_perm_0, x = key_states_155_cast_fp16)[name = string("transpose_124")]; tensor key_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = key_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = key_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_16_squeeze_mask_0, stride = key_cache_internal_tensor_assign_16_stride_0, update = key_states_157_cast_fp16, x = coreml_update_state_84)[name = string("key_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_16_cast_fp16, input = key_cache)[name = string("coreml_update_state_86_write_state")]; tensor coreml_update_state_86 = read_state(input = key_cache)[name = string("coreml_update_state_86")]; tensor value_states_93_perm_0 = const()[name = string("value_states_93_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_16_stride_0 = const()[name = string("value_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_93_cast_fp16 = transpose(perm = value_states_93_perm_0, x = var_5806_cast_fp16)[name = string("transpose_123")]; tensor value_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = value_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = value_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_16_squeeze_mask_0, stride = value_cache_internal_tensor_assign_16_stride_0, update = value_states_93_cast_fp16, x = coreml_update_state_85)[name = string("value_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_16_cast_fp16, input = value_cache)[name = string("coreml_update_state_87_write_state")]; tensor coreml_update_state_87 = read_state(input = value_cache)[name = string("coreml_update_state_87")]; tensor var_5900_begin_0 = const()[name = string("op_5900_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5900_end_0 = const()[name = string("op_5900_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5900_end_mask_0 = const()[name = string("op_5900_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5900_cast_fp16 = slice_by_index(begin = var_5900_begin_0, end = var_5900_end_0, end_mask = var_5900_end_mask_0, x = coreml_update_state_86)[name = string("op_5900_cast_fp16")]; tensor tile_30 = const()[name = string("tile_30"), val = tensor([1, 1])]; int32 var_5903_axis_0 = const()[name = string("op_5903_axis_0"), val = int32(1)]; tensor var_5903_cast_fp16_0, tensor var_5903_cast_fp16_1 = split(axis = var_5903_axis_0, split_sizes = tile_30, x = var_5900_cast_fp16)[name = string("op_5903_cast_fp16")]; tensor var_5910_begin_0 = const()[name = string("op_5910_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5910_end_0 = const()[name = string("op_5910_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5910_end_mask_0 = const()[name = string("op_5910_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5910_cast_fp16 = slice_by_index(begin = var_5910_begin_0, end = var_5910_end_0, end_mask = var_5910_end_mask_0, x = coreml_update_state_87)[name = string("op_5910_cast_fp16")]; tensor tile_31 = const()[name = string("tile_31"), val = tensor([1, 1])]; int32 var_5913_axis_0 = const()[name = string("op_5913_axis_0"), val = int32(1)]; tensor var_5913_cast_fp16_0, tensor var_5913_cast_fp16_1 = split(axis = var_5913_axis_0, split_sizes = tile_31, x = var_5910_cast_fp16)[name = string("op_5913_cast_fp16")]; tensor var_5916_split_sizes_0 = const()[name = string("op_5916_split_sizes_0"), val = tensor([8, 8])]; int32 var_5916_axis_0 = const()[name = string("op_5916_axis_0"), val = int32(1)]; tensor var_5916_0, tensor var_5916_1 = split(axis = var_5916_axis_0, split_sizes = var_5916_split_sizes_0, x = query_states_93_cast_fp16)[name = string("op_5916")]; bool attn_weights_241_transpose_x_0 = const()[name = string("attn_weights_241_transpose_x_0"), val = bool(false)]; bool attn_weights_241_transpose_y_0 = const()[name = string("attn_weights_241_transpose_y_0"), val = bool(false)]; tensor attn_weights_241_cast_fp16 = matmul(transpose_x = attn_weights_241_transpose_x_0, transpose_y = attn_weights_241_transpose_y_0, x = var_5903_cast_fp16_0, y = var_5916_0)[name = string("attn_weights_241_cast_fp16")]; fp16 var_5919_to_fp16 = const()[name = string("op_5919_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_243_cast_fp16 = mul(x = attn_weights_241_cast_fp16, y = var_5919_to_fp16)[name = string("attn_weights_243_cast_fp16")]; tensor attn_weights_245_cast_fp16 = add(x = attn_weights_243_cast_fp16, y = attn_mask_1)[name = string("attn_weights_245_cast_fp16")]; int32 var_5923 = const()[name = string("op_5923"), val = int32(-2)]; tensor attn_weights_247_cast_fp16 = softmax(axis = var_5923, x = attn_weights_245_cast_fp16)[name = string("attn_weights_247_cast_fp16")]; bool var_5929_transpose_x_1 = const()[name = string("op_5929_transpose_x_1"), val = bool(true)]; bool var_5929_transpose_y_1 = const()[name = string("op_5929_transpose_y_1"), val = bool(false)]; tensor var_5929_cast_fp16 = matmul(transpose_x = var_5929_transpose_x_1, transpose_y = var_5929_transpose_y_1, x = attn_weights_247_cast_fp16, y = var_5913_cast_fp16_0)[name = string("op_5929_cast_fp16")]; bool attn_weights_249_transpose_x_0 = const()[name = string("attn_weights_249_transpose_x_0"), val = bool(false)]; bool attn_weights_249_transpose_y_0 = const()[name = string("attn_weights_249_transpose_y_0"), val = bool(false)]; tensor attn_weights_249_cast_fp16 = matmul(transpose_x = attn_weights_249_transpose_x_0, transpose_y = attn_weights_249_transpose_y_0, x = var_5903_cast_fp16_1, y = var_5916_1)[name = string("attn_weights_249_cast_fp16")]; fp16 var_5931_to_fp16 = const()[name = string("op_5931_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_251_cast_fp16 = mul(x = attn_weights_249_cast_fp16, y = var_5931_to_fp16)[name = string("attn_weights_251_cast_fp16")]; tensor attn_weights_253_cast_fp16 = add(x = attn_weights_251_cast_fp16, y = attn_mask_1)[name = string("attn_weights_253_cast_fp16")]; int32 var_5935 = const()[name = string("op_5935"), val = int32(-2)]; tensor attn_weights_255_cast_fp16 = softmax(axis = var_5935, x = attn_weights_253_cast_fp16)[name = string("attn_weights_255_cast_fp16")]; bool attn_output_121_transpose_x_1 = const()[name = string("attn_output_121_transpose_x_1"), val = bool(true)]; bool attn_output_121_transpose_y_1 = const()[name = string("attn_output_121_transpose_y_1"), val = bool(false)]; tensor attn_output_121_cast_fp16 = matmul(transpose_x = attn_output_121_transpose_x_1, transpose_y = attn_output_121_transpose_y_1, x = attn_weights_255_cast_fp16, y = var_5913_cast_fp16_1)[name = string("attn_output_121_cast_fp16")]; int32 var_5943 = const()[name = string("op_5943"), val = int32(1)]; bool attn_output_123_interleave_0 = const()[name = string("attn_output_123_interleave_0"), val = bool(false)]; tensor attn_output_123_cast_fp16 = concat(axis = var_5943, interleave = attn_output_123_interleave_0, values = (var_5929_cast_fp16, attn_output_121_cast_fp16))[name = string("attn_output_123_cast_fp16")]; tensor var_5947_perm_0 = const()[name = string("op_5947_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_191x = const()[name = string("concat_191x"), val = tensor([1, 2048, 1, -1])]; tensor var_5947_cast_fp16 = transpose(perm = var_5947_perm_0, x = attn_output_123_cast_fp16)[name = string("transpose_122")]; tensor attn_output_127_cast_fp16 = reshape(shape = concat_191x, x = var_5947_cast_fp16)[name = string("attn_output_127_cast_fp16")]; tensor hidden_states_153_strides_0 = const()[name = string("hidden_states_153_strides_0"), val = tensor([1, 1])]; string hidden_states_153_pad_type_0 = const()[name = string("hidden_states_153_pad_type_0"), val = string("valid")]; tensor hidden_states_153_pad_0 = const()[name = string("hidden_states_153_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_153_dilations_0 = const()[name = string("hidden_states_153_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_153_groups_0 = const()[name = string("hidden_states_153_groups_0"), val = int32(1)]; tensor hidden_states_153_cast_fp16 = conv(dilations = hidden_states_153_dilations_0, groups = hidden_states_153_groups_0, pad = hidden_states_153_pad_0, pad_type = hidden_states_153_pad_type_0, strides = hidden_states_153_strides_0, weight = layers_15_self_attn_o_proj_weight_cast_fp16, x = attn_output_127_cast_fp16)[name = string("hidden_states_153_cast_fp16")]; tensor hidden_states_155_cast_fp16 = add(x = hidden_states_149_cast_fp16, y = hidden_states_153_cast_fp16)[name = string("hidden_states_155_cast_fp16")]; fp16 const_158_promoted_to_fp16 = const()[name = string("const_158_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5980_cast_fp16 = mul(x = hidden_states_155_cast_fp16, y = const_158_promoted_to_fp16)[name = string("op_5980_cast_fp16")]; int32 var_5978 = const()[name = string("op_5978"), val = int32(1)]; bool doubled_125_interleave_0 = const()[name = string("doubled_125_interleave_0"), val = bool(false)]; tensor doubled_125_cast_fp16 = concat(axis = var_5978, interleave = doubled_125_interleave_0, values = (hidden_states_155_cast_fp16, var_5980_cast_fp16))[name = string("doubled_125_cast_fp16")]; tensor out_63_axes_0 = const()[name = string("out_63_axes_0"), val = tensor([1])]; tensor out_63_gamma_0_to_fp16 = const()[name = string("out_63_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441646976)))]; fp16 var_5990_to_fp16 = const()[name = string("op_5990_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_63_cast_fp16 = layer_norm(axes = out_63_axes_0, epsilon = var_5990_to_fp16, gamma = out_63_gamma_0_to_fp16, x = doubled_125_cast_fp16)[name = string("out_63_cast_fp16")]; tensor var_6001_split_sizes_0 = const()[name = string("op_6001_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6001_axis_0 = const()[name = string("op_6001_axis_0"), val = int32(1)]; tensor var_6001_cast_fp16_0, tensor var_6001_cast_fp16_1 = split(axis = var_6001_axis_0, split_sizes = var_6001_split_sizes_0, x = out_63_cast_fp16)[name = string("op_6001_cast_fp16")]; tensor input_31_strides_0 = const()[name = string("input_31_strides_0"), val = tensor([1, 1])]; string input_31_pad_type_0 = const()[name = string("input_31_pad_type_0"), val = string("valid")]; tensor input_31_pad_0 = const()[name = string("input_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_31_dilations_0 = const()[name = string("input_31_dilations_0"), val = tensor([1, 1])]; int32 input_31_groups_0 = const()[name = string("input_31_groups_0"), val = int32(1)]; tensor input_31_cast_fp16 = conv(dilations = input_31_dilations_0, groups = input_31_groups_0, pad = input_31_pad_0, pad_type = input_31_pad_type_0, strides = input_31_strides_0, weight = layers_15_mlp_gate_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("input_31_cast_fp16")]; tensor var_6018_cast_fp16 = silu(x = input_31_cast_fp16)[name = string("op_6018_cast_fp16")]; tensor var_6024_strides_0 = const()[name = string("op_6024_strides_0"), val = tensor([1, 1])]; string var_6024_pad_type_0 = const()[name = string("op_6024_pad_type_0"), val = string("valid")]; tensor var_6024_pad_0 = const()[name = string("op_6024_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6024_dilations_0 = const()[name = string("op_6024_dilations_0"), val = tensor([1, 1])]; int32 var_6024_groups_0 = const()[name = string("op_6024_groups_0"), val = int32(1)]; tensor var_6024_cast_fp16 = conv(dilations = var_6024_dilations_0, groups = var_6024_groups_0, pad = var_6024_pad_0, pad_type = var_6024_pad_type_0, strides = var_6024_strides_0, weight = layers_15_mlp_up_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("op_6024_cast_fp16")]; tensor x_159_cast_fp16 = mul(x = var_6018_cast_fp16, y = var_6024_cast_fp16)[name = string("x_159_cast_fp16")]; tensor hidden_states_157_strides_0 = const()[name = string("hidden_states_157_strides_0"), val = tensor([1, 1])]; string hidden_states_157_pad_type_0 = const()[name = string("hidden_states_157_pad_type_0"), val = string("valid")]; tensor hidden_states_157_pad_0 = const()[name = string("hidden_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_157_dilations_0 = const()[name = string("hidden_states_157_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_157_groups_0 = const()[name = string("hidden_states_157_groups_0"), val = int32(1)]; tensor hidden_states_157_cast_fp16 = conv(dilations = hidden_states_157_dilations_0, groups = hidden_states_157_groups_0, pad = hidden_states_157_pad_0, pad_type = hidden_states_157_pad_type_0, strides = hidden_states_157_strides_0, weight = layers_15_mlp_down_proj_weight_cast_fp16, x = x_159_cast_fp16)[name = string("hidden_states_157_cast_fp16")]; tensor hidden_states_159_cast_fp16 = add(x = hidden_states_155_cast_fp16, y = hidden_states_157_cast_fp16)[name = string("hidden_states_159_cast_fp16")]; fp16 const_160_promoted_to_fp16 = const()[name = string("const_160_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6042_cast_fp16 = mul(x = hidden_states_159_cast_fp16, y = const_160_promoted_to_fp16)[name = string("op_6042_cast_fp16")]; int32 var_6040 = const()[name = string("op_6040"), val = int32(1)]; bool doubled_129_interleave_0 = const()[name = string("doubled_129_interleave_0"), val = bool(false)]; tensor doubled_129_cast_fp16 = concat(axis = var_6040, interleave = doubled_129_interleave_0, values = (hidden_states_159_cast_fp16, var_6042_cast_fp16))[name = string("doubled_129_cast_fp16")]; tensor out_65_axes_0 = const()[name = string("out_65_axes_0"), val = tensor([1])]; tensor out_65_gamma_0_to_fp16 = const()[name = string("out_65_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441655232)))]; fp16 var_6052_to_fp16 = const()[name = string("op_6052_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_65_cast_fp16 = layer_norm(axes = out_65_axes_0, epsilon = var_6052_to_fp16, gamma = out_65_gamma_0_to_fp16, x = doubled_129_cast_fp16)[name = string("out_65_cast_fp16")]; tensor var_6063_split_sizes_0 = const()[name = string("op_6063_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6063_axis_0 = const()[name = string("op_6063_axis_0"), val = int32(1)]; tensor var_6063_cast_fp16_0, tensor var_6063_cast_fp16_1 = split(axis = var_6063_axis_0, split_sizes = var_6063_split_sizes_0, x = out_65_cast_fp16)[name = string("op_6063_cast_fp16")]; tensor query_states_97_strides_0 = const()[name = string("query_states_97_strides_0"), val = tensor([1, 1])]; string query_states_97_pad_type_0 = const()[name = string("query_states_97_pad_type_0"), val = string("valid")]; tensor query_states_97_pad_0 = const()[name = string("query_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_97_dilations_0 = const()[name = string("query_states_97_dilations_0"), val = tensor([1, 1])]; int32 query_states_97_groups_0 = const()[name = string("query_states_97_groups_0"), val = int32(1)]; tensor query_states_97_cast_fp16 = conv(dilations = query_states_97_dilations_0, groups = query_states_97_groups_0, pad = query_states_97_pad_0, pad_type = query_states_97_pad_type_0, strides = query_states_97_strides_0, weight = layers_16_self_attn_q_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("query_states_97_cast_fp16")]; tensor key_states_161_strides_0 = const()[name = string("key_states_161_strides_0"), val = tensor([1, 1])]; string key_states_161_pad_type_0 = const()[name = string("key_states_161_pad_type_0"), val = string("valid")]; tensor key_states_161_pad_0 = const()[name = string("key_states_161_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_161_dilations_0 = const()[name = string("key_states_161_dilations_0"), val = tensor([1, 1])]; int32 key_states_161_groups_0 = const()[name = string("key_states_161_groups_0"), val = int32(1)]; tensor key_states_161_cast_fp16 = conv(dilations = key_states_161_dilations_0, groups = key_states_161_groups_0, pad = key_states_161_pad_0, pad_type = key_states_161_pad_type_0, strides = key_states_161_strides_0, weight = layers_16_self_attn_k_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("key_states_161_cast_fp16")]; tensor value_states_97_strides_0 = const()[name = string("value_states_97_strides_0"), val = tensor([1, 1])]; string value_states_97_pad_type_0 = const()[name = string("value_states_97_pad_type_0"), val = string("valid")]; tensor value_states_97_pad_0 = const()[name = string("value_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_97_dilations_0 = const()[name = string("value_states_97_dilations_0"), val = tensor([1, 1])]; int32 value_states_97_groups_0 = const()[name = string("value_states_97_groups_0"), val = int32(1)]; tensor value_states_97_cast_fp16 = conv(dilations = value_states_97_dilations_0, groups = value_states_97_groups_0, pad = value_states_97_pad_0, pad_type = value_states_97_pad_type_0, strides = value_states_97_strides_0, weight = layers_16_self_attn_v_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("value_states_97_cast_fp16")]; tensor concat_192x = const()[name = string("concat_192x"), val = tensor([1, 16, 128, -1])]; tensor x_161_cast_fp16 = reshape(shape = concat_192x, x = query_states_97_cast_fp16)[name = string("x_161_cast_fp16")]; tensor concat_193x = const()[name = string("concat_193x"), val = tensor([1, 2, 128, -1])]; tensor var_6120_cast_fp16 = reshape(shape = concat_193x, x = key_states_161_cast_fp16)[name = string("op_6120_cast_fp16")]; tensor concat_194x = const()[name = string("concat_194x"), val = tensor([1, 2, 128, -1])]; tensor var_6127_cast_fp16 = reshape(shape = concat_194x, x = value_states_97_cast_fp16)[name = string("op_6127_cast_fp16")]; tensor var_6131_cast_fp16 = mul(x = x_161_cast_fp16, y = var_869_cast_fp16)[name = string("op_6131_cast_fp16")]; tensor var_6132_split_sizes_0 = const()[name = string("op_6132_split_sizes_0"), val = tensor([64, 64])]; int32 var_6132_axis_0 = const()[name = string("op_6132_axis_0"), val = int32(-2)]; tensor var_6132_cast_fp16_0, tensor var_6132_cast_fp16_1 = split(axis = var_6132_axis_0, split_sizes = var_6132_split_sizes_0, x = x_161_cast_fp16)[name = string("op_6132_cast_fp16")]; fp16 const_162_promoted_to_fp16 = const()[name = string("const_162_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6134_cast_fp16 = mul(x = var_6132_cast_fp16_1, y = const_162_promoted_to_fp16)[name = string("op_6134_cast_fp16")]; int32 var_6136 = const()[name = string("op_6136"), val = int32(-2)]; bool var_6137_interleave_0 = const()[name = string("op_6137_interleave_0"), val = bool(false)]; tensor var_6137_cast_fp16 = concat(axis = var_6136, interleave = var_6137_interleave_0, values = (var_6134_cast_fp16, var_6132_cast_fp16_0))[name = string("op_6137_cast_fp16")]; tensor var_6138_cast_fp16 = mul(x = var_6137_cast_fp16, y = var_878_cast_fp16)[name = string("op_6138_cast_fp16")]; tensor query_states_99_cast_fp16 = add(x = var_6131_cast_fp16, y = var_6138_cast_fp16)[name = string("query_states_99_cast_fp16")]; tensor var_6144_cast_fp16 = mul(x = var_6120_cast_fp16, y = var_869_cast_fp16)[name = string("op_6144_cast_fp16")]; tensor var_6145_split_sizes_0 = const()[name = string("op_6145_split_sizes_0"), val = tensor([64, 64])]; int32 var_6145_axis_0 = const()[name = string("op_6145_axis_0"), val = int32(-2)]; tensor var_6145_cast_fp16_0, tensor var_6145_cast_fp16_1 = split(axis = var_6145_axis_0, split_sizes = var_6145_split_sizes_0, x = var_6120_cast_fp16)[name = string("op_6145_cast_fp16")]; fp16 const_163_promoted_to_fp16 = const()[name = string("const_163_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6147_cast_fp16 = mul(x = var_6145_cast_fp16_1, y = const_163_promoted_to_fp16)[name = string("op_6147_cast_fp16")]; int32 var_6149 = const()[name = string("op_6149"), val = int32(-2)]; bool var_6150_interleave_0 = const()[name = string("op_6150_interleave_0"), val = bool(false)]; tensor var_6150_cast_fp16 = concat(axis = var_6149, interleave = var_6150_interleave_0, values = (var_6147_cast_fp16, var_6145_cast_fp16_0))[name = string("op_6150_cast_fp16")]; tensor var_6151_cast_fp16 = mul(x = var_6150_cast_fp16, y = var_878_cast_fp16)[name = string("op_6151_cast_fp16")]; tensor key_states_165_cast_fp16 = add(x = var_6144_cast_fp16, y = var_6151_cast_fp16)[name = string("key_states_165_cast_fp16")]; tensor expand_dims_192 = const()[name = string("expand_dims_192"), val = tensor([16])]; tensor expand_dims_193 = const()[name = string("expand_dims_193"), val = tensor([0])]; tensor expand_dims_195 = const()[name = string("expand_dims_195"), val = tensor([0])]; int32 concat_197_axis_0 = const()[name = string("concat_197_axis_0"), val = int32(0)]; bool concat_197_interleave_0 = const()[name = string("concat_197_interleave_0"), val = bool(false)]; tensor concat_197 = concat(axis = concat_197_axis_0, interleave = concat_197_interleave_0, values = (expand_dims_192, expand_dims_193, position_id, expand_dims_195))[name = string("concat_197")]; tensor expand_dims_196 = const()[name = string("expand_dims_196"), val = tensor([17])]; tensor concat_198_values1_0 = const()[name = string("concat_198_values1_0"), val = tensor([0])]; tensor concat_198_values3_0 = const()[name = string("concat_198_values3_0"), val = tensor([0])]; int32 concat_198_axis_0 = const()[name = string("concat_198_axis_0"), val = int32(0)]; bool concat_198_interleave_0 = const()[name = string("concat_198_interleave_0"), val = bool(false)]; tensor concat_198 = concat(axis = concat_198_axis_0, interleave = concat_198_interleave_0, values = (expand_dims_196, concat_198_values1_0, cache_position_end, concat_198_values3_0))[name = string("concat_198")]; tensor key_states_167_perm_0 = const()[name = string("key_states_167_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_17_stride_0 = const()[name = string("key_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_167_cast_fp16 = transpose(perm = key_states_167_perm_0, x = key_states_165_cast_fp16)[name = string("transpose_121")]; tensor key_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = key_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = key_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_17_squeeze_mask_0, stride = key_cache_internal_tensor_assign_17_stride_0, update = key_states_167_cast_fp16, x = coreml_update_state_86)[name = string("key_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_17_cast_fp16, input = key_cache)[name = string("coreml_update_state_88_write_state")]; tensor coreml_update_state_88 = read_state(input = key_cache)[name = string("coreml_update_state_88")]; tensor value_states_99_perm_0 = const()[name = string("value_states_99_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_17_stride_0 = const()[name = string("value_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_99_cast_fp16 = transpose(perm = value_states_99_perm_0, x = var_6127_cast_fp16)[name = string("transpose_120")]; tensor value_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = value_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = value_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_17_squeeze_mask_0, stride = value_cache_internal_tensor_assign_17_stride_0, update = value_states_99_cast_fp16, x = coreml_update_state_87)[name = string("value_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_17_cast_fp16, input = value_cache)[name = string("coreml_update_state_89_write_state")]; tensor coreml_update_state_89 = read_state(input = value_cache)[name = string("coreml_update_state_89")]; tensor var_6221_begin_0 = const()[name = string("op_6221_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6221_end_0 = const()[name = string("op_6221_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6221_end_mask_0 = const()[name = string("op_6221_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6221_cast_fp16 = slice_by_index(begin = var_6221_begin_0, end = var_6221_end_0, end_mask = var_6221_end_mask_0, x = coreml_update_state_88)[name = string("op_6221_cast_fp16")]; tensor tile_32 = const()[name = string("tile_32"), val = tensor([1, 1])]; int32 var_6224_axis_0 = const()[name = string("op_6224_axis_0"), val = int32(1)]; tensor var_6224_cast_fp16_0, tensor var_6224_cast_fp16_1 = split(axis = var_6224_axis_0, split_sizes = tile_32, x = var_6221_cast_fp16)[name = string("op_6224_cast_fp16")]; tensor var_6231_begin_0 = const()[name = string("op_6231_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6231_end_0 = const()[name = string("op_6231_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6231_end_mask_0 = const()[name = string("op_6231_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6231_cast_fp16 = slice_by_index(begin = var_6231_begin_0, end = var_6231_end_0, end_mask = var_6231_end_mask_0, x = coreml_update_state_89)[name = string("op_6231_cast_fp16")]; tensor tile_33 = const()[name = string("tile_33"), val = tensor([1, 1])]; int32 var_6234_axis_0 = const()[name = string("op_6234_axis_0"), val = int32(1)]; tensor var_6234_cast_fp16_0, tensor var_6234_cast_fp16_1 = split(axis = var_6234_axis_0, split_sizes = tile_33, x = var_6231_cast_fp16)[name = string("op_6234_cast_fp16")]; tensor var_6237_split_sizes_0 = const()[name = string("op_6237_split_sizes_0"), val = tensor([8, 8])]; int32 var_6237_axis_0 = const()[name = string("op_6237_axis_0"), val = int32(1)]; tensor var_6237_0, tensor var_6237_1 = split(axis = var_6237_axis_0, split_sizes = var_6237_split_sizes_0, x = query_states_99_cast_fp16)[name = string("op_6237")]; bool attn_weights_257_transpose_x_0 = const()[name = string("attn_weights_257_transpose_x_0"), val = bool(false)]; bool attn_weights_257_transpose_y_0 = const()[name = string("attn_weights_257_transpose_y_0"), val = bool(false)]; tensor attn_weights_257_cast_fp16 = matmul(transpose_x = attn_weights_257_transpose_x_0, transpose_y = attn_weights_257_transpose_y_0, x = var_6224_cast_fp16_0, y = var_6237_0)[name = string("attn_weights_257_cast_fp16")]; fp16 var_6240_to_fp16 = const()[name = string("op_6240_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_259_cast_fp16 = mul(x = attn_weights_257_cast_fp16, y = var_6240_to_fp16)[name = string("attn_weights_259_cast_fp16")]; tensor attn_weights_261_cast_fp16 = add(x = attn_weights_259_cast_fp16, y = attn_mask_1)[name = string("attn_weights_261_cast_fp16")]; int32 var_6244 = const()[name = string("op_6244"), val = int32(-2)]; tensor attn_weights_263_cast_fp16 = softmax(axis = var_6244, x = attn_weights_261_cast_fp16)[name = string("attn_weights_263_cast_fp16")]; bool var_6250_transpose_x_1 = const()[name = string("op_6250_transpose_x_1"), val = bool(true)]; bool var_6250_transpose_y_1 = const()[name = string("op_6250_transpose_y_1"), val = bool(false)]; tensor var_6250_cast_fp16 = matmul(transpose_x = var_6250_transpose_x_1, transpose_y = var_6250_transpose_y_1, x = attn_weights_263_cast_fp16, y = var_6234_cast_fp16_0)[name = string("op_6250_cast_fp16")]; bool attn_weights_265_transpose_x_0 = const()[name = string("attn_weights_265_transpose_x_0"), val = bool(false)]; bool attn_weights_265_transpose_y_0 = const()[name = string("attn_weights_265_transpose_y_0"), val = bool(false)]; tensor attn_weights_265_cast_fp16 = matmul(transpose_x = attn_weights_265_transpose_x_0, transpose_y = attn_weights_265_transpose_y_0, x = var_6224_cast_fp16_1, y = var_6237_1)[name = string("attn_weights_265_cast_fp16")]; fp16 var_6252_to_fp16 = const()[name = string("op_6252_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_267_cast_fp16 = mul(x = attn_weights_265_cast_fp16, y = var_6252_to_fp16)[name = string("attn_weights_267_cast_fp16")]; tensor attn_weights_269_cast_fp16 = add(x = attn_weights_267_cast_fp16, y = attn_mask_1)[name = string("attn_weights_269_cast_fp16")]; int32 var_6256 = const()[name = string("op_6256"), val = int32(-2)]; tensor attn_weights_271_cast_fp16 = softmax(axis = var_6256, x = attn_weights_269_cast_fp16)[name = string("attn_weights_271_cast_fp16")]; bool attn_output_129_transpose_x_1 = const()[name = string("attn_output_129_transpose_x_1"), val = bool(true)]; bool attn_output_129_transpose_y_1 = const()[name = string("attn_output_129_transpose_y_1"), val = bool(false)]; tensor attn_output_129_cast_fp16 = matmul(transpose_x = attn_output_129_transpose_x_1, transpose_y = attn_output_129_transpose_y_1, x = attn_weights_271_cast_fp16, y = var_6234_cast_fp16_1)[name = string("attn_output_129_cast_fp16")]; int32 var_6264 = const()[name = string("op_6264"), val = int32(1)]; bool attn_output_131_interleave_0 = const()[name = string("attn_output_131_interleave_0"), val = bool(false)]; tensor attn_output_131_cast_fp16 = concat(axis = var_6264, interleave = attn_output_131_interleave_0, values = (var_6250_cast_fp16, attn_output_129_cast_fp16))[name = string("attn_output_131_cast_fp16")]; tensor var_6268_perm_0 = const()[name = string("op_6268_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_203x = const()[name = string("concat_203x"), val = tensor([1, 2048, 1, -1])]; tensor var_6268_cast_fp16 = transpose(perm = var_6268_perm_0, x = attn_output_131_cast_fp16)[name = string("transpose_119")]; tensor attn_output_135_cast_fp16 = reshape(shape = concat_203x, x = var_6268_cast_fp16)[name = string("attn_output_135_cast_fp16")]; tensor hidden_states_163_strides_0 = const()[name = string("hidden_states_163_strides_0"), val = tensor([1, 1])]; string hidden_states_163_pad_type_0 = const()[name = string("hidden_states_163_pad_type_0"), val = string("valid")]; tensor hidden_states_163_pad_0 = const()[name = string("hidden_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_163_dilations_0 = const()[name = string("hidden_states_163_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_163_groups_0 = const()[name = string("hidden_states_163_groups_0"), val = int32(1)]; tensor hidden_states_163_cast_fp16 = conv(dilations = hidden_states_163_dilations_0, groups = hidden_states_163_groups_0, pad = hidden_states_163_pad_0, pad_type = hidden_states_163_pad_type_0, strides = hidden_states_163_strides_0, weight = layers_16_self_attn_o_proj_weight_cast_fp16, x = attn_output_135_cast_fp16)[name = string("hidden_states_163_cast_fp16")]; tensor hidden_states_165_cast_fp16 = add(x = hidden_states_159_cast_fp16, y = hidden_states_163_cast_fp16)[name = string("hidden_states_165_cast_fp16")]; fp16 const_168_promoted_to_fp16 = const()[name = string("const_168_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6301_cast_fp16 = mul(x = hidden_states_165_cast_fp16, y = const_168_promoted_to_fp16)[name = string("op_6301_cast_fp16")]; int32 var_6299 = const()[name = string("op_6299"), val = int32(1)]; bool doubled_133_interleave_0 = const()[name = string("doubled_133_interleave_0"), val = bool(false)]; tensor doubled_133_cast_fp16 = concat(axis = var_6299, interleave = doubled_133_interleave_0, values = (hidden_states_165_cast_fp16, var_6301_cast_fp16))[name = string("doubled_133_cast_fp16")]; tensor out_67_axes_0 = const()[name = string("out_67_axes_0"), val = tensor([1])]; tensor out_67_gamma_0_to_fp16 = const()[name = string("out_67_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441663488)))]; fp16 var_6311_to_fp16 = const()[name = string("op_6311_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_67_cast_fp16 = layer_norm(axes = out_67_axes_0, epsilon = var_6311_to_fp16, gamma = out_67_gamma_0_to_fp16, x = doubled_133_cast_fp16)[name = string("out_67_cast_fp16")]; tensor var_6322_split_sizes_0 = const()[name = string("op_6322_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6322_axis_0 = const()[name = string("op_6322_axis_0"), val = int32(1)]; tensor var_6322_cast_fp16_0, tensor var_6322_cast_fp16_1 = split(axis = var_6322_axis_0, split_sizes = var_6322_split_sizes_0, x = out_67_cast_fp16)[name = string("op_6322_cast_fp16")]; tensor layers_16_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441671744)))]; tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; tensor input_33_cast_fp16 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = layers_16_mlp_gate_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("input_33_cast_fp16")]; tensor var_6339_cast_fp16 = silu(x = input_33_cast_fp16)[name = string("op_6339_cast_fp16")]; tensor layers_16_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1466837632)))]; tensor var_6345_strides_0 = const()[name = string("op_6345_strides_0"), val = tensor([1, 1])]; string var_6345_pad_type_0 = const()[name = string("op_6345_pad_type_0"), val = string("valid")]; tensor var_6345_pad_0 = const()[name = string("op_6345_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6345_dilations_0 = const()[name = string("op_6345_dilations_0"), val = tensor([1, 1])]; int32 var_6345_groups_0 = const()[name = string("op_6345_groups_0"), val = int32(1)]; tensor var_6345_cast_fp16 = conv(dilations = var_6345_dilations_0, groups = var_6345_groups_0, pad = var_6345_pad_0, pad_type = var_6345_pad_type_0, strides = var_6345_strides_0, weight = layers_16_mlp_up_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("op_6345_cast_fp16")]; tensor x_169_cast_fp16 = mul(x = var_6339_cast_fp16, y = var_6345_cast_fp16)[name = string("x_169_cast_fp16")]; tensor hidden_states_167_strides_0 = const()[name = string("hidden_states_167_strides_0"), val = tensor([1, 1])]; string hidden_states_167_pad_type_0 = const()[name = string("hidden_states_167_pad_type_0"), val = string("valid")]; tensor hidden_states_167_pad_0 = const()[name = string("hidden_states_167_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_167_dilations_0 = const()[name = string("hidden_states_167_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_167_groups_0 = const()[name = string("hidden_states_167_groups_0"), val = int32(1)]; tensor hidden_states_167_cast_fp16 = conv(dilations = hidden_states_167_dilations_0, groups = hidden_states_167_groups_0, pad = hidden_states_167_pad_0, pad_type = hidden_states_167_pad_type_0, strides = hidden_states_167_strides_0, weight = layers_16_mlp_down_proj_weight_cast_fp16, x = x_169_cast_fp16)[name = string("hidden_states_167_cast_fp16")]; tensor hidden_states_169_cast_fp16 = add(x = hidden_states_165_cast_fp16, y = hidden_states_167_cast_fp16)[name = string("hidden_states_169_cast_fp16")]; fp16 const_170_promoted_to_fp16 = const()[name = string("const_170_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6363_cast_fp16 = mul(x = hidden_states_169_cast_fp16, y = const_170_promoted_to_fp16)[name = string("op_6363_cast_fp16")]; int32 var_6361 = const()[name = string("op_6361"), val = int32(1)]; bool doubled_137_interleave_0 = const()[name = string("doubled_137_interleave_0"), val = bool(false)]; tensor doubled_137_cast_fp16 = concat(axis = var_6361, interleave = doubled_137_interleave_0, values = (hidden_states_169_cast_fp16, var_6363_cast_fp16))[name = string("doubled_137_cast_fp16")]; tensor out_69_axes_0 = const()[name = string("out_69_axes_0"), val = tensor([1])]; tensor out_69_gamma_0_to_fp16 = const()[name = string("out_69_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492003520)))]; fp16 var_6373_to_fp16 = const()[name = string("op_6373_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_69_cast_fp16 = layer_norm(axes = out_69_axes_0, epsilon = var_6373_to_fp16, gamma = out_69_gamma_0_to_fp16, x = doubled_137_cast_fp16)[name = string("out_69_cast_fp16")]; tensor var_6384_split_sizes_0 = const()[name = string("op_6384_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6384_axis_0 = const()[name = string("op_6384_axis_0"), val = int32(1)]; tensor var_6384_cast_fp16_0, tensor var_6384_cast_fp16_1 = split(axis = var_6384_axis_0, split_sizes = var_6384_split_sizes_0, x = out_69_cast_fp16)[name = string("op_6384_cast_fp16")]; tensor query_states_103_strides_0 = const()[name = string("query_states_103_strides_0"), val = tensor([1, 1])]; string query_states_103_pad_type_0 = const()[name = string("query_states_103_pad_type_0"), val = string("valid")]; tensor query_states_103_pad_0 = const()[name = string("query_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_103_dilations_0 = const()[name = string("query_states_103_dilations_0"), val = tensor([1, 1])]; int32 query_states_103_groups_0 = const()[name = string("query_states_103_groups_0"), val = int32(1)]; tensor query_states_103_cast_fp16 = conv(dilations = query_states_103_dilations_0, groups = query_states_103_groups_0, pad = query_states_103_pad_0, pad_type = query_states_103_pad_type_0, strides = query_states_103_strides_0, weight = layers_17_self_attn_q_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("query_states_103_cast_fp16")]; tensor key_states_171_strides_0 = const()[name = string("key_states_171_strides_0"), val = tensor([1, 1])]; string key_states_171_pad_type_0 = const()[name = string("key_states_171_pad_type_0"), val = string("valid")]; tensor key_states_171_pad_0 = const()[name = string("key_states_171_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_171_dilations_0 = const()[name = string("key_states_171_dilations_0"), val = tensor([1, 1])]; int32 key_states_171_groups_0 = const()[name = string("key_states_171_groups_0"), val = int32(1)]; tensor key_states_171_cast_fp16 = conv(dilations = key_states_171_dilations_0, groups = key_states_171_groups_0, pad = key_states_171_pad_0, pad_type = key_states_171_pad_type_0, strides = key_states_171_strides_0, weight = layers_17_self_attn_k_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("key_states_171_cast_fp16")]; tensor value_states_103_strides_0 = const()[name = string("value_states_103_strides_0"), val = tensor([1, 1])]; string value_states_103_pad_type_0 = const()[name = string("value_states_103_pad_type_0"), val = string("valid")]; tensor value_states_103_pad_0 = const()[name = string("value_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_103_dilations_0 = const()[name = string("value_states_103_dilations_0"), val = tensor([1, 1])]; int32 value_states_103_groups_0 = const()[name = string("value_states_103_groups_0"), val = int32(1)]; tensor value_states_103_cast_fp16 = conv(dilations = value_states_103_dilations_0, groups = value_states_103_groups_0, pad = value_states_103_pad_0, pad_type = value_states_103_pad_type_0, strides = value_states_103_strides_0, weight = layers_17_self_attn_v_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("value_states_103_cast_fp16")]; tensor concat_204x = const()[name = string("concat_204x"), val = tensor([1, 16, 128, -1])]; tensor x_171_cast_fp16 = reshape(shape = concat_204x, x = query_states_103_cast_fp16)[name = string("x_171_cast_fp16")]; tensor concat_205x = const()[name = string("concat_205x"), val = tensor([1, 2, 128, -1])]; tensor var_6441_cast_fp16 = reshape(shape = concat_205x, x = key_states_171_cast_fp16)[name = string("op_6441_cast_fp16")]; tensor concat_206x = const()[name = string("concat_206x"), val = tensor([1, 2, 128, -1])]; tensor var_6448_cast_fp16 = reshape(shape = concat_206x, x = value_states_103_cast_fp16)[name = string("op_6448_cast_fp16")]; tensor var_6452_cast_fp16 = mul(x = x_171_cast_fp16, y = var_869_cast_fp16)[name = string("op_6452_cast_fp16")]; tensor var_6453_split_sizes_0 = const()[name = string("op_6453_split_sizes_0"), val = tensor([64, 64])]; int32 var_6453_axis_0 = const()[name = string("op_6453_axis_0"), val = int32(-2)]; tensor var_6453_cast_fp16_0, tensor var_6453_cast_fp16_1 = split(axis = var_6453_axis_0, split_sizes = var_6453_split_sizes_0, x = x_171_cast_fp16)[name = string("op_6453_cast_fp16")]; fp16 const_172_promoted_to_fp16 = const()[name = string("const_172_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6455_cast_fp16 = mul(x = var_6453_cast_fp16_1, y = const_172_promoted_to_fp16)[name = string("op_6455_cast_fp16")]; int32 var_6457 = const()[name = string("op_6457"), val = int32(-2)]; bool var_6458_interleave_0 = const()[name = string("op_6458_interleave_0"), val = bool(false)]; tensor var_6458_cast_fp16 = concat(axis = var_6457, interleave = var_6458_interleave_0, values = (var_6455_cast_fp16, var_6453_cast_fp16_0))[name = string("op_6458_cast_fp16")]; tensor var_6459_cast_fp16 = mul(x = var_6458_cast_fp16, y = var_878_cast_fp16)[name = string("op_6459_cast_fp16")]; tensor query_states_105_cast_fp16 = add(x = var_6452_cast_fp16, y = var_6459_cast_fp16)[name = string("query_states_105_cast_fp16")]; tensor var_6465_cast_fp16 = mul(x = var_6441_cast_fp16, y = var_869_cast_fp16)[name = string("op_6465_cast_fp16")]; tensor var_6466_split_sizes_0 = const()[name = string("op_6466_split_sizes_0"), val = tensor([64, 64])]; int32 var_6466_axis_0 = const()[name = string("op_6466_axis_0"), val = int32(-2)]; tensor var_6466_cast_fp16_0, tensor var_6466_cast_fp16_1 = split(axis = var_6466_axis_0, split_sizes = var_6466_split_sizes_0, x = var_6441_cast_fp16)[name = string("op_6466_cast_fp16")]; fp16 const_173_promoted_to_fp16 = const()[name = string("const_173_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6468_cast_fp16 = mul(x = var_6466_cast_fp16_1, y = const_173_promoted_to_fp16)[name = string("op_6468_cast_fp16")]; int32 var_6470 = const()[name = string("op_6470"), val = int32(-2)]; bool var_6471_interleave_0 = const()[name = string("op_6471_interleave_0"), val = bool(false)]; tensor var_6471_cast_fp16 = concat(axis = var_6470, interleave = var_6471_interleave_0, values = (var_6468_cast_fp16, var_6466_cast_fp16_0))[name = string("op_6471_cast_fp16")]; tensor var_6472_cast_fp16 = mul(x = var_6471_cast_fp16, y = var_878_cast_fp16)[name = string("op_6472_cast_fp16")]; tensor key_states_175_cast_fp16 = add(x = var_6465_cast_fp16, y = var_6472_cast_fp16)[name = string("key_states_175_cast_fp16")]; tensor expand_dims_204 = const()[name = string("expand_dims_204"), val = tensor([17])]; tensor expand_dims_205 = const()[name = string("expand_dims_205"), val = tensor([0])]; tensor expand_dims_207 = const()[name = string("expand_dims_207"), val = tensor([0])]; int32 concat_209_axis_0 = const()[name = string("concat_209_axis_0"), val = int32(0)]; bool concat_209_interleave_0 = const()[name = string("concat_209_interleave_0"), val = bool(false)]; tensor concat_209 = concat(axis = concat_209_axis_0, interleave = concat_209_interleave_0, values = (expand_dims_204, expand_dims_205, position_id, expand_dims_207))[name = string("concat_209")]; tensor expand_dims_208 = const()[name = string("expand_dims_208"), val = tensor([18])]; tensor concat_210_values1_0 = const()[name = string("concat_210_values1_0"), val = tensor([0])]; tensor concat_210_values3_0 = const()[name = string("concat_210_values3_0"), val = tensor([0])]; int32 concat_210_axis_0 = const()[name = string("concat_210_axis_0"), val = int32(0)]; bool concat_210_interleave_0 = const()[name = string("concat_210_interleave_0"), val = bool(false)]; tensor concat_210 = concat(axis = concat_210_axis_0, interleave = concat_210_interleave_0, values = (expand_dims_208, concat_210_values1_0, cache_position_end, concat_210_values3_0))[name = string("concat_210")]; tensor key_states_177_perm_0 = const()[name = string("key_states_177_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_18_stride_0 = const()[name = string("key_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_177_cast_fp16 = transpose(perm = key_states_177_perm_0, x = key_states_175_cast_fp16)[name = string("transpose_118")]; tensor key_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = key_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = key_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_18_squeeze_mask_0, stride = key_cache_internal_tensor_assign_18_stride_0, update = key_states_177_cast_fp16, x = coreml_update_state_88)[name = string("key_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_18_cast_fp16, input = key_cache)[name = string("coreml_update_state_90_write_state")]; tensor coreml_update_state_90 = read_state(input = key_cache)[name = string("coreml_update_state_90")]; tensor value_states_105_perm_0 = const()[name = string("value_states_105_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_18_stride_0 = const()[name = string("value_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_105_cast_fp16 = transpose(perm = value_states_105_perm_0, x = var_6448_cast_fp16)[name = string("transpose_117")]; tensor value_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = value_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = value_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_18_squeeze_mask_0, stride = value_cache_internal_tensor_assign_18_stride_0, update = value_states_105_cast_fp16, x = coreml_update_state_89)[name = string("value_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_18_cast_fp16, input = value_cache)[name = string("coreml_update_state_91_write_state")]; tensor coreml_update_state_91 = read_state(input = value_cache)[name = string("coreml_update_state_91")]; tensor var_6542_begin_0 = const()[name = string("op_6542_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6542_end_0 = const()[name = string("op_6542_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6542_end_mask_0 = const()[name = string("op_6542_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6542_cast_fp16 = slice_by_index(begin = var_6542_begin_0, end = var_6542_end_0, end_mask = var_6542_end_mask_0, x = coreml_update_state_90)[name = string("op_6542_cast_fp16")]; tensor tile_34 = const()[name = string("tile_34"), val = tensor([1, 1])]; int32 var_6545_axis_0 = const()[name = string("op_6545_axis_0"), val = int32(1)]; tensor var_6545_cast_fp16_0, tensor var_6545_cast_fp16_1 = split(axis = var_6545_axis_0, split_sizes = tile_34, x = var_6542_cast_fp16)[name = string("op_6545_cast_fp16")]; tensor var_6552_begin_0 = const()[name = string("op_6552_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6552_end_0 = const()[name = string("op_6552_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6552_end_mask_0 = const()[name = string("op_6552_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6552_cast_fp16 = slice_by_index(begin = var_6552_begin_0, end = var_6552_end_0, end_mask = var_6552_end_mask_0, x = coreml_update_state_91)[name = string("op_6552_cast_fp16")]; tensor tile_35 = const()[name = string("tile_35"), val = tensor([1, 1])]; int32 var_6555_axis_0 = const()[name = string("op_6555_axis_0"), val = int32(1)]; tensor var_6555_cast_fp16_0, tensor var_6555_cast_fp16_1 = split(axis = var_6555_axis_0, split_sizes = tile_35, x = var_6552_cast_fp16)[name = string("op_6555_cast_fp16")]; tensor var_6558_split_sizes_0 = const()[name = string("op_6558_split_sizes_0"), val = tensor([8, 8])]; int32 var_6558_axis_0 = const()[name = string("op_6558_axis_0"), val = int32(1)]; tensor var_6558_0, tensor var_6558_1 = split(axis = var_6558_axis_0, split_sizes = var_6558_split_sizes_0, x = query_states_105_cast_fp16)[name = string("op_6558")]; bool attn_weights_273_transpose_x_0 = const()[name = string("attn_weights_273_transpose_x_0"), val = bool(false)]; bool attn_weights_273_transpose_y_0 = const()[name = string("attn_weights_273_transpose_y_0"), val = bool(false)]; tensor attn_weights_273_cast_fp16 = matmul(transpose_x = attn_weights_273_transpose_x_0, transpose_y = attn_weights_273_transpose_y_0, x = var_6545_cast_fp16_0, y = var_6558_0)[name = string("attn_weights_273_cast_fp16")]; fp16 var_6561_to_fp16 = const()[name = string("op_6561_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_275_cast_fp16 = mul(x = attn_weights_273_cast_fp16, y = var_6561_to_fp16)[name = string("attn_weights_275_cast_fp16")]; tensor attn_weights_277_cast_fp16 = add(x = attn_weights_275_cast_fp16, y = attn_mask_1)[name = string("attn_weights_277_cast_fp16")]; int32 var_6565 = const()[name = string("op_6565"), val = int32(-2)]; tensor attn_weights_279_cast_fp16 = softmax(axis = var_6565, x = attn_weights_277_cast_fp16)[name = string("attn_weights_279_cast_fp16")]; bool var_6571_transpose_x_1 = const()[name = string("op_6571_transpose_x_1"), val = bool(true)]; bool var_6571_transpose_y_1 = const()[name = string("op_6571_transpose_y_1"), val = bool(false)]; tensor var_6571_cast_fp16 = matmul(transpose_x = var_6571_transpose_x_1, transpose_y = var_6571_transpose_y_1, x = attn_weights_279_cast_fp16, y = var_6555_cast_fp16_0)[name = string("op_6571_cast_fp16")]; bool attn_weights_281_transpose_x_0 = const()[name = string("attn_weights_281_transpose_x_0"), val = bool(false)]; bool attn_weights_281_transpose_y_0 = const()[name = string("attn_weights_281_transpose_y_0"), val = bool(false)]; tensor attn_weights_281_cast_fp16 = matmul(transpose_x = attn_weights_281_transpose_x_0, transpose_y = attn_weights_281_transpose_y_0, x = var_6545_cast_fp16_1, y = var_6558_1)[name = string("attn_weights_281_cast_fp16")]; fp16 var_6573_to_fp16 = const()[name = string("op_6573_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_283_cast_fp16 = mul(x = attn_weights_281_cast_fp16, y = var_6573_to_fp16)[name = string("attn_weights_283_cast_fp16")]; tensor attn_weights_285_cast_fp16 = add(x = attn_weights_283_cast_fp16, y = attn_mask_1)[name = string("attn_weights_285_cast_fp16")]; int32 var_6577 = const()[name = string("op_6577"), val = int32(-2)]; tensor attn_weights_287_cast_fp16 = softmax(axis = var_6577, x = attn_weights_285_cast_fp16)[name = string("attn_weights_287_cast_fp16")]; bool attn_output_137_transpose_x_1 = const()[name = string("attn_output_137_transpose_x_1"), val = bool(true)]; bool attn_output_137_transpose_y_1 = const()[name = string("attn_output_137_transpose_y_1"), val = bool(false)]; tensor attn_output_137_cast_fp16 = matmul(transpose_x = attn_output_137_transpose_x_1, transpose_y = attn_output_137_transpose_y_1, x = attn_weights_287_cast_fp16, y = var_6555_cast_fp16_1)[name = string("attn_output_137_cast_fp16")]; int32 var_6585 = const()[name = string("op_6585"), val = int32(1)]; bool attn_output_139_interleave_0 = const()[name = string("attn_output_139_interleave_0"), val = bool(false)]; tensor attn_output_139_cast_fp16 = concat(axis = var_6585, interleave = attn_output_139_interleave_0, values = (var_6571_cast_fp16, attn_output_137_cast_fp16))[name = string("attn_output_139_cast_fp16")]; tensor var_6589_perm_0 = const()[name = string("op_6589_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_215x = const()[name = string("concat_215x"), val = tensor([1, 2048, 1, -1])]; tensor var_6589_cast_fp16 = transpose(perm = var_6589_perm_0, x = attn_output_139_cast_fp16)[name = string("transpose_116")]; tensor attn_output_143_cast_fp16 = reshape(shape = concat_215x, x = var_6589_cast_fp16)[name = string("attn_output_143_cast_fp16")]; tensor hidden_states_173_strides_0 = const()[name = string("hidden_states_173_strides_0"), val = tensor([1, 1])]; string hidden_states_173_pad_type_0 = const()[name = string("hidden_states_173_pad_type_0"), val = string("valid")]; tensor hidden_states_173_pad_0 = const()[name = string("hidden_states_173_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_173_dilations_0 = const()[name = string("hidden_states_173_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_173_groups_0 = const()[name = string("hidden_states_173_groups_0"), val = int32(1)]; tensor hidden_states_173_cast_fp16 = conv(dilations = hidden_states_173_dilations_0, groups = hidden_states_173_groups_0, pad = hidden_states_173_pad_0, pad_type = hidden_states_173_pad_type_0, strides = hidden_states_173_strides_0, weight = layers_17_self_attn_o_proj_weight_cast_fp16, x = attn_output_143_cast_fp16)[name = string("hidden_states_173_cast_fp16")]; tensor hidden_states_175_cast_fp16 = add(x = hidden_states_169_cast_fp16, y = hidden_states_173_cast_fp16)[name = string("hidden_states_175_cast_fp16")]; fp16 const_178_promoted_to_fp16 = const()[name = string("const_178_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6622_cast_fp16 = mul(x = hidden_states_175_cast_fp16, y = const_178_promoted_to_fp16)[name = string("op_6622_cast_fp16")]; int32 var_6620 = const()[name = string("op_6620"), val = int32(1)]; bool doubled_141_interleave_0 = const()[name = string("doubled_141_interleave_0"), val = bool(false)]; tensor doubled_141_cast_fp16 = concat(axis = var_6620, interleave = doubled_141_interleave_0, values = (hidden_states_175_cast_fp16, var_6622_cast_fp16))[name = string("doubled_141_cast_fp16")]; tensor out_71_axes_0 = const()[name = string("out_71_axes_0"), val = tensor([1])]; tensor out_71_gamma_0_to_fp16 = const()[name = string("out_71_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492011776)))]; fp16 var_6632_to_fp16 = const()[name = string("op_6632_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_71_cast_fp16 = layer_norm(axes = out_71_axes_0, epsilon = var_6632_to_fp16, gamma = out_71_gamma_0_to_fp16, x = doubled_141_cast_fp16)[name = string("out_71_cast_fp16")]; tensor var_6643_split_sizes_0 = const()[name = string("op_6643_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6643_axis_0 = const()[name = string("op_6643_axis_0"), val = int32(1)]; tensor var_6643_cast_fp16_0, tensor var_6643_cast_fp16_1 = split(axis = var_6643_axis_0, split_sizes = var_6643_split_sizes_0, x = out_71_cast_fp16)[name = string("op_6643_cast_fp16")]; tensor input_35_strides_0 = const()[name = string("input_35_strides_0"), val = tensor([1, 1])]; string input_35_pad_type_0 = const()[name = string("input_35_pad_type_0"), val = string("valid")]; tensor input_35_pad_0 = const()[name = string("input_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_35_dilations_0 = const()[name = string("input_35_dilations_0"), val = tensor([1, 1])]; int32 input_35_groups_0 = const()[name = string("input_35_groups_0"), val = int32(1)]; tensor input_35_cast_fp16 = conv(dilations = input_35_dilations_0, groups = input_35_groups_0, pad = input_35_pad_0, pad_type = input_35_pad_type_0, strides = input_35_strides_0, weight = layers_17_mlp_gate_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("input_35_cast_fp16")]; tensor var_6660_cast_fp16 = silu(x = input_35_cast_fp16)[name = string("op_6660_cast_fp16")]; tensor var_6666_strides_0 = const()[name = string("op_6666_strides_0"), val = tensor([1, 1])]; string var_6666_pad_type_0 = const()[name = string("op_6666_pad_type_0"), val = string("valid")]; tensor var_6666_pad_0 = const()[name = string("op_6666_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6666_dilations_0 = const()[name = string("op_6666_dilations_0"), val = tensor([1, 1])]; int32 var_6666_groups_0 = const()[name = string("op_6666_groups_0"), val = int32(1)]; tensor var_6666_cast_fp16 = conv(dilations = var_6666_dilations_0, groups = var_6666_groups_0, pad = var_6666_pad_0, pad_type = var_6666_pad_type_0, strides = var_6666_strides_0, weight = layers_17_mlp_up_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("op_6666_cast_fp16")]; tensor x_179_cast_fp16 = mul(x = var_6660_cast_fp16, y = var_6666_cast_fp16)[name = string("x_179_cast_fp16")]; tensor hidden_states_177_strides_0 = const()[name = string("hidden_states_177_strides_0"), val = tensor([1, 1])]; string hidden_states_177_pad_type_0 = const()[name = string("hidden_states_177_pad_type_0"), val = string("valid")]; tensor hidden_states_177_pad_0 = const()[name = string("hidden_states_177_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_177_dilations_0 = const()[name = string("hidden_states_177_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_177_groups_0 = const()[name = string("hidden_states_177_groups_0"), val = int32(1)]; tensor hidden_states_177_cast_fp16 = conv(dilations = hidden_states_177_dilations_0, groups = hidden_states_177_groups_0, pad = hidden_states_177_pad_0, pad_type = hidden_states_177_pad_type_0, strides = hidden_states_177_strides_0, weight = layers_17_mlp_down_proj_weight_cast_fp16, x = x_179_cast_fp16)[name = string("hidden_states_177_cast_fp16")]; tensor hidden_states_179_cast_fp16 = add(x = hidden_states_175_cast_fp16, y = hidden_states_177_cast_fp16)[name = string("hidden_states_179_cast_fp16")]; fp16 const_180_promoted_to_fp16 = const()[name = string("const_180_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6684_cast_fp16 = mul(x = hidden_states_179_cast_fp16, y = const_180_promoted_to_fp16)[name = string("op_6684_cast_fp16")]; int32 var_6682 = const()[name = string("op_6682"), val = int32(1)]; bool doubled_145_interleave_0 = const()[name = string("doubled_145_interleave_0"), val = bool(false)]; tensor doubled_145_cast_fp16 = concat(axis = var_6682, interleave = doubled_145_interleave_0, values = (hidden_states_179_cast_fp16, var_6684_cast_fp16))[name = string("doubled_145_cast_fp16")]; tensor out_73_axes_0 = const()[name = string("out_73_axes_0"), val = tensor([1])]; tensor out_73_gamma_0_to_fp16 = const()[name = string("out_73_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492020032)))]; fp16 var_6694_to_fp16 = const()[name = string("op_6694_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_73_cast_fp16 = layer_norm(axes = out_73_axes_0, epsilon = var_6694_to_fp16, gamma = out_73_gamma_0_to_fp16, x = doubled_145_cast_fp16)[name = string("out_73_cast_fp16")]; tensor var_6705_split_sizes_0 = const()[name = string("op_6705_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6705_axis_0 = const()[name = string("op_6705_axis_0"), val = int32(1)]; tensor var_6705_cast_fp16_0, tensor var_6705_cast_fp16_1 = split(axis = var_6705_axis_0, split_sizes = var_6705_split_sizes_0, x = out_73_cast_fp16)[name = string("op_6705_cast_fp16")]; tensor query_states_109_strides_0 = const()[name = string("query_states_109_strides_0"), val = tensor([1, 1])]; string query_states_109_pad_type_0 = const()[name = string("query_states_109_pad_type_0"), val = string("valid")]; tensor query_states_109_pad_0 = const()[name = string("query_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_109_dilations_0 = const()[name = string("query_states_109_dilations_0"), val = tensor([1, 1])]; int32 query_states_109_groups_0 = const()[name = string("query_states_109_groups_0"), val = int32(1)]; tensor query_states_109_cast_fp16 = conv(dilations = query_states_109_dilations_0, groups = query_states_109_groups_0, pad = query_states_109_pad_0, pad_type = query_states_109_pad_type_0, strides = query_states_109_strides_0, weight = layers_18_self_attn_q_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("query_states_109_cast_fp16")]; tensor key_states_181_strides_0 = const()[name = string("key_states_181_strides_0"), val = tensor([1, 1])]; string key_states_181_pad_type_0 = const()[name = string("key_states_181_pad_type_0"), val = string("valid")]; tensor key_states_181_pad_0 = const()[name = string("key_states_181_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_181_dilations_0 = const()[name = string("key_states_181_dilations_0"), val = tensor([1, 1])]; int32 key_states_181_groups_0 = const()[name = string("key_states_181_groups_0"), val = int32(1)]; tensor key_states_181_cast_fp16 = conv(dilations = key_states_181_dilations_0, groups = key_states_181_groups_0, pad = key_states_181_pad_0, pad_type = key_states_181_pad_type_0, strides = key_states_181_strides_0, weight = layers_18_self_attn_k_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("key_states_181_cast_fp16")]; tensor value_states_109_strides_0 = const()[name = string("value_states_109_strides_0"), val = tensor([1, 1])]; string value_states_109_pad_type_0 = const()[name = string("value_states_109_pad_type_0"), val = string("valid")]; tensor value_states_109_pad_0 = const()[name = string("value_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_109_dilations_0 = const()[name = string("value_states_109_dilations_0"), val = tensor([1, 1])]; int32 value_states_109_groups_0 = const()[name = string("value_states_109_groups_0"), val = int32(1)]; tensor value_states_109_cast_fp16 = conv(dilations = value_states_109_dilations_0, groups = value_states_109_groups_0, pad = value_states_109_pad_0, pad_type = value_states_109_pad_type_0, strides = value_states_109_strides_0, weight = layers_18_self_attn_v_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("value_states_109_cast_fp16")]; tensor concat_216x = const()[name = string("concat_216x"), val = tensor([1, 16, 128, -1])]; tensor x_181_cast_fp16 = reshape(shape = concat_216x, x = query_states_109_cast_fp16)[name = string("x_181_cast_fp16")]; tensor concat_217x = const()[name = string("concat_217x"), val = tensor([1, 2, 128, -1])]; tensor var_6762_cast_fp16 = reshape(shape = concat_217x, x = key_states_181_cast_fp16)[name = string("op_6762_cast_fp16")]; tensor concat_218x = const()[name = string("concat_218x"), val = tensor([1, 2, 128, -1])]; tensor var_6769_cast_fp16 = reshape(shape = concat_218x, x = value_states_109_cast_fp16)[name = string("op_6769_cast_fp16")]; tensor var_6773_cast_fp16 = mul(x = x_181_cast_fp16, y = var_869_cast_fp16)[name = string("op_6773_cast_fp16")]; tensor var_6774_split_sizes_0 = const()[name = string("op_6774_split_sizes_0"), val = tensor([64, 64])]; int32 var_6774_axis_0 = const()[name = string("op_6774_axis_0"), val = int32(-2)]; tensor var_6774_cast_fp16_0, tensor var_6774_cast_fp16_1 = split(axis = var_6774_axis_0, split_sizes = var_6774_split_sizes_0, x = x_181_cast_fp16)[name = string("op_6774_cast_fp16")]; fp16 const_182_promoted_to_fp16 = const()[name = string("const_182_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6776_cast_fp16 = mul(x = var_6774_cast_fp16_1, y = const_182_promoted_to_fp16)[name = string("op_6776_cast_fp16")]; int32 var_6778 = const()[name = string("op_6778"), val = int32(-2)]; bool var_6779_interleave_0 = const()[name = string("op_6779_interleave_0"), val = bool(false)]; tensor var_6779_cast_fp16 = concat(axis = var_6778, interleave = var_6779_interleave_0, values = (var_6776_cast_fp16, var_6774_cast_fp16_0))[name = string("op_6779_cast_fp16")]; tensor var_6780_cast_fp16 = mul(x = var_6779_cast_fp16, y = var_878_cast_fp16)[name = string("op_6780_cast_fp16")]; tensor query_states_111_cast_fp16 = add(x = var_6773_cast_fp16, y = var_6780_cast_fp16)[name = string("query_states_111_cast_fp16")]; tensor var_6786_cast_fp16 = mul(x = var_6762_cast_fp16, y = var_869_cast_fp16)[name = string("op_6786_cast_fp16")]; tensor var_6787_split_sizes_0 = const()[name = string("op_6787_split_sizes_0"), val = tensor([64, 64])]; int32 var_6787_axis_0 = const()[name = string("op_6787_axis_0"), val = int32(-2)]; tensor var_6787_cast_fp16_0, tensor var_6787_cast_fp16_1 = split(axis = var_6787_axis_0, split_sizes = var_6787_split_sizes_0, x = var_6762_cast_fp16)[name = string("op_6787_cast_fp16")]; fp16 const_183_promoted_to_fp16 = const()[name = string("const_183_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6789_cast_fp16 = mul(x = var_6787_cast_fp16_1, y = const_183_promoted_to_fp16)[name = string("op_6789_cast_fp16")]; int32 var_6791 = const()[name = string("op_6791"), val = int32(-2)]; bool var_6792_interleave_0 = const()[name = string("op_6792_interleave_0"), val = bool(false)]; tensor var_6792_cast_fp16 = concat(axis = var_6791, interleave = var_6792_interleave_0, values = (var_6789_cast_fp16, var_6787_cast_fp16_0))[name = string("op_6792_cast_fp16")]; tensor var_6793_cast_fp16 = mul(x = var_6792_cast_fp16, y = var_878_cast_fp16)[name = string("op_6793_cast_fp16")]; tensor key_states_185_cast_fp16 = add(x = var_6786_cast_fp16, y = var_6793_cast_fp16)[name = string("key_states_185_cast_fp16")]; tensor expand_dims_216 = const()[name = string("expand_dims_216"), val = tensor([18])]; tensor expand_dims_217 = const()[name = string("expand_dims_217"), val = tensor([0])]; tensor expand_dims_219 = const()[name = string("expand_dims_219"), val = tensor([0])]; int32 concat_221_axis_0 = const()[name = string("concat_221_axis_0"), val = int32(0)]; bool concat_221_interleave_0 = const()[name = string("concat_221_interleave_0"), val = bool(false)]; tensor concat_221 = concat(axis = concat_221_axis_0, interleave = concat_221_interleave_0, values = (expand_dims_216, expand_dims_217, position_id, expand_dims_219))[name = string("concat_221")]; tensor expand_dims_220 = const()[name = string("expand_dims_220"), val = tensor([19])]; tensor concat_222_values1_0 = const()[name = string("concat_222_values1_0"), val = tensor([0])]; tensor concat_222_values3_0 = const()[name = string("concat_222_values3_0"), val = tensor([0])]; int32 concat_222_axis_0 = const()[name = string("concat_222_axis_0"), val = int32(0)]; bool concat_222_interleave_0 = const()[name = string("concat_222_interleave_0"), val = bool(false)]; tensor concat_222 = concat(axis = concat_222_axis_0, interleave = concat_222_interleave_0, values = (expand_dims_220, concat_222_values1_0, cache_position_end, concat_222_values3_0))[name = string("concat_222")]; tensor key_states_187_perm_0 = const()[name = string("key_states_187_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_19_stride_0 = const()[name = string("key_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_187_cast_fp16 = transpose(perm = key_states_187_perm_0, x = key_states_185_cast_fp16)[name = string("transpose_115")]; tensor key_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = key_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = key_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_19_squeeze_mask_0, stride = key_cache_internal_tensor_assign_19_stride_0, update = key_states_187_cast_fp16, x = coreml_update_state_90)[name = string("key_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_19_cast_fp16, input = key_cache)[name = string("coreml_update_state_92_write_state")]; tensor coreml_update_state_92 = read_state(input = key_cache)[name = string("coreml_update_state_92")]; tensor value_states_111_perm_0 = const()[name = string("value_states_111_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_19_stride_0 = const()[name = string("value_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_111_cast_fp16 = transpose(perm = value_states_111_perm_0, x = var_6769_cast_fp16)[name = string("transpose_114")]; tensor value_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = value_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = value_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_19_squeeze_mask_0, stride = value_cache_internal_tensor_assign_19_stride_0, update = value_states_111_cast_fp16, x = coreml_update_state_91)[name = string("value_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_19_cast_fp16, input = value_cache)[name = string("coreml_update_state_93_write_state")]; tensor coreml_update_state_93 = read_state(input = value_cache)[name = string("coreml_update_state_93")]; tensor var_6863_begin_0 = const()[name = string("op_6863_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6863_end_0 = const()[name = string("op_6863_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6863_end_mask_0 = const()[name = string("op_6863_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6863_cast_fp16 = slice_by_index(begin = var_6863_begin_0, end = var_6863_end_0, end_mask = var_6863_end_mask_0, x = coreml_update_state_92)[name = string("op_6863_cast_fp16")]; tensor tile_36 = const()[name = string("tile_36"), val = tensor([1, 1])]; int32 var_6866_axis_0 = const()[name = string("op_6866_axis_0"), val = int32(1)]; tensor var_6866_cast_fp16_0, tensor var_6866_cast_fp16_1 = split(axis = var_6866_axis_0, split_sizes = tile_36, x = var_6863_cast_fp16)[name = string("op_6866_cast_fp16")]; tensor var_6873_begin_0 = const()[name = string("op_6873_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6873_end_0 = const()[name = string("op_6873_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6873_end_mask_0 = const()[name = string("op_6873_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6873_cast_fp16 = slice_by_index(begin = var_6873_begin_0, end = var_6873_end_0, end_mask = var_6873_end_mask_0, x = coreml_update_state_93)[name = string("op_6873_cast_fp16")]; tensor tile_37 = const()[name = string("tile_37"), val = tensor([1, 1])]; int32 var_6876_axis_0 = const()[name = string("op_6876_axis_0"), val = int32(1)]; tensor var_6876_cast_fp16_0, tensor var_6876_cast_fp16_1 = split(axis = var_6876_axis_0, split_sizes = tile_37, x = var_6873_cast_fp16)[name = string("op_6876_cast_fp16")]; tensor var_6879_split_sizes_0 = const()[name = string("op_6879_split_sizes_0"), val = tensor([8, 8])]; int32 var_6879_axis_0 = const()[name = string("op_6879_axis_0"), val = int32(1)]; tensor var_6879_0, tensor var_6879_1 = split(axis = var_6879_axis_0, split_sizes = var_6879_split_sizes_0, x = query_states_111_cast_fp16)[name = string("op_6879")]; bool attn_weights_289_transpose_x_0 = const()[name = string("attn_weights_289_transpose_x_0"), val = bool(false)]; bool attn_weights_289_transpose_y_0 = const()[name = string("attn_weights_289_transpose_y_0"), val = bool(false)]; tensor attn_weights_289_cast_fp16 = matmul(transpose_x = attn_weights_289_transpose_x_0, transpose_y = attn_weights_289_transpose_y_0, x = var_6866_cast_fp16_0, y = var_6879_0)[name = string("attn_weights_289_cast_fp16")]; fp16 var_6882_to_fp16 = const()[name = string("op_6882_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_291_cast_fp16 = mul(x = attn_weights_289_cast_fp16, y = var_6882_to_fp16)[name = string("attn_weights_291_cast_fp16")]; tensor attn_weights_293_cast_fp16 = add(x = attn_weights_291_cast_fp16, y = attn_mask_1)[name = string("attn_weights_293_cast_fp16")]; int32 var_6886 = const()[name = string("op_6886"), val = int32(-2)]; tensor attn_weights_295_cast_fp16 = softmax(axis = var_6886, x = attn_weights_293_cast_fp16)[name = string("attn_weights_295_cast_fp16")]; bool var_6892_transpose_x_1 = const()[name = string("op_6892_transpose_x_1"), val = bool(true)]; bool var_6892_transpose_y_1 = const()[name = string("op_6892_transpose_y_1"), val = bool(false)]; tensor var_6892_cast_fp16 = matmul(transpose_x = var_6892_transpose_x_1, transpose_y = var_6892_transpose_y_1, x = attn_weights_295_cast_fp16, y = var_6876_cast_fp16_0)[name = string("op_6892_cast_fp16")]; bool attn_weights_297_transpose_x_0 = const()[name = string("attn_weights_297_transpose_x_0"), val = bool(false)]; bool attn_weights_297_transpose_y_0 = const()[name = string("attn_weights_297_transpose_y_0"), val = bool(false)]; tensor attn_weights_297_cast_fp16 = matmul(transpose_x = attn_weights_297_transpose_x_0, transpose_y = attn_weights_297_transpose_y_0, x = var_6866_cast_fp16_1, y = var_6879_1)[name = string("attn_weights_297_cast_fp16")]; fp16 var_6894_to_fp16 = const()[name = string("op_6894_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_299_cast_fp16 = mul(x = attn_weights_297_cast_fp16, y = var_6894_to_fp16)[name = string("attn_weights_299_cast_fp16")]; tensor attn_weights_301_cast_fp16 = add(x = attn_weights_299_cast_fp16, y = attn_mask_1)[name = string("attn_weights_301_cast_fp16")]; int32 var_6898 = const()[name = string("op_6898"), val = int32(-2)]; tensor attn_weights_303_cast_fp16 = softmax(axis = var_6898, x = attn_weights_301_cast_fp16)[name = string("attn_weights_303_cast_fp16")]; bool attn_output_145_transpose_x_1 = const()[name = string("attn_output_145_transpose_x_1"), val = bool(true)]; bool attn_output_145_transpose_y_1 = const()[name = string("attn_output_145_transpose_y_1"), val = bool(false)]; tensor attn_output_145_cast_fp16 = matmul(transpose_x = attn_output_145_transpose_x_1, transpose_y = attn_output_145_transpose_y_1, x = attn_weights_303_cast_fp16, y = var_6876_cast_fp16_1)[name = string("attn_output_145_cast_fp16")]; int32 var_6906 = const()[name = string("op_6906"), val = int32(1)]; bool attn_output_147_interleave_0 = const()[name = string("attn_output_147_interleave_0"), val = bool(false)]; tensor attn_output_147_cast_fp16 = concat(axis = var_6906, interleave = attn_output_147_interleave_0, values = (var_6892_cast_fp16, attn_output_145_cast_fp16))[name = string("attn_output_147_cast_fp16")]; tensor var_6910_perm_0 = const()[name = string("op_6910_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_227x = const()[name = string("concat_227x"), val = tensor([1, 2048, 1, -1])]; tensor var_6910_cast_fp16 = transpose(perm = var_6910_perm_0, x = attn_output_147_cast_fp16)[name = string("transpose_113")]; tensor attn_output_151_cast_fp16 = reshape(shape = concat_227x, x = var_6910_cast_fp16)[name = string("attn_output_151_cast_fp16")]; tensor hidden_states_183_strides_0 = const()[name = string("hidden_states_183_strides_0"), val = tensor([1, 1])]; string hidden_states_183_pad_type_0 = const()[name = string("hidden_states_183_pad_type_0"), val = string("valid")]; tensor hidden_states_183_pad_0 = const()[name = string("hidden_states_183_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_183_dilations_0 = const()[name = string("hidden_states_183_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_183_groups_0 = const()[name = string("hidden_states_183_groups_0"), val = int32(1)]; tensor hidden_states_183_cast_fp16 = conv(dilations = hidden_states_183_dilations_0, groups = hidden_states_183_groups_0, pad = hidden_states_183_pad_0, pad_type = hidden_states_183_pad_type_0, strides = hidden_states_183_strides_0, weight = layers_18_self_attn_o_proj_weight_cast_fp16, x = attn_output_151_cast_fp16)[name = string("hidden_states_183_cast_fp16")]; tensor hidden_states_185_cast_fp16 = add(x = hidden_states_179_cast_fp16, y = hidden_states_183_cast_fp16)[name = string("hidden_states_185_cast_fp16")]; fp16 const_188_promoted_to_fp16 = const()[name = string("const_188_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6943_cast_fp16 = mul(x = hidden_states_185_cast_fp16, y = const_188_promoted_to_fp16)[name = string("op_6943_cast_fp16")]; int32 var_6941 = const()[name = string("op_6941"), val = int32(1)]; bool doubled_149_interleave_0 = const()[name = string("doubled_149_interleave_0"), val = bool(false)]; tensor doubled_149_cast_fp16 = concat(axis = var_6941, interleave = doubled_149_interleave_0, values = (hidden_states_185_cast_fp16, var_6943_cast_fp16))[name = string("doubled_149_cast_fp16")]; tensor out_75_axes_0 = const()[name = string("out_75_axes_0"), val = tensor([1])]; tensor out_75_gamma_0_to_fp16 = const()[name = string("out_75_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492028288)))]; fp16 var_6953_to_fp16 = const()[name = string("op_6953_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_75_cast_fp16 = layer_norm(axes = out_75_axes_0, epsilon = var_6953_to_fp16, gamma = out_75_gamma_0_to_fp16, x = doubled_149_cast_fp16)[name = string("out_75_cast_fp16")]; tensor var_6964_split_sizes_0 = const()[name = string("op_6964_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6964_axis_0 = const()[name = string("op_6964_axis_0"), val = int32(1)]; tensor var_6964_cast_fp16_0, tensor var_6964_cast_fp16_1 = split(axis = var_6964_axis_0, split_sizes = var_6964_split_sizes_0, x = out_75_cast_fp16)[name = string("op_6964_cast_fp16")]; tensor input_37_strides_0 = const()[name = string("input_37_strides_0"), val = tensor([1, 1])]; string input_37_pad_type_0 = const()[name = string("input_37_pad_type_0"), val = string("valid")]; tensor input_37_pad_0 = const()[name = string("input_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_37_dilations_0 = const()[name = string("input_37_dilations_0"), val = tensor([1, 1])]; int32 input_37_groups_0 = const()[name = string("input_37_groups_0"), val = int32(1)]; tensor input_37_cast_fp16 = conv(dilations = input_37_dilations_0, groups = input_37_groups_0, pad = input_37_pad_0, pad_type = input_37_pad_type_0, strides = input_37_strides_0, weight = layers_18_mlp_gate_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("input_37_cast_fp16")]; tensor var_6981_cast_fp16 = silu(x = input_37_cast_fp16)[name = string("op_6981_cast_fp16")]; tensor var_6987_strides_0 = const()[name = string("op_6987_strides_0"), val = tensor([1, 1])]; string var_6987_pad_type_0 = const()[name = string("op_6987_pad_type_0"), val = string("valid")]; tensor var_6987_pad_0 = const()[name = string("op_6987_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6987_dilations_0 = const()[name = string("op_6987_dilations_0"), val = tensor([1, 1])]; int32 var_6987_groups_0 = const()[name = string("op_6987_groups_0"), val = int32(1)]; tensor var_6987_cast_fp16 = conv(dilations = var_6987_dilations_0, groups = var_6987_groups_0, pad = var_6987_pad_0, pad_type = var_6987_pad_type_0, strides = var_6987_strides_0, weight = layers_18_mlp_up_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("op_6987_cast_fp16")]; tensor x_189_cast_fp16 = mul(x = var_6981_cast_fp16, y = var_6987_cast_fp16)[name = string("x_189_cast_fp16")]; tensor hidden_states_187_strides_0 = const()[name = string("hidden_states_187_strides_0"), val = tensor([1, 1])]; string hidden_states_187_pad_type_0 = const()[name = string("hidden_states_187_pad_type_0"), val = string("valid")]; tensor hidden_states_187_pad_0 = const()[name = string("hidden_states_187_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_187_dilations_0 = const()[name = string("hidden_states_187_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_187_groups_0 = const()[name = string("hidden_states_187_groups_0"), val = int32(1)]; tensor hidden_states_187_cast_fp16 = conv(dilations = hidden_states_187_dilations_0, groups = hidden_states_187_groups_0, pad = hidden_states_187_pad_0, pad_type = hidden_states_187_pad_type_0, strides = hidden_states_187_strides_0, weight = layers_18_mlp_down_proj_weight_cast_fp16, x = x_189_cast_fp16)[name = string("hidden_states_187_cast_fp16")]; tensor hidden_states_189_cast_fp16 = add(x = hidden_states_185_cast_fp16, y = hidden_states_187_cast_fp16)[name = string("hidden_states_189_cast_fp16")]; fp16 const_190_promoted_to_fp16 = const()[name = string("const_190_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7005_cast_fp16 = mul(x = hidden_states_189_cast_fp16, y = const_190_promoted_to_fp16)[name = string("op_7005_cast_fp16")]; int32 var_7003 = const()[name = string("op_7003"), val = int32(1)]; bool doubled_153_interleave_0 = const()[name = string("doubled_153_interleave_0"), val = bool(false)]; tensor doubled_153_cast_fp16 = concat(axis = var_7003, interleave = doubled_153_interleave_0, values = (hidden_states_189_cast_fp16, var_7005_cast_fp16))[name = string("doubled_153_cast_fp16")]; tensor out_77_axes_0 = const()[name = string("out_77_axes_0"), val = tensor([1])]; tensor out_77_gamma_0_to_fp16 = const()[name = string("out_77_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492036544)))]; fp16 var_7015_to_fp16 = const()[name = string("op_7015_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_77_cast_fp16 = layer_norm(axes = out_77_axes_0, epsilon = var_7015_to_fp16, gamma = out_77_gamma_0_to_fp16, x = doubled_153_cast_fp16)[name = string("out_77_cast_fp16")]; tensor var_7026_split_sizes_0 = const()[name = string("op_7026_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7026_axis_0 = const()[name = string("op_7026_axis_0"), val = int32(1)]; tensor var_7026_cast_fp16_0, tensor var_7026_cast_fp16_1 = split(axis = var_7026_axis_0, split_sizes = var_7026_split_sizes_0, x = out_77_cast_fp16)[name = string("op_7026_cast_fp16")]; tensor query_states_115_strides_0 = const()[name = string("query_states_115_strides_0"), val = tensor([1, 1])]; string query_states_115_pad_type_0 = const()[name = string("query_states_115_pad_type_0"), val = string("valid")]; tensor query_states_115_pad_0 = const()[name = string("query_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_115_dilations_0 = const()[name = string("query_states_115_dilations_0"), val = tensor([1, 1])]; int32 query_states_115_groups_0 = const()[name = string("query_states_115_groups_0"), val = int32(1)]; tensor query_states_115_cast_fp16 = conv(dilations = query_states_115_dilations_0, groups = query_states_115_groups_0, pad = query_states_115_pad_0, pad_type = query_states_115_pad_type_0, strides = query_states_115_strides_0, weight = layers_19_self_attn_q_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("query_states_115_cast_fp16")]; tensor key_states_191_strides_0 = const()[name = string("key_states_191_strides_0"), val = tensor([1, 1])]; string key_states_191_pad_type_0 = const()[name = string("key_states_191_pad_type_0"), val = string("valid")]; tensor key_states_191_pad_0 = const()[name = string("key_states_191_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_191_dilations_0 = const()[name = string("key_states_191_dilations_0"), val = tensor([1, 1])]; int32 key_states_191_groups_0 = const()[name = string("key_states_191_groups_0"), val = int32(1)]; tensor key_states_191_cast_fp16 = conv(dilations = key_states_191_dilations_0, groups = key_states_191_groups_0, pad = key_states_191_pad_0, pad_type = key_states_191_pad_type_0, strides = key_states_191_strides_0, weight = layers_19_self_attn_k_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("key_states_191_cast_fp16")]; tensor layers_19_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492044800)))]; tensor value_states_115_strides_0 = const()[name = string("value_states_115_strides_0"), val = tensor([1, 1])]; string value_states_115_pad_type_0 = const()[name = string("value_states_115_pad_type_0"), val = string("valid")]; tensor value_states_115_pad_0 = const()[name = string("value_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_115_dilations_0 = const()[name = string("value_states_115_dilations_0"), val = tensor([1, 1])]; int32 value_states_115_groups_0 = const()[name = string("value_states_115_groups_0"), val = int32(1)]; tensor value_states_115_cast_fp16 = conv(dilations = value_states_115_dilations_0, groups = value_states_115_groups_0, pad = value_states_115_pad_0, pad_type = value_states_115_pad_type_0, strides = value_states_115_strides_0, weight = layers_19_self_attn_v_proj_weight_to_fp16, x = var_7026_cast_fp16_0)[name = string("value_states_115_cast_fp16")]; tensor concat_228x = const()[name = string("concat_228x"), val = tensor([1, 16, 128, -1])]; tensor x_191_cast_fp16 = reshape(shape = concat_228x, x = query_states_115_cast_fp16)[name = string("x_191_cast_fp16")]; tensor concat_229x = const()[name = string("concat_229x"), val = tensor([1, 2, 128, -1])]; tensor var_7083_cast_fp16 = reshape(shape = concat_229x, x = key_states_191_cast_fp16)[name = string("op_7083_cast_fp16")]; tensor concat_230x = const()[name = string("concat_230x"), val = tensor([1, 2, 128, -1])]; tensor var_7090_cast_fp16 = reshape(shape = concat_230x, x = value_states_115_cast_fp16)[name = string("op_7090_cast_fp16")]; tensor var_7094_cast_fp16 = mul(x = x_191_cast_fp16, y = var_869_cast_fp16)[name = string("op_7094_cast_fp16")]; tensor var_7095_split_sizes_0 = const()[name = string("op_7095_split_sizes_0"), val = tensor([64, 64])]; int32 var_7095_axis_0 = const()[name = string("op_7095_axis_0"), val = int32(-2)]; tensor var_7095_cast_fp16_0, tensor var_7095_cast_fp16_1 = split(axis = var_7095_axis_0, split_sizes = var_7095_split_sizes_0, x = x_191_cast_fp16)[name = string("op_7095_cast_fp16")]; fp16 const_192_promoted_to_fp16 = const()[name = string("const_192_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7097_cast_fp16 = mul(x = var_7095_cast_fp16_1, y = const_192_promoted_to_fp16)[name = string("op_7097_cast_fp16")]; int32 var_7099 = const()[name = string("op_7099"), val = int32(-2)]; bool var_7100_interleave_0 = const()[name = string("op_7100_interleave_0"), val = bool(false)]; tensor var_7100_cast_fp16 = concat(axis = var_7099, interleave = var_7100_interleave_0, values = (var_7097_cast_fp16, var_7095_cast_fp16_0))[name = string("op_7100_cast_fp16")]; tensor var_7101_cast_fp16 = mul(x = var_7100_cast_fp16, y = var_878_cast_fp16)[name = string("op_7101_cast_fp16")]; tensor query_states_117_cast_fp16 = add(x = var_7094_cast_fp16, y = var_7101_cast_fp16)[name = string("query_states_117_cast_fp16")]; tensor var_7107_cast_fp16 = mul(x = var_7083_cast_fp16, y = var_869_cast_fp16)[name = string("op_7107_cast_fp16")]; tensor var_7108_split_sizes_0 = const()[name = string("op_7108_split_sizes_0"), val = tensor([64, 64])]; int32 var_7108_axis_0 = const()[name = string("op_7108_axis_0"), val = int32(-2)]; tensor var_7108_cast_fp16_0, tensor var_7108_cast_fp16_1 = split(axis = var_7108_axis_0, split_sizes = var_7108_split_sizes_0, x = var_7083_cast_fp16)[name = string("op_7108_cast_fp16")]; fp16 const_193_promoted_to_fp16 = const()[name = string("const_193_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7110_cast_fp16 = mul(x = var_7108_cast_fp16_1, y = const_193_promoted_to_fp16)[name = string("op_7110_cast_fp16")]; int32 var_7112 = const()[name = string("op_7112"), val = int32(-2)]; bool var_7113_interleave_0 = const()[name = string("op_7113_interleave_0"), val = bool(false)]; tensor var_7113_cast_fp16 = concat(axis = var_7112, interleave = var_7113_interleave_0, values = (var_7110_cast_fp16, var_7108_cast_fp16_0))[name = string("op_7113_cast_fp16")]; tensor var_7114_cast_fp16 = mul(x = var_7113_cast_fp16, y = var_878_cast_fp16)[name = string("op_7114_cast_fp16")]; tensor key_states_195_cast_fp16 = add(x = var_7107_cast_fp16, y = var_7114_cast_fp16)[name = string("key_states_195_cast_fp16")]; tensor expand_dims_228 = const()[name = string("expand_dims_228"), val = tensor([19])]; tensor expand_dims_229 = const()[name = string("expand_dims_229"), val = tensor([0])]; tensor expand_dims_231 = const()[name = string("expand_dims_231"), val = tensor([0])]; int32 concat_233_axis_0 = const()[name = string("concat_233_axis_0"), val = int32(0)]; bool concat_233_interleave_0 = const()[name = string("concat_233_interleave_0"), val = bool(false)]; tensor concat_233 = concat(axis = concat_233_axis_0, interleave = concat_233_interleave_0, values = (expand_dims_228, expand_dims_229, position_id, expand_dims_231))[name = string("concat_233")]; tensor expand_dims_232 = const()[name = string("expand_dims_232"), val = tensor([20])]; tensor concat_234_values1_0 = const()[name = string("concat_234_values1_0"), val = tensor([0])]; tensor concat_234_values3_0 = const()[name = string("concat_234_values3_0"), val = tensor([0])]; int32 concat_234_axis_0 = const()[name = string("concat_234_axis_0"), val = int32(0)]; bool concat_234_interleave_0 = const()[name = string("concat_234_interleave_0"), val = bool(false)]; tensor concat_234 = concat(axis = concat_234_axis_0, interleave = concat_234_interleave_0, values = (expand_dims_232, concat_234_values1_0, cache_position_end, concat_234_values3_0))[name = string("concat_234")]; tensor key_states_197_perm_0 = const()[name = string("key_states_197_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_20_stride_0 = const()[name = string("key_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_197_cast_fp16 = transpose(perm = key_states_197_perm_0, x = key_states_195_cast_fp16)[name = string("transpose_112")]; tensor key_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = key_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = key_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_20_squeeze_mask_0, stride = key_cache_internal_tensor_assign_20_stride_0, update = key_states_197_cast_fp16, x = coreml_update_state_92)[name = string("key_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_20_cast_fp16, input = key_cache)[name = string("coreml_update_state_94_write_state")]; tensor coreml_update_state_94 = read_state(input = key_cache)[name = string("coreml_update_state_94")]; tensor value_states_117_perm_0 = const()[name = string("value_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_20_stride_0 = const()[name = string("value_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_117_cast_fp16 = transpose(perm = value_states_117_perm_0, x = var_7090_cast_fp16)[name = string("transpose_111")]; tensor value_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = value_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = value_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_20_squeeze_mask_0, stride = value_cache_internal_tensor_assign_20_stride_0, update = value_states_117_cast_fp16, x = coreml_update_state_93)[name = string("value_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_20_cast_fp16, input = value_cache)[name = string("coreml_update_state_95_write_state")]; tensor coreml_update_state_95 = read_state(input = value_cache)[name = string("coreml_update_state_95")]; tensor var_7184_begin_0 = const()[name = string("op_7184_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7184_end_0 = const()[name = string("op_7184_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7184_end_mask_0 = const()[name = string("op_7184_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7184_cast_fp16 = slice_by_index(begin = var_7184_begin_0, end = var_7184_end_0, end_mask = var_7184_end_mask_0, x = coreml_update_state_94)[name = string("op_7184_cast_fp16")]; tensor tile_38 = const()[name = string("tile_38"), val = tensor([1, 1])]; int32 var_7187_axis_0 = const()[name = string("op_7187_axis_0"), val = int32(1)]; tensor var_7187_cast_fp16_0, tensor var_7187_cast_fp16_1 = split(axis = var_7187_axis_0, split_sizes = tile_38, x = var_7184_cast_fp16)[name = string("op_7187_cast_fp16")]; tensor var_7194_begin_0 = const()[name = string("op_7194_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7194_end_0 = const()[name = string("op_7194_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7194_end_mask_0 = const()[name = string("op_7194_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7194_cast_fp16 = slice_by_index(begin = var_7194_begin_0, end = var_7194_end_0, end_mask = var_7194_end_mask_0, x = coreml_update_state_95)[name = string("op_7194_cast_fp16")]; tensor tile_39 = const()[name = string("tile_39"), val = tensor([1, 1])]; int32 var_7197_axis_0 = const()[name = string("op_7197_axis_0"), val = int32(1)]; tensor var_7197_cast_fp16_0, tensor var_7197_cast_fp16_1 = split(axis = var_7197_axis_0, split_sizes = tile_39, x = var_7194_cast_fp16)[name = string("op_7197_cast_fp16")]; tensor var_7200_split_sizes_0 = const()[name = string("op_7200_split_sizes_0"), val = tensor([8, 8])]; int32 var_7200_axis_0 = const()[name = string("op_7200_axis_0"), val = int32(1)]; tensor var_7200_0, tensor var_7200_1 = split(axis = var_7200_axis_0, split_sizes = var_7200_split_sizes_0, x = query_states_117_cast_fp16)[name = string("op_7200")]; bool attn_weights_305_transpose_x_0 = const()[name = string("attn_weights_305_transpose_x_0"), val = bool(false)]; bool attn_weights_305_transpose_y_0 = const()[name = string("attn_weights_305_transpose_y_0"), val = bool(false)]; tensor attn_weights_305_cast_fp16 = matmul(transpose_x = attn_weights_305_transpose_x_0, transpose_y = attn_weights_305_transpose_y_0, x = var_7187_cast_fp16_0, y = var_7200_0)[name = string("attn_weights_305_cast_fp16")]; fp16 var_7203_to_fp16 = const()[name = string("op_7203_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_307_cast_fp16 = mul(x = attn_weights_305_cast_fp16, y = var_7203_to_fp16)[name = string("attn_weights_307_cast_fp16")]; tensor attn_weights_309_cast_fp16 = add(x = attn_weights_307_cast_fp16, y = attn_mask_1)[name = string("attn_weights_309_cast_fp16")]; int32 var_7207 = const()[name = string("op_7207"), val = int32(-2)]; tensor attn_weights_311_cast_fp16 = softmax(axis = var_7207, x = attn_weights_309_cast_fp16)[name = string("attn_weights_311_cast_fp16")]; bool var_7213_transpose_x_1 = const()[name = string("op_7213_transpose_x_1"), val = bool(true)]; bool var_7213_transpose_y_1 = const()[name = string("op_7213_transpose_y_1"), val = bool(false)]; tensor var_7213_cast_fp16 = matmul(transpose_x = var_7213_transpose_x_1, transpose_y = var_7213_transpose_y_1, x = attn_weights_311_cast_fp16, y = var_7197_cast_fp16_0)[name = string("op_7213_cast_fp16")]; bool attn_weights_313_transpose_x_0 = const()[name = string("attn_weights_313_transpose_x_0"), val = bool(false)]; bool attn_weights_313_transpose_y_0 = const()[name = string("attn_weights_313_transpose_y_0"), val = bool(false)]; tensor attn_weights_313_cast_fp16 = matmul(transpose_x = attn_weights_313_transpose_x_0, transpose_y = attn_weights_313_transpose_y_0, x = var_7187_cast_fp16_1, y = var_7200_1)[name = string("attn_weights_313_cast_fp16")]; fp16 var_7215_to_fp16 = const()[name = string("op_7215_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_315_cast_fp16 = mul(x = attn_weights_313_cast_fp16, y = var_7215_to_fp16)[name = string("attn_weights_315_cast_fp16")]; tensor attn_weights_317_cast_fp16 = add(x = attn_weights_315_cast_fp16, y = attn_mask_1)[name = string("attn_weights_317_cast_fp16")]; int32 var_7219 = const()[name = string("op_7219"), val = int32(-2)]; tensor attn_weights_319_cast_fp16 = softmax(axis = var_7219, x = attn_weights_317_cast_fp16)[name = string("attn_weights_319_cast_fp16")]; bool attn_output_153_transpose_x_1 = const()[name = string("attn_output_153_transpose_x_1"), val = bool(true)]; bool attn_output_153_transpose_y_1 = const()[name = string("attn_output_153_transpose_y_1"), val = bool(false)]; tensor attn_output_153_cast_fp16 = matmul(transpose_x = attn_output_153_transpose_x_1, transpose_y = attn_output_153_transpose_y_1, x = attn_weights_319_cast_fp16, y = var_7197_cast_fp16_1)[name = string("attn_output_153_cast_fp16")]; int32 var_7227 = const()[name = string("op_7227"), val = int32(1)]; bool attn_output_155_interleave_0 = const()[name = string("attn_output_155_interleave_0"), val = bool(false)]; tensor attn_output_155_cast_fp16 = concat(axis = var_7227, interleave = attn_output_155_interleave_0, values = (var_7213_cast_fp16, attn_output_153_cast_fp16))[name = string("attn_output_155_cast_fp16")]; tensor var_7231_perm_0 = const()[name = string("op_7231_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_239x = const()[name = string("concat_239x"), val = tensor([1, 2048, 1, -1])]; tensor var_7231_cast_fp16 = transpose(perm = var_7231_perm_0, x = attn_output_155_cast_fp16)[name = string("transpose_110")]; tensor attn_output_159_cast_fp16 = reshape(shape = concat_239x, x = var_7231_cast_fp16)[name = string("attn_output_159_cast_fp16")]; tensor layers_19_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1493093440)))]; tensor hidden_states_193_strides_0 = const()[name = string("hidden_states_193_strides_0"), val = tensor([1, 1])]; string hidden_states_193_pad_type_0 = const()[name = string("hidden_states_193_pad_type_0"), val = string("valid")]; tensor hidden_states_193_pad_0 = const()[name = string("hidden_states_193_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_193_dilations_0 = const()[name = string("hidden_states_193_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_193_groups_0 = const()[name = string("hidden_states_193_groups_0"), val = int32(1)]; tensor hidden_states_193_cast_fp16 = conv(dilations = hidden_states_193_dilations_0, groups = hidden_states_193_groups_0, pad = hidden_states_193_pad_0, pad_type = hidden_states_193_pad_type_0, strides = hidden_states_193_strides_0, weight = layers_19_self_attn_o_proj_weight_to_fp16, x = attn_output_159_cast_fp16)[name = string("hidden_states_193_cast_fp16")]; tensor hidden_states_195_cast_fp16 = add(x = hidden_states_189_cast_fp16, y = hidden_states_193_cast_fp16)[name = string("hidden_states_195_cast_fp16")]; fp16 const_198_promoted_to_fp16 = const()[name = string("const_198_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7264_cast_fp16 = mul(x = hidden_states_195_cast_fp16, y = const_198_promoted_to_fp16)[name = string("op_7264_cast_fp16")]; int32 var_7262 = const()[name = string("op_7262"), val = int32(1)]; bool doubled_157_interleave_0 = const()[name = string("doubled_157_interleave_0"), val = bool(false)]; tensor doubled_157_cast_fp16 = concat(axis = var_7262, interleave = doubled_157_interleave_0, values = (hidden_states_195_cast_fp16, var_7264_cast_fp16))[name = string("doubled_157_cast_fp16")]; tensor out_79_axes_0 = const()[name = string("out_79_axes_0"), val = tensor([1])]; tensor out_79_gamma_0_to_fp16 = const()[name = string("out_79_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501482112)))]; fp16 var_7274_to_fp16 = const()[name = string("op_7274_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_79_cast_fp16 = layer_norm(axes = out_79_axes_0, epsilon = var_7274_to_fp16, gamma = out_79_gamma_0_to_fp16, x = doubled_157_cast_fp16)[name = string("out_79_cast_fp16")]; tensor var_7285_split_sizes_0 = const()[name = string("op_7285_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7285_axis_0 = const()[name = string("op_7285_axis_0"), val = int32(1)]; tensor var_7285_cast_fp16_0, tensor var_7285_cast_fp16_1 = split(axis = var_7285_axis_0, split_sizes = var_7285_split_sizes_0, x = out_79_cast_fp16)[name = string("op_7285_cast_fp16")]; tensor input_39_strides_0 = const()[name = string("input_39_strides_0"), val = tensor([1, 1])]; string input_39_pad_type_0 = const()[name = string("input_39_pad_type_0"), val = string("valid")]; tensor input_39_pad_0 = const()[name = string("input_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_39_dilations_0 = const()[name = string("input_39_dilations_0"), val = tensor([1, 1])]; int32 input_39_groups_0 = const()[name = string("input_39_groups_0"), val = int32(1)]; tensor input_39_cast_fp16 = conv(dilations = input_39_dilations_0, groups = input_39_groups_0, pad = input_39_pad_0, pad_type = input_39_pad_type_0, strides = input_39_strides_0, weight = layers_19_mlp_gate_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("input_39_cast_fp16")]; tensor var_7302_cast_fp16 = silu(x = input_39_cast_fp16)[name = string("op_7302_cast_fp16")]; tensor var_7308_strides_0 = const()[name = string("op_7308_strides_0"), val = tensor([1, 1])]; string var_7308_pad_type_0 = const()[name = string("op_7308_pad_type_0"), val = string("valid")]; tensor var_7308_pad_0 = const()[name = string("op_7308_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7308_dilations_0 = const()[name = string("op_7308_dilations_0"), val = tensor([1, 1])]; int32 var_7308_groups_0 = const()[name = string("op_7308_groups_0"), val = int32(1)]; tensor var_7308_cast_fp16 = conv(dilations = var_7308_dilations_0, groups = var_7308_groups_0, pad = var_7308_pad_0, pad_type = var_7308_pad_type_0, strides = var_7308_strides_0, weight = layers_19_mlp_up_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("op_7308_cast_fp16")]; tensor x_199_cast_fp16 = mul(x = var_7302_cast_fp16, y = var_7308_cast_fp16)[name = string("x_199_cast_fp16")]; tensor hidden_states_197_strides_0 = const()[name = string("hidden_states_197_strides_0"), val = tensor([1, 1])]; string hidden_states_197_pad_type_0 = const()[name = string("hidden_states_197_pad_type_0"), val = string("valid")]; tensor hidden_states_197_pad_0 = const()[name = string("hidden_states_197_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_197_dilations_0 = const()[name = string("hidden_states_197_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_197_groups_0 = const()[name = string("hidden_states_197_groups_0"), val = int32(1)]; tensor hidden_states_197_cast_fp16 = conv(dilations = hidden_states_197_dilations_0, groups = hidden_states_197_groups_0, pad = hidden_states_197_pad_0, pad_type = hidden_states_197_pad_type_0, strides = hidden_states_197_strides_0, weight = layers_19_mlp_down_proj_weight_cast_fp16, x = x_199_cast_fp16)[name = string("hidden_states_197_cast_fp16")]; tensor hidden_states_199_cast_fp16 = add(x = hidden_states_195_cast_fp16, y = hidden_states_197_cast_fp16)[name = string("hidden_states_199_cast_fp16")]; fp16 const_200_promoted_to_fp16 = const()[name = string("const_200_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7326_cast_fp16 = mul(x = hidden_states_199_cast_fp16, y = const_200_promoted_to_fp16)[name = string("op_7326_cast_fp16")]; int32 var_7324 = const()[name = string("op_7324"), val = int32(1)]; bool doubled_161_interleave_0 = const()[name = string("doubled_161_interleave_0"), val = bool(false)]; tensor doubled_161_cast_fp16 = concat(axis = var_7324, interleave = doubled_161_interleave_0, values = (hidden_states_199_cast_fp16, var_7326_cast_fp16))[name = string("doubled_161_cast_fp16")]; tensor out_81_axes_0 = const()[name = string("out_81_axes_0"), val = tensor([1])]; tensor out_81_gamma_0_to_fp16 = const()[name = string("out_81_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501490368)))]; fp16 var_7336_to_fp16 = const()[name = string("op_7336_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_81_cast_fp16 = layer_norm(axes = out_81_axes_0, epsilon = var_7336_to_fp16, gamma = out_81_gamma_0_to_fp16, x = doubled_161_cast_fp16)[name = string("out_81_cast_fp16")]; tensor var_7347_split_sizes_0 = const()[name = string("op_7347_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7347_axis_0 = const()[name = string("op_7347_axis_0"), val = int32(1)]; tensor var_7347_cast_fp16_0, tensor var_7347_cast_fp16_1 = split(axis = var_7347_axis_0, split_sizes = var_7347_split_sizes_0, x = out_81_cast_fp16)[name = string("op_7347_cast_fp16")]; tensor query_states_121_strides_0 = const()[name = string("query_states_121_strides_0"), val = tensor([1, 1])]; string query_states_121_pad_type_0 = const()[name = string("query_states_121_pad_type_0"), val = string("valid")]; tensor query_states_121_pad_0 = const()[name = string("query_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_121_dilations_0 = const()[name = string("query_states_121_dilations_0"), val = tensor([1, 1])]; int32 query_states_121_groups_0 = const()[name = string("query_states_121_groups_0"), val = int32(1)]; tensor query_states_121_cast_fp16 = conv(dilations = query_states_121_dilations_0, groups = query_states_121_groups_0, pad = query_states_121_pad_0, pad_type = query_states_121_pad_type_0, strides = query_states_121_strides_0, weight = layers_20_self_attn_q_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("query_states_121_cast_fp16")]; tensor key_states_201_strides_0 = const()[name = string("key_states_201_strides_0"), val = tensor([1, 1])]; string key_states_201_pad_type_0 = const()[name = string("key_states_201_pad_type_0"), val = string("valid")]; tensor key_states_201_pad_0 = const()[name = string("key_states_201_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_201_dilations_0 = const()[name = string("key_states_201_dilations_0"), val = tensor([1, 1])]; int32 key_states_201_groups_0 = const()[name = string("key_states_201_groups_0"), val = int32(1)]; tensor key_states_201_cast_fp16 = conv(dilations = key_states_201_dilations_0, groups = key_states_201_groups_0, pad = key_states_201_pad_0, pad_type = key_states_201_pad_type_0, strides = key_states_201_strides_0, weight = layers_20_self_attn_k_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("key_states_201_cast_fp16")]; tensor layers_20_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_20_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501498624)))]; tensor value_states_121_strides_0 = const()[name = string("value_states_121_strides_0"), val = tensor([1, 1])]; string value_states_121_pad_type_0 = const()[name = string("value_states_121_pad_type_0"), val = string("valid")]; tensor value_states_121_pad_0 = const()[name = string("value_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_121_dilations_0 = const()[name = string("value_states_121_dilations_0"), val = tensor([1, 1])]; int32 value_states_121_groups_0 = const()[name = string("value_states_121_groups_0"), val = int32(1)]; tensor value_states_121_cast_fp16 = conv(dilations = value_states_121_dilations_0, groups = value_states_121_groups_0, pad = value_states_121_pad_0, pad_type = value_states_121_pad_type_0, strides = value_states_121_strides_0, weight = layers_20_self_attn_v_proj_weight_to_fp16, x = var_7347_cast_fp16_0)[name = string("value_states_121_cast_fp16")]; tensor concat_240x = const()[name = string("concat_240x"), val = tensor([1, 16, 128, -1])]; tensor x_201_cast_fp16 = reshape(shape = concat_240x, x = query_states_121_cast_fp16)[name = string("x_201_cast_fp16")]; tensor concat_241x = const()[name = string("concat_241x"), val = tensor([1, 2, 128, -1])]; tensor var_7404_cast_fp16 = reshape(shape = concat_241x, x = key_states_201_cast_fp16)[name = string("op_7404_cast_fp16")]; tensor concat_242x = const()[name = string("concat_242x"), val = tensor([1, 2, 128, -1])]; tensor var_7411_cast_fp16 = reshape(shape = concat_242x, x = value_states_121_cast_fp16)[name = string("op_7411_cast_fp16")]; tensor var_7415_cast_fp16 = mul(x = x_201_cast_fp16, y = var_869_cast_fp16)[name = string("op_7415_cast_fp16")]; tensor var_7416_split_sizes_0 = const()[name = string("op_7416_split_sizes_0"), val = tensor([64, 64])]; int32 var_7416_axis_0 = const()[name = string("op_7416_axis_0"), val = int32(-2)]; tensor var_7416_cast_fp16_0, tensor var_7416_cast_fp16_1 = split(axis = var_7416_axis_0, split_sizes = var_7416_split_sizes_0, x = x_201_cast_fp16)[name = string("op_7416_cast_fp16")]; fp16 const_202_promoted_to_fp16 = const()[name = string("const_202_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7418_cast_fp16 = mul(x = var_7416_cast_fp16_1, y = const_202_promoted_to_fp16)[name = string("op_7418_cast_fp16")]; int32 var_7420 = const()[name = string("op_7420"), val = int32(-2)]; bool var_7421_interleave_0 = const()[name = string("op_7421_interleave_0"), val = bool(false)]; tensor var_7421_cast_fp16 = concat(axis = var_7420, interleave = var_7421_interleave_0, values = (var_7418_cast_fp16, var_7416_cast_fp16_0))[name = string("op_7421_cast_fp16")]; tensor var_7422_cast_fp16 = mul(x = var_7421_cast_fp16, y = var_878_cast_fp16)[name = string("op_7422_cast_fp16")]; tensor query_states_123_cast_fp16 = add(x = var_7415_cast_fp16, y = var_7422_cast_fp16)[name = string("query_states_123_cast_fp16")]; tensor var_7428_cast_fp16 = mul(x = var_7404_cast_fp16, y = var_869_cast_fp16)[name = string("op_7428_cast_fp16")]; tensor var_7429_split_sizes_0 = const()[name = string("op_7429_split_sizes_0"), val = tensor([64, 64])]; int32 var_7429_axis_0 = const()[name = string("op_7429_axis_0"), val = int32(-2)]; tensor var_7429_cast_fp16_0, tensor var_7429_cast_fp16_1 = split(axis = var_7429_axis_0, split_sizes = var_7429_split_sizes_0, x = var_7404_cast_fp16)[name = string("op_7429_cast_fp16")]; fp16 const_203_promoted_to_fp16 = const()[name = string("const_203_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7431_cast_fp16 = mul(x = var_7429_cast_fp16_1, y = const_203_promoted_to_fp16)[name = string("op_7431_cast_fp16")]; int32 var_7433 = const()[name = string("op_7433"), val = int32(-2)]; bool var_7434_interleave_0 = const()[name = string("op_7434_interleave_0"), val = bool(false)]; tensor var_7434_cast_fp16 = concat(axis = var_7433, interleave = var_7434_interleave_0, values = (var_7431_cast_fp16, var_7429_cast_fp16_0))[name = string("op_7434_cast_fp16")]; tensor var_7435_cast_fp16 = mul(x = var_7434_cast_fp16, y = var_878_cast_fp16)[name = string("op_7435_cast_fp16")]; tensor key_states_205_cast_fp16 = add(x = var_7428_cast_fp16, y = var_7435_cast_fp16)[name = string("key_states_205_cast_fp16")]; tensor expand_dims_240 = const()[name = string("expand_dims_240"), val = tensor([20])]; tensor expand_dims_241 = const()[name = string("expand_dims_241"), val = tensor([0])]; tensor expand_dims_243 = const()[name = string("expand_dims_243"), val = tensor([0])]; int32 concat_245_axis_0 = const()[name = string("concat_245_axis_0"), val = int32(0)]; bool concat_245_interleave_0 = const()[name = string("concat_245_interleave_0"), val = bool(false)]; tensor concat_245 = concat(axis = concat_245_axis_0, interleave = concat_245_interleave_0, values = (expand_dims_240, expand_dims_241, position_id, expand_dims_243))[name = string("concat_245")]; tensor expand_dims_244 = const()[name = string("expand_dims_244"), val = tensor([21])]; tensor concat_246_values1_0 = const()[name = string("concat_246_values1_0"), val = tensor([0])]; tensor concat_246_values3_0 = const()[name = string("concat_246_values3_0"), val = tensor([0])]; int32 concat_246_axis_0 = const()[name = string("concat_246_axis_0"), val = int32(0)]; bool concat_246_interleave_0 = const()[name = string("concat_246_interleave_0"), val = bool(false)]; tensor concat_246 = concat(axis = concat_246_axis_0, interleave = concat_246_interleave_0, values = (expand_dims_244, concat_246_values1_0, cache_position_end, concat_246_values3_0))[name = string("concat_246")]; tensor key_states_207_perm_0 = const()[name = string("key_states_207_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_21_stride_0 = const()[name = string("key_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_207_cast_fp16 = transpose(perm = key_states_207_perm_0, x = key_states_205_cast_fp16)[name = string("transpose_109")]; tensor key_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = key_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = key_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_21_squeeze_mask_0, stride = key_cache_internal_tensor_assign_21_stride_0, update = key_states_207_cast_fp16, x = coreml_update_state_94)[name = string("key_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_21_cast_fp16, input = key_cache)[name = string("coreml_update_state_96_write_state")]; tensor coreml_update_state_96 = read_state(input = key_cache)[name = string("coreml_update_state_96")]; tensor value_states_123_perm_0 = const()[name = string("value_states_123_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_21_stride_0 = const()[name = string("value_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_123_cast_fp16 = transpose(perm = value_states_123_perm_0, x = var_7411_cast_fp16)[name = string("transpose_108")]; tensor value_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = value_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = value_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_21_squeeze_mask_0, stride = value_cache_internal_tensor_assign_21_stride_0, update = value_states_123_cast_fp16, x = coreml_update_state_95)[name = string("value_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_21_cast_fp16, input = value_cache)[name = string("coreml_update_state_97_write_state")]; tensor coreml_update_state_97 = read_state(input = value_cache)[name = string("coreml_update_state_97")]; tensor var_7505_begin_0 = const()[name = string("op_7505_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7505_end_0 = const()[name = string("op_7505_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7505_end_mask_0 = const()[name = string("op_7505_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7505_cast_fp16 = slice_by_index(begin = var_7505_begin_0, end = var_7505_end_0, end_mask = var_7505_end_mask_0, x = coreml_update_state_96)[name = string("op_7505_cast_fp16")]; tensor tile_40 = const()[name = string("tile_40"), val = tensor([1, 1])]; int32 var_7508_axis_0 = const()[name = string("op_7508_axis_0"), val = int32(1)]; tensor var_7508_cast_fp16_0, tensor var_7508_cast_fp16_1 = split(axis = var_7508_axis_0, split_sizes = tile_40, x = var_7505_cast_fp16)[name = string("op_7508_cast_fp16")]; tensor var_7515_begin_0 = const()[name = string("op_7515_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7515_end_0 = const()[name = string("op_7515_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7515_end_mask_0 = const()[name = string("op_7515_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7515_cast_fp16 = slice_by_index(begin = var_7515_begin_0, end = var_7515_end_0, end_mask = var_7515_end_mask_0, x = coreml_update_state_97)[name = string("op_7515_cast_fp16")]; tensor tile_41 = const()[name = string("tile_41"), val = tensor([1, 1])]; int32 var_7518_axis_0 = const()[name = string("op_7518_axis_0"), val = int32(1)]; tensor var_7518_cast_fp16_0, tensor var_7518_cast_fp16_1 = split(axis = var_7518_axis_0, split_sizes = tile_41, x = var_7515_cast_fp16)[name = string("op_7518_cast_fp16")]; tensor var_7521_split_sizes_0 = const()[name = string("op_7521_split_sizes_0"), val = tensor([8, 8])]; int32 var_7521_axis_0 = const()[name = string("op_7521_axis_0"), val = int32(1)]; tensor var_7521_0, tensor var_7521_1 = split(axis = var_7521_axis_0, split_sizes = var_7521_split_sizes_0, x = query_states_123_cast_fp16)[name = string("op_7521")]; bool attn_weights_321_transpose_x_0 = const()[name = string("attn_weights_321_transpose_x_0"), val = bool(false)]; bool attn_weights_321_transpose_y_0 = const()[name = string("attn_weights_321_transpose_y_0"), val = bool(false)]; tensor attn_weights_321_cast_fp16 = matmul(transpose_x = attn_weights_321_transpose_x_0, transpose_y = attn_weights_321_transpose_y_0, x = var_7508_cast_fp16_0, y = var_7521_0)[name = string("attn_weights_321_cast_fp16")]; fp16 var_7524_to_fp16 = const()[name = string("op_7524_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_323_cast_fp16 = mul(x = attn_weights_321_cast_fp16, y = var_7524_to_fp16)[name = string("attn_weights_323_cast_fp16")]; tensor attn_weights_325_cast_fp16 = add(x = attn_weights_323_cast_fp16, y = attn_mask_1)[name = string("attn_weights_325_cast_fp16")]; int32 var_7528 = const()[name = string("op_7528"), val = int32(-2)]; tensor attn_weights_327_cast_fp16 = softmax(axis = var_7528, x = attn_weights_325_cast_fp16)[name = string("attn_weights_327_cast_fp16")]; bool var_7534_transpose_x_1 = const()[name = string("op_7534_transpose_x_1"), val = bool(true)]; bool var_7534_transpose_y_1 = const()[name = string("op_7534_transpose_y_1"), val = bool(false)]; tensor var_7534_cast_fp16 = matmul(transpose_x = var_7534_transpose_x_1, transpose_y = var_7534_transpose_y_1, x = attn_weights_327_cast_fp16, y = var_7518_cast_fp16_0)[name = string("op_7534_cast_fp16")]; bool attn_weights_329_transpose_x_0 = const()[name = string("attn_weights_329_transpose_x_0"), val = bool(false)]; bool attn_weights_329_transpose_y_0 = const()[name = string("attn_weights_329_transpose_y_0"), val = bool(false)]; tensor attn_weights_329_cast_fp16 = matmul(transpose_x = attn_weights_329_transpose_x_0, transpose_y = attn_weights_329_transpose_y_0, x = var_7508_cast_fp16_1, y = var_7521_1)[name = string("attn_weights_329_cast_fp16")]; fp16 var_7536_to_fp16 = const()[name = string("op_7536_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_331_cast_fp16 = mul(x = attn_weights_329_cast_fp16, y = var_7536_to_fp16)[name = string("attn_weights_331_cast_fp16")]; tensor attn_weights_333_cast_fp16 = add(x = attn_weights_331_cast_fp16, y = attn_mask_1)[name = string("attn_weights_333_cast_fp16")]; int32 var_7540 = const()[name = string("op_7540"), val = int32(-2)]; tensor attn_weights_335_cast_fp16 = softmax(axis = var_7540, x = attn_weights_333_cast_fp16)[name = string("attn_weights_335_cast_fp16")]; bool attn_output_161_transpose_x_1 = const()[name = string("attn_output_161_transpose_x_1"), val = bool(true)]; bool attn_output_161_transpose_y_1 = const()[name = string("attn_output_161_transpose_y_1"), val = bool(false)]; tensor attn_output_161_cast_fp16 = matmul(transpose_x = attn_output_161_transpose_x_1, transpose_y = attn_output_161_transpose_y_1, x = attn_weights_335_cast_fp16, y = var_7518_cast_fp16_1)[name = string("attn_output_161_cast_fp16")]; int32 var_7548 = const()[name = string("op_7548"), val = int32(1)]; bool attn_output_163_interleave_0 = const()[name = string("attn_output_163_interleave_0"), val = bool(false)]; tensor attn_output_163_cast_fp16 = concat(axis = var_7548, interleave = attn_output_163_interleave_0, values = (var_7534_cast_fp16, attn_output_161_cast_fp16))[name = string("attn_output_163_cast_fp16")]; tensor var_7552_perm_0 = const()[name = string("op_7552_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_251x = const()[name = string("concat_251x"), val = tensor([1, 2048, 1, -1])]; tensor var_7552_cast_fp16 = transpose(perm = var_7552_perm_0, x = attn_output_163_cast_fp16)[name = string("transpose_107")]; tensor attn_output_167_cast_fp16 = reshape(shape = concat_251x, x = var_7552_cast_fp16)[name = string("attn_output_167_cast_fp16")]; tensor hidden_states_203_strides_0 = const()[name = string("hidden_states_203_strides_0"), val = tensor([1, 1])]; string hidden_states_203_pad_type_0 = const()[name = string("hidden_states_203_pad_type_0"), val = string("valid")]; tensor hidden_states_203_pad_0 = const()[name = string("hidden_states_203_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_203_dilations_0 = const()[name = string("hidden_states_203_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_203_groups_0 = const()[name = string("hidden_states_203_groups_0"), val = int32(1)]; tensor hidden_states_203_cast_fp16 = conv(dilations = hidden_states_203_dilations_0, groups = hidden_states_203_groups_0, pad = hidden_states_203_pad_0, pad_type = hidden_states_203_pad_type_0, strides = hidden_states_203_strides_0, weight = layers_20_self_attn_o_proj_weight_cast_fp16, x = attn_output_167_cast_fp16)[name = string("hidden_states_203_cast_fp16")]; tensor hidden_states_205_cast_fp16 = add(x = hidden_states_199_cast_fp16, y = hidden_states_203_cast_fp16)[name = string("hidden_states_205_cast_fp16")]; fp16 const_208_promoted_to_fp16 = const()[name = string("const_208_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7585_cast_fp16 = mul(x = hidden_states_205_cast_fp16, y = const_208_promoted_to_fp16)[name = string("op_7585_cast_fp16")]; int32 var_7583 = const()[name = string("op_7583"), val = int32(1)]; bool doubled_165_interleave_0 = const()[name = string("doubled_165_interleave_0"), val = bool(false)]; tensor doubled_165_cast_fp16 = concat(axis = var_7583, interleave = doubled_165_interleave_0, values = (hidden_states_205_cast_fp16, var_7585_cast_fp16))[name = string("doubled_165_cast_fp16")]; tensor out_83_axes_0 = const()[name = string("out_83_axes_0"), val = tensor([1])]; tensor out_83_gamma_0_to_fp16 = const()[name = string("out_83_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502547264)))]; fp16 var_7595_to_fp16 = const()[name = string("op_7595_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_83_cast_fp16 = layer_norm(axes = out_83_axes_0, epsilon = var_7595_to_fp16, gamma = out_83_gamma_0_to_fp16, x = doubled_165_cast_fp16)[name = string("out_83_cast_fp16")]; tensor var_7606_split_sizes_0 = const()[name = string("op_7606_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7606_axis_0 = const()[name = string("op_7606_axis_0"), val = int32(1)]; tensor var_7606_cast_fp16_0, tensor var_7606_cast_fp16_1 = split(axis = var_7606_axis_0, split_sizes = var_7606_split_sizes_0, x = out_83_cast_fp16)[name = string("op_7606_cast_fp16")]; tensor input_41_strides_0 = const()[name = string("input_41_strides_0"), val = tensor([1, 1])]; string input_41_pad_type_0 = const()[name = string("input_41_pad_type_0"), val = string("valid")]; tensor input_41_pad_0 = const()[name = string("input_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_41_dilations_0 = const()[name = string("input_41_dilations_0"), val = tensor([1, 1])]; int32 input_41_groups_0 = const()[name = string("input_41_groups_0"), val = int32(1)]; tensor input_41_cast_fp16 = conv(dilations = input_41_dilations_0, groups = input_41_groups_0, pad = input_41_pad_0, pad_type = input_41_pad_type_0, strides = input_41_strides_0, weight = layers_20_mlp_gate_proj_weight_cast_fp16, x = var_7606_cast_fp16_0)[name = string("input_41_cast_fp16")]; tensor var_7623_cast_fp16 = silu(x = input_41_cast_fp16)[name = string("op_7623_cast_fp16")]; tensor layers_20_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_20_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502555520)))]; tensor var_7629_strides_0 = const()[name = string("op_7629_strides_0"), val = tensor([1, 1])]; string var_7629_pad_type_0 = const()[name = string("op_7629_pad_type_0"), val = string("valid")]; tensor var_7629_pad_0 = const()[name = string("op_7629_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7629_dilations_0 = const()[name = string("op_7629_dilations_0"), val = tensor([1, 1])]; int32 var_7629_groups_0 = const()[name = string("op_7629_groups_0"), val = int32(1)]; tensor var_7629_cast_fp16 = conv(dilations = var_7629_dilations_0, groups = var_7629_groups_0, pad = var_7629_pad_0, pad_type = var_7629_pad_type_0, strides = var_7629_strides_0, weight = layers_20_mlp_up_proj_weight_to_fp16, x = var_7606_cast_fp16_0)[name = string("op_7629_cast_fp16")]; tensor x_209_cast_fp16 = mul(x = var_7623_cast_fp16, y = var_7629_cast_fp16)[name = string("x_209_cast_fp16")]; tensor hidden_states_207_strides_0 = const()[name = string("hidden_states_207_strides_0"), val = tensor([1, 1])]; string hidden_states_207_pad_type_0 = const()[name = string("hidden_states_207_pad_type_0"), val = string("valid")]; tensor hidden_states_207_pad_0 = const()[name = string("hidden_states_207_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_207_dilations_0 = const()[name = string("hidden_states_207_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_207_groups_0 = const()[name = string("hidden_states_207_groups_0"), val = int32(1)]; tensor hidden_states_207_cast_fp16 = conv(dilations = hidden_states_207_dilations_0, groups = hidden_states_207_groups_0, pad = hidden_states_207_pad_0, pad_type = hidden_states_207_pad_type_0, strides = hidden_states_207_strides_0, weight = layers_20_mlp_down_proj_weight_cast_fp16, x = x_209_cast_fp16)[name = string("hidden_states_207_cast_fp16")]; tensor hidden_states_209_cast_fp16 = add(x = hidden_states_205_cast_fp16, y = hidden_states_207_cast_fp16)[name = string("hidden_states_209_cast_fp16")]; fp16 const_210_promoted_to_fp16 = const()[name = string("const_210_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7647_cast_fp16 = mul(x = hidden_states_209_cast_fp16, y = const_210_promoted_to_fp16)[name = string("op_7647_cast_fp16")]; int32 var_7645 = const()[name = string("op_7645"), val = int32(1)]; bool doubled_169_interleave_0 = const()[name = string("doubled_169_interleave_0"), val = bool(false)]; tensor doubled_169_cast_fp16 = concat(axis = var_7645, interleave = doubled_169_interleave_0, values = (hidden_states_209_cast_fp16, var_7647_cast_fp16))[name = string("doubled_169_cast_fp16")]; tensor out_85_axes_0 = const()[name = string("out_85_axes_0"), val = tensor([1])]; tensor out_85_gamma_0_to_fp16 = const()[name = string("out_85_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527721408)))]; fp16 var_7657_to_fp16 = const()[name = string("op_7657_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_85_cast_fp16 = layer_norm(axes = out_85_axes_0, epsilon = var_7657_to_fp16, gamma = out_85_gamma_0_to_fp16, x = doubled_169_cast_fp16)[name = string("out_85_cast_fp16")]; tensor var_7668_split_sizes_0 = const()[name = string("op_7668_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7668_axis_0 = const()[name = string("op_7668_axis_0"), val = int32(1)]; tensor var_7668_cast_fp16_0, tensor var_7668_cast_fp16_1 = split(axis = var_7668_axis_0, split_sizes = var_7668_split_sizes_0, x = out_85_cast_fp16)[name = string("op_7668_cast_fp16")]; tensor query_states_127_strides_0 = const()[name = string("query_states_127_strides_0"), val = tensor([1, 1])]; string query_states_127_pad_type_0 = const()[name = string("query_states_127_pad_type_0"), val = string("valid")]; tensor query_states_127_pad_0 = const()[name = string("query_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_127_dilations_0 = const()[name = string("query_states_127_dilations_0"), val = tensor([1, 1])]; int32 query_states_127_groups_0 = const()[name = string("query_states_127_groups_0"), val = int32(1)]; tensor query_states_127_cast_fp16 = conv(dilations = query_states_127_dilations_0, groups = query_states_127_groups_0, pad = query_states_127_pad_0, pad_type = query_states_127_pad_type_0, strides = query_states_127_strides_0, weight = layers_21_self_attn_q_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("query_states_127_cast_fp16")]; tensor key_states_211_strides_0 = const()[name = string("key_states_211_strides_0"), val = tensor([1, 1])]; string key_states_211_pad_type_0 = const()[name = string("key_states_211_pad_type_0"), val = string("valid")]; tensor key_states_211_pad_0 = const()[name = string("key_states_211_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_211_dilations_0 = const()[name = string("key_states_211_dilations_0"), val = tensor([1, 1])]; int32 key_states_211_groups_0 = const()[name = string("key_states_211_groups_0"), val = int32(1)]; tensor key_states_211_cast_fp16 = conv(dilations = key_states_211_dilations_0, groups = key_states_211_groups_0, pad = key_states_211_pad_0, pad_type = key_states_211_pad_type_0, strides = key_states_211_strides_0, weight = layers_21_self_attn_k_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("key_states_211_cast_fp16")]; tensor layers_21_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_21_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527729664)))]; tensor value_states_127_strides_0 = const()[name = string("value_states_127_strides_0"), val = tensor([1, 1])]; string value_states_127_pad_type_0 = const()[name = string("value_states_127_pad_type_0"), val = string("valid")]; tensor value_states_127_pad_0 = const()[name = string("value_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_127_dilations_0 = const()[name = string("value_states_127_dilations_0"), val = tensor([1, 1])]; int32 value_states_127_groups_0 = const()[name = string("value_states_127_groups_0"), val = int32(1)]; tensor value_states_127_cast_fp16 = conv(dilations = value_states_127_dilations_0, groups = value_states_127_groups_0, pad = value_states_127_pad_0, pad_type = value_states_127_pad_type_0, strides = value_states_127_strides_0, weight = layers_21_self_attn_v_proj_weight_to_fp16, x = var_7668_cast_fp16_0)[name = string("value_states_127_cast_fp16")]; tensor concat_252x = const()[name = string("concat_252x"), val = tensor([1, 16, 128, -1])]; tensor x_211_cast_fp16 = reshape(shape = concat_252x, x = query_states_127_cast_fp16)[name = string("x_211_cast_fp16")]; tensor concat_253x = const()[name = string("concat_253x"), val = tensor([1, 2, 128, -1])]; tensor var_7725_cast_fp16 = reshape(shape = concat_253x, x = key_states_211_cast_fp16)[name = string("op_7725_cast_fp16")]; tensor concat_254x = const()[name = string("concat_254x"), val = tensor([1, 2, 128, -1])]; tensor var_7732_cast_fp16 = reshape(shape = concat_254x, x = value_states_127_cast_fp16)[name = string("op_7732_cast_fp16")]; tensor var_7736_cast_fp16 = mul(x = x_211_cast_fp16, y = var_869_cast_fp16)[name = string("op_7736_cast_fp16")]; tensor var_7737_split_sizes_0 = const()[name = string("op_7737_split_sizes_0"), val = tensor([64, 64])]; int32 var_7737_axis_0 = const()[name = string("op_7737_axis_0"), val = int32(-2)]; tensor var_7737_cast_fp16_0, tensor var_7737_cast_fp16_1 = split(axis = var_7737_axis_0, split_sizes = var_7737_split_sizes_0, x = x_211_cast_fp16)[name = string("op_7737_cast_fp16")]; fp16 const_212_promoted_to_fp16 = const()[name = string("const_212_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7739_cast_fp16 = mul(x = var_7737_cast_fp16_1, y = const_212_promoted_to_fp16)[name = string("op_7739_cast_fp16")]; int32 var_7741 = const()[name = string("op_7741"), val = int32(-2)]; bool var_7742_interleave_0 = const()[name = string("op_7742_interleave_0"), val = bool(false)]; tensor var_7742_cast_fp16 = concat(axis = var_7741, interleave = var_7742_interleave_0, values = (var_7739_cast_fp16, var_7737_cast_fp16_0))[name = string("op_7742_cast_fp16")]; tensor var_7743_cast_fp16 = mul(x = var_7742_cast_fp16, y = var_878_cast_fp16)[name = string("op_7743_cast_fp16")]; tensor query_states_129_cast_fp16 = add(x = var_7736_cast_fp16, y = var_7743_cast_fp16)[name = string("query_states_129_cast_fp16")]; tensor var_7749_cast_fp16 = mul(x = var_7725_cast_fp16, y = var_869_cast_fp16)[name = string("op_7749_cast_fp16")]; tensor var_7750_split_sizes_0 = const()[name = string("op_7750_split_sizes_0"), val = tensor([64, 64])]; int32 var_7750_axis_0 = const()[name = string("op_7750_axis_0"), val = int32(-2)]; tensor var_7750_cast_fp16_0, tensor var_7750_cast_fp16_1 = split(axis = var_7750_axis_0, split_sizes = var_7750_split_sizes_0, x = var_7725_cast_fp16)[name = string("op_7750_cast_fp16")]; fp16 const_213_promoted_to_fp16 = const()[name = string("const_213_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7752_cast_fp16 = mul(x = var_7750_cast_fp16_1, y = const_213_promoted_to_fp16)[name = string("op_7752_cast_fp16")]; int32 var_7754 = const()[name = string("op_7754"), val = int32(-2)]; bool var_7755_interleave_0 = const()[name = string("op_7755_interleave_0"), val = bool(false)]; tensor var_7755_cast_fp16 = concat(axis = var_7754, interleave = var_7755_interleave_0, values = (var_7752_cast_fp16, var_7750_cast_fp16_0))[name = string("op_7755_cast_fp16")]; tensor var_7756_cast_fp16 = mul(x = var_7755_cast_fp16, y = var_878_cast_fp16)[name = string("op_7756_cast_fp16")]; tensor key_states_215_cast_fp16 = add(x = var_7749_cast_fp16, y = var_7756_cast_fp16)[name = string("key_states_215_cast_fp16")]; tensor expand_dims_252 = const()[name = string("expand_dims_252"), val = tensor([21])]; tensor expand_dims_253 = const()[name = string("expand_dims_253"), val = tensor([0])]; tensor expand_dims_255 = const()[name = string("expand_dims_255"), val = tensor([0])]; int32 concat_257_axis_0 = const()[name = string("concat_257_axis_0"), val = int32(0)]; bool concat_257_interleave_0 = const()[name = string("concat_257_interleave_0"), val = bool(false)]; tensor concat_257 = concat(axis = concat_257_axis_0, interleave = concat_257_interleave_0, values = (expand_dims_252, expand_dims_253, position_id, expand_dims_255))[name = string("concat_257")]; tensor expand_dims_256 = const()[name = string("expand_dims_256"), val = tensor([22])]; tensor concat_258_values1_0 = const()[name = string("concat_258_values1_0"), val = tensor([0])]; tensor concat_258_values3_0 = const()[name = string("concat_258_values3_0"), val = tensor([0])]; int32 concat_258_axis_0 = const()[name = string("concat_258_axis_0"), val = int32(0)]; bool concat_258_interleave_0 = const()[name = string("concat_258_interleave_0"), val = bool(false)]; tensor concat_258 = concat(axis = concat_258_axis_0, interleave = concat_258_interleave_0, values = (expand_dims_256, concat_258_values1_0, cache_position_end, concat_258_values3_0))[name = string("concat_258")]; tensor key_states_217_perm_0 = const()[name = string("key_states_217_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_22_stride_0 = const()[name = string("key_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_217_cast_fp16 = transpose(perm = key_states_217_perm_0, x = key_states_215_cast_fp16)[name = string("transpose_106")]; tensor key_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = key_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = key_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_22_squeeze_mask_0, stride = key_cache_internal_tensor_assign_22_stride_0, update = key_states_217_cast_fp16, x = coreml_update_state_96)[name = string("key_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_22_cast_fp16, input = key_cache)[name = string("coreml_update_state_98_write_state")]; tensor coreml_update_state_98 = read_state(input = key_cache)[name = string("coreml_update_state_98")]; tensor value_states_129_perm_0 = const()[name = string("value_states_129_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_22_stride_0 = const()[name = string("value_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_129_cast_fp16 = transpose(perm = value_states_129_perm_0, x = var_7732_cast_fp16)[name = string("transpose_105")]; tensor value_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = value_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = value_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_22_squeeze_mask_0, stride = value_cache_internal_tensor_assign_22_stride_0, update = value_states_129_cast_fp16, x = coreml_update_state_97)[name = string("value_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_22_cast_fp16, input = value_cache)[name = string("coreml_update_state_99_write_state")]; tensor coreml_update_state_99 = read_state(input = value_cache)[name = string("coreml_update_state_99")]; tensor var_7826_begin_0 = const()[name = string("op_7826_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7826_end_0 = const()[name = string("op_7826_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7826_end_mask_0 = const()[name = string("op_7826_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7826_cast_fp16 = slice_by_index(begin = var_7826_begin_0, end = var_7826_end_0, end_mask = var_7826_end_mask_0, x = coreml_update_state_98)[name = string("op_7826_cast_fp16")]; tensor tile_42 = const()[name = string("tile_42"), val = tensor([1, 1])]; int32 var_7829_axis_0 = const()[name = string("op_7829_axis_0"), val = int32(1)]; tensor var_7829_cast_fp16_0, tensor var_7829_cast_fp16_1 = split(axis = var_7829_axis_0, split_sizes = tile_42, x = var_7826_cast_fp16)[name = string("op_7829_cast_fp16")]; tensor var_7836_begin_0 = const()[name = string("op_7836_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7836_end_0 = const()[name = string("op_7836_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7836_end_mask_0 = const()[name = string("op_7836_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7836_cast_fp16 = slice_by_index(begin = var_7836_begin_0, end = var_7836_end_0, end_mask = var_7836_end_mask_0, x = coreml_update_state_99)[name = string("op_7836_cast_fp16")]; tensor tile_43 = const()[name = string("tile_43"), val = tensor([1, 1])]; int32 var_7839_axis_0 = const()[name = string("op_7839_axis_0"), val = int32(1)]; tensor var_7839_cast_fp16_0, tensor var_7839_cast_fp16_1 = split(axis = var_7839_axis_0, split_sizes = tile_43, x = var_7836_cast_fp16)[name = string("op_7839_cast_fp16")]; tensor var_7842_split_sizes_0 = const()[name = string("op_7842_split_sizes_0"), val = tensor([8, 8])]; int32 var_7842_axis_0 = const()[name = string("op_7842_axis_0"), val = int32(1)]; tensor var_7842_0, tensor var_7842_1 = split(axis = var_7842_axis_0, split_sizes = var_7842_split_sizes_0, x = query_states_129_cast_fp16)[name = string("op_7842")]; bool attn_weights_337_transpose_x_0 = const()[name = string("attn_weights_337_transpose_x_0"), val = bool(false)]; bool attn_weights_337_transpose_y_0 = const()[name = string("attn_weights_337_transpose_y_0"), val = bool(false)]; tensor attn_weights_337_cast_fp16 = matmul(transpose_x = attn_weights_337_transpose_x_0, transpose_y = attn_weights_337_transpose_y_0, x = var_7829_cast_fp16_0, y = var_7842_0)[name = string("attn_weights_337_cast_fp16")]; fp16 var_7845_to_fp16 = const()[name = string("op_7845_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_339_cast_fp16 = mul(x = attn_weights_337_cast_fp16, y = var_7845_to_fp16)[name = string("attn_weights_339_cast_fp16")]; tensor attn_weights_341_cast_fp16 = add(x = attn_weights_339_cast_fp16, y = attn_mask_1)[name = string("attn_weights_341_cast_fp16")]; int32 var_7849 = const()[name = string("op_7849"), val = int32(-2)]; tensor attn_weights_343_cast_fp16 = softmax(axis = var_7849, x = attn_weights_341_cast_fp16)[name = string("attn_weights_343_cast_fp16")]; bool var_7855_transpose_x_1 = const()[name = string("op_7855_transpose_x_1"), val = bool(true)]; bool var_7855_transpose_y_1 = const()[name = string("op_7855_transpose_y_1"), val = bool(false)]; tensor var_7855_cast_fp16 = matmul(transpose_x = var_7855_transpose_x_1, transpose_y = var_7855_transpose_y_1, x = attn_weights_343_cast_fp16, y = var_7839_cast_fp16_0)[name = string("op_7855_cast_fp16")]; bool attn_weights_345_transpose_x_0 = const()[name = string("attn_weights_345_transpose_x_0"), val = bool(false)]; bool attn_weights_345_transpose_y_0 = const()[name = string("attn_weights_345_transpose_y_0"), val = bool(false)]; tensor attn_weights_345_cast_fp16 = matmul(transpose_x = attn_weights_345_transpose_x_0, transpose_y = attn_weights_345_transpose_y_0, x = var_7829_cast_fp16_1, y = var_7842_1)[name = string("attn_weights_345_cast_fp16")]; fp16 var_7857_to_fp16 = const()[name = string("op_7857_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_347_cast_fp16 = mul(x = attn_weights_345_cast_fp16, y = var_7857_to_fp16)[name = string("attn_weights_347_cast_fp16")]; tensor attn_weights_349_cast_fp16 = add(x = attn_weights_347_cast_fp16, y = attn_mask_1)[name = string("attn_weights_349_cast_fp16")]; int32 var_7861 = const()[name = string("op_7861"), val = int32(-2)]; tensor attn_weights_351_cast_fp16 = softmax(axis = var_7861, x = attn_weights_349_cast_fp16)[name = string("attn_weights_351_cast_fp16")]; bool attn_output_169_transpose_x_1 = const()[name = string("attn_output_169_transpose_x_1"), val = bool(true)]; bool attn_output_169_transpose_y_1 = const()[name = string("attn_output_169_transpose_y_1"), val = bool(false)]; tensor attn_output_169_cast_fp16 = matmul(transpose_x = attn_output_169_transpose_x_1, transpose_y = attn_output_169_transpose_y_1, x = attn_weights_351_cast_fp16, y = var_7839_cast_fp16_1)[name = string("attn_output_169_cast_fp16")]; int32 var_7869 = const()[name = string("op_7869"), val = int32(1)]; bool attn_output_171_interleave_0 = const()[name = string("attn_output_171_interleave_0"), val = bool(false)]; tensor attn_output_171_cast_fp16 = concat(axis = var_7869, interleave = attn_output_171_interleave_0, values = (var_7855_cast_fp16, attn_output_169_cast_fp16))[name = string("attn_output_171_cast_fp16")]; tensor var_7873_perm_0 = const()[name = string("op_7873_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_263x = const()[name = string("concat_263x"), val = tensor([1, 2048, 1, -1])]; tensor var_7873_cast_fp16 = transpose(perm = var_7873_perm_0, x = attn_output_171_cast_fp16)[name = string("transpose_104")]; tensor attn_output_175_cast_fp16 = reshape(shape = concat_263x, x = var_7873_cast_fp16)[name = string("attn_output_175_cast_fp16")]; tensor hidden_states_213_strides_0 = const()[name = string("hidden_states_213_strides_0"), val = tensor([1, 1])]; string hidden_states_213_pad_type_0 = const()[name = string("hidden_states_213_pad_type_0"), val = string("valid")]; tensor hidden_states_213_pad_0 = const()[name = string("hidden_states_213_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_213_dilations_0 = const()[name = string("hidden_states_213_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_213_groups_0 = const()[name = string("hidden_states_213_groups_0"), val = int32(1)]; tensor hidden_states_213_cast_fp16 = conv(dilations = hidden_states_213_dilations_0, groups = hidden_states_213_groups_0, pad = hidden_states_213_pad_0, pad_type = hidden_states_213_pad_type_0, strides = hidden_states_213_strides_0, weight = layers_21_self_attn_o_proj_weight_cast_fp16, x = attn_output_175_cast_fp16)[name = string("hidden_states_213_cast_fp16")]; tensor hidden_states_215_cast_fp16 = add(x = hidden_states_209_cast_fp16, y = hidden_states_213_cast_fp16)[name = string("hidden_states_215_cast_fp16")]; fp16 const_218_promoted_to_fp16 = const()[name = string("const_218_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7906_cast_fp16 = mul(x = hidden_states_215_cast_fp16, y = const_218_promoted_to_fp16)[name = string("op_7906_cast_fp16")]; int32 var_7904 = const()[name = string("op_7904"), val = int32(1)]; bool doubled_173_interleave_0 = const()[name = string("doubled_173_interleave_0"), val = bool(false)]; tensor doubled_173_cast_fp16 = concat(axis = var_7904, interleave = doubled_173_interleave_0, values = (hidden_states_215_cast_fp16, var_7906_cast_fp16))[name = string("doubled_173_cast_fp16")]; tensor out_87_axes_0 = const()[name = string("out_87_axes_0"), val = tensor([1])]; tensor out_87_gamma_0_to_fp16 = const()[name = string("out_87_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528778304)))]; fp16 var_7916_to_fp16 = const()[name = string("op_7916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_87_cast_fp16 = layer_norm(axes = out_87_axes_0, epsilon = var_7916_to_fp16, gamma = out_87_gamma_0_to_fp16, x = doubled_173_cast_fp16)[name = string("out_87_cast_fp16")]; tensor var_7927_split_sizes_0 = const()[name = string("op_7927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7927_axis_0 = const()[name = string("op_7927_axis_0"), val = int32(1)]; tensor var_7927_cast_fp16_0, tensor var_7927_cast_fp16_1 = split(axis = var_7927_axis_0, split_sizes = var_7927_split_sizes_0, x = out_87_cast_fp16)[name = string("op_7927_cast_fp16")]; tensor input_43_strides_0 = const()[name = string("input_43_strides_0"), val = tensor([1, 1])]; string input_43_pad_type_0 = const()[name = string("input_43_pad_type_0"), val = string("valid")]; tensor input_43_pad_0 = const()[name = string("input_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_43_dilations_0 = const()[name = string("input_43_dilations_0"), val = tensor([1, 1])]; int32 input_43_groups_0 = const()[name = string("input_43_groups_0"), val = int32(1)]; tensor input_43_cast_fp16 = conv(dilations = input_43_dilations_0, groups = input_43_groups_0, pad = input_43_pad_0, pad_type = input_43_pad_type_0, strides = input_43_strides_0, weight = layers_21_mlp_gate_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("input_43_cast_fp16")]; tensor var_7944_cast_fp16 = silu(x = input_43_cast_fp16)[name = string("op_7944_cast_fp16")]; tensor var_7950_strides_0 = const()[name = string("op_7950_strides_0"), val = tensor([1, 1])]; string var_7950_pad_type_0 = const()[name = string("op_7950_pad_type_0"), val = string("valid")]; tensor var_7950_pad_0 = const()[name = string("op_7950_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7950_dilations_0 = const()[name = string("op_7950_dilations_0"), val = tensor([1, 1])]; int32 var_7950_groups_0 = const()[name = string("op_7950_groups_0"), val = int32(1)]; tensor var_7950_cast_fp16 = conv(dilations = var_7950_dilations_0, groups = var_7950_groups_0, pad = var_7950_pad_0, pad_type = var_7950_pad_type_0, strides = var_7950_strides_0, weight = layers_21_mlp_up_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("op_7950_cast_fp16")]; tensor x_219_cast_fp16 = mul(x = var_7944_cast_fp16, y = var_7950_cast_fp16)[name = string("x_219_cast_fp16")]; tensor hidden_states_217_strides_0 = const()[name = string("hidden_states_217_strides_0"), val = tensor([1, 1])]; string hidden_states_217_pad_type_0 = const()[name = string("hidden_states_217_pad_type_0"), val = string("valid")]; tensor hidden_states_217_pad_0 = const()[name = string("hidden_states_217_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_217_dilations_0 = const()[name = string("hidden_states_217_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_217_groups_0 = const()[name = string("hidden_states_217_groups_0"), val = int32(1)]; tensor hidden_states_217_cast_fp16 = conv(dilations = hidden_states_217_dilations_0, groups = hidden_states_217_groups_0, pad = hidden_states_217_pad_0, pad_type = hidden_states_217_pad_type_0, strides = hidden_states_217_strides_0, weight = layers_21_mlp_down_proj_weight_cast_fp16, x = x_219_cast_fp16)[name = string("hidden_states_217_cast_fp16")]; tensor hidden_states_219_cast_fp16 = add(x = hidden_states_215_cast_fp16, y = hidden_states_217_cast_fp16)[name = string("hidden_states_219_cast_fp16")]; fp16 const_220_promoted_to_fp16 = const()[name = string("const_220_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7968_cast_fp16 = mul(x = hidden_states_219_cast_fp16, y = const_220_promoted_to_fp16)[name = string("op_7968_cast_fp16")]; int32 var_7966 = const()[name = string("op_7966"), val = int32(1)]; bool doubled_177_interleave_0 = const()[name = string("doubled_177_interleave_0"), val = bool(false)]; tensor doubled_177_cast_fp16 = concat(axis = var_7966, interleave = doubled_177_interleave_0, values = (hidden_states_219_cast_fp16, var_7968_cast_fp16))[name = string("doubled_177_cast_fp16")]; tensor out_89_axes_0 = const()[name = string("out_89_axes_0"), val = tensor([1])]; tensor out_89_gamma_0_to_fp16 = const()[name = string("out_89_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528786560)))]; fp16 var_7978_to_fp16 = const()[name = string("op_7978_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_89_cast_fp16 = layer_norm(axes = out_89_axes_0, epsilon = var_7978_to_fp16, gamma = out_89_gamma_0_to_fp16, x = doubled_177_cast_fp16)[name = string("out_89_cast_fp16")]; tensor var_7989_split_sizes_0 = const()[name = string("op_7989_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7989_axis_0 = const()[name = string("op_7989_axis_0"), val = int32(1)]; tensor var_7989_cast_fp16_0, tensor var_7989_cast_fp16_1 = split(axis = var_7989_axis_0, split_sizes = var_7989_split_sizes_0, x = out_89_cast_fp16)[name = string("op_7989_cast_fp16")]; tensor query_states_133_strides_0 = const()[name = string("query_states_133_strides_0"), val = tensor([1, 1])]; string query_states_133_pad_type_0 = const()[name = string("query_states_133_pad_type_0"), val = string("valid")]; tensor query_states_133_pad_0 = const()[name = string("query_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_133_dilations_0 = const()[name = string("query_states_133_dilations_0"), val = tensor([1, 1])]; int32 query_states_133_groups_0 = const()[name = string("query_states_133_groups_0"), val = int32(1)]; tensor query_states_133_cast_fp16 = conv(dilations = query_states_133_dilations_0, groups = query_states_133_groups_0, pad = query_states_133_pad_0, pad_type = query_states_133_pad_type_0, strides = query_states_133_strides_0, weight = layers_22_self_attn_q_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("query_states_133_cast_fp16")]; tensor key_states_221_strides_0 = const()[name = string("key_states_221_strides_0"), val = tensor([1, 1])]; string key_states_221_pad_type_0 = const()[name = string("key_states_221_pad_type_0"), val = string("valid")]; tensor key_states_221_pad_0 = const()[name = string("key_states_221_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_221_dilations_0 = const()[name = string("key_states_221_dilations_0"), val = tensor([1, 1])]; int32 key_states_221_groups_0 = const()[name = string("key_states_221_groups_0"), val = int32(1)]; tensor key_states_221_cast_fp16 = conv(dilations = key_states_221_dilations_0, groups = key_states_221_groups_0, pad = key_states_221_pad_0, pad_type = key_states_221_pad_type_0, strides = key_states_221_strides_0, weight = layers_22_self_attn_k_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("key_states_221_cast_fp16")]; tensor layers_22_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528794816)))]; tensor value_states_133_strides_0 = const()[name = string("value_states_133_strides_0"), val = tensor([1, 1])]; string value_states_133_pad_type_0 = const()[name = string("value_states_133_pad_type_0"), val = string("valid")]; tensor value_states_133_pad_0 = const()[name = string("value_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_133_dilations_0 = const()[name = string("value_states_133_dilations_0"), val = tensor([1, 1])]; int32 value_states_133_groups_0 = const()[name = string("value_states_133_groups_0"), val = int32(1)]; tensor value_states_133_cast_fp16 = conv(dilations = value_states_133_dilations_0, groups = value_states_133_groups_0, pad = value_states_133_pad_0, pad_type = value_states_133_pad_type_0, strides = value_states_133_strides_0, weight = layers_22_self_attn_v_proj_weight_to_fp16, x = var_7989_cast_fp16_0)[name = string("value_states_133_cast_fp16")]; tensor concat_264x = const()[name = string("concat_264x"), val = tensor([1, 16, 128, -1])]; tensor x_221_cast_fp16 = reshape(shape = concat_264x, x = query_states_133_cast_fp16)[name = string("x_221_cast_fp16")]; tensor concat_265x = const()[name = string("concat_265x"), val = tensor([1, 2, 128, -1])]; tensor var_8046_cast_fp16 = reshape(shape = concat_265x, x = key_states_221_cast_fp16)[name = string("op_8046_cast_fp16")]; tensor concat_266x = const()[name = string("concat_266x"), val = tensor([1, 2, 128, -1])]; tensor var_8053_cast_fp16 = reshape(shape = concat_266x, x = value_states_133_cast_fp16)[name = string("op_8053_cast_fp16")]; tensor var_8057_cast_fp16 = mul(x = x_221_cast_fp16, y = var_869_cast_fp16)[name = string("op_8057_cast_fp16")]; tensor var_8058_split_sizes_0 = const()[name = string("op_8058_split_sizes_0"), val = tensor([64, 64])]; int32 var_8058_axis_0 = const()[name = string("op_8058_axis_0"), val = int32(-2)]; tensor var_8058_cast_fp16_0, tensor var_8058_cast_fp16_1 = split(axis = var_8058_axis_0, split_sizes = var_8058_split_sizes_0, x = x_221_cast_fp16)[name = string("op_8058_cast_fp16")]; fp16 const_222_promoted_to_fp16 = const()[name = string("const_222_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8060_cast_fp16 = mul(x = var_8058_cast_fp16_1, y = const_222_promoted_to_fp16)[name = string("op_8060_cast_fp16")]; int32 var_8062 = const()[name = string("op_8062"), val = int32(-2)]; bool var_8063_interleave_0 = const()[name = string("op_8063_interleave_0"), val = bool(false)]; tensor var_8063_cast_fp16 = concat(axis = var_8062, interleave = var_8063_interleave_0, values = (var_8060_cast_fp16, var_8058_cast_fp16_0))[name = string("op_8063_cast_fp16")]; tensor var_8064_cast_fp16 = mul(x = var_8063_cast_fp16, y = var_878_cast_fp16)[name = string("op_8064_cast_fp16")]; tensor query_states_135_cast_fp16 = add(x = var_8057_cast_fp16, y = var_8064_cast_fp16)[name = string("query_states_135_cast_fp16")]; tensor var_8070_cast_fp16 = mul(x = var_8046_cast_fp16, y = var_869_cast_fp16)[name = string("op_8070_cast_fp16")]; tensor var_8071_split_sizes_0 = const()[name = string("op_8071_split_sizes_0"), val = tensor([64, 64])]; int32 var_8071_axis_0 = const()[name = string("op_8071_axis_0"), val = int32(-2)]; tensor var_8071_cast_fp16_0, tensor var_8071_cast_fp16_1 = split(axis = var_8071_axis_0, split_sizes = var_8071_split_sizes_0, x = var_8046_cast_fp16)[name = string("op_8071_cast_fp16")]; fp16 const_223_promoted_to_fp16 = const()[name = string("const_223_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8073_cast_fp16 = mul(x = var_8071_cast_fp16_1, y = const_223_promoted_to_fp16)[name = string("op_8073_cast_fp16")]; int32 var_8075 = const()[name = string("op_8075"), val = int32(-2)]; bool var_8076_interleave_0 = const()[name = string("op_8076_interleave_0"), val = bool(false)]; tensor var_8076_cast_fp16 = concat(axis = var_8075, interleave = var_8076_interleave_0, values = (var_8073_cast_fp16, var_8071_cast_fp16_0))[name = string("op_8076_cast_fp16")]; tensor var_8077_cast_fp16 = mul(x = var_8076_cast_fp16, y = var_878_cast_fp16)[name = string("op_8077_cast_fp16")]; tensor key_states_225_cast_fp16 = add(x = var_8070_cast_fp16, y = var_8077_cast_fp16)[name = string("key_states_225_cast_fp16")]; tensor expand_dims_264 = const()[name = string("expand_dims_264"), val = tensor([22])]; tensor expand_dims_265 = const()[name = string("expand_dims_265"), val = tensor([0])]; tensor expand_dims_267 = const()[name = string("expand_dims_267"), val = tensor([0])]; int32 concat_269_axis_0 = const()[name = string("concat_269_axis_0"), val = int32(0)]; bool concat_269_interleave_0 = const()[name = string("concat_269_interleave_0"), val = bool(false)]; tensor concat_269 = concat(axis = concat_269_axis_0, interleave = concat_269_interleave_0, values = (expand_dims_264, expand_dims_265, position_id, expand_dims_267))[name = string("concat_269")]; tensor expand_dims_268 = const()[name = string("expand_dims_268"), val = tensor([23])]; tensor concat_270_values1_0 = const()[name = string("concat_270_values1_0"), val = tensor([0])]; tensor concat_270_values3_0 = const()[name = string("concat_270_values3_0"), val = tensor([0])]; int32 concat_270_axis_0 = const()[name = string("concat_270_axis_0"), val = int32(0)]; bool concat_270_interleave_0 = const()[name = string("concat_270_interleave_0"), val = bool(false)]; tensor concat_270 = concat(axis = concat_270_axis_0, interleave = concat_270_interleave_0, values = (expand_dims_268, concat_270_values1_0, cache_position_end, concat_270_values3_0))[name = string("concat_270")]; tensor key_states_227_perm_0 = const()[name = string("key_states_227_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_23_stride_0 = const()[name = string("key_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_227_cast_fp16 = transpose(perm = key_states_227_perm_0, x = key_states_225_cast_fp16)[name = string("transpose_103")]; tensor key_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = key_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = key_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_23_squeeze_mask_0, stride = key_cache_internal_tensor_assign_23_stride_0, update = key_states_227_cast_fp16, x = coreml_update_state_98)[name = string("key_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_23_cast_fp16, input = key_cache)[name = string("coreml_update_state_100_write_state")]; tensor coreml_update_state_100 = read_state(input = key_cache)[name = string("coreml_update_state_100")]; tensor value_states_135_perm_0 = const()[name = string("value_states_135_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_23_stride_0 = const()[name = string("value_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_135_cast_fp16 = transpose(perm = value_states_135_perm_0, x = var_8053_cast_fp16)[name = string("transpose_102")]; tensor value_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = value_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = value_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_23_squeeze_mask_0, stride = value_cache_internal_tensor_assign_23_stride_0, update = value_states_135_cast_fp16, x = coreml_update_state_99)[name = string("value_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_23_cast_fp16, input = value_cache)[name = string("coreml_update_state_101_write_state")]; tensor coreml_update_state_101 = read_state(input = value_cache)[name = string("coreml_update_state_101")]; tensor var_8147_begin_0 = const()[name = string("op_8147_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8147_end_0 = const()[name = string("op_8147_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8147_end_mask_0 = const()[name = string("op_8147_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8147_cast_fp16 = slice_by_index(begin = var_8147_begin_0, end = var_8147_end_0, end_mask = var_8147_end_mask_0, x = coreml_update_state_100)[name = string("op_8147_cast_fp16")]; tensor tile_44 = const()[name = string("tile_44"), val = tensor([1, 1])]; int32 var_8150_axis_0 = const()[name = string("op_8150_axis_0"), val = int32(1)]; tensor var_8150_cast_fp16_0, tensor var_8150_cast_fp16_1 = split(axis = var_8150_axis_0, split_sizes = tile_44, x = var_8147_cast_fp16)[name = string("op_8150_cast_fp16")]; tensor var_8157_begin_0 = const()[name = string("op_8157_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8157_end_0 = const()[name = string("op_8157_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8157_end_mask_0 = const()[name = string("op_8157_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8157_cast_fp16 = slice_by_index(begin = var_8157_begin_0, end = var_8157_end_0, end_mask = var_8157_end_mask_0, x = coreml_update_state_101)[name = string("op_8157_cast_fp16")]; tensor tile_45 = const()[name = string("tile_45"), val = tensor([1, 1])]; int32 var_8160_axis_0 = const()[name = string("op_8160_axis_0"), val = int32(1)]; tensor var_8160_cast_fp16_0, tensor var_8160_cast_fp16_1 = split(axis = var_8160_axis_0, split_sizes = tile_45, x = var_8157_cast_fp16)[name = string("op_8160_cast_fp16")]; tensor var_8163_split_sizes_0 = const()[name = string("op_8163_split_sizes_0"), val = tensor([8, 8])]; int32 var_8163_axis_0 = const()[name = string("op_8163_axis_0"), val = int32(1)]; tensor var_8163_0, tensor var_8163_1 = split(axis = var_8163_axis_0, split_sizes = var_8163_split_sizes_0, x = query_states_135_cast_fp16)[name = string("op_8163")]; bool attn_weights_353_transpose_x_0 = const()[name = string("attn_weights_353_transpose_x_0"), val = bool(false)]; bool attn_weights_353_transpose_y_0 = const()[name = string("attn_weights_353_transpose_y_0"), val = bool(false)]; tensor attn_weights_353_cast_fp16 = matmul(transpose_x = attn_weights_353_transpose_x_0, transpose_y = attn_weights_353_transpose_y_0, x = var_8150_cast_fp16_0, y = var_8163_0)[name = string("attn_weights_353_cast_fp16")]; fp16 var_8166_to_fp16 = const()[name = string("op_8166_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_355_cast_fp16 = mul(x = attn_weights_353_cast_fp16, y = var_8166_to_fp16)[name = string("attn_weights_355_cast_fp16")]; tensor attn_weights_357_cast_fp16 = add(x = attn_weights_355_cast_fp16, y = attn_mask_1)[name = string("attn_weights_357_cast_fp16")]; int32 var_8170 = const()[name = string("op_8170"), val = int32(-2)]; tensor attn_weights_359_cast_fp16 = softmax(axis = var_8170, x = attn_weights_357_cast_fp16)[name = string("attn_weights_359_cast_fp16")]; bool var_8176_transpose_x_1 = const()[name = string("op_8176_transpose_x_1"), val = bool(true)]; bool var_8176_transpose_y_1 = const()[name = string("op_8176_transpose_y_1"), val = bool(false)]; tensor var_8176_cast_fp16 = matmul(transpose_x = var_8176_transpose_x_1, transpose_y = var_8176_transpose_y_1, x = attn_weights_359_cast_fp16, y = var_8160_cast_fp16_0)[name = string("op_8176_cast_fp16")]; bool attn_weights_361_transpose_x_0 = const()[name = string("attn_weights_361_transpose_x_0"), val = bool(false)]; bool attn_weights_361_transpose_y_0 = const()[name = string("attn_weights_361_transpose_y_0"), val = bool(false)]; tensor attn_weights_361_cast_fp16 = matmul(transpose_x = attn_weights_361_transpose_x_0, transpose_y = attn_weights_361_transpose_y_0, x = var_8150_cast_fp16_1, y = var_8163_1)[name = string("attn_weights_361_cast_fp16")]; fp16 var_8178_to_fp16 = const()[name = string("op_8178_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_363_cast_fp16 = mul(x = attn_weights_361_cast_fp16, y = var_8178_to_fp16)[name = string("attn_weights_363_cast_fp16")]; tensor attn_weights_365_cast_fp16 = add(x = attn_weights_363_cast_fp16, y = attn_mask_1)[name = string("attn_weights_365_cast_fp16")]; int32 var_8182 = const()[name = string("op_8182"), val = int32(-2)]; tensor attn_weights_367_cast_fp16 = softmax(axis = var_8182, x = attn_weights_365_cast_fp16)[name = string("attn_weights_367_cast_fp16")]; bool attn_output_177_transpose_x_1 = const()[name = string("attn_output_177_transpose_x_1"), val = bool(true)]; bool attn_output_177_transpose_y_1 = const()[name = string("attn_output_177_transpose_y_1"), val = bool(false)]; tensor attn_output_177_cast_fp16 = matmul(transpose_x = attn_output_177_transpose_x_1, transpose_y = attn_output_177_transpose_y_1, x = attn_weights_367_cast_fp16, y = var_8160_cast_fp16_1)[name = string("attn_output_177_cast_fp16")]; int32 var_8190 = const()[name = string("op_8190"), val = int32(1)]; bool attn_output_179_interleave_0 = const()[name = string("attn_output_179_interleave_0"), val = bool(false)]; tensor attn_output_179_cast_fp16 = concat(axis = var_8190, interleave = attn_output_179_interleave_0, values = (var_8176_cast_fp16, attn_output_177_cast_fp16))[name = string("attn_output_179_cast_fp16")]; tensor var_8194_perm_0 = const()[name = string("op_8194_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_275x = const()[name = string("concat_275x"), val = tensor([1, 2048, 1, -1])]; tensor var_8194_cast_fp16 = transpose(perm = var_8194_perm_0, x = attn_output_179_cast_fp16)[name = string("transpose_101")]; tensor attn_output_183_cast_fp16 = reshape(shape = concat_275x, x = var_8194_cast_fp16)[name = string("attn_output_183_cast_fp16")]; tensor layers_22_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1529843456)))]; tensor hidden_states_223_strides_0 = const()[name = string("hidden_states_223_strides_0"), val = tensor([1, 1])]; string hidden_states_223_pad_type_0 = const()[name = string("hidden_states_223_pad_type_0"), val = string("valid")]; tensor hidden_states_223_pad_0 = const()[name = string("hidden_states_223_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_223_dilations_0 = const()[name = string("hidden_states_223_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_223_groups_0 = const()[name = string("hidden_states_223_groups_0"), val = int32(1)]; tensor hidden_states_223_cast_fp16 = conv(dilations = hidden_states_223_dilations_0, groups = hidden_states_223_groups_0, pad = hidden_states_223_pad_0, pad_type = hidden_states_223_pad_type_0, strides = hidden_states_223_strides_0, weight = layers_22_self_attn_o_proj_weight_to_fp16, x = attn_output_183_cast_fp16)[name = string("hidden_states_223_cast_fp16")]; tensor hidden_states_225_cast_fp16 = add(x = hidden_states_219_cast_fp16, y = hidden_states_223_cast_fp16)[name = string("hidden_states_225_cast_fp16")]; fp16 const_228_promoted_to_fp16 = const()[name = string("const_228_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8227_cast_fp16 = mul(x = hidden_states_225_cast_fp16, y = const_228_promoted_to_fp16)[name = string("op_8227_cast_fp16")]; int32 var_8225 = const()[name = string("op_8225"), val = int32(1)]; bool doubled_181_interleave_0 = const()[name = string("doubled_181_interleave_0"), val = bool(false)]; tensor doubled_181_cast_fp16 = concat(axis = var_8225, interleave = doubled_181_interleave_0, values = (hidden_states_225_cast_fp16, var_8227_cast_fp16))[name = string("doubled_181_cast_fp16")]; tensor out_91_axes_0 = const()[name = string("out_91_axes_0"), val = tensor([1])]; tensor out_91_gamma_0_to_fp16 = const()[name = string("out_91_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538232128)))]; fp16 var_8237_to_fp16 = const()[name = string("op_8237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_91_cast_fp16 = layer_norm(axes = out_91_axes_0, epsilon = var_8237_to_fp16, gamma = out_91_gamma_0_to_fp16, x = doubled_181_cast_fp16)[name = string("out_91_cast_fp16")]; tensor var_8248_split_sizes_0 = const()[name = string("op_8248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8248_axis_0 = const()[name = string("op_8248_axis_0"), val = int32(1)]; tensor var_8248_cast_fp16_0, tensor var_8248_cast_fp16_1 = split(axis = var_8248_axis_0, split_sizes = var_8248_split_sizes_0, x = out_91_cast_fp16)[name = string("op_8248_cast_fp16")]; tensor input_45_strides_0 = const()[name = string("input_45_strides_0"), val = tensor([1, 1])]; string input_45_pad_type_0 = const()[name = string("input_45_pad_type_0"), val = string("valid")]; tensor input_45_pad_0 = const()[name = string("input_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_45_dilations_0 = const()[name = string("input_45_dilations_0"), val = tensor([1, 1])]; int32 input_45_groups_0 = const()[name = string("input_45_groups_0"), val = int32(1)]; tensor input_45_cast_fp16 = conv(dilations = input_45_dilations_0, groups = input_45_groups_0, pad = input_45_pad_0, pad_type = input_45_pad_type_0, strides = input_45_strides_0, weight = layers_22_mlp_gate_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("input_45_cast_fp16")]; tensor var_8265_cast_fp16 = silu(x = input_45_cast_fp16)[name = string("op_8265_cast_fp16")]; tensor var_8271_strides_0 = const()[name = string("op_8271_strides_0"), val = tensor([1, 1])]; string var_8271_pad_type_0 = const()[name = string("op_8271_pad_type_0"), val = string("valid")]; tensor var_8271_pad_0 = const()[name = string("op_8271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8271_dilations_0 = const()[name = string("op_8271_dilations_0"), val = tensor([1, 1])]; int32 var_8271_groups_0 = const()[name = string("op_8271_groups_0"), val = int32(1)]; tensor var_8271_cast_fp16 = conv(dilations = var_8271_dilations_0, groups = var_8271_groups_0, pad = var_8271_pad_0, pad_type = var_8271_pad_type_0, strides = var_8271_strides_0, weight = layers_22_mlp_up_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("op_8271_cast_fp16")]; tensor x_229_cast_fp16 = mul(x = var_8265_cast_fp16, y = var_8271_cast_fp16)[name = string("x_229_cast_fp16")]; tensor hidden_states_227_strides_0 = const()[name = string("hidden_states_227_strides_0"), val = tensor([1, 1])]; string hidden_states_227_pad_type_0 = const()[name = string("hidden_states_227_pad_type_0"), val = string("valid")]; tensor hidden_states_227_pad_0 = const()[name = string("hidden_states_227_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_227_dilations_0 = const()[name = string("hidden_states_227_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_227_groups_0 = const()[name = string("hidden_states_227_groups_0"), val = int32(1)]; tensor hidden_states_227_cast_fp16 = conv(dilations = hidden_states_227_dilations_0, groups = hidden_states_227_groups_0, pad = hidden_states_227_pad_0, pad_type = hidden_states_227_pad_type_0, strides = hidden_states_227_strides_0, weight = layers_22_mlp_down_proj_weight_cast_fp16, x = x_229_cast_fp16)[name = string("hidden_states_227_cast_fp16")]; tensor hidden_states_229_cast_fp16 = add(x = hidden_states_225_cast_fp16, y = hidden_states_227_cast_fp16)[name = string("hidden_states_229_cast_fp16")]; fp16 const_230_promoted_to_fp16 = const()[name = string("const_230_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8289_cast_fp16 = mul(x = hidden_states_229_cast_fp16, y = const_230_promoted_to_fp16)[name = string("op_8289_cast_fp16")]; int32 var_8287 = const()[name = string("op_8287"), val = int32(1)]; bool doubled_185_interleave_0 = const()[name = string("doubled_185_interleave_0"), val = bool(false)]; tensor doubled_185_cast_fp16 = concat(axis = var_8287, interleave = doubled_185_interleave_0, values = (hidden_states_229_cast_fp16, var_8289_cast_fp16))[name = string("doubled_185_cast_fp16")]; tensor out_93_axes_0 = const()[name = string("out_93_axes_0"), val = tensor([1])]; tensor out_93_gamma_0_to_fp16 = const()[name = string("out_93_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538240384)))]; fp16 var_8299_to_fp16 = const()[name = string("op_8299_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_93_cast_fp16 = layer_norm(axes = out_93_axes_0, epsilon = var_8299_to_fp16, gamma = out_93_gamma_0_to_fp16, x = doubled_185_cast_fp16)[name = string("out_93_cast_fp16")]; tensor var_8310_split_sizes_0 = const()[name = string("op_8310_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8310_axis_0 = const()[name = string("op_8310_axis_0"), val = int32(1)]; tensor var_8310_cast_fp16_0, tensor var_8310_cast_fp16_1 = split(axis = var_8310_axis_0, split_sizes = var_8310_split_sizes_0, x = out_93_cast_fp16)[name = string("op_8310_cast_fp16")]; tensor query_states_139_strides_0 = const()[name = string("query_states_139_strides_0"), val = tensor([1, 1])]; string query_states_139_pad_type_0 = const()[name = string("query_states_139_pad_type_0"), val = string("valid")]; tensor query_states_139_pad_0 = const()[name = string("query_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_139_dilations_0 = const()[name = string("query_states_139_dilations_0"), val = tensor([1, 1])]; int32 query_states_139_groups_0 = const()[name = string("query_states_139_groups_0"), val = int32(1)]; tensor query_states_139_cast_fp16 = conv(dilations = query_states_139_dilations_0, groups = query_states_139_groups_0, pad = query_states_139_pad_0, pad_type = query_states_139_pad_type_0, strides = query_states_139_strides_0, weight = layers_23_self_attn_q_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("query_states_139_cast_fp16")]; tensor key_states_231_strides_0 = const()[name = string("key_states_231_strides_0"), val = tensor([1, 1])]; string key_states_231_pad_type_0 = const()[name = string("key_states_231_pad_type_0"), val = string("valid")]; tensor key_states_231_pad_0 = const()[name = string("key_states_231_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_231_dilations_0 = const()[name = string("key_states_231_dilations_0"), val = tensor([1, 1])]; int32 key_states_231_groups_0 = const()[name = string("key_states_231_groups_0"), val = int32(1)]; tensor key_states_231_cast_fp16 = conv(dilations = key_states_231_dilations_0, groups = key_states_231_groups_0, pad = key_states_231_pad_0, pad_type = key_states_231_pad_type_0, strides = key_states_231_strides_0, weight = layers_23_self_attn_k_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("key_states_231_cast_fp16")]; tensor layers_23_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_23_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538248640)))]; tensor value_states_139_strides_0 = const()[name = string("value_states_139_strides_0"), val = tensor([1, 1])]; string value_states_139_pad_type_0 = const()[name = string("value_states_139_pad_type_0"), val = string("valid")]; tensor value_states_139_pad_0 = const()[name = string("value_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_139_dilations_0 = const()[name = string("value_states_139_dilations_0"), val = tensor([1, 1])]; int32 value_states_139_groups_0 = const()[name = string("value_states_139_groups_0"), val = int32(1)]; tensor value_states_139_cast_fp16 = conv(dilations = value_states_139_dilations_0, groups = value_states_139_groups_0, pad = value_states_139_pad_0, pad_type = value_states_139_pad_type_0, strides = value_states_139_strides_0, weight = layers_23_self_attn_v_proj_weight_to_fp16, x = var_8310_cast_fp16_0)[name = string("value_states_139_cast_fp16")]; tensor concat_276x = const()[name = string("concat_276x"), val = tensor([1, 16, 128, -1])]; tensor x_231_cast_fp16 = reshape(shape = concat_276x, x = query_states_139_cast_fp16)[name = string("x_231_cast_fp16")]; tensor concat_277x = const()[name = string("concat_277x"), val = tensor([1, 2, 128, -1])]; tensor var_8367_cast_fp16 = reshape(shape = concat_277x, x = key_states_231_cast_fp16)[name = string("op_8367_cast_fp16")]; tensor concat_278x = const()[name = string("concat_278x"), val = tensor([1, 2, 128, -1])]; tensor var_8374_cast_fp16 = reshape(shape = concat_278x, x = value_states_139_cast_fp16)[name = string("op_8374_cast_fp16")]; tensor var_8378_cast_fp16 = mul(x = x_231_cast_fp16, y = var_869_cast_fp16)[name = string("op_8378_cast_fp16")]; tensor var_8379_split_sizes_0 = const()[name = string("op_8379_split_sizes_0"), val = tensor([64, 64])]; int32 var_8379_axis_0 = const()[name = string("op_8379_axis_0"), val = int32(-2)]; tensor var_8379_cast_fp16_0, tensor var_8379_cast_fp16_1 = split(axis = var_8379_axis_0, split_sizes = var_8379_split_sizes_0, x = x_231_cast_fp16)[name = string("op_8379_cast_fp16")]; fp16 const_232_promoted_to_fp16 = const()[name = string("const_232_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8381_cast_fp16 = mul(x = var_8379_cast_fp16_1, y = const_232_promoted_to_fp16)[name = string("op_8381_cast_fp16")]; int32 var_8383 = const()[name = string("op_8383"), val = int32(-2)]; bool var_8384_interleave_0 = const()[name = string("op_8384_interleave_0"), val = bool(false)]; tensor var_8384_cast_fp16 = concat(axis = var_8383, interleave = var_8384_interleave_0, values = (var_8381_cast_fp16, var_8379_cast_fp16_0))[name = string("op_8384_cast_fp16")]; tensor var_8385_cast_fp16 = mul(x = var_8384_cast_fp16, y = var_878_cast_fp16)[name = string("op_8385_cast_fp16")]; tensor query_states_141_cast_fp16 = add(x = var_8378_cast_fp16, y = var_8385_cast_fp16)[name = string("query_states_141_cast_fp16")]; tensor var_8391_cast_fp16 = mul(x = var_8367_cast_fp16, y = var_869_cast_fp16)[name = string("op_8391_cast_fp16")]; tensor var_8392_split_sizes_0 = const()[name = string("op_8392_split_sizes_0"), val = tensor([64, 64])]; int32 var_8392_axis_0 = const()[name = string("op_8392_axis_0"), val = int32(-2)]; tensor var_8392_cast_fp16_0, tensor var_8392_cast_fp16_1 = split(axis = var_8392_axis_0, split_sizes = var_8392_split_sizes_0, x = var_8367_cast_fp16)[name = string("op_8392_cast_fp16")]; fp16 const_233_promoted_to_fp16 = const()[name = string("const_233_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8394_cast_fp16 = mul(x = var_8392_cast_fp16_1, y = const_233_promoted_to_fp16)[name = string("op_8394_cast_fp16")]; int32 var_8396 = const()[name = string("op_8396"), val = int32(-2)]; bool var_8397_interleave_0 = const()[name = string("op_8397_interleave_0"), val = bool(false)]; tensor var_8397_cast_fp16 = concat(axis = var_8396, interleave = var_8397_interleave_0, values = (var_8394_cast_fp16, var_8392_cast_fp16_0))[name = string("op_8397_cast_fp16")]; tensor var_8398_cast_fp16 = mul(x = var_8397_cast_fp16, y = var_878_cast_fp16)[name = string("op_8398_cast_fp16")]; tensor key_states_235_cast_fp16 = add(x = var_8391_cast_fp16, y = var_8398_cast_fp16)[name = string("key_states_235_cast_fp16")]; tensor expand_dims_276 = const()[name = string("expand_dims_276"), val = tensor([23])]; tensor expand_dims_277 = const()[name = string("expand_dims_277"), val = tensor([0])]; tensor expand_dims_279 = const()[name = string("expand_dims_279"), val = tensor([0])]; int32 concat_281_axis_0 = const()[name = string("concat_281_axis_0"), val = int32(0)]; bool concat_281_interleave_0 = const()[name = string("concat_281_interleave_0"), val = bool(false)]; tensor concat_281 = concat(axis = concat_281_axis_0, interleave = concat_281_interleave_0, values = (expand_dims_276, expand_dims_277, position_id, expand_dims_279))[name = string("concat_281")]; tensor expand_dims_280 = const()[name = string("expand_dims_280"), val = tensor([24])]; tensor concat_282_values1_0 = const()[name = string("concat_282_values1_0"), val = tensor([0])]; tensor concat_282_values3_0 = const()[name = string("concat_282_values3_0"), val = tensor([0])]; int32 concat_282_axis_0 = const()[name = string("concat_282_axis_0"), val = int32(0)]; bool concat_282_interleave_0 = const()[name = string("concat_282_interleave_0"), val = bool(false)]; tensor concat_282 = concat(axis = concat_282_axis_0, interleave = concat_282_interleave_0, values = (expand_dims_280, concat_282_values1_0, cache_position_end, concat_282_values3_0))[name = string("concat_282")]; tensor key_states_237_perm_0 = const()[name = string("key_states_237_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_24_stride_0 = const()[name = string("key_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_237_cast_fp16 = transpose(perm = key_states_237_perm_0, x = key_states_235_cast_fp16)[name = string("transpose_100")]; tensor key_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = key_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = key_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_24_squeeze_mask_0, stride = key_cache_internal_tensor_assign_24_stride_0, update = key_states_237_cast_fp16, x = coreml_update_state_100)[name = string("key_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_24_cast_fp16, input = key_cache)[name = string("coreml_update_state_102_write_state")]; tensor coreml_update_state_102 = read_state(input = key_cache)[name = string("coreml_update_state_102")]; tensor value_states_141_perm_0 = const()[name = string("value_states_141_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_24_stride_0 = const()[name = string("value_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_141_cast_fp16 = transpose(perm = value_states_141_perm_0, x = var_8374_cast_fp16)[name = string("transpose_99")]; tensor value_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = value_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = value_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_24_squeeze_mask_0, stride = value_cache_internal_tensor_assign_24_stride_0, update = value_states_141_cast_fp16, x = coreml_update_state_101)[name = string("value_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_24_cast_fp16, input = value_cache)[name = string("coreml_update_state_103_write_state")]; tensor coreml_update_state_103 = read_state(input = value_cache)[name = string("coreml_update_state_103")]; tensor var_8468_begin_0 = const()[name = string("op_8468_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8468_end_0 = const()[name = string("op_8468_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8468_end_mask_0 = const()[name = string("op_8468_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8468_cast_fp16 = slice_by_index(begin = var_8468_begin_0, end = var_8468_end_0, end_mask = var_8468_end_mask_0, x = coreml_update_state_102)[name = string("op_8468_cast_fp16")]; tensor tile_46 = const()[name = string("tile_46"), val = tensor([1, 1])]; int32 var_8471_axis_0 = const()[name = string("op_8471_axis_0"), val = int32(1)]; tensor var_8471_cast_fp16_0, tensor var_8471_cast_fp16_1 = split(axis = var_8471_axis_0, split_sizes = tile_46, x = var_8468_cast_fp16)[name = string("op_8471_cast_fp16")]; tensor var_8478_begin_0 = const()[name = string("op_8478_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8478_end_0 = const()[name = string("op_8478_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8478_end_mask_0 = const()[name = string("op_8478_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8478_cast_fp16 = slice_by_index(begin = var_8478_begin_0, end = var_8478_end_0, end_mask = var_8478_end_mask_0, x = coreml_update_state_103)[name = string("op_8478_cast_fp16")]; tensor tile_47 = const()[name = string("tile_47"), val = tensor([1, 1])]; int32 var_8481_axis_0 = const()[name = string("op_8481_axis_0"), val = int32(1)]; tensor var_8481_cast_fp16_0, tensor var_8481_cast_fp16_1 = split(axis = var_8481_axis_0, split_sizes = tile_47, x = var_8478_cast_fp16)[name = string("op_8481_cast_fp16")]; tensor var_8484_split_sizes_0 = const()[name = string("op_8484_split_sizes_0"), val = tensor([8, 8])]; int32 var_8484_axis_0 = const()[name = string("op_8484_axis_0"), val = int32(1)]; tensor var_8484_0, tensor var_8484_1 = split(axis = var_8484_axis_0, split_sizes = var_8484_split_sizes_0, x = query_states_141_cast_fp16)[name = string("op_8484")]; bool attn_weights_369_transpose_x_0 = const()[name = string("attn_weights_369_transpose_x_0"), val = bool(false)]; bool attn_weights_369_transpose_y_0 = const()[name = string("attn_weights_369_transpose_y_0"), val = bool(false)]; tensor attn_weights_369_cast_fp16 = matmul(transpose_x = attn_weights_369_transpose_x_0, transpose_y = attn_weights_369_transpose_y_0, x = var_8471_cast_fp16_0, y = var_8484_0)[name = string("attn_weights_369_cast_fp16")]; fp16 var_8487_to_fp16 = const()[name = string("op_8487_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_371_cast_fp16 = mul(x = attn_weights_369_cast_fp16, y = var_8487_to_fp16)[name = string("attn_weights_371_cast_fp16")]; tensor attn_weights_373_cast_fp16 = add(x = attn_weights_371_cast_fp16, y = attn_mask_1)[name = string("attn_weights_373_cast_fp16")]; int32 var_8491 = const()[name = string("op_8491"), val = int32(-2)]; tensor attn_weights_375_cast_fp16 = softmax(axis = var_8491, x = attn_weights_373_cast_fp16)[name = string("attn_weights_375_cast_fp16")]; bool var_8497_transpose_x_1 = const()[name = string("op_8497_transpose_x_1"), val = bool(true)]; bool var_8497_transpose_y_1 = const()[name = string("op_8497_transpose_y_1"), val = bool(false)]; tensor var_8497_cast_fp16 = matmul(transpose_x = var_8497_transpose_x_1, transpose_y = var_8497_transpose_y_1, x = attn_weights_375_cast_fp16, y = var_8481_cast_fp16_0)[name = string("op_8497_cast_fp16")]; bool attn_weights_377_transpose_x_0 = const()[name = string("attn_weights_377_transpose_x_0"), val = bool(false)]; bool attn_weights_377_transpose_y_0 = const()[name = string("attn_weights_377_transpose_y_0"), val = bool(false)]; tensor attn_weights_377_cast_fp16 = matmul(transpose_x = attn_weights_377_transpose_x_0, transpose_y = attn_weights_377_transpose_y_0, x = var_8471_cast_fp16_1, y = var_8484_1)[name = string("attn_weights_377_cast_fp16")]; fp16 var_8499_to_fp16 = const()[name = string("op_8499_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_379_cast_fp16 = mul(x = attn_weights_377_cast_fp16, y = var_8499_to_fp16)[name = string("attn_weights_379_cast_fp16")]; tensor attn_weights_381_cast_fp16 = add(x = attn_weights_379_cast_fp16, y = attn_mask_1)[name = string("attn_weights_381_cast_fp16")]; int32 var_8503 = const()[name = string("op_8503"), val = int32(-2)]; tensor attn_weights_383_cast_fp16 = softmax(axis = var_8503, x = attn_weights_381_cast_fp16)[name = string("attn_weights_383_cast_fp16")]; bool attn_output_185_transpose_x_1 = const()[name = string("attn_output_185_transpose_x_1"), val = bool(true)]; bool attn_output_185_transpose_y_1 = const()[name = string("attn_output_185_transpose_y_1"), val = bool(false)]; tensor attn_output_185_cast_fp16 = matmul(transpose_x = attn_output_185_transpose_x_1, transpose_y = attn_output_185_transpose_y_1, x = attn_weights_383_cast_fp16, y = var_8481_cast_fp16_1)[name = string("attn_output_185_cast_fp16")]; int32 var_8511 = const()[name = string("op_8511"), val = int32(1)]; bool attn_output_187_interleave_0 = const()[name = string("attn_output_187_interleave_0"), val = bool(false)]; tensor attn_output_187_cast_fp16 = concat(axis = var_8511, interleave = attn_output_187_interleave_0, values = (var_8497_cast_fp16, attn_output_185_cast_fp16))[name = string("attn_output_187_cast_fp16")]; tensor var_8515_perm_0 = const()[name = string("op_8515_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_287x = const()[name = string("concat_287x"), val = tensor([1, 2048, 1, -1])]; tensor var_8515_cast_fp16 = transpose(perm = var_8515_perm_0, x = attn_output_187_cast_fp16)[name = string("transpose_98")]; tensor attn_output_191_cast_fp16 = reshape(shape = concat_287x, x = var_8515_cast_fp16)[name = string("attn_output_191_cast_fp16")]; tensor hidden_states_233_strides_0 = const()[name = string("hidden_states_233_strides_0"), val = tensor([1, 1])]; string hidden_states_233_pad_type_0 = const()[name = string("hidden_states_233_pad_type_0"), val = string("valid")]; tensor hidden_states_233_pad_0 = const()[name = string("hidden_states_233_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_233_dilations_0 = const()[name = string("hidden_states_233_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_233_groups_0 = const()[name = string("hidden_states_233_groups_0"), val = int32(1)]; tensor hidden_states_233_cast_fp16 = conv(dilations = hidden_states_233_dilations_0, groups = hidden_states_233_groups_0, pad = hidden_states_233_pad_0, pad_type = hidden_states_233_pad_type_0, strides = hidden_states_233_strides_0, weight = layers_23_self_attn_o_proj_weight_cast_fp16, x = attn_output_191_cast_fp16)[name = string("hidden_states_233_cast_fp16")]; tensor hidden_states_235_cast_fp16 = add(x = hidden_states_229_cast_fp16, y = hidden_states_233_cast_fp16)[name = string("hidden_states_235_cast_fp16")]; fp16 const_238_promoted_to_fp16 = const()[name = string("const_238_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8548_cast_fp16 = mul(x = hidden_states_235_cast_fp16, y = const_238_promoted_to_fp16)[name = string("op_8548_cast_fp16")]; int32 var_8546 = const()[name = string("op_8546"), val = int32(1)]; bool doubled_189_interleave_0 = const()[name = string("doubled_189_interleave_0"), val = bool(false)]; tensor doubled_189_cast_fp16 = concat(axis = var_8546, interleave = doubled_189_interleave_0, values = (hidden_states_235_cast_fp16, var_8548_cast_fp16))[name = string("doubled_189_cast_fp16")]; tensor out_95_axes_0 = const()[name = string("out_95_axes_0"), val = tensor([1])]; tensor out_95_gamma_0_to_fp16 = const()[name = string("out_95_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539297280)))]; fp16 var_8558_to_fp16 = const()[name = string("op_8558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_95_cast_fp16 = layer_norm(axes = out_95_axes_0, epsilon = var_8558_to_fp16, gamma = out_95_gamma_0_to_fp16, x = doubled_189_cast_fp16)[name = string("out_95_cast_fp16")]; tensor var_8569_split_sizes_0 = const()[name = string("op_8569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8569_axis_0 = const()[name = string("op_8569_axis_0"), val = int32(1)]; tensor var_8569_cast_fp16_0, tensor var_8569_cast_fp16_1 = split(axis = var_8569_axis_0, split_sizes = var_8569_split_sizes_0, x = out_95_cast_fp16)[name = string("op_8569_cast_fp16")]; tensor input_47_strides_0 = const()[name = string("input_47_strides_0"), val = tensor([1, 1])]; string input_47_pad_type_0 = const()[name = string("input_47_pad_type_0"), val = string("valid")]; tensor input_47_pad_0 = const()[name = string("input_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_47_dilations_0 = const()[name = string("input_47_dilations_0"), val = tensor([1, 1])]; int32 input_47_groups_0 = const()[name = string("input_47_groups_0"), val = int32(1)]; tensor input_47_cast_fp16 = conv(dilations = input_47_dilations_0, groups = input_47_groups_0, pad = input_47_pad_0, pad_type = input_47_pad_type_0, strides = input_47_strides_0, weight = layers_23_mlp_gate_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("input_47_cast_fp16")]; tensor var_8586_cast_fp16 = silu(x = input_47_cast_fp16)[name = string("op_8586_cast_fp16")]; tensor var_8592_strides_0 = const()[name = string("op_8592_strides_0"), val = tensor([1, 1])]; string var_8592_pad_type_0 = const()[name = string("op_8592_pad_type_0"), val = string("valid")]; tensor var_8592_pad_0 = const()[name = string("op_8592_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8592_dilations_0 = const()[name = string("op_8592_dilations_0"), val = tensor([1, 1])]; int32 var_8592_groups_0 = const()[name = string("op_8592_groups_0"), val = int32(1)]; tensor var_8592_cast_fp16 = conv(dilations = var_8592_dilations_0, groups = var_8592_groups_0, pad = var_8592_pad_0, pad_type = var_8592_pad_type_0, strides = var_8592_strides_0, weight = layers_23_mlp_up_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("op_8592_cast_fp16")]; tensor x_239_cast_fp16 = mul(x = var_8586_cast_fp16, y = var_8592_cast_fp16)[name = string("x_239_cast_fp16")]; tensor hidden_states_237_strides_0 = const()[name = string("hidden_states_237_strides_0"), val = tensor([1, 1])]; string hidden_states_237_pad_type_0 = const()[name = string("hidden_states_237_pad_type_0"), val = string("valid")]; tensor hidden_states_237_pad_0 = const()[name = string("hidden_states_237_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_237_dilations_0 = const()[name = string("hidden_states_237_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_237_groups_0 = const()[name = string("hidden_states_237_groups_0"), val = int32(1)]; tensor hidden_states_237_cast_fp16 = conv(dilations = hidden_states_237_dilations_0, groups = hidden_states_237_groups_0, pad = hidden_states_237_pad_0, pad_type = hidden_states_237_pad_type_0, strides = hidden_states_237_strides_0, weight = layers_23_mlp_down_proj_weight_cast_fp16, x = x_239_cast_fp16)[name = string("hidden_states_237_cast_fp16")]; tensor hidden_states_239_cast_fp16 = add(x = hidden_states_235_cast_fp16, y = hidden_states_237_cast_fp16)[name = string("hidden_states_239_cast_fp16")]; fp16 const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8610_cast_fp16 = mul(x = hidden_states_239_cast_fp16, y = const_240_promoted_to_fp16)[name = string("op_8610_cast_fp16")]; int32 var_8608 = const()[name = string("op_8608"), val = int32(1)]; bool doubled_193_interleave_0 = const()[name = string("doubled_193_interleave_0"), val = bool(false)]; tensor doubled_193_cast_fp16 = concat(axis = var_8608, interleave = doubled_193_interleave_0, values = (hidden_states_239_cast_fp16, var_8610_cast_fp16))[name = string("doubled_193_cast_fp16")]; tensor out_97_axes_0 = const()[name = string("out_97_axes_0"), val = tensor([1])]; tensor out_97_gamma_0_to_fp16 = const()[name = string("out_97_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539305536)))]; fp16 var_8620_to_fp16 = const()[name = string("op_8620_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_97_cast_fp16 = layer_norm(axes = out_97_axes_0, epsilon = var_8620_to_fp16, gamma = out_97_gamma_0_to_fp16, x = doubled_193_cast_fp16)[name = string("out_97_cast_fp16")]; tensor var_8631_split_sizes_0 = const()[name = string("op_8631_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8631_axis_0 = const()[name = string("op_8631_axis_0"), val = int32(1)]; tensor var_8631_cast_fp16_0, tensor var_8631_cast_fp16_1 = split(axis = var_8631_axis_0, split_sizes = var_8631_split_sizes_0, x = out_97_cast_fp16)[name = string("op_8631_cast_fp16")]; tensor query_states_145_strides_0 = const()[name = string("query_states_145_strides_0"), val = tensor([1, 1])]; string query_states_145_pad_type_0 = const()[name = string("query_states_145_pad_type_0"), val = string("valid")]; tensor query_states_145_pad_0 = const()[name = string("query_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_145_dilations_0 = const()[name = string("query_states_145_dilations_0"), val = tensor([1, 1])]; int32 query_states_145_groups_0 = const()[name = string("query_states_145_groups_0"), val = int32(1)]; tensor query_states_145_cast_fp16 = conv(dilations = query_states_145_dilations_0, groups = query_states_145_groups_0, pad = query_states_145_pad_0, pad_type = query_states_145_pad_type_0, strides = query_states_145_strides_0, weight = layers_24_self_attn_q_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("query_states_145_cast_fp16")]; tensor key_states_241_strides_0 = const()[name = string("key_states_241_strides_0"), val = tensor([1, 1])]; string key_states_241_pad_type_0 = const()[name = string("key_states_241_pad_type_0"), val = string("valid")]; tensor key_states_241_pad_0 = const()[name = string("key_states_241_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_241_dilations_0 = const()[name = string("key_states_241_dilations_0"), val = tensor([1, 1])]; int32 key_states_241_groups_0 = const()[name = string("key_states_241_groups_0"), val = int32(1)]; tensor key_states_241_cast_fp16 = conv(dilations = key_states_241_dilations_0, groups = key_states_241_groups_0, pad = key_states_241_pad_0, pad_type = key_states_241_pad_type_0, strides = key_states_241_strides_0, weight = layers_24_self_attn_k_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("key_states_241_cast_fp16")]; tensor layers_24_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_24_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539313792)))]; tensor value_states_145_strides_0 = const()[name = string("value_states_145_strides_0"), val = tensor([1, 1])]; string value_states_145_pad_type_0 = const()[name = string("value_states_145_pad_type_0"), val = string("valid")]; tensor value_states_145_pad_0 = const()[name = string("value_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_145_dilations_0 = const()[name = string("value_states_145_dilations_0"), val = tensor([1, 1])]; int32 value_states_145_groups_0 = const()[name = string("value_states_145_groups_0"), val = int32(1)]; tensor value_states_145_cast_fp16 = conv(dilations = value_states_145_dilations_0, groups = value_states_145_groups_0, pad = value_states_145_pad_0, pad_type = value_states_145_pad_type_0, strides = value_states_145_strides_0, weight = layers_24_self_attn_v_proj_weight_to_fp16, x = var_8631_cast_fp16_0)[name = string("value_states_145_cast_fp16")]; tensor concat_288x = const()[name = string("concat_288x"), val = tensor([1, 16, 128, -1])]; tensor x_241_cast_fp16 = reshape(shape = concat_288x, x = query_states_145_cast_fp16)[name = string("x_241_cast_fp16")]; tensor concat_289x = const()[name = string("concat_289x"), val = tensor([1, 2, 128, -1])]; tensor var_8688_cast_fp16 = reshape(shape = concat_289x, x = key_states_241_cast_fp16)[name = string("op_8688_cast_fp16")]; tensor concat_290x = const()[name = string("concat_290x"), val = tensor([1, 2, 128, -1])]; tensor var_8695_cast_fp16 = reshape(shape = concat_290x, x = value_states_145_cast_fp16)[name = string("op_8695_cast_fp16")]; tensor var_8699_cast_fp16 = mul(x = x_241_cast_fp16, y = var_869_cast_fp16)[name = string("op_8699_cast_fp16")]; tensor var_8700_split_sizes_0 = const()[name = string("op_8700_split_sizes_0"), val = tensor([64, 64])]; int32 var_8700_axis_0 = const()[name = string("op_8700_axis_0"), val = int32(-2)]; tensor var_8700_cast_fp16_0, tensor var_8700_cast_fp16_1 = split(axis = var_8700_axis_0, split_sizes = var_8700_split_sizes_0, x = x_241_cast_fp16)[name = string("op_8700_cast_fp16")]; fp16 const_242_promoted_to_fp16 = const()[name = string("const_242_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8702_cast_fp16 = mul(x = var_8700_cast_fp16_1, y = const_242_promoted_to_fp16)[name = string("op_8702_cast_fp16")]; int32 var_8704 = const()[name = string("op_8704"), val = int32(-2)]; bool var_8705_interleave_0 = const()[name = string("op_8705_interleave_0"), val = bool(false)]; tensor var_8705_cast_fp16 = concat(axis = var_8704, interleave = var_8705_interleave_0, values = (var_8702_cast_fp16, var_8700_cast_fp16_0))[name = string("op_8705_cast_fp16")]; tensor var_8706_cast_fp16 = mul(x = var_8705_cast_fp16, y = var_878_cast_fp16)[name = string("op_8706_cast_fp16")]; tensor query_states_147_cast_fp16 = add(x = var_8699_cast_fp16, y = var_8706_cast_fp16)[name = string("query_states_147_cast_fp16")]; tensor var_8712_cast_fp16 = mul(x = var_8688_cast_fp16, y = var_869_cast_fp16)[name = string("op_8712_cast_fp16")]; tensor var_8713_split_sizes_0 = const()[name = string("op_8713_split_sizes_0"), val = tensor([64, 64])]; int32 var_8713_axis_0 = const()[name = string("op_8713_axis_0"), val = int32(-2)]; tensor var_8713_cast_fp16_0, tensor var_8713_cast_fp16_1 = split(axis = var_8713_axis_0, split_sizes = var_8713_split_sizes_0, x = var_8688_cast_fp16)[name = string("op_8713_cast_fp16")]; fp16 const_243_promoted_to_fp16 = const()[name = string("const_243_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8715_cast_fp16 = mul(x = var_8713_cast_fp16_1, y = const_243_promoted_to_fp16)[name = string("op_8715_cast_fp16")]; int32 var_8717 = const()[name = string("op_8717"), val = int32(-2)]; bool var_8718_interleave_0 = const()[name = string("op_8718_interleave_0"), val = bool(false)]; tensor var_8718_cast_fp16 = concat(axis = var_8717, interleave = var_8718_interleave_0, values = (var_8715_cast_fp16, var_8713_cast_fp16_0))[name = string("op_8718_cast_fp16")]; tensor var_8719_cast_fp16 = mul(x = var_8718_cast_fp16, y = var_878_cast_fp16)[name = string("op_8719_cast_fp16")]; tensor key_states_245_cast_fp16 = add(x = var_8712_cast_fp16, y = var_8719_cast_fp16)[name = string("key_states_245_cast_fp16")]; tensor expand_dims_288 = const()[name = string("expand_dims_288"), val = tensor([24])]; tensor expand_dims_289 = const()[name = string("expand_dims_289"), val = tensor([0])]; tensor expand_dims_291 = const()[name = string("expand_dims_291"), val = tensor([0])]; int32 concat_293_axis_0 = const()[name = string("concat_293_axis_0"), val = int32(0)]; bool concat_293_interleave_0 = const()[name = string("concat_293_interleave_0"), val = bool(false)]; tensor concat_293 = concat(axis = concat_293_axis_0, interleave = concat_293_interleave_0, values = (expand_dims_288, expand_dims_289, position_id, expand_dims_291))[name = string("concat_293")]; tensor expand_dims_292 = const()[name = string("expand_dims_292"), val = tensor([25])]; tensor concat_294_values1_0 = const()[name = string("concat_294_values1_0"), val = tensor([0])]; tensor concat_294_values3_0 = const()[name = string("concat_294_values3_0"), val = tensor([0])]; int32 concat_294_axis_0 = const()[name = string("concat_294_axis_0"), val = int32(0)]; bool concat_294_interleave_0 = const()[name = string("concat_294_interleave_0"), val = bool(false)]; tensor concat_294 = concat(axis = concat_294_axis_0, interleave = concat_294_interleave_0, values = (expand_dims_292, concat_294_values1_0, cache_position_end, concat_294_values3_0))[name = string("concat_294")]; tensor key_states_247_perm_0 = const()[name = string("key_states_247_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_25_stride_0 = const()[name = string("key_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_247_cast_fp16 = transpose(perm = key_states_247_perm_0, x = key_states_245_cast_fp16)[name = string("transpose_97")]; tensor key_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = key_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = key_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_25_squeeze_mask_0, stride = key_cache_internal_tensor_assign_25_stride_0, update = key_states_247_cast_fp16, x = coreml_update_state_102)[name = string("key_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_25_cast_fp16, input = key_cache)[name = string("coreml_update_state_104_write_state")]; tensor coreml_update_state_104 = read_state(input = key_cache)[name = string("coreml_update_state_104")]; tensor value_states_147_perm_0 = const()[name = string("value_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_25_stride_0 = const()[name = string("value_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_147_cast_fp16 = transpose(perm = value_states_147_perm_0, x = var_8695_cast_fp16)[name = string("transpose_96")]; tensor value_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = value_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = value_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_25_squeeze_mask_0, stride = value_cache_internal_tensor_assign_25_stride_0, update = value_states_147_cast_fp16, x = coreml_update_state_103)[name = string("value_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_25_cast_fp16, input = value_cache)[name = string("coreml_update_state_105_write_state")]; tensor coreml_update_state_105 = read_state(input = value_cache)[name = string("coreml_update_state_105")]; tensor var_8789_begin_0 = const()[name = string("op_8789_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8789_end_0 = const()[name = string("op_8789_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8789_end_mask_0 = const()[name = string("op_8789_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8789_cast_fp16 = slice_by_index(begin = var_8789_begin_0, end = var_8789_end_0, end_mask = var_8789_end_mask_0, x = coreml_update_state_104)[name = string("op_8789_cast_fp16")]; tensor tile_48 = const()[name = string("tile_48"), val = tensor([1, 1])]; int32 var_8792_axis_0 = const()[name = string("op_8792_axis_0"), val = int32(1)]; tensor var_8792_cast_fp16_0, tensor var_8792_cast_fp16_1 = split(axis = var_8792_axis_0, split_sizes = tile_48, x = var_8789_cast_fp16)[name = string("op_8792_cast_fp16")]; tensor var_8799_begin_0 = const()[name = string("op_8799_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8799_end_0 = const()[name = string("op_8799_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8799_end_mask_0 = const()[name = string("op_8799_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8799_cast_fp16 = slice_by_index(begin = var_8799_begin_0, end = var_8799_end_0, end_mask = var_8799_end_mask_0, x = coreml_update_state_105)[name = string("op_8799_cast_fp16")]; tensor tile_49 = const()[name = string("tile_49"), val = tensor([1, 1])]; int32 var_8802_axis_0 = const()[name = string("op_8802_axis_0"), val = int32(1)]; tensor var_8802_cast_fp16_0, tensor var_8802_cast_fp16_1 = split(axis = var_8802_axis_0, split_sizes = tile_49, x = var_8799_cast_fp16)[name = string("op_8802_cast_fp16")]; tensor var_8805_split_sizes_0 = const()[name = string("op_8805_split_sizes_0"), val = tensor([8, 8])]; int32 var_8805_axis_0 = const()[name = string("op_8805_axis_0"), val = int32(1)]; tensor var_8805_0, tensor var_8805_1 = split(axis = var_8805_axis_0, split_sizes = var_8805_split_sizes_0, x = query_states_147_cast_fp16)[name = string("op_8805")]; bool attn_weights_385_transpose_x_0 = const()[name = string("attn_weights_385_transpose_x_0"), val = bool(false)]; bool attn_weights_385_transpose_y_0 = const()[name = string("attn_weights_385_transpose_y_0"), val = bool(false)]; tensor attn_weights_385_cast_fp16 = matmul(transpose_x = attn_weights_385_transpose_x_0, transpose_y = attn_weights_385_transpose_y_0, x = var_8792_cast_fp16_0, y = var_8805_0)[name = string("attn_weights_385_cast_fp16")]; fp16 var_8808_to_fp16 = const()[name = string("op_8808_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_387_cast_fp16 = mul(x = attn_weights_385_cast_fp16, y = var_8808_to_fp16)[name = string("attn_weights_387_cast_fp16")]; tensor attn_weights_389_cast_fp16 = add(x = attn_weights_387_cast_fp16, y = attn_mask_1)[name = string("attn_weights_389_cast_fp16")]; int32 var_8812 = const()[name = string("op_8812"), val = int32(-2)]; tensor attn_weights_391_cast_fp16 = softmax(axis = var_8812, x = attn_weights_389_cast_fp16)[name = string("attn_weights_391_cast_fp16")]; bool var_8818_transpose_x_1 = const()[name = string("op_8818_transpose_x_1"), val = bool(true)]; bool var_8818_transpose_y_1 = const()[name = string("op_8818_transpose_y_1"), val = bool(false)]; tensor var_8818_cast_fp16 = matmul(transpose_x = var_8818_transpose_x_1, transpose_y = var_8818_transpose_y_1, x = attn_weights_391_cast_fp16, y = var_8802_cast_fp16_0)[name = string("op_8818_cast_fp16")]; bool attn_weights_393_transpose_x_0 = const()[name = string("attn_weights_393_transpose_x_0"), val = bool(false)]; bool attn_weights_393_transpose_y_0 = const()[name = string("attn_weights_393_transpose_y_0"), val = bool(false)]; tensor attn_weights_393_cast_fp16 = matmul(transpose_x = attn_weights_393_transpose_x_0, transpose_y = attn_weights_393_transpose_y_0, x = var_8792_cast_fp16_1, y = var_8805_1)[name = string("attn_weights_393_cast_fp16")]; fp16 var_8820_to_fp16 = const()[name = string("op_8820_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_395_cast_fp16 = mul(x = attn_weights_393_cast_fp16, y = var_8820_to_fp16)[name = string("attn_weights_395_cast_fp16")]; tensor attn_weights_397_cast_fp16 = add(x = attn_weights_395_cast_fp16, y = attn_mask_1)[name = string("attn_weights_397_cast_fp16")]; int32 var_8824 = const()[name = string("op_8824"), val = int32(-2)]; tensor attn_weights_399_cast_fp16 = softmax(axis = var_8824, x = attn_weights_397_cast_fp16)[name = string("attn_weights_399_cast_fp16")]; bool attn_output_193_transpose_x_1 = const()[name = string("attn_output_193_transpose_x_1"), val = bool(true)]; bool attn_output_193_transpose_y_1 = const()[name = string("attn_output_193_transpose_y_1"), val = bool(false)]; tensor attn_output_193_cast_fp16 = matmul(transpose_x = attn_output_193_transpose_x_1, transpose_y = attn_output_193_transpose_y_1, x = attn_weights_399_cast_fp16, y = var_8802_cast_fp16_1)[name = string("attn_output_193_cast_fp16")]; int32 var_8832 = const()[name = string("op_8832"), val = int32(1)]; bool attn_output_195_interleave_0 = const()[name = string("attn_output_195_interleave_0"), val = bool(false)]; tensor attn_output_195_cast_fp16 = concat(axis = var_8832, interleave = attn_output_195_interleave_0, values = (var_8818_cast_fp16, attn_output_193_cast_fp16))[name = string("attn_output_195_cast_fp16")]; tensor var_8836_perm_0 = const()[name = string("op_8836_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_299x = const()[name = string("concat_299x"), val = tensor([1, 2048, 1, -1])]; tensor var_8836_cast_fp16 = transpose(perm = var_8836_perm_0, x = attn_output_195_cast_fp16)[name = string("transpose_95")]; tensor attn_output_199_cast_fp16 = reshape(shape = concat_299x, x = var_8836_cast_fp16)[name = string("attn_output_199_cast_fp16")]; tensor hidden_states_243_strides_0 = const()[name = string("hidden_states_243_strides_0"), val = tensor([1, 1])]; string hidden_states_243_pad_type_0 = const()[name = string("hidden_states_243_pad_type_0"), val = string("valid")]; tensor hidden_states_243_pad_0 = const()[name = string("hidden_states_243_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_243_dilations_0 = const()[name = string("hidden_states_243_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_243_groups_0 = const()[name = string("hidden_states_243_groups_0"), val = int32(1)]; tensor hidden_states_243_cast_fp16 = conv(dilations = hidden_states_243_dilations_0, groups = hidden_states_243_groups_0, pad = hidden_states_243_pad_0, pad_type = hidden_states_243_pad_type_0, strides = hidden_states_243_strides_0, weight = layers_24_self_attn_o_proj_weight_cast_fp16, x = attn_output_199_cast_fp16)[name = string("hidden_states_243_cast_fp16")]; tensor hidden_states_245_cast_fp16 = add(x = hidden_states_239_cast_fp16, y = hidden_states_243_cast_fp16)[name = string("hidden_states_245_cast_fp16")]; fp16 const_248_promoted_to_fp16 = const()[name = string("const_248_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8869_cast_fp16 = mul(x = hidden_states_245_cast_fp16, y = const_248_promoted_to_fp16)[name = string("op_8869_cast_fp16")]; int32 var_8867 = const()[name = string("op_8867"), val = int32(1)]; bool doubled_197_interleave_0 = const()[name = string("doubled_197_interleave_0"), val = bool(false)]; tensor doubled_197_cast_fp16 = concat(axis = var_8867, interleave = doubled_197_interleave_0, values = (hidden_states_245_cast_fp16, var_8869_cast_fp16))[name = string("doubled_197_cast_fp16")]; tensor out_99_axes_0 = const()[name = string("out_99_axes_0"), val = tensor([1])]; tensor out_99_gamma_0_to_fp16 = const()[name = string("out_99_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540362432)))]; fp16 var_8879_to_fp16 = const()[name = string("op_8879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_99_cast_fp16 = layer_norm(axes = out_99_axes_0, epsilon = var_8879_to_fp16, gamma = out_99_gamma_0_to_fp16, x = doubled_197_cast_fp16)[name = string("out_99_cast_fp16")]; tensor var_8890_split_sizes_0 = const()[name = string("op_8890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8890_axis_0 = const()[name = string("op_8890_axis_0"), val = int32(1)]; tensor var_8890_cast_fp16_0, tensor var_8890_cast_fp16_1 = split(axis = var_8890_axis_0, split_sizes = var_8890_split_sizes_0, x = out_99_cast_fp16)[name = string("op_8890_cast_fp16")]; tensor input_49_strides_0 = const()[name = string("input_49_strides_0"), val = tensor([1, 1])]; string input_49_pad_type_0 = const()[name = string("input_49_pad_type_0"), val = string("valid")]; tensor input_49_pad_0 = const()[name = string("input_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_49_dilations_0 = const()[name = string("input_49_dilations_0"), val = tensor([1, 1])]; int32 input_49_groups_0 = const()[name = string("input_49_groups_0"), val = int32(1)]; tensor input_49_cast_fp16 = conv(dilations = input_49_dilations_0, groups = input_49_groups_0, pad = input_49_pad_0, pad_type = input_49_pad_type_0, strides = input_49_strides_0, weight = layers_24_mlp_gate_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("input_49_cast_fp16")]; tensor var_8907_cast_fp16 = silu(x = input_49_cast_fp16)[name = string("op_8907_cast_fp16")]; tensor var_8913_strides_0 = const()[name = string("op_8913_strides_0"), val = tensor([1, 1])]; string var_8913_pad_type_0 = const()[name = string("op_8913_pad_type_0"), val = string("valid")]; tensor var_8913_pad_0 = const()[name = string("op_8913_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8913_dilations_0 = const()[name = string("op_8913_dilations_0"), val = tensor([1, 1])]; int32 var_8913_groups_0 = const()[name = string("op_8913_groups_0"), val = int32(1)]; tensor var_8913_cast_fp16 = conv(dilations = var_8913_dilations_0, groups = var_8913_groups_0, pad = var_8913_pad_0, pad_type = var_8913_pad_type_0, strides = var_8913_strides_0, weight = layers_24_mlp_up_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("op_8913_cast_fp16")]; tensor x_249_cast_fp16 = mul(x = var_8907_cast_fp16, y = var_8913_cast_fp16)[name = string("x_249_cast_fp16")]; tensor hidden_states_247_strides_0 = const()[name = string("hidden_states_247_strides_0"), val = tensor([1, 1])]; string hidden_states_247_pad_type_0 = const()[name = string("hidden_states_247_pad_type_0"), val = string("valid")]; tensor hidden_states_247_pad_0 = const()[name = string("hidden_states_247_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_247_dilations_0 = const()[name = string("hidden_states_247_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_247_groups_0 = const()[name = string("hidden_states_247_groups_0"), val = int32(1)]; tensor hidden_states_247_cast_fp16 = conv(dilations = hidden_states_247_dilations_0, groups = hidden_states_247_groups_0, pad = hidden_states_247_pad_0, pad_type = hidden_states_247_pad_type_0, strides = hidden_states_247_strides_0, weight = layers_24_mlp_down_proj_weight_cast_fp16, x = x_249_cast_fp16)[name = string("hidden_states_247_cast_fp16")]; tensor hidden_states_249_cast_fp16 = add(x = hidden_states_245_cast_fp16, y = hidden_states_247_cast_fp16)[name = string("hidden_states_249_cast_fp16")]; fp16 const_250_promoted_to_fp16 = const()[name = string("const_250_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8931_cast_fp16 = mul(x = hidden_states_249_cast_fp16, y = const_250_promoted_to_fp16)[name = string("op_8931_cast_fp16")]; int32 var_8929 = const()[name = string("op_8929"), val = int32(1)]; bool doubled_201_interleave_0 = const()[name = string("doubled_201_interleave_0"), val = bool(false)]; tensor doubled_201_cast_fp16 = concat(axis = var_8929, interleave = doubled_201_interleave_0, values = (hidden_states_249_cast_fp16, var_8931_cast_fp16))[name = string("doubled_201_cast_fp16")]; tensor out_101_axes_0 = const()[name = string("out_101_axes_0"), val = tensor([1])]; tensor out_101_gamma_0_to_fp16 = const()[name = string("out_101_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540370688)))]; fp16 var_8941_to_fp16 = const()[name = string("op_8941_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_101_cast_fp16 = layer_norm(axes = out_101_axes_0, epsilon = var_8941_to_fp16, gamma = out_101_gamma_0_to_fp16, x = doubled_201_cast_fp16)[name = string("out_101_cast_fp16")]; tensor var_8952_split_sizes_0 = const()[name = string("op_8952_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8952_axis_0 = const()[name = string("op_8952_axis_0"), val = int32(1)]; tensor var_8952_cast_fp16_0, tensor var_8952_cast_fp16_1 = split(axis = var_8952_axis_0, split_sizes = var_8952_split_sizes_0, x = out_101_cast_fp16)[name = string("op_8952_cast_fp16")]; tensor query_states_151_strides_0 = const()[name = string("query_states_151_strides_0"), val = tensor([1, 1])]; string query_states_151_pad_type_0 = const()[name = string("query_states_151_pad_type_0"), val = string("valid")]; tensor query_states_151_pad_0 = const()[name = string("query_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_151_dilations_0 = const()[name = string("query_states_151_dilations_0"), val = tensor([1, 1])]; int32 query_states_151_groups_0 = const()[name = string("query_states_151_groups_0"), val = int32(1)]; tensor query_states_151_cast_fp16 = conv(dilations = query_states_151_dilations_0, groups = query_states_151_groups_0, pad = query_states_151_pad_0, pad_type = query_states_151_pad_type_0, strides = query_states_151_strides_0, weight = layers_25_self_attn_q_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("query_states_151_cast_fp16")]; tensor key_states_251_strides_0 = const()[name = string("key_states_251_strides_0"), val = tensor([1, 1])]; string key_states_251_pad_type_0 = const()[name = string("key_states_251_pad_type_0"), val = string("valid")]; tensor key_states_251_pad_0 = const()[name = string("key_states_251_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_251_dilations_0 = const()[name = string("key_states_251_dilations_0"), val = tensor([1, 1])]; int32 key_states_251_groups_0 = const()[name = string("key_states_251_groups_0"), val = int32(1)]; tensor key_states_251_cast_fp16 = conv(dilations = key_states_251_dilations_0, groups = key_states_251_groups_0, pad = key_states_251_pad_0, pad_type = key_states_251_pad_type_0, strides = key_states_251_strides_0, weight = layers_25_self_attn_k_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("key_states_251_cast_fp16")]; tensor layers_25_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_25_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540378944)))]; tensor value_states_151_strides_0 = const()[name = string("value_states_151_strides_0"), val = tensor([1, 1])]; string value_states_151_pad_type_0 = const()[name = string("value_states_151_pad_type_0"), val = string("valid")]; tensor value_states_151_pad_0 = const()[name = string("value_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_151_dilations_0 = const()[name = string("value_states_151_dilations_0"), val = tensor([1, 1])]; int32 value_states_151_groups_0 = const()[name = string("value_states_151_groups_0"), val = int32(1)]; tensor value_states_151_cast_fp16 = conv(dilations = value_states_151_dilations_0, groups = value_states_151_groups_0, pad = value_states_151_pad_0, pad_type = value_states_151_pad_type_0, strides = value_states_151_strides_0, weight = layers_25_self_attn_v_proj_weight_to_fp16, x = var_8952_cast_fp16_0)[name = string("value_states_151_cast_fp16")]; tensor concat_300x = const()[name = string("concat_300x"), val = tensor([1, 16, 128, -1])]; tensor x_251_cast_fp16 = reshape(shape = concat_300x, x = query_states_151_cast_fp16)[name = string("x_251_cast_fp16")]; tensor concat_301x = const()[name = string("concat_301x"), val = tensor([1, 2, 128, -1])]; tensor var_9009_cast_fp16 = reshape(shape = concat_301x, x = key_states_251_cast_fp16)[name = string("op_9009_cast_fp16")]; tensor concat_302x = const()[name = string("concat_302x"), val = tensor([1, 2, 128, -1])]; tensor var_9016_cast_fp16 = reshape(shape = concat_302x, x = value_states_151_cast_fp16)[name = string("op_9016_cast_fp16")]; tensor var_9020_cast_fp16 = mul(x = x_251_cast_fp16, y = var_869_cast_fp16)[name = string("op_9020_cast_fp16")]; tensor var_9021_split_sizes_0 = const()[name = string("op_9021_split_sizes_0"), val = tensor([64, 64])]; int32 var_9021_axis_0 = const()[name = string("op_9021_axis_0"), val = int32(-2)]; tensor var_9021_cast_fp16_0, tensor var_9021_cast_fp16_1 = split(axis = var_9021_axis_0, split_sizes = var_9021_split_sizes_0, x = x_251_cast_fp16)[name = string("op_9021_cast_fp16")]; fp16 const_252_promoted_to_fp16 = const()[name = string("const_252_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9023_cast_fp16 = mul(x = var_9021_cast_fp16_1, y = const_252_promoted_to_fp16)[name = string("op_9023_cast_fp16")]; int32 var_9025 = const()[name = string("op_9025"), val = int32(-2)]; bool var_9026_interleave_0 = const()[name = string("op_9026_interleave_0"), val = bool(false)]; tensor var_9026_cast_fp16 = concat(axis = var_9025, interleave = var_9026_interleave_0, values = (var_9023_cast_fp16, var_9021_cast_fp16_0))[name = string("op_9026_cast_fp16")]; tensor var_9027_cast_fp16 = mul(x = var_9026_cast_fp16, y = var_878_cast_fp16)[name = string("op_9027_cast_fp16")]; tensor query_states_153_cast_fp16 = add(x = var_9020_cast_fp16, y = var_9027_cast_fp16)[name = string("query_states_153_cast_fp16")]; tensor var_9033_cast_fp16 = mul(x = var_9009_cast_fp16, y = var_869_cast_fp16)[name = string("op_9033_cast_fp16")]; tensor var_9034_split_sizes_0 = const()[name = string("op_9034_split_sizes_0"), val = tensor([64, 64])]; int32 var_9034_axis_0 = const()[name = string("op_9034_axis_0"), val = int32(-2)]; tensor var_9034_cast_fp16_0, tensor var_9034_cast_fp16_1 = split(axis = var_9034_axis_0, split_sizes = var_9034_split_sizes_0, x = var_9009_cast_fp16)[name = string("op_9034_cast_fp16")]; fp16 const_253_promoted_to_fp16 = const()[name = string("const_253_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9036_cast_fp16 = mul(x = var_9034_cast_fp16_1, y = const_253_promoted_to_fp16)[name = string("op_9036_cast_fp16")]; int32 var_9038 = const()[name = string("op_9038"), val = int32(-2)]; bool var_9039_interleave_0 = const()[name = string("op_9039_interleave_0"), val = bool(false)]; tensor var_9039_cast_fp16 = concat(axis = var_9038, interleave = var_9039_interleave_0, values = (var_9036_cast_fp16, var_9034_cast_fp16_0))[name = string("op_9039_cast_fp16")]; tensor var_9040_cast_fp16 = mul(x = var_9039_cast_fp16, y = var_878_cast_fp16)[name = string("op_9040_cast_fp16")]; tensor key_states_255_cast_fp16 = add(x = var_9033_cast_fp16, y = var_9040_cast_fp16)[name = string("key_states_255_cast_fp16")]; tensor expand_dims_300 = const()[name = string("expand_dims_300"), val = tensor([25])]; tensor expand_dims_301 = const()[name = string("expand_dims_301"), val = tensor([0])]; tensor expand_dims_303 = const()[name = string("expand_dims_303"), val = tensor([0])]; int32 concat_305_axis_0 = const()[name = string("concat_305_axis_0"), val = int32(0)]; bool concat_305_interleave_0 = const()[name = string("concat_305_interleave_0"), val = bool(false)]; tensor concat_305 = concat(axis = concat_305_axis_0, interleave = concat_305_interleave_0, values = (expand_dims_300, expand_dims_301, position_id, expand_dims_303))[name = string("concat_305")]; tensor expand_dims_304 = const()[name = string("expand_dims_304"), val = tensor([26])]; tensor concat_306_values1_0 = const()[name = string("concat_306_values1_0"), val = tensor([0])]; tensor concat_306_values3_0 = const()[name = string("concat_306_values3_0"), val = tensor([0])]; int32 concat_306_axis_0 = const()[name = string("concat_306_axis_0"), val = int32(0)]; bool concat_306_interleave_0 = const()[name = string("concat_306_interleave_0"), val = bool(false)]; tensor concat_306 = concat(axis = concat_306_axis_0, interleave = concat_306_interleave_0, values = (expand_dims_304, concat_306_values1_0, cache_position_end, concat_306_values3_0))[name = string("concat_306")]; tensor key_states_257_perm_0 = const()[name = string("key_states_257_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_26_stride_0 = const()[name = string("key_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_257_cast_fp16 = transpose(perm = key_states_257_perm_0, x = key_states_255_cast_fp16)[name = string("transpose_94")]; tensor key_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = key_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = key_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_26_squeeze_mask_0, stride = key_cache_internal_tensor_assign_26_stride_0, update = key_states_257_cast_fp16, x = coreml_update_state_104)[name = string("key_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_26_cast_fp16, input = key_cache)[name = string("coreml_update_state_106_write_state")]; tensor coreml_update_state_106 = read_state(input = key_cache)[name = string("coreml_update_state_106")]; tensor value_states_153_perm_0 = const()[name = string("value_states_153_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_26_stride_0 = const()[name = string("value_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_153_cast_fp16 = transpose(perm = value_states_153_perm_0, x = var_9016_cast_fp16)[name = string("transpose_93")]; tensor value_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = value_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = value_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_26_squeeze_mask_0, stride = value_cache_internal_tensor_assign_26_stride_0, update = value_states_153_cast_fp16, x = coreml_update_state_105)[name = string("value_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_26_cast_fp16, input = value_cache)[name = string("coreml_update_state_107_write_state")]; tensor coreml_update_state_107 = read_state(input = value_cache)[name = string("coreml_update_state_107")]; tensor var_9110_begin_0 = const()[name = string("op_9110_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9110_end_0 = const()[name = string("op_9110_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9110_end_mask_0 = const()[name = string("op_9110_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9110_cast_fp16 = slice_by_index(begin = var_9110_begin_0, end = var_9110_end_0, end_mask = var_9110_end_mask_0, x = coreml_update_state_106)[name = string("op_9110_cast_fp16")]; tensor tile_50 = const()[name = string("tile_50"), val = tensor([1, 1])]; int32 var_9113_axis_0 = const()[name = string("op_9113_axis_0"), val = int32(1)]; tensor var_9113_cast_fp16_0, tensor var_9113_cast_fp16_1 = split(axis = var_9113_axis_0, split_sizes = tile_50, x = var_9110_cast_fp16)[name = string("op_9113_cast_fp16")]; tensor var_9120_begin_0 = const()[name = string("op_9120_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9120_end_0 = const()[name = string("op_9120_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9120_end_mask_0 = const()[name = string("op_9120_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9120_cast_fp16 = slice_by_index(begin = var_9120_begin_0, end = var_9120_end_0, end_mask = var_9120_end_mask_0, x = coreml_update_state_107)[name = string("op_9120_cast_fp16")]; tensor tile_51 = const()[name = string("tile_51"), val = tensor([1, 1])]; int32 var_9123_axis_0 = const()[name = string("op_9123_axis_0"), val = int32(1)]; tensor var_9123_cast_fp16_0, tensor var_9123_cast_fp16_1 = split(axis = var_9123_axis_0, split_sizes = tile_51, x = var_9120_cast_fp16)[name = string("op_9123_cast_fp16")]; tensor var_9126_split_sizes_0 = const()[name = string("op_9126_split_sizes_0"), val = tensor([8, 8])]; int32 var_9126_axis_0 = const()[name = string("op_9126_axis_0"), val = int32(1)]; tensor var_9126_0, tensor var_9126_1 = split(axis = var_9126_axis_0, split_sizes = var_9126_split_sizes_0, x = query_states_153_cast_fp16)[name = string("op_9126")]; bool attn_weights_401_transpose_x_0 = const()[name = string("attn_weights_401_transpose_x_0"), val = bool(false)]; bool attn_weights_401_transpose_y_0 = const()[name = string("attn_weights_401_transpose_y_0"), val = bool(false)]; tensor attn_weights_401_cast_fp16 = matmul(transpose_x = attn_weights_401_transpose_x_0, transpose_y = attn_weights_401_transpose_y_0, x = var_9113_cast_fp16_0, y = var_9126_0)[name = string("attn_weights_401_cast_fp16")]; fp16 var_9129_to_fp16 = const()[name = string("op_9129_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_403_cast_fp16 = mul(x = attn_weights_401_cast_fp16, y = var_9129_to_fp16)[name = string("attn_weights_403_cast_fp16")]; tensor attn_weights_405_cast_fp16 = add(x = attn_weights_403_cast_fp16, y = attn_mask_1)[name = string("attn_weights_405_cast_fp16")]; int32 var_9133 = const()[name = string("op_9133"), val = int32(-2)]; tensor attn_weights_407_cast_fp16 = softmax(axis = var_9133, x = attn_weights_405_cast_fp16)[name = string("attn_weights_407_cast_fp16")]; bool var_9139_transpose_x_1 = const()[name = string("op_9139_transpose_x_1"), val = bool(true)]; bool var_9139_transpose_y_1 = const()[name = string("op_9139_transpose_y_1"), val = bool(false)]; tensor var_9139_cast_fp16 = matmul(transpose_x = var_9139_transpose_x_1, transpose_y = var_9139_transpose_y_1, x = attn_weights_407_cast_fp16, y = var_9123_cast_fp16_0)[name = string("op_9139_cast_fp16")]; bool attn_weights_409_transpose_x_0 = const()[name = string("attn_weights_409_transpose_x_0"), val = bool(false)]; bool attn_weights_409_transpose_y_0 = const()[name = string("attn_weights_409_transpose_y_0"), val = bool(false)]; tensor attn_weights_409_cast_fp16 = matmul(transpose_x = attn_weights_409_transpose_x_0, transpose_y = attn_weights_409_transpose_y_0, x = var_9113_cast_fp16_1, y = var_9126_1)[name = string("attn_weights_409_cast_fp16")]; fp16 var_9141_to_fp16 = const()[name = string("op_9141_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_411_cast_fp16 = mul(x = attn_weights_409_cast_fp16, y = var_9141_to_fp16)[name = string("attn_weights_411_cast_fp16")]; tensor attn_weights_413_cast_fp16 = add(x = attn_weights_411_cast_fp16, y = attn_mask_1)[name = string("attn_weights_413_cast_fp16")]; int32 var_9145 = const()[name = string("op_9145"), val = int32(-2)]; tensor attn_weights_415_cast_fp16 = softmax(axis = var_9145, x = attn_weights_413_cast_fp16)[name = string("attn_weights_415_cast_fp16")]; bool attn_output_201_transpose_x_1 = const()[name = string("attn_output_201_transpose_x_1"), val = bool(true)]; bool attn_output_201_transpose_y_1 = const()[name = string("attn_output_201_transpose_y_1"), val = bool(false)]; tensor attn_output_201_cast_fp16 = matmul(transpose_x = attn_output_201_transpose_x_1, transpose_y = attn_output_201_transpose_y_1, x = attn_weights_415_cast_fp16, y = var_9123_cast_fp16_1)[name = string("attn_output_201_cast_fp16")]; int32 var_9153 = const()[name = string("op_9153"), val = int32(1)]; bool attn_output_203_interleave_0 = const()[name = string("attn_output_203_interleave_0"), val = bool(false)]; tensor attn_output_203_cast_fp16 = concat(axis = var_9153, interleave = attn_output_203_interleave_0, values = (var_9139_cast_fp16, attn_output_201_cast_fp16))[name = string("attn_output_203_cast_fp16")]; tensor var_9157_perm_0 = const()[name = string("op_9157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_311x = const()[name = string("concat_311x"), val = tensor([1, 2048, 1, -1])]; tensor var_9157_cast_fp16 = transpose(perm = var_9157_perm_0, x = attn_output_203_cast_fp16)[name = string("transpose_92")]; tensor attn_output_207_cast_fp16 = reshape(shape = concat_311x, x = var_9157_cast_fp16)[name = string("attn_output_207_cast_fp16")]; tensor hidden_states_253_strides_0 = const()[name = string("hidden_states_253_strides_0"), val = tensor([1, 1])]; string hidden_states_253_pad_type_0 = const()[name = string("hidden_states_253_pad_type_0"), val = string("valid")]; tensor hidden_states_253_pad_0 = const()[name = string("hidden_states_253_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_253_dilations_0 = const()[name = string("hidden_states_253_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_253_groups_0 = const()[name = string("hidden_states_253_groups_0"), val = int32(1)]; tensor hidden_states_253_cast_fp16 = conv(dilations = hidden_states_253_dilations_0, groups = hidden_states_253_groups_0, pad = hidden_states_253_pad_0, pad_type = hidden_states_253_pad_type_0, strides = hidden_states_253_strides_0, weight = layers_25_self_attn_o_proj_weight_cast_fp16, x = attn_output_207_cast_fp16)[name = string("hidden_states_253_cast_fp16")]; tensor hidden_states_255_cast_fp16 = add(x = hidden_states_249_cast_fp16, y = hidden_states_253_cast_fp16)[name = string("hidden_states_255_cast_fp16")]; fp16 const_258_promoted_to_fp16 = const()[name = string("const_258_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9190_cast_fp16 = mul(x = hidden_states_255_cast_fp16, y = const_258_promoted_to_fp16)[name = string("op_9190_cast_fp16")]; int32 var_9188 = const()[name = string("op_9188"), val = int32(1)]; bool doubled_205_interleave_0 = const()[name = string("doubled_205_interleave_0"), val = bool(false)]; tensor doubled_205_cast_fp16 = concat(axis = var_9188, interleave = doubled_205_interleave_0, values = (hidden_states_255_cast_fp16, var_9190_cast_fp16))[name = string("doubled_205_cast_fp16")]; tensor out_103_axes_0 = const()[name = string("out_103_axes_0"), val = tensor([1])]; tensor out_103_gamma_0_to_fp16 = const()[name = string("out_103_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541427584)))]; fp16 var_9200_to_fp16 = const()[name = string("op_9200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_103_cast_fp16 = layer_norm(axes = out_103_axes_0, epsilon = var_9200_to_fp16, gamma = out_103_gamma_0_to_fp16, x = doubled_205_cast_fp16)[name = string("out_103_cast_fp16")]; tensor var_9211_split_sizes_0 = const()[name = string("op_9211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9211_axis_0 = const()[name = string("op_9211_axis_0"), val = int32(1)]; tensor var_9211_cast_fp16_0, tensor var_9211_cast_fp16_1 = split(axis = var_9211_axis_0, split_sizes = var_9211_split_sizes_0, x = out_103_cast_fp16)[name = string("op_9211_cast_fp16")]; tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; tensor input_51_cast_fp16 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = layers_25_mlp_gate_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("input_51_cast_fp16")]; tensor var_9228_cast_fp16 = silu(x = input_51_cast_fp16)[name = string("op_9228_cast_fp16")]; tensor var_9234_strides_0 = const()[name = string("op_9234_strides_0"), val = tensor([1, 1])]; string var_9234_pad_type_0 = const()[name = string("op_9234_pad_type_0"), val = string("valid")]; tensor var_9234_pad_0 = const()[name = string("op_9234_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9234_dilations_0 = const()[name = string("op_9234_dilations_0"), val = tensor([1, 1])]; int32 var_9234_groups_0 = const()[name = string("op_9234_groups_0"), val = int32(1)]; tensor var_9234_cast_fp16 = conv(dilations = var_9234_dilations_0, groups = var_9234_groups_0, pad = var_9234_pad_0, pad_type = var_9234_pad_type_0, strides = var_9234_strides_0, weight = layers_25_mlp_up_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("op_9234_cast_fp16")]; tensor x_259_cast_fp16 = mul(x = var_9228_cast_fp16, y = var_9234_cast_fp16)[name = string("x_259_cast_fp16")]; tensor layers_25_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_25_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541435840)))]; tensor hidden_states_257_strides_0 = const()[name = string("hidden_states_257_strides_0"), val = tensor([1, 1])]; string hidden_states_257_pad_type_0 = const()[name = string("hidden_states_257_pad_type_0"), val = string("valid")]; tensor hidden_states_257_pad_0 = const()[name = string("hidden_states_257_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_257_dilations_0 = const()[name = string("hidden_states_257_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_257_groups_0 = const()[name = string("hidden_states_257_groups_0"), val = int32(1)]; tensor hidden_states_257_cast_fp16 = conv(dilations = hidden_states_257_dilations_0, groups = hidden_states_257_groups_0, pad = hidden_states_257_pad_0, pad_type = hidden_states_257_pad_type_0, strides = hidden_states_257_strides_0, weight = layers_25_mlp_down_proj_weight_to_fp16, x = x_259_cast_fp16)[name = string("hidden_states_257_cast_fp16")]; tensor hidden_states_259_cast_fp16 = add(x = hidden_states_255_cast_fp16, y = hidden_states_257_cast_fp16)[name = string("hidden_states_259_cast_fp16")]; fp16 const_260_promoted_to_fp16 = const()[name = string("const_260_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9252_cast_fp16 = mul(x = hidden_states_259_cast_fp16, y = const_260_promoted_to_fp16)[name = string("op_9252_cast_fp16")]; int32 var_9250 = const()[name = string("op_9250"), val = int32(1)]; bool doubled_209_interleave_0 = const()[name = string("doubled_209_interleave_0"), val = bool(false)]; tensor doubled_209_cast_fp16 = concat(axis = var_9250, interleave = doubled_209_interleave_0, values = (hidden_states_259_cast_fp16, var_9252_cast_fp16))[name = string("doubled_209_cast_fp16")]; tensor out_105_axes_0 = const()[name = string("out_105_axes_0"), val = tensor([1])]; tensor out_105_gamma_0_to_fp16 = const()[name = string("out_105_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566601728)))]; fp16 var_9262_to_fp16 = const()[name = string("op_9262_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_105_cast_fp16 = layer_norm(axes = out_105_axes_0, epsilon = var_9262_to_fp16, gamma = out_105_gamma_0_to_fp16, x = doubled_209_cast_fp16)[name = string("out_105_cast_fp16")]; tensor var_9273_split_sizes_0 = const()[name = string("op_9273_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9273_axis_0 = const()[name = string("op_9273_axis_0"), val = int32(1)]; tensor var_9273_cast_fp16_0, tensor var_9273_cast_fp16_1 = split(axis = var_9273_axis_0, split_sizes = var_9273_split_sizes_0, x = out_105_cast_fp16)[name = string("op_9273_cast_fp16")]; tensor query_states_157_strides_0 = const()[name = string("query_states_157_strides_0"), val = tensor([1, 1])]; string query_states_157_pad_type_0 = const()[name = string("query_states_157_pad_type_0"), val = string("valid")]; tensor query_states_157_pad_0 = const()[name = string("query_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_157_dilations_0 = const()[name = string("query_states_157_dilations_0"), val = tensor([1, 1])]; int32 query_states_157_groups_0 = const()[name = string("query_states_157_groups_0"), val = int32(1)]; tensor query_states_157_cast_fp16 = conv(dilations = query_states_157_dilations_0, groups = query_states_157_groups_0, pad = query_states_157_pad_0, pad_type = query_states_157_pad_type_0, strides = query_states_157_strides_0, weight = layers_26_self_attn_q_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("query_states_157_cast_fp16")]; tensor key_states_261_strides_0 = const()[name = string("key_states_261_strides_0"), val = tensor([1, 1])]; string key_states_261_pad_type_0 = const()[name = string("key_states_261_pad_type_0"), val = string("valid")]; tensor key_states_261_pad_0 = const()[name = string("key_states_261_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_261_dilations_0 = const()[name = string("key_states_261_dilations_0"), val = tensor([1, 1])]; int32 key_states_261_groups_0 = const()[name = string("key_states_261_groups_0"), val = int32(1)]; tensor key_states_261_cast_fp16 = conv(dilations = key_states_261_dilations_0, groups = key_states_261_groups_0, pad = key_states_261_pad_0, pad_type = key_states_261_pad_type_0, strides = key_states_261_strides_0, weight = layers_26_self_attn_k_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("key_states_261_cast_fp16")]; tensor layers_26_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_26_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566609984)))]; tensor value_states_157_strides_0 = const()[name = string("value_states_157_strides_0"), val = tensor([1, 1])]; string value_states_157_pad_type_0 = const()[name = string("value_states_157_pad_type_0"), val = string("valid")]; tensor value_states_157_pad_0 = const()[name = string("value_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_157_dilations_0 = const()[name = string("value_states_157_dilations_0"), val = tensor([1, 1])]; int32 value_states_157_groups_0 = const()[name = string("value_states_157_groups_0"), val = int32(1)]; tensor value_states_157_cast_fp16 = conv(dilations = value_states_157_dilations_0, groups = value_states_157_groups_0, pad = value_states_157_pad_0, pad_type = value_states_157_pad_type_0, strides = value_states_157_strides_0, weight = layers_26_self_attn_v_proj_weight_to_fp16, x = var_9273_cast_fp16_0)[name = string("value_states_157_cast_fp16")]; tensor concat_312x = const()[name = string("concat_312x"), val = tensor([1, 16, 128, -1])]; tensor x_261_cast_fp16 = reshape(shape = concat_312x, x = query_states_157_cast_fp16)[name = string("x_261_cast_fp16")]; tensor concat_313x = const()[name = string("concat_313x"), val = tensor([1, 2, 128, -1])]; tensor var_9330_cast_fp16 = reshape(shape = concat_313x, x = key_states_261_cast_fp16)[name = string("op_9330_cast_fp16")]; tensor concat_314x = const()[name = string("concat_314x"), val = tensor([1, 2, 128, -1])]; tensor var_9337_cast_fp16 = reshape(shape = concat_314x, x = value_states_157_cast_fp16)[name = string("op_9337_cast_fp16")]; tensor var_9341_cast_fp16 = mul(x = x_261_cast_fp16, y = var_869_cast_fp16)[name = string("op_9341_cast_fp16")]; tensor var_9342_split_sizes_0 = const()[name = string("op_9342_split_sizes_0"), val = tensor([64, 64])]; int32 var_9342_axis_0 = const()[name = string("op_9342_axis_0"), val = int32(-2)]; tensor var_9342_cast_fp16_0, tensor var_9342_cast_fp16_1 = split(axis = var_9342_axis_0, split_sizes = var_9342_split_sizes_0, x = x_261_cast_fp16)[name = string("op_9342_cast_fp16")]; fp16 const_262_promoted_to_fp16 = const()[name = string("const_262_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9344_cast_fp16 = mul(x = var_9342_cast_fp16_1, y = const_262_promoted_to_fp16)[name = string("op_9344_cast_fp16")]; int32 var_9346 = const()[name = string("op_9346"), val = int32(-2)]; bool var_9347_interleave_0 = const()[name = string("op_9347_interleave_0"), val = bool(false)]; tensor var_9347_cast_fp16 = concat(axis = var_9346, interleave = var_9347_interleave_0, values = (var_9344_cast_fp16, var_9342_cast_fp16_0))[name = string("op_9347_cast_fp16")]; tensor var_9348_cast_fp16 = mul(x = var_9347_cast_fp16, y = var_878_cast_fp16)[name = string("op_9348_cast_fp16")]; tensor query_states_159_cast_fp16 = add(x = var_9341_cast_fp16, y = var_9348_cast_fp16)[name = string("query_states_159_cast_fp16")]; tensor var_9354_cast_fp16 = mul(x = var_9330_cast_fp16, y = var_869_cast_fp16)[name = string("op_9354_cast_fp16")]; tensor var_9355_split_sizes_0 = const()[name = string("op_9355_split_sizes_0"), val = tensor([64, 64])]; int32 var_9355_axis_0 = const()[name = string("op_9355_axis_0"), val = int32(-2)]; tensor var_9355_cast_fp16_0, tensor var_9355_cast_fp16_1 = split(axis = var_9355_axis_0, split_sizes = var_9355_split_sizes_0, x = var_9330_cast_fp16)[name = string("op_9355_cast_fp16")]; fp16 const_263_promoted_to_fp16 = const()[name = string("const_263_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9357_cast_fp16 = mul(x = var_9355_cast_fp16_1, y = const_263_promoted_to_fp16)[name = string("op_9357_cast_fp16")]; int32 var_9359 = const()[name = string("op_9359"), val = int32(-2)]; bool var_9360_interleave_0 = const()[name = string("op_9360_interleave_0"), val = bool(false)]; tensor var_9360_cast_fp16 = concat(axis = var_9359, interleave = var_9360_interleave_0, values = (var_9357_cast_fp16, var_9355_cast_fp16_0))[name = string("op_9360_cast_fp16")]; tensor var_9361_cast_fp16 = mul(x = var_9360_cast_fp16, y = var_878_cast_fp16)[name = string("op_9361_cast_fp16")]; tensor key_states_265_cast_fp16 = add(x = var_9354_cast_fp16, y = var_9361_cast_fp16)[name = string("key_states_265_cast_fp16")]; tensor expand_dims_312 = const()[name = string("expand_dims_312"), val = tensor([26])]; tensor expand_dims_313 = const()[name = string("expand_dims_313"), val = tensor([0])]; tensor expand_dims_315 = const()[name = string("expand_dims_315"), val = tensor([0])]; int32 concat_317_axis_0 = const()[name = string("concat_317_axis_0"), val = int32(0)]; bool concat_317_interleave_0 = const()[name = string("concat_317_interleave_0"), val = bool(false)]; tensor concat_317 = concat(axis = concat_317_axis_0, interleave = concat_317_interleave_0, values = (expand_dims_312, expand_dims_313, position_id, expand_dims_315))[name = string("concat_317")]; tensor expand_dims_316 = const()[name = string("expand_dims_316"), val = tensor([27])]; tensor concat_318_values1_0 = const()[name = string("concat_318_values1_0"), val = tensor([0])]; tensor concat_318_values3_0 = const()[name = string("concat_318_values3_0"), val = tensor([0])]; int32 concat_318_axis_0 = const()[name = string("concat_318_axis_0"), val = int32(0)]; bool concat_318_interleave_0 = const()[name = string("concat_318_interleave_0"), val = bool(false)]; tensor concat_318 = concat(axis = concat_318_axis_0, interleave = concat_318_interleave_0, values = (expand_dims_316, concat_318_values1_0, cache_position_end, concat_318_values3_0))[name = string("concat_318")]; tensor key_states_267_perm_0 = const()[name = string("key_states_267_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_27_stride_0 = const()[name = string("key_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_267_cast_fp16 = transpose(perm = key_states_267_perm_0, x = key_states_265_cast_fp16)[name = string("transpose_91")]; tensor key_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = key_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = key_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_27_squeeze_mask_0, stride = key_cache_internal_tensor_assign_27_stride_0, update = key_states_267_cast_fp16, x = coreml_update_state_106)[name = string("key_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_27_cast_fp16, input = key_cache)[name = string("coreml_update_state_108_write_state")]; tensor coreml_update_state_108 = read_state(input = key_cache)[name = string("coreml_update_state_108")]; tensor value_states_159_perm_0 = const()[name = string("value_states_159_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_27_stride_0 = const()[name = string("value_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_159_cast_fp16 = transpose(perm = value_states_159_perm_0, x = var_9337_cast_fp16)[name = string("transpose_90")]; tensor value_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = value_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = value_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_27_squeeze_mask_0, stride = value_cache_internal_tensor_assign_27_stride_0, update = value_states_159_cast_fp16, x = coreml_update_state_107)[name = string("value_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_27_cast_fp16, input = value_cache)[name = string("coreml_update_state_109_write_state")]; tensor coreml_update_state_109 = read_state(input = value_cache)[name = string("coreml_update_state_109")]; tensor var_9431_begin_0 = const()[name = string("op_9431_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9431_end_0 = const()[name = string("op_9431_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9431_end_mask_0 = const()[name = string("op_9431_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9431_cast_fp16 = slice_by_index(begin = var_9431_begin_0, end = var_9431_end_0, end_mask = var_9431_end_mask_0, x = coreml_update_state_108)[name = string("op_9431_cast_fp16")]; tensor tile_52 = const()[name = string("tile_52"), val = tensor([1, 1])]; int32 var_9434_axis_0 = const()[name = string("op_9434_axis_0"), val = int32(1)]; tensor var_9434_cast_fp16_0, tensor var_9434_cast_fp16_1 = split(axis = var_9434_axis_0, split_sizes = tile_52, x = var_9431_cast_fp16)[name = string("op_9434_cast_fp16")]; tensor var_9441_begin_0 = const()[name = string("op_9441_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9441_end_0 = const()[name = string("op_9441_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9441_end_mask_0 = const()[name = string("op_9441_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9441_cast_fp16 = slice_by_index(begin = var_9441_begin_0, end = var_9441_end_0, end_mask = var_9441_end_mask_0, x = coreml_update_state_109)[name = string("op_9441_cast_fp16")]; tensor tile_53 = const()[name = string("tile_53"), val = tensor([1, 1])]; int32 var_9444_axis_0 = const()[name = string("op_9444_axis_0"), val = int32(1)]; tensor var_9444_cast_fp16_0, tensor var_9444_cast_fp16_1 = split(axis = var_9444_axis_0, split_sizes = tile_53, x = var_9441_cast_fp16)[name = string("op_9444_cast_fp16")]; tensor var_9447_split_sizes_0 = const()[name = string("op_9447_split_sizes_0"), val = tensor([8, 8])]; int32 var_9447_axis_0 = const()[name = string("op_9447_axis_0"), val = int32(1)]; tensor var_9447_0, tensor var_9447_1 = split(axis = var_9447_axis_0, split_sizes = var_9447_split_sizes_0, x = query_states_159_cast_fp16)[name = string("op_9447")]; bool attn_weights_417_transpose_x_0 = const()[name = string("attn_weights_417_transpose_x_0"), val = bool(false)]; bool attn_weights_417_transpose_y_0 = const()[name = string("attn_weights_417_transpose_y_0"), val = bool(false)]; tensor attn_weights_417_cast_fp16 = matmul(transpose_x = attn_weights_417_transpose_x_0, transpose_y = attn_weights_417_transpose_y_0, x = var_9434_cast_fp16_0, y = var_9447_0)[name = string("attn_weights_417_cast_fp16")]; fp16 var_9450_to_fp16 = const()[name = string("op_9450_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_419_cast_fp16 = mul(x = attn_weights_417_cast_fp16, y = var_9450_to_fp16)[name = string("attn_weights_419_cast_fp16")]; tensor attn_weights_421_cast_fp16 = add(x = attn_weights_419_cast_fp16, y = attn_mask_1)[name = string("attn_weights_421_cast_fp16")]; int32 var_9454 = const()[name = string("op_9454"), val = int32(-2)]; tensor attn_weights_423_cast_fp16 = softmax(axis = var_9454, x = attn_weights_421_cast_fp16)[name = string("attn_weights_423_cast_fp16")]; bool var_9460_transpose_x_1 = const()[name = string("op_9460_transpose_x_1"), val = bool(true)]; bool var_9460_transpose_y_1 = const()[name = string("op_9460_transpose_y_1"), val = bool(false)]; tensor var_9460_cast_fp16 = matmul(transpose_x = var_9460_transpose_x_1, transpose_y = var_9460_transpose_y_1, x = attn_weights_423_cast_fp16, y = var_9444_cast_fp16_0)[name = string("op_9460_cast_fp16")]; bool attn_weights_425_transpose_x_0 = const()[name = string("attn_weights_425_transpose_x_0"), val = bool(false)]; bool attn_weights_425_transpose_y_0 = const()[name = string("attn_weights_425_transpose_y_0"), val = bool(false)]; tensor attn_weights_425_cast_fp16 = matmul(transpose_x = attn_weights_425_transpose_x_0, transpose_y = attn_weights_425_transpose_y_0, x = var_9434_cast_fp16_1, y = var_9447_1)[name = string("attn_weights_425_cast_fp16")]; fp16 var_9462_to_fp16 = const()[name = string("op_9462_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_427_cast_fp16 = mul(x = attn_weights_425_cast_fp16, y = var_9462_to_fp16)[name = string("attn_weights_427_cast_fp16")]; tensor attn_weights_429_cast_fp16 = add(x = attn_weights_427_cast_fp16, y = attn_mask_1)[name = string("attn_weights_429_cast_fp16")]; int32 var_9466 = const()[name = string("op_9466"), val = int32(-2)]; tensor attn_weights_431_cast_fp16 = softmax(axis = var_9466, x = attn_weights_429_cast_fp16)[name = string("attn_weights_431_cast_fp16")]; bool attn_output_209_transpose_x_1 = const()[name = string("attn_output_209_transpose_x_1"), val = bool(true)]; bool attn_output_209_transpose_y_1 = const()[name = string("attn_output_209_transpose_y_1"), val = bool(false)]; tensor attn_output_209_cast_fp16 = matmul(transpose_x = attn_output_209_transpose_x_1, transpose_y = attn_output_209_transpose_y_1, x = attn_weights_431_cast_fp16, y = var_9444_cast_fp16_1)[name = string("attn_output_209_cast_fp16")]; int32 var_9474 = const()[name = string("op_9474"), val = int32(1)]; bool attn_output_211_interleave_0 = const()[name = string("attn_output_211_interleave_0"), val = bool(false)]; tensor attn_output_211_cast_fp16 = concat(axis = var_9474, interleave = attn_output_211_interleave_0, values = (var_9460_cast_fp16, attn_output_209_cast_fp16))[name = string("attn_output_211_cast_fp16")]; tensor var_9478_perm_0 = const()[name = string("op_9478_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_323x = const()[name = string("concat_323x"), val = tensor([1, 2048, 1, -1])]; tensor var_9478_cast_fp16 = transpose(perm = var_9478_perm_0, x = attn_output_211_cast_fp16)[name = string("transpose_89")]; tensor attn_output_215_cast_fp16 = reshape(shape = concat_323x, x = var_9478_cast_fp16)[name = string("attn_output_215_cast_fp16")]; tensor hidden_states_263_strides_0 = const()[name = string("hidden_states_263_strides_0"), val = tensor([1, 1])]; string hidden_states_263_pad_type_0 = const()[name = string("hidden_states_263_pad_type_0"), val = string("valid")]; tensor hidden_states_263_pad_0 = const()[name = string("hidden_states_263_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_263_dilations_0 = const()[name = string("hidden_states_263_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_263_groups_0 = const()[name = string("hidden_states_263_groups_0"), val = int32(1)]; tensor hidden_states_263_cast_fp16 = conv(dilations = hidden_states_263_dilations_0, groups = hidden_states_263_groups_0, pad = hidden_states_263_pad_0, pad_type = hidden_states_263_pad_type_0, strides = hidden_states_263_strides_0, weight = layers_26_self_attn_o_proj_weight_cast_fp16, x = attn_output_215_cast_fp16)[name = string("hidden_states_263_cast_fp16")]; tensor hidden_states_265_cast_fp16 = add(x = hidden_states_259_cast_fp16, y = hidden_states_263_cast_fp16)[name = string("hidden_states_265_cast_fp16")]; fp16 const_268_promoted_to_fp16 = const()[name = string("const_268_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9511_cast_fp16 = mul(x = hidden_states_265_cast_fp16, y = const_268_promoted_to_fp16)[name = string("op_9511_cast_fp16")]; int32 var_9509 = const()[name = string("op_9509"), val = int32(1)]; bool doubled_213_interleave_0 = const()[name = string("doubled_213_interleave_0"), val = bool(false)]; tensor doubled_213_cast_fp16 = concat(axis = var_9509, interleave = doubled_213_interleave_0, values = (hidden_states_265_cast_fp16, var_9511_cast_fp16))[name = string("doubled_213_cast_fp16")]; tensor out_107_axes_0 = const()[name = string("out_107_axes_0"), val = tensor([1])]; tensor out_107_gamma_0_to_fp16 = const()[name = string("out_107_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567658624)))]; fp16 var_9521_to_fp16 = const()[name = string("op_9521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_107_cast_fp16 = layer_norm(axes = out_107_axes_0, epsilon = var_9521_to_fp16, gamma = out_107_gamma_0_to_fp16, x = doubled_213_cast_fp16)[name = string("out_107_cast_fp16")]; tensor var_9532_split_sizes_0 = const()[name = string("op_9532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9532_axis_0 = const()[name = string("op_9532_axis_0"), val = int32(1)]; tensor var_9532_cast_fp16_0, tensor var_9532_cast_fp16_1 = split(axis = var_9532_axis_0, split_sizes = var_9532_split_sizes_0, x = out_107_cast_fp16)[name = string("op_9532_cast_fp16")]; tensor input_53_strides_0 = const()[name = string("input_53_strides_0"), val = tensor([1, 1])]; string input_53_pad_type_0 = const()[name = string("input_53_pad_type_0"), val = string("valid")]; tensor input_53_pad_0 = const()[name = string("input_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_53_dilations_0 = const()[name = string("input_53_dilations_0"), val = tensor([1, 1])]; int32 input_53_groups_0 = const()[name = string("input_53_groups_0"), val = int32(1)]; tensor input_53_cast_fp16 = conv(dilations = input_53_dilations_0, groups = input_53_groups_0, pad = input_53_pad_0, pad_type = input_53_pad_type_0, strides = input_53_strides_0, weight = layers_26_mlp_gate_proj_weight_cast_fp16, x = var_9532_cast_fp16_0)[name = string("input_53_cast_fp16")]; tensor var_9549_cast_fp16 = silu(x = input_53_cast_fp16)[name = string("op_9549_cast_fp16")]; tensor layers_26_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567666880)))]; tensor var_9555_strides_0 = const()[name = string("op_9555_strides_0"), val = tensor([1, 1])]; string var_9555_pad_type_0 = const()[name = string("op_9555_pad_type_0"), val = string("valid")]; tensor var_9555_pad_0 = const()[name = string("op_9555_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9555_dilations_0 = const()[name = string("op_9555_dilations_0"), val = tensor([1, 1])]; int32 var_9555_groups_0 = const()[name = string("op_9555_groups_0"), val = int32(1)]; tensor var_9555_cast_fp16 = conv(dilations = var_9555_dilations_0, groups = var_9555_groups_0, pad = var_9555_pad_0, pad_type = var_9555_pad_type_0, strides = var_9555_strides_0, weight = layers_26_mlp_up_proj_weight_to_fp16, x = var_9532_cast_fp16_0)[name = string("op_9555_cast_fp16")]; tensor x_269_cast_fp16 = mul(x = var_9549_cast_fp16, y = var_9555_cast_fp16)[name = string("x_269_cast_fp16")]; tensor layers_26_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1592832768)))]; tensor hidden_states_267_strides_0 = const()[name = string("hidden_states_267_strides_0"), val = tensor([1, 1])]; string hidden_states_267_pad_type_0 = const()[name = string("hidden_states_267_pad_type_0"), val = string("valid")]; tensor hidden_states_267_pad_0 = const()[name = string("hidden_states_267_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_267_dilations_0 = const()[name = string("hidden_states_267_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_267_groups_0 = const()[name = string("hidden_states_267_groups_0"), val = int32(1)]; tensor hidden_states_267_cast_fp16 = conv(dilations = hidden_states_267_dilations_0, groups = hidden_states_267_groups_0, pad = hidden_states_267_pad_0, pad_type = hidden_states_267_pad_type_0, strides = hidden_states_267_strides_0, weight = layers_26_mlp_down_proj_weight_to_fp16, x = x_269_cast_fp16)[name = string("hidden_states_267_cast_fp16")]; tensor hidden_states_269_cast_fp16 = add(x = hidden_states_265_cast_fp16, y = hidden_states_267_cast_fp16)[name = string("hidden_states_269_cast_fp16")]; fp16 const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9573_cast_fp16 = mul(x = hidden_states_269_cast_fp16, y = const_270_promoted_to_fp16)[name = string("op_9573_cast_fp16")]; int32 var_9571 = const()[name = string("op_9571"), val = int32(1)]; bool doubled_217_interleave_0 = const()[name = string("doubled_217_interleave_0"), val = bool(false)]; tensor doubled_217_cast_fp16 = concat(axis = var_9571, interleave = doubled_217_interleave_0, values = (hidden_states_269_cast_fp16, var_9573_cast_fp16))[name = string("doubled_217_cast_fp16")]; tensor out_109_axes_0 = const()[name = string("out_109_axes_0"), val = tensor([1])]; tensor out_109_gamma_0_to_fp16 = const()[name = string("out_109_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1617998656)))]; fp16 var_9583_to_fp16 = const()[name = string("op_9583_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_109_cast_fp16 = layer_norm(axes = out_109_axes_0, epsilon = var_9583_to_fp16, gamma = out_109_gamma_0_to_fp16, x = doubled_217_cast_fp16)[name = string("out_109_cast_fp16")]; tensor var_9594_split_sizes_0 = const()[name = string("op_9594_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9594_axis_0 = const()[name = string("op_9594_axis_0"), val = int32(1)]; tensor var_9594_cast_fp16_0, tensor var_9594_cast_fp16_1 = split(axis = var_9594_axis_0, split_sizes = var_9594_split_sizes_0, x = out_109_cast_fp16)[name = string("op_9594_cast_fp16")]; tensor layers_27_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1618006912)))]; tensor query_states_163_strides_0 = const()[name = string("query_states_163_strides_0"), val = tensor([1, 1])]; string query_states_163_pad_type_0 = const()[name = string("query_states_163_pad_type_0"), val = string("valid")]; tensor query_states_163_pad_0 = const()[name = string("query_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_163_dilations_0 = const()[name = string("query_states_163_dilations_0"), val = tensor([1, 1])]; int32 query_states_163_groups_0 = const()[name = string("query_states_163_groups_0"), val = int32(1)]; tensor query_states_163_cast_fp16 = conv(dilations = query_states_163_dilations_0, groups = query_states_163_groups_0, pad = query_states_163_pad_0, pad_type = query_states_163_pad_type_0, strides = query_states_163_strides_0, weight = layers_27_self_attn_q_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("query_states_163_cast_fp16")]; tensor layers_27_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1626395584)))]; tensor key_states_271_strides_0 = const()[name = string("key_states_271_strides_0"), val = tensor([1, 1])]; string key_states_271_pad_type_0 = const()[name = string("key_states_271_pad_type_0"), val = string("valid")]; tensor key_states_271_pad_0 = const()[name = string("key_states_271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_271_dilations_0 = const()[name = string("key_states_271_dilations_0"), val = tensor([1, 1])]; int32 key_states_271_groups_0 = const()[name = string("key_states_271_groups_0"), val = int32(1)]; tensor key_states_271_cast_fp16 = conv(dilations = key_states_271_dilations_0, groups = key_states_271_groups_0, pad = key_states_271_pad_0, pad_type = key_states_271_pad_type_0, strides = key_states_271_strides_0, weight = layers_27_self_attn_k_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("key_states_271_cast_fp16")]; tensor layers_27_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1627444224)))]; tensor value_states_163_strides_0 = const()[name = string("value_states_163_strides_0"), val = tensor([1, 1])]; string value_states_163_pad_type_0 = const()[name = string("value_states_163_pad_type_0"), val = string("valid")]; tensor value_states_163_pad_0 = const()[name = string("value_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_163_dilations_0 = const()[name = string("value_states_163_dilations_0"), val = tensor([1, 1])]; int32 value_states_163_groups_0 = const()[name = string("value_states_163_groups_0"), val = int32(1)]; tensor value_states_163_cast_fp16 = conv(dilations = value_states_163_dilations_0, groups = value_states_163_groups_0, pad = value_states_163_pad_0, pad_type = value_states_163_pad_type_0, strides = value_states_163_strides_0, weight = layers_27_self_attn_v_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("value_states_163_cast_fp16")]; tensor concat_324x = const()[name = string("concat_324x"), val = tensor([1, 16, 128, -1])]; tensor x_271_cast_fp16 = reshape(shape = concat_324x, x = query_states_163_cast_fp16)[name = string("x_271_cast_fp16")]; tensor concat_325x = const()[name = string("concat_325x"), val = tensor([1, 2, 128, -1])]; tensor var_9651_cast_fp16 = reshape(shape = concat_325x, x = key_states_271_cast_fp16)[name = string("op_9651_cast_fp16")]; tensor concat_326x = const()[name = string("concat_326x"), val = tensor([1, 2, 128, -1])]; tensor var_9658_cast_fp16 = reshape(shape = concat_326x, x = value_states_163_cast_fp16)[name = string("op_9658_cast_fp16")]; tensor var_9662_cast_fp16 = mul(x = x_271_cast_fp16, y = var_869_cast_fp16)[name = string("op_9662_cast_fp16")]; tensor var_9663_split_sizes_0 = const()[name = string("op_9663_split_sizes_0"), val = tensor([64, 64])]; int32 var_9663_axis_0 = const()[name = string("op_9663_axis_0"), val = int32(-2)]; tensor var_9663_cast_fp16_0, tensor var_9663_cast_fp16_1 = split(axis = var_9663_axis_0, split_sizes = var_9663_split_sizes_0, x = x_271_cast_fp16)[name = string("op_9663_cast_fp16")]; fp16 const_272_promoted_to_fp16 = const()[name = string("const_272_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9665_cast_fp16 = mul(x = var_9663_cast_fp16_1, y = const_272_promoted_to_fp16)[name = string("op_9665_cast_fp16")]; int32 var_9667 = const()[name = string("op_9667"), val = int32(-2)]; bool var_9668_interleave_0 = const()[name = string("op_9668_interleave_0"), val = bool(false)]; tensor var_9668_cast_fp16 = concat(axis = var_9667, interleave = var_9668_interleave_0, values = (var_9665_cast_fp16, var_9663_cast_fp16_0))[name = string("op_9668_cast_fp16")]; tensor var_9669_cast_fp16 = mul(x = var_9668_cast_fp16, y = var_878_cast_fp16)[name = string("op_9669_cast_fp16")]; tensor query_states_165_cast_fp16 = add(x = var_9662_cast_fp16, y = var_9669_cast_fp16)[name = string("query_states_165_cast_fp16")]; tensor var_9675_cast_fp16 = mul(x = var_9651_cast_fp16, y = var_869_cast_fp16)[name = string("op_9675_cast_fp16")]; tensor var_9676_split_sizes_0 = const()[name = string("op_9676_split_sizes_0"), val = tensor([64, 64])]; int32 var_9676_axis_0 = const()[name = string("op_9676_axis_0"), val = int32(-2)]; tensor var_9676_cast_fp16_0, tensor var_9676_cast_fp16_1 = split(axis = var_9676_axis_0, split_sizes = var_9676_split_sizes_0, x = var_9651_cast_fp16)[name = string("op_9676_cast_fp16")]; fp16 const_273_promoted_to_fp16 = const()[name = string("const_273_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9678_cast_fp16 = mul(x = var_9676_cast_fp16_1, y = const_273_promoted_to_fp16)[name = string("op_9678_cast_fp16")]; int32 var_9680 = const()[name = string("op_9680"), val = int32(-2)]; bool var_9681_interleave_0 = const()[name = string("op_9681_interleave_0"), val = bool(false)]; tensor var_9681_cast_fp16 = concat(axis = var_9680, interleave = var_9681_interleave_0, values = (var_9678_cast_fp16, var_9676_cast_fp16_0))[name = string("op_9681_cast_fp16")]; tensor var_9682_cast_fp16 = mul(x = var_9681_cast_fp16, y = var_878_cast_fp16)[name = string("op_9682_cast_fp16")]; tensor key_states_275_cast_fp16 = add(x = var_9675_cast_fp16, y = var_9682_cast_fp16)[name = string("key_states_275_cast_fp16")]; tensor expand_dims_324 = const()[name = string("expand_dims_324"), val = tensor([27])]; tensor expand_dims_325 = const()[name = string("expand_dims_325"), val = tensor([0])]; tensor expand_dims_327 = const()[name = string("expand_dims_327"), val = tensor([0])]; int32 concat_329_axis_0 = const()[name = string("concat_329_axis_0"), val = int32(0)]; bool concat_329_interleave_0 = const()[name = string("concat_329_interleave_0"), val = bool(false)]; tensor concat_329 = concat(axis = concat_329_axis_0, interleave = concat_329_interleave_0, values = (expand_dims_324, expand_dims_325, position_id, expand_dims_327))[name = string("concat_329")]; tensor expand_dims_328 = const()[name = string("expand_dims_328"), val = tensor([28])]; tensor concat_330_values1_0 = const()[name = string("concat_330_values1_0"), val = tensor([0])]; tensor concat_330_values3_0 = const()[name = string("concat_330_values3_0"), val = tensor([0])]; int32 concat_330_axis_0 = const()[name = string("concat_330_axis_0"), val = int32(0)]; bool concat_330_interleave_0 = const()[name = string("concat_330_interleave_0"), val = bool(false)]; tensor concat_330 = concat(axis = concat_330_axis_0, interleave = concat_330_interleave_0, values = (expand_dims_328, concat_330_values1_0, cache_position_end, concat_330_values3_0))[name = string("concat_330")]; tensor key_states_277_perm_0 = const()[name = string("key_states_277_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_28_stride_0 = const()[name = string("key_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_277_cast_fp16 = transpose(perm = key_states_277_perm_0, x = key_states_275_cast_fp16)[name = string("transpose_88")]; tensor key_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = key_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = key_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_28_squeeze_mask_0, stride = key_cache_internal_tensor_assign_28_stride_0, update = key_states_277_cast_fp16, x = coreml_update_state_108)[name = string("key_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_28_cast_fp16, input = key_cache)[name = string("coreml_update_state_110_write_state")]; tensor coreml_update_state_110 = read_state(input = key_cache)[name = string("coreml_update_state_110")]; tensor value_states_165_perm_0 = const()[name = string("value_states_165_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_28_stride_0 = const()[name = string("value_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_165_cast_fp16 = transpose(perm = value_states_165_perm_0, x = var_9658_cast_fp16)[name = string("transpose_87")]; tensor value_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = value_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = value_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_28_squeeze_mask_0, stride = value_cache_internal_tensor_assign_28_stride_0, update = value_states_165_cast_fp16, x = coreml_update_state_109)[name = string("value_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_28_cast_fp16, input = value_cache)[name = string("coreml_update_state_111_write_state")]; tensor coreml_update_state_111 = read_state(input = value_cache)[name = string("coreml_update_state_111")]; tensor var_9752_begin_0 = const()[name = string("op_9752_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9752_end_0 = const()[name = string("op_9752_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9752_end_mask_0 = const()[name = string("op_9752_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9752_cast_fp16 = slice_by_index(begin = var_9752_begin_0, end = var_9752_end_0, end_mask = var_9752_end_mask_0, x = coreml_update_state_110)[name = string("op_9752_cast_fp16")]; tensor tile_54 = const()[name = string("tile_54"), val = tensor([1, 1])]; int32 var_9755_axis_0 = const()[name = string("op_9755_axis_0"), val = int32(1)]; tensor var_9755_cast_fp16_0, tensor var_9755_cast_fp16_1 = split(axis = var_9755_axis_0, split_sizes = tile_54, x = var_9752_cast_fp16)[name = string("op_9755_cast_fp16")]; tensor var_9762_begin_0 = const()[name = string("op_9762_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9762_end_0 = const()[name = string("op_9762_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9762_end_mask_0 = const()[name = string("op_9762_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9762_cast_fp16 = slice_by_index(begin = var_9762_begin_0, end = var_9762_end_0, end_mask = var_9762_end_mask_0, x = coreml_update_state_111)[name = string("op_9762_cast_fp16")]; tensor tile_55 = const()[name = string("tile_55"), val = tensor([1, 1])]; int32 var_9765_axis_0 = const()[name = string("op_9765_axis_0"), val = int32(1)]; tensor var_9765_cast_fp16_0, tensor var_9765_cast_fp16_1 = split(axis = var_9765_axis_0, split_sizes = tile_55, x = var_9762_cast_fp16)[name = string("op_9765_cast_fp16")]; tensor var_9768_split_sizes_0 = const()[name = string("op_9768_split_sizes_0"), val = tensor([8, 8])]; int32 var_9768_axis_0 = const()[name = string("op_9768_axis_0"), val = int32(1)]; tensor var_9768_0, tensor var_9768_1 = split(axis = var_9768_axis_0, split_sizes = var_9768_split_sizes_0, x = query_states_165_cast_fp16)[name = string("op_9768")]; bool attn_weights_433_transpose_x_0 = const()[name = string("attn_weights_433_transpose_x_0"), val = bool(false)]; bool attn_weights_433_transpose_y_0 = const()[name = string("attn_weights_433_transpose_y_0"), val = bool(false)]; tensor attn_weights_433_cast_fp16 = matmul(transpose_x = attn_weights_433_transpose_x_0, transpose_y = attn_weights_433_transpose_y_0, x = var_9755_cast_fp16_0, y = var_9768_0)[name = string("attn_weights_433_cast_fp16")]; fp16 var_9771_to_fp16 = const()[name = string("op_9771_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_435_cast_fp16 = mul(x = attn_weights_433_cast_fp16, y = var_9771_to_fp16)[name = string("attn_weights_435_cast_fp16")]; tensor attn_weights_437_cast_fp16 = add(x = attn_weights_435_cast_fp16, y = attn_mask_1)[name = string("attn_weights_437_cast_fp16")]; int32 var_9775 = const()[name = string("op_9775"), val = int32(-2)]; tensor attn_weights_439_cast_fp16 = softmax(axis = var_9775, x = attn_weights_437_cast_fp16)[name = string("attn_weights_439_cast_fp16")]; bool var_9781_transpose_x_1 = const()[name = string("op_9781_transpose_x_1"), val = bool(true)]; bool var_9781_transpose_y_1 = const()[name = string("op_9781_transpose_y_1"), val = bool(false)]; tensor var_9781_cast_fp16 = matmul(transpose_x = var_9781_transpose_x_1, transpose_y = var_9781_transpose_y_1, x = attn_weights_439_cast_fp16, y = var_9765_cast_fp16_0)[name = string("op_9781_cast_fp16")]; bool attn_weights_441_transpose_x_0 = const()[name = string("attn_weights_441_transpose_x_0"), val = bool(false)]; bool attn_weights_441_transpose_y_0 = const()[name = string("attn_weights_441_transpose_y_0"), val = bool(false)]; tensor attn_weights_441_cast_fp16 = matmul(transpose_x = attn_weights_441_transpose_x_0, transpose_y = attn_weights_441_transpose_y_0, x = var_9755_cast_fp16_1, y = var_9768_1)[name = string("attn_weights_441_cast_fp16")]; fp16 var_9783_to_fp16 = const()[name = string("op_9783_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_443_cast_fp16 = mul(x = attn_weights_441_cast_fp16, y = var_9783_to_fp16)[name = string("attn_weights_443_cast_fp16")]; tensor attn_weights_445_cast_fp16 = add(x = attn_weights_443_cast_fp16, y = attn_mask_1)[name = string("attn_weights_445_cast_fp16")]; int32 var_9787 = const()[name = string("op_9787"), val = int32(-2)]; tensor attn_weights_cast_fp16 = softmax(axis = var_9787, x = attn_weights_445_cast_fp16)[name = string("attn_weights_cast_fp16")]; bool attn_output_217_transpose_x_1 = const()[name = string("attn_output_217_transpose_x_1"), val = bool(true)]; bool attn_output_217_transpose_y_1 = const()[name = string("attn_output_217_transpose_y_1"), val = bool(false)]; tensor attn_output_217_cast_fp16 = matmul(transpose_x = attn_output_217_transpose_x_1, transpose_y = attn_output_217_transpose_y_1, x = attn_weights_cast_fp16, y = var_9765_cast_fp16_1)[name = string("attn_output_217_cast_fp16")]; int32 var_9795 = const()[name = string("op_9795"), val = int32(1)]; bool attn_output_219_interleave_0 = const()[name = string("attn_output_219_interleave_0"), val = bool(false)]; tensor attn_output_219_cast_fp16 = concat(axis = var_9795, interleave = attn_output_219_interleave_0, values = (var_9781_cast_fp16, attn_output_217_cast_fp16))[name = string("attn_output_219_cast_fp16")]; tensor var_9799_perm_0 = const()[name = string("op_9799_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_335x = const()[name = string("concat_335x"), val = tensor([1, 2048, 1, -1])]; tensor var_9799_cast_fp16 = transpose(perm = var_9799_perm_0, x = attn_output_219_cast_fp16)[name = string("transpose_86")]; tensor attn_output_cast_fp16 = reshape(shape = concat_335x, x = var_9799_cast_fp16)[name = string("attn_output_cast_fp16")]; tensor layers_27_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1628492864)))]; tensor hidden_states_273_strides_0 = const()[name = string("hidden_states_273_strides_0"), val = tensor([1, 1])]; string hidden_states_273_pad_type_0 = const()[name = string("hidden_states_273_pad_type_0"), val = string("valid")]; tensor hidden_states_273_pad_0 = const()[name = string("hidden_states_273_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_273_dilations_0 = const()[name = string("hidden_states_273_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_273_groups_0 = const()[name = string("hidden_states_273_groups_0"), val = int32(1)]; tensor hidden_states_273_cast_fp16 = conv(dilations = hidden_states_273_dilations_0, groups = hidden_states_273_groups_0, pad = hidden_states_273_pad_0, pad_type = hidden_states_273_pad_type_0, strides = hidden_states_273_strides_0, weight = layers_27_self_attn_o_proj_weight_to_fp16, x = attn_output_cast_fp16)[name = string("hidden_states_273_cast_fp16")]; tensor hidden_states_275_cast_fp16 = add(x = hidden_states_269_cast_fp16, y = hidden_states_273_cast_fp16)[name = string("hidden_states_275_cast_fp16")]; fp16 const_278_promoted_to_fp16 = const()[name = string("const_278_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9832_cast_fp16 = mul(x = hidden_states_275_cast_fp16, y = const_278_promoted_to_fp16)[name = string("op_9832_cast_fp16")]; int32 var_9830 = const()[name = string("op_9830"), val = int32(1)]; bool doubled_221_interleave_0 = const()[name = string("doubled_221_interleave_0"), val = bool(false)]; tensor doubled_221_cast_fp16 = concat(axis = var_9830, interleave = doubled_221_interleave_0, values = (hidden_states_275_cast_fp16, var_9832_cast_fp16))[name = string("doubled_221_cast_fp16")]; tensor out_111_axes_0 = const()[name = string("out_111_axes_0"), val = tensor([1])]; tensor out_111_gamma_0_to_fp16 = const()[name = string("out_111_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636881536)))]; fp16 var_9842_to_fp16 = const()[name = string("op_9842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_111_cast_fp16 = layer_norm(axes = out_111_axes_0, epsilon = var_9842_to_fp16, gamma = out_111_gamma_0_to_fp16, x = doubled_221_cast_fp16)[name = string("out_111_cast_fp16")]; tensor var_9853_split_sizes_0 = const()[name = string("op_9853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9853_axis_0 = const()[name = string("op_9853_axis_0"), val = int32(1)]; tensor var_9853_cast_fp16_0, tensor var_9853_cast_fp16_1 = split(axis = var_9853_axis_0, split_sizes = var_9853_split_sizes_0, x = out_111_cast_fp16)[name = string("op_9853_cast_fp16")]; tensor layers_27_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636889792)))]; tensor input_strides_0 = const()[name = string("input_strides_0"), val = tensor([1, 1])]; string input_pad_type_0 = const()[name = string("input_pad_type_0"), val = string("valid")]; tensor input_pad_0 = const()[name = string("input_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_dilations_0 = const()[name = string("input_dilations_0"), val = tensor([1, 1])]; int32 input_groups_0 = const()[name = string("input_groups_0"), val = int32(1)]; tensor input_cast_fp16 = conv(dilations = input_dilations_0, groups = input_groups_0, pad = input_pad_0, pad_type = input_pad_type_0, strides = input_strides_0, weight = layers_27_mlp_gate_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("input_cast_fp16")]; tensor var_9870_cast_fp16 = silu(x = input_cast_fp16)[name = string("op_9870_cast_fp16")]; tensor layers_27_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1662055680)))]; tensor var_9876_strides_0 = const()[name = string("op_9876_strides_0"), val = tensor([1, 1])]; string var_9876_pad_type_0 = const()[name = string("op_9876_pad_type_0"), val = string("valid")]; tensor var_9876_pad_0 = const()[name = string("op_9876_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9876_dilations_0 = const()[name = string("op_9876_dilations_0"), val = tensor([1, 1])]; int32 var_9876_groups_0 = const()[name = string("op_9876_groups_0"), val = int32(1)]; tensor var_9876_cast_fp16 = conv(dilations = var_9876_dilations_0, groups = var_9876_groups_0, pad = var_9876_pad_0, pad_type = var_9876_pad_type_0, strides = var_9876_strides_0, weight = layers_27_mlp_up_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("op_9876_cast_fp16")]; tensor x_cast_fp16 = mul(x = var_9870_cast_fp16, y = var_9876_cast_fp16)[name = string("x_cast_fp16")]; tensor layers_27_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1687221568)))]; tensor hidden_states_277_strides_0 = const()[name = string("hidden_states_277_strides_0"), val = tensor([1, 1])]; string hidden_states_277_pad_type_0 = const()[name = string("hidden_states_277_pad_type_0"), val = string("valid")]; tensor hidden_states_277_pad_0 = const()[name = string("hidden_states_277_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_277_dilations_0 = const()[name = string("hidden_states_277_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_277_groups_0 = const()[name = string("hidden_states_277_groups_0"), val = int32(1)]; tensor hidden_states_277_cast_fp16 = conv(dilations = hidden_states_277_dilations_0, groups = hidden_states_277_groups_0, pad = hidden_states_277_pad_0, pad_type = hidden_states_277_pad_type_0, strides = hidden_states_277_strides_0, weight = layers_27_mlp_down_proj_weight_to_fp16, x = x_cast_fp16)[name = string("hidden_states_277_cast_fp16")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_275_cast_fp16, y = hidden_states_277_cast_fp16)[name = string("hidden_states_cast_fp16")]; fp16 const_280_promoted_to_fp16 = const()[name = string("const_280_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9894_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_280_promoted_to_fp16)[name = string("op_9894_cast_fp16")]; int32 var_9892 = const()[name = string("op_9892"), val = int32(1)]; bool doubled_225_interleave_0 = const()[name = string("doubled_225_interleave_0"), val = bool(false)]; tensor doubled_225_cast_fp16 = concat(axis = var_9892, interleave = doubled_225_interleave_0, values = (hidden_states_cast_fp16, var_9894_cast_fp16))[name = string("doubled_225_cast_fp16")]; tensor out_axes_0 = const()[name = string("out_axes_0"), val = tensor([1])]; tensor out_gamma_0_to_fp16 = const()[name = string("out_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1712387456)))]; fp16 var_9904_to_fp16 = const()[name = string("op_9904_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_cast_fp16 = layer_norm(axes = out_axes_0, epsilon = var_9904_to_fp16, gamma = out_gamma_0_to_fp16, x = doubled_225_cast_fp16)[name = string("out_cast_fp16")]; tensor var_9915_split_sizes_0 = const()[name = string("op_9915_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9915_axis_0 = const()[name = string("op_9915_axis_0"), val = int32(1)]; tensor hidden_states, tensor var_9915_cast_fp16_1 = split(axis = var_9915_axis_0, split_sizes = var_9915_split_sizes_0, x = out_cast_fp16)[name = string("op_9915_cast_fp16")]; } -> (hidden_states); func length_128(tensor inputs_embeds, state> key_cache, tensor position_id, tensor position_index_seed, state> value_cache) { tensor layers_1_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524992))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524416))))[name = string("layers_1_self_attn_v_proj_weight_cast_fp16")]; tensor layers_1_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(525312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13120640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13108288))))[name = string("layers_1_mlp_up_proj_weight_cast_fp16")]; tensor layers_2_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13126848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651776))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651200))))[name = string("layers_2_self_attn_v_proj_weight_cast_fp16")]; tensor layers_2_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13652096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26247424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26235072))))[name = string("layers_2_mlp_up_proj_weight_cast_fp16")]; tensor layers_3_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26253632))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778560))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26777984))))[name = string("layers_3_self_attn_v_proj_weight_cast_fp16")]; tensor layers_3_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30977408))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30973248))))[name = string("layers_3_self_attn_o_proj_weight_cast_fp16")]; tensor layers_3_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30979520))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43566656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43562496))))[name = string("layers_3_mlp_down_proj_weight_cast_fp16")]; tensor layers_4_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43568768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093696))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093120))))[name = string("layers_4_self_attn_v_proj_weight_cast_fp16")]; tensor layers_4_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44094016))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48292544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48288384))))[name = string("layers_4_self_attn_o_proj_weight_cast_fp16")]; tensor layers_4_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48294656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60877632))))[name = string("layers_4_mlp_gate_proj_weight_cast_fp16")]; tensor layers_4_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60896192))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73491520))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73479168))))[name = string("layers_4_mlp_up_proj_weight_cast_fp16")]; tensor layers_4_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73497728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86084864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86080704))))[name = string("layers_4_mlp_down_proj_weight_cast_fp16")]; tensor layers_5_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86086976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611328))))[name = string("layers_5_self_attn_v_proj_weight_cast_fp16")]; tensor layers_5_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86612224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90810752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90806592))))[name = string("layers_5_self_attn_o_proj_weight_cast_fp16")]; tensor layers_5_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90812864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103408192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103395840))))[name = string("layers_5_mlp_up_proj_weight_cast_fp16")]; tensor layers_5_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103414400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116001536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115997376))))[name = string("layers_5_mlp_down_proj_weight_cast_fp16")]; tensor layers_6_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116003648))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528576))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528000))))[name = string("layers_6_self_attn_v_proj_weight_cast_fp16")]; tensor layers_6_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528896))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120727424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120723264))))[name = string("layers_6_self_attn_o_proj_weight_cast_fp16")]; tensor layers_6_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120729536))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133324864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133312512))))[name = string("layers_6_mlp_gate_proj_weight_cast_fp16")]; tensor layers_6_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133331072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145926400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145914048))))[name = string("layers_6_mlp_up_proj_weight_cast_fp16")]; tensor layers_6_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145932608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158519744))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158515584))))[name = string("layers_6_mlp_down_proj_weight_cast_fp16")]; tensor layers_7_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158521856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046208))))[name = string("layers_7_self_attn_v_proj_weight_cast_fp16")]; tensor layers_7_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159047104))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163245632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163241472))))[name = string("layers_7_self_attn_o_proj_weight_cast_fp16")]; tensor layers_7_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163247744))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175843072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175830720))))[name = string("layers_7_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175849280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374208))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176373632))))[name = string("layers_8_self_attn_v_proj_weight_cast_fp16")]; tensor layers_8_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374528))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180573056))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180568896))))[name = string("layers_8_self_attn_o_proj_weight_cast_fp16")]; tensor layers_8_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180575168))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193170496))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193158144))))[name = string("layers_8_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193176704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205772032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205759680))))[name = string("layers_8_mlp_up_proj_weight_cast_fp16")]; tensor layers_8_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205778240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218365376))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218361216))))[name = string("layers_8_mlp_down_proj_weight_cast_fp16")]; tensor layers_9_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218367488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218891840))))[name = string("layers_9_self_attn_v_proj_weight_cast_fp16")]; tensor layers_9_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223091264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223087104))))[name = string("layers_9_self_attn_o_proj_weight_cast_fp16")]; tensor layers_9_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223093376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235688704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235676352))))[name = string("layers_9_mlp_gate_proj_weight_cast_fp16")]; tensor layers_9_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235694912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248290240))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248277888))))[name = string("layers_9_mlp_up_proj_weight_cast_fp16")]; tensor layers_9_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248296448))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260883584))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260879424))))[name = string("layers_9_mlp_down_proj_weight_cast_fp16")]; tensor layers_10_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260885696))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410048))))[name = string("layers_10_self_attn_v_proj_weight_cast_fp16")]; tensor layers_10_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410944))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265609472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265605312))))[name = string("layers_10_self_attn_o_proj_weight_cast_fp16")]; tensor layers_10_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265611584))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278206912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278194560))))[name = string("layers_10_mlp_gate_proj_weight_cast_fp16")]; tensor layers_10_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278213120))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290808448))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290796096))))[name = string("layers_10_mlp_up_proj_weight_cast_fp16")]; tensor layers_10_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290814656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303401792))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303397632))))[name = string("layers_10_mlp_down_proj_weight_cast_fp16")]; tensor layers_11_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303403904))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307602432))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307598272))))[name = string("layers_11_self_attn_q_proj_weight_cast_fp16")]; tensor layers_11_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307604544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308128896))))[name = string("layers_11_self_attn_k_proj_weight_cast_fp16")]; tensor layers_11_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129792))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654720))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654144))))[name = string("layers_11_self_attn_v_proj_weight_cast_fp16")]; tensor layers_11_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308655040))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312853568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312849408))))[name = string("layers_11_self_attn_o_proj_weight_cast_fp16")]; tensor layers_11_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312855680))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325451008))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325438656))))[name = string("layers_11_mlp_gate_proj_weight_cast_fp16")]; tensor layers_11_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325457216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338052544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338040192))))[name = string("layers_11_mlp_up_proj_weight_cast_fp16")]; tensor layers_11_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338058752))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350645888))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350641728))))[name = string("layers_11_mlp_down_proj_weight_cast_fp16")]; tensor layers_12_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350648000))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354846528))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354842368))))[name = string("layers_12_self_attn_q_proj_weight_cast_fp16")]; tensor layers_12_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354848640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355372992))))[name = string("layers_12_self_attn_k_proj_weight_cast_fp16")]; tensor layers_12_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373888))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898816))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898240))))[name = string("layers_12_self_attn_v_proj_weight_cast_fp16")]; tensor layers_12_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355899136))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360097664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360093504))))[name = string("layers_12_self_attn_o_proj_weight_cast_fp16")]; tensor layers_12_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360099776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372695104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372682752))))[name = string("layers_12_mlp_gate_proj_weight_cast_fp16")]; tensor layers_12_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372701312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385296640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385284288))))[name = string("layers_12_mlp_up_proj_weight_cast_fp16")]; tensor layers_12_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385302848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397885824))))[name = string("layers_12_mlp_down_proj_weight_cast_fp16")]; tensor layers_13_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397892096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402090624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402086464))))[name = string("layers_13_self_attn_q_proj_weight_cast_fp16")]; tensor layers_13_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402092736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617088))))[name = string("layers_13_self_attn_k_proj_weight_cast_fp16")]; tensor layers_13_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617984))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142336))))[name = string("layers_13_self_attn_v_proj_weight_cast_fp16")]; tensor layers_13_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403143232))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407341760))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407337600))))[name = string("layers_13_self_attn_o_proj_weight_cast_fp16")]; tensor layers_13_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407343872))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419939200))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419926848))))[name = string("layers_13_mlp_gate_proj_weight_cast_fp16")]; tensor layers_13_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419945408))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432532544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432528384))))[name = string("layers_13_mlp_down_proj_weight_cast_fp16")]; tensor layers_14_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432534656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436733184))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436729024))))[name = string("layers_14_self_attn_q_proj_weight_cast_fp16")]; tensor layers_14_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436735296))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437259648))))[name = string("layers_14_self_attn_v_proj_weight_cast_fp16")]; tensor layers_14_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441459072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441454912))))[name = string("layers_14_self_attn_o_proj_weight_cast_fp16")]; tensor layers_14_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441461184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454056512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454044160))))[name = string("layers_14_mlp_gate_proj_weight_cast_fp16")]; tensor layers_14_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454062720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466658048))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466645696))))[name = string("layers_14_mlp_up_proj_weight_cast_fp16")]; tensor layers_14_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466664256))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479251392))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479247232))))[name = string("layers_14_mlp_down_proj_weight_cast_fp16")]; tensor layers_15_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479253504))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483452032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483447872))))[name = string("layers_15_self_attn_q_proj_weight_cast_fp16")]; tensor layers_15_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483454144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483978496))))[name = string("layers_15_self_attn_k_proj_weight_cast_fp16")]; tensor layers_15_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504320))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484503744))))[name = string("layers_15_self_attn_v_proj_weight_cast_fp16")]; tensor layers_15_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488703168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488699008))))[name = string("layers_15_self_attn_o_proj_weight_cast_fp16")]; tensor layers_15_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488705280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501300608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501288256))))[name = string("layers_15_mlp_gate_proj_weight_cast_fp16")]; tensor layers_15_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501306816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513902144))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513889792))))[name = string("layers_15_mlp_up_proj_weight_cast_fp16")]; tensor layers_15_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513908352))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526495488))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526491328))))[name = string("layers_15_mlp_down_proj_weight_cast_fp16")]; tensor layers_16_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526497600))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530696128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530691968))))[name = string("layers_16_self_attn_q_proj_weight_cast_fp16")]; tensor layers_16_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530698240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531222592))))[name = string("layers_16_self_attn_k_proj_weight_cast_fp16")]; tensor layers_16_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531747840))))[name = string("layers_16_self_attn_v_proj_weight_cast_fp16")]; tensor layers_16_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535947264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535943104))))[name = string("layers_16_self_attn_o_proj_weight_cast_fp16")]; tensor layers_16_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535949376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548536512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548532352))))[name = string("layers_16_mlp_down_proj_weight_cast_fp16")]; tensor layers_17_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548538624))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552737152))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552732992))))[name = string("layers_17_self_attn_q_proj_weight_cast_fp16")]; tensor layers_17_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552739264))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553263616))))[name = string("layers_17_self_attn_k_proj_weight_cast_fp16")]; tensor layers_17_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264512))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553788864))))[name = string("layers_17_self_attn_v_proj_weight_cast_fp16")]; tensor layers_17_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789760))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557988288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557984128))))[name = string("layers_17_self_attn_o_proj_weight_cast_fp16")]; tensor layers_17_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557990400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570585728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570573376))))[name = string("layers_17_mlp_gate_proj_weight_cast_fp16")]; tensor layers_17_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570591936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583187264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583174912))))[name = string("layers_17_mlp_up_proj_weight_cast_fp16")]; tensor layers_17_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583193472))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595780608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595776448))))[name = string("layers_17_mlp_down_proj_weight_cast_fp16")]; tensor layers_18_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595782720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599981248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599977088))))[name = string("layers_18_self_attn_q_proj_weight_cast_fp16")]; tensor layers_18_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599983360))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600507712))))[name = string("layers_18_self_attn_k_proj_weight_cast_fp16")]; tensor layers_18_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601032960))))[name = string("layers_18_self_attn_v_proj_weight_cast_fp16")]; tensor layers_18_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605232384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605228224))))[name = string("layers_18_self_attn_o_proj_weight_cast_fp16")]; tensor layers_18_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605234496))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617829824))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617817472))))[name = string("layers_18_mlp_gate_proj_weight_cast_fp16")]; tensor layers_18_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617836032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630431360))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630419008))))[name = string("layers_18_mlp_up_proj_weight_cast_fp16")]; tensor layers_18_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630437568))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643024704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643020544))))[name = string("layers_18_mlp_down_proj_weight_cast_fp16")]; tensor layers_19_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643026816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647225344))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647221184))))[name = string("layers_19_self_attn_q_proj_weight_cast_fp16")]; tensor layers_19_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647227456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647751808))))[name = string("layers_19_self_attn_k_proj_weight_cast_fp16")]; tensor layers_19_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660348032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660335680))))[name = string("layers_19_mlp_gate_proj_weight_cast_fp16")]; tensor layers_19_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660354240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672949568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672937216))))[name = string("layers_19_mlp_up_proj_weight_cast_fp16")]; tensor layers_19_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672955776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685542912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685538752))))[name = string("layers_19_mlp_down_proj_weight_cast_fp16")]; tensor layers_20_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685545024))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689743552))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689739392))))[name = string("layers_20_self_attn_q_proj_weight_cast_fp16")]; tensor layers_20_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689745664))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270592))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270016))))[name = string("layers_20_self_attn_k_proj_weight_cast_fp16")]; tensor layers_20_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694469440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694465280))))[name = string("layers_20_self_attn_o_proj_weight_cast_fp16")]; tensor layers_20_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694471552))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707066880))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707054528))))[name = string("layers_20_mlp_gate_proj_weight_cast_fp16")]; tensor layers_20_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707073088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719660224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719656064))))[name = string("layers_20_mlp_down_proj_weight_cast_fp16")]; tensor layers_21_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719662336))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723860864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723856704))))[name = string("layers_21_self_attn_q_proj_weight_cast_fp16")]; tensor layers_21_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723862976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387328))))[name = string("layers_21_self_attn_k_proj_weight_cast_fp16")]; tensor layers_21_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724388224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728586752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728582592))))[name = string("layers_21_self_attn_o_proj_weight_cast_fp16")]; tensor layers_21_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728588864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741184192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741171840))))[name = string("layers_21_mlp_gate_proj_weight_cast_fp16")]; tensor layers_21_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741190400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753785728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753773376))))[name = string("layers_21_mlp_up_proj_weight_cast_fp16")]; tensor layers_21_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753791936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766379072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766374912))))[name = string("layers_21_mlp_down_proj_weight_cast_fp16")]; tensor layers_22_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766381184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770579712))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770575552))))[name = string("layers_22_self_attn_q_proj_weight_cast_fp16")]; tensor layers_22_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770581824))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106176))))[name = string("layers_22_self_attn_k_proj_weight_cast_fp16")]; tensor layers_22_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771107072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783702400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783690048))))[name = string("layers_22_mlp_gate_proj_weight_cast_fp16")]; tensor layers_22_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783708608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796303936))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796291584))))[name = string("layers_22_mlp_up_proj_weight_cast_fp16")]; tensor layers_22_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796310144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808897280))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808893120))))[name = string("layers_22_mlp_down_proj_weight_cast_fp16")]; tensor layers_23_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808899392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813097920))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813093760))))[name = string("layers_23_self_attn_q_proj_weight_cast_fp16")]; tensor layers_23_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813100032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624960))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624384))))[name = string("layers_23_self_attn_k_proj_weight_cast_fp16")]; tensor layers_23_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813625280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817823808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817819648))))[name = string("layers_23_self_attn_o_proj_weight_cast_fp16")]; tensor layers_23_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817825920))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830421248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830408896))))[name = string("layers_23_mlp_gate_proj_weight_cast_fp16")]; tensor layers_23_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830427456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843022784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843010432))))[name = string("layers_23_mlp_up_proj_weight_cast_fp16")]; tensor layers_23_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843028992))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855616128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855611968))))[name = string("layers_23_mlp_down_proj_weight_cast_fp16")]; tensor layers_24_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855618240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859816768))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859812608))))[name = string("layers_24_self_attn_q_proj_weight_cast_fp16")]; tensor layers_24_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859818880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343232))))[name = string("layers_24_self_attn_k_proj_weight_cast_fp16")]; tensor layers_24_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860344128))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864542656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864538496))))[name = string("layers_24_self_attn_o_proj_weight_cast_fp16")]; tensor layers_24_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864544768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877140096))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877127744))))[name = string("layers_24_mlp_gate_proj_weight_cast_fp16")]; tensor layers_24_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877146304))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889741632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889729280))))[name = string("layers_24_mlp_up_proj_weight_cast_fp16")]; tensor layers_24_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889747840))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902334976))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902330816))))[name = string("layers_24_mlp_down_proj_weight_cast_fp16")]; tensor layers_25_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902337088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906535616))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906531456))))[name = string("layers_25_self_attn_q_proj_weight_cast_fp16")]; tensor layers_25_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906537728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062080))))[name = string("layers_25_self_attn_k_proj_weight_cast_fp16")]; tensor layers_25_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911261504))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911257344))))[name = string("layers_25_self_attn_o_proj_weight_cast_fp16")]; tensor layers_25_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911263616))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923858944))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923846592))))[name = string("layers_25_mlp_gate_proj_weight_cast_fp16")]; tensor layers_25_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923865152))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936460480))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936448128))))[name = string("layers_25_mlp_up_proj_weight_cast_fp16")]; tensor layers_26_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936466688))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940665216))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940661056))))[name = string("layers_26_self_attn_q_proj_weight_cast_fp16")]; tensor layers_26_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940667328))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192256))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941191680))))[name = string("layers_26_self_attn_k_proj_weight_cast_fp16")]; tensor layers_26_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192576))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945391104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945386944))))[name = string("layers_26_self_attn_o_proj_weight_cast_fp16")]; tensor layers_26_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945393216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957988544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957976192))))[name = string("layers_26_mlp_gate_proj_weight_cast_fp16")]; int32 var_765 = const()[name = string("op_765"), val = int32(0)]; tensor var_766 = mul(x = position_index_seed, y = var_765)[name = string("op_766")]; int32 var_768 = const()[name = string("op_768"), val = int32(1)]; tensor ones = add(x = var_766, y = var_768)[name = string("ones")]; int32 var_770 = const()[name = string("op_770"), val = int32(0)]; bool var_772_exclusive_0 = const()[name = string("op_772_exclusive_0"), val = bool(false)]; bool var_772_reverse_0 = const()[name = string("op_772_reverse_0"), val = bool(false)]; tensor var_772 = cumsum(axis = var_770, exclusive = var_772_exclusive_0, reverse = var_772_reverse_0, x = ones)[name = string("op_772")]; int32 var_774 = const()[name = string("op_774"), val = int32(1)]; tensor position_offsets = sub(x = var_772, y = var_774)[name = string("position_offsets")]; tensor position_ids_1 = add(x = position_offsets, y = position_id)[name = string("position_ids_1")]; bool var_784_keep_dims_0 = const()[name = string("op_784_keep_dims_0"), val = bool(false)]; int32 var_784 = reduce_sum(keep_dims = var_784_keep_dims_0, x = ones)[name = string("op_784")]; int32 var_786 = const()[name = string("op_786"), val = int32(1)]; int32 offset = sub(x = var_784, y = var_786)[name = string("offset")]; tensor var_789 = add(x = position_id, y = offset)[name = string("op_789")]; int32 var_791 = const()[name = string("op_791"), val = int32(1)]; tensor cache_position_end = add(x = var_789, y = var_791)[name = string("cache_position_end")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor greater_equal_0 = greater_equal(x = position_ids_1, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(32768)]; tensor add_0 = add(x = position_ids_1, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = position_ids_1, b = add_0, cond = greater_equal_0)[name = string("select_0")]; tensor rope_emb_cos_cached_to_fp16 = const()[name = string("rope_emb_cos_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957994752)))]; int32 cos_1_batch_dims_0 = const()[name = string("cos_1_batch_dims_0"), val = int32(0)]; bool cos_1_validate_indices_0 = const()[name = string("cos_1_validate_indices_0"), val = bool(false)]; int32 greater_equal_12_y_0 = const()[name = string("greater_equal_12_y_0"), val = int32(0)]; tensor greater_equal_12 = greater_equal(x = select_0, y = greater_equal_12_y_0)[name = string("greater_equal_12")]; int32 slice_by_index_12 = const()[name = string("slice_by_index_12"), val = int32(32768)]; tensor add_12 = add(x = select_0, y = slice_by_index_12)[name = string("add_12")]; tensor select_12 = select(a = select_0, b = add_12, cond = greater_equal_12)[name = string("select_12")]; int32 cos_1_cast_fp16_axis_6 = const()[name = string("cos_1_cast_fp16_axis_6"), val = int32(0)]; tensor cos_1_cast_fp16 = gather(axis = cos_1_cast_fp16_axis_6, batch_dims = cos_1_batch_dims_0, indices = select_12, validate_indices = cos_1_validate_indices_0, x = rope_emb_cos_cached_to_fp16)[name = string("cos_1_cast_fp16")]; tensor rope_emb_sin_cached_to_fp16 = const()[name = string("rope_emb_sin_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(966383424)))]; int32 sin_1_batch_dims_0 = const()[name = string("sin_1_batch_dims_0"), val = int32(0)]; bool sin_1_validate_indices_0 = const()[name = string("sin_1_validate_indices_0"), val = bool(false)]; int32 sin_1_cast_fp16_axis_6 = const()[name = string("sin_1_cast_fp16_axis_6"), val = int32(0)]; tensor sin_1_cast_fp16 = gather(axis = sin_1_cast_fp16_axis_6, batch_dims = sin_1_batch_dims_0, indices = select_12, validate_indices = sin_1_validate_indices_0, x = rope_emb_sin_cached_to_fp16)[name = string("sin_1_cast_fp16")]; tensor var_865_perm_0 = const()[name = string("op_865_perm_0"), val = tensor([-1, -2])]; tensor var_867_axes_0 = const()[name = string("op_867_axes_0"), val = tensor([0])]; tensor var_865_cast_fp16 = transpose(perm = var_865_perm_0, x = cos_1_cast_fp16)[name = string("transpose_601")]; tensor var_867_cast_fp16 = expand_dims(axes = var_867_axes_0, x = var_865_cast_fp16)[name = string("op_867_cast_fp16")]; tensor var_869_axes_0 = const()[name = string("op_869_axes_0"), val = tensor([0])]; tensor var_869_cast_fp16 = expand_dims(axes = var_869_axes_0, x = var_867_cast_fp16)[name = string("op_869_cast_fp16")]; tensor var_874_perm_0 = const()[name = string("op_874_perm_0"), val = tensor([-1, -2])]; tensor var_876_axes_0 = const()[name = string("op_876_axes_0"), val = tensor([0])]; tensor var_874_cast_fp16 = transpose(perm = var_874_perm_0, x = sin_1_cast_fp16)[name = string("transpose_600")]; tensor var_876_cast_fp16 = expand_dims(axes = var_876_axes_0, x = var_874_cast_fp16)[name = string("op_876_cast_fp16")]; tensor var_878_axes_0 = const()[name = string("op_878_axes_0"), val = tensor([0])]; tensor var_878_cast_fp16 = expand_dims(axes = var_878_axes_0, x = var_876_cast_fp16)[name = string("op_878_cast_fp16")]; string position_ids_1_to_uint16_dtype_0 = const()[name = string("position_ids_1_to_uint16_dtype_0"), val = string("uint16")]; tensor causal_mask = const()[name = string("causal_mask"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(974772096)))]; int32 mask_axis_0 = const()[name = string("mask_axis_0"), val = int32(1)]; int32 mask_batch_dims_0 = const()[name = string("mask_batch_dims_0"), val = int32(0)]; bool mask_validate_indices_0 = const()[name = string("mask_validate_indices_0"), val = bool(false)]; tensor position_ids_1_to_uint16 = cast(dtype = position_ids_1_to_uint16_dtype_0, x = position_ids_1)[name = string("cast_13")]; tensor mask_cast_uint16 = gather(axis = mask_axis_0, batch_dims = mask_batch_dims_0, indices = position_ids_1_to_uint16, validate_indices = mask_validate_indices_0, x = causal_mask)[name = string("mask_cast_uint16")]; tensor var_895_axes_0 = const()[name = string("op_895_axes_0"), val = tensor([0])]; tensor var_895 = expand_dims(axes = var_895_axes_0, x = mask_cast_uint16)[name = string("op_895")]; tensor attn_mask_1_axes_0 = const()[name = string("attn_mask_1_axes_0"), val = tensor([0])]; tensor attn_mask_1 = expand_dims(axes = attn_mask_1_axes_0, x = var_895)[name = string("attn_mask_1")]; string inputs_embeds_to_fp16_dtype_0 = const()[name = string("inputs_embeds_to_fp16_dtype_0"), val = string("fp16")]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor inputs_embeds_to_fp16 = cast(dtype = inputs_embeds_to_fp16_dtype_0, x = inputs_embeds)[name = string("cast_12")]; tensor var_906_cast_fp16 = mul(x = inputs_embeds_to_fp16, y = const_0_promoted_to_fp16)[name = string("op_906_cast_fp16")]; int32 var_904 = const()[name = string("op_904"), val = int32(1)]; bool doubled_1_interleave_0 = const()[name = string("doubled_1_interleave_0"), val = bool(false)]; tensor doubled_1_cast_fp16 = concat(axis = var_904, interleave = doubled_1_interleave_0, values = (inputs_embeds_to_fp16, var_906_cast_fp16))[name = string("doubled_1_cast_fp16")]; tensor out_1_axes_0 = const()[name = string("out_1_axes_0"), val = tensor([1])]; tensor out_1_gamma_0_to_fp16 = const()[name = string("out_1_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983160768)))]; fp16 var_916_to_fp16 = const()[name = string("op_916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_1_cast_fp16 = layer_norm(axes = out_1_axes_0, epsilon = var_916_to_fp16, gamma = out_1_gamma_0_to_fp16, x = doubled_1_cast_fp16)[name = string("out_1_cast_fp16")]; tensor var_927_split_sizes_0 = const()[name = string("op_927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_927_axis_0 = const()[name = string("op_927_axis_0"), val = int32(1)]; tensor var_927_cast_fp16_0, tensor var_927_cast_fp16_1 = split(axis = var_927_axis_0, split_sizes = var_927_split_sizes_0, x = out_1_cast_fp16)[name = string("op_927_cast_fp16")]; tensor layers_0_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983169024)))]; tensor query_states_1_strides_0 = const()[name = string("query_states_1_strides_0"), val = tensor([1, 1])]; string query_states_1_pad_type_0 = const()[name = string("query_states_1_pad_type_0"), val = string("valid")]; tensor query_states_1_pad_0 = const()[name = string("query_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_1_dilations_0 = const()[name = string("query_states_1_dilations_0"), val = tensor([1, 1])]; int32 query_states_1_groups_0 = const()[name = string("query_states_1_groups_0"), val = int32(1)]; tensor query_states_1_cast_fp16 = conv(dilations = query_states_1_dilations_0, groups = query_states_1_groups_0, pad = query_states_1_pad_0, pad_type = query_states_1_pad_type_0, strides = query_states_1_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("query_states_1_cast_fp16")]; tensor layers_0_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(991557696)))]; tensor key_states_1_strides_0 = const()[name = string("key_states_1_strides_0"), val = tensor([1, 1])]; string key_states_1_pad_type_0 = const()[name = string("key_states_1_pad_type_0"), val = string("valid")]; tensor key_states_1_pad_0 = const()[name = string("key_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_1_dilations_0 = const()[name = string("key_states_1_dilations_0"), val = tensor([1, 1])]; int32 key_states_1_groups_0 = const()[name = string("key_states_1_groups_0"), val = int32(1)]; tensor key_states_1_cast_fp16 = conv(dilations = key_states_1_dilations_0, groups = key_states_1_groups_0, pad = key_states_1_pad_0, pad_type = key_states_1_pad_type_0, strides = key_states_1_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("key_states_1_cast_fp16")]; tensor layers_0_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(992606336)))]; tensor value_states_1_strides_0 = const()[name = string("value_states_1_strides_0"), val = tensor([1, 1])]; string value_states_1_pad_type_0 = const()[name = string("value_states_1_pad_type_0"), val = string("valid")]; tensor value_states_1_pad_0 = const()[name = string("value_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_1_dilations_0 = const()[name = string("value_states_1_dilations_0"), val = tensor([1, 1])]; int32 value_states_1_groups_0 = const()[name = string("value_states_1_groups_0"), val = int32(1)]; tensor value_states_1_cast_fp16 = conv(dilations = value_states_1_dilations_0, groups = value_states_1_groups_0, pad = value_states_1_pad_0, pad_type = value_states_1_pad_type_0, strides = value_states_1_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("value_states_1_cast_fp16")]; tensor concat_0x = const()[name = string("concat_0x"), val = tensor([1, 16, 128, -1])]; tensor x_1_cast_fp16 = reshape(shape = concat_0x, x = query_states_1_cast_fp16)[name = string("x_1_cast_fp16")]; tensor concat_1x = const()[name = string("concat_1x"), val = tensor([1, 2, 128, -1])]; tensor var_984_cast_fp16 = reshape(shape = concat_1x, x = key_states_1_cast_fp16)[name = string("op_984_cast_fp16")]; tensor concat_2x = const()[name = string("concat_2x"), val = tensor([1, 2, 128, -1])]; tensor var_991_cast_fp16 = reshape(shape = concat_2x, x = value_states_1_cast_fp16)[name = string("op_991_cast_fp16")]; tensor var_995_cast_fp16 = mul(x = x_1_cast_fp16, y = var_869_cast_fp16)[name = string("op_995_cast_fp16")]; tensor var_996_split_sizes_0 = const()[name = string("op_996_split_sizes_0"), val = tensor([64, 64])]; int32 var_996_axis_0 = const()[name = string("op_996_axis_0"), val = int32(-2)]; tensor var_996_cast_fp16_0, tensor var_996_cast_fp16_1 = split(axis = var_996_axis_0, split_sizes = var_996_split_sizes_0, x = x_1_cast_fp16)[name = string("op_996_cast_fp16")]; fp16 const_2_promoted_to_fp16 = const()[name = string("const_2_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_998_cast_fp16 = mul(x = var_996_cast_fp16_1, y = const_2_promoted_to_fp16)[name = string("op_998_cast_fp16")]; int32 var_1000 = const()[name = string("op_1000"), val = int32(-2)]; bool var_1001_interleave_0 = const()[name = string("op_1001_interleave_0"), val = bool(false)]; tensor var_1001_cast_fp16 = concat(axis = var_1000, interleave = var_1001_interleave_0, values = (var_998_cast_fp16, var_996_cast_fp16_0))[name = string("op_1001_cast_fp16")]; tensor var_1002_cast_fp16 = mul(x = var_1001_cast_fp16, y = var_878_cast_fp16)[name = string("op_1002_cast_fp16")]; tensor query_states_3_cast_fp16 = add(x = var_995_cast_fp16, y = var_1002_cast_fp16)[name = string("query_states_3_cast_fp16")]; tensor var_1008_cast_fp16 = mul(x = var_984_cast_fp16, y = var_869_cast_fp16)[name = string("op_1008_cast_fp16")]; tensor var_1009_split_sizes_0 = const()[name = string("op_1009_split_sizes_0"), val = tensor([64, 64])]; int32 var_1009_axis_0 = const()[name = string("op_1009_axis_0"), val = int32(-2)]; tensor var_1009_cast_fp16_0, tensor var_1009_cast_fp16_1 = split(axis = var_1009_axis_0, split_sizes = var_1009_split_sizes_0, x = var_984_cast_fp16)[name = string("op_1009_cast_fp16")]; fp16 const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1011_cast_fp16 = mul(x = var_1009_cast_fp16_1, y = const_3_promoted_to_fp16)[name = string("op_1011_cast_fp16")]; int32 var_1013 = const()[name = string("op_1013"), val = int32(-2)]; bool var_1014_interleave_0 = const()[name = string("op_1014_interleave_0"), val = bool(false)]; tensor var_1014_cast_fp16 = concat(axis = var_1013, interleave = var_1014_interleave_0, values = (var_1011_cast_fp16, var_1009_cast_fp16_0))[name = string("op_1014_cast_fp16")]; tensor var_1015_cast_fp16 = mul(x = var_1014_cast_fp16, y = var_878_cast_fp16)[name = string("op_1015_cast_fp16")]; tensor key_states_5_cast_fp16 = add(x = var_1008_cast_fp16, y = var_1015_cast_fp16)[name = string("key_states_5_cast_fp16")]; tensor read_state_0 = read_state(input = key_cache)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; int32 concat_5_axis_0 = const()[name = string("concat_5_axis_0"), val = int32(0)]; bool concat_5_interleave_0 = const()[name = string("concat_5_interleave_0"), val = bool(false)]; tensor concat_5 = concat(axis = concat_5_axis_0, interleave = concat_5_interleave_0, values = (expand_dims_0, expand_dims_1, position_id, expand_dims_3))[name = string("concat_5")]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; tensor concat_6_values1_0 = const()[name = string("concat_6_values1_0"), val = tensor([0])]; tensor concat_6_values3_0 = const()[name = string("concat_6_values3_0"), val = tensor([0])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_4, concat_6_values1_0, cache_position_end, concat_6_values3_0))[name = string("concat_6")]; tensor key_states_7_perm_0 = const()[name = string("key_states_7_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_1_stride_0 = const()[name = string("key_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_7_cast_fp16 = transpose(perm = key_states_7_perm_0, x = key_states_5_cast_fp16)[name = string("transpose_599")]; tensor key_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = key_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = key_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_1_squeeze_mask_0, stride = key_cache_internal_tensor_assign_1_stride_0, update = key_states_7_cast_fp16, x = read_state_0)[name = string("key_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_1_cast_fp16, input = key_cache)[name = string("coreml_update_state_336_write_state")]; tensor coreml_update_state_336 = read_state(input = key_cache)[name = string("coreml_update_state_336")]; tensor read_state_1 = read_state(input = value_cache)[name = string("read_state_1")]; tensor value_states_3_perm_0 = const()[name = string("value_states_3_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_1_stride_0 = const()[name = string("value_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_3_cast_fp16 = transpose(perm = value_states_3_perm_0, x = var_991_cast_fp16)[name = string("transpose_598")]; tensor value_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = value_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = value_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_1_squeeze_mask_0, stride = value_cache_internal_tensor_assign_1_stride_0, update = value_states_3_cast_fp16, x = read_state_1)[name = string("value_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_1_cast_fp16, input = value_cache)[name = string("coreml_update_state_337_write_state")]; tensor coreml_update_state_337 = read_state(input = value_cache)[name = string("coreml_update_state_337")]; tensor var_1085_begin_0 = const()[name = string("op_1085_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1085_end_0 = const()[name = string("op_1085_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1085_end_mask_0 = const()[name = string("op_1085_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1085_cast_fp16 = slice_by_index(begin = var_1085_begin_0, end = var_1085_end_0, end_mask = var_1085_end_mask_0, x = coreml_update_state_336)[name = string("op_1085_cast_fp16")]; tensor tile_0 = const()[name = string("tile_0"), val = tensor([1, 1])]; int32 var_1088_axis_0 = const()[name = string("op_1088_axis_0"), val = int32(1)]; tensor var_1088_cast_fp16_0, tensor var_1088_cast_fp16_1 = split(axis = var_1088_axis_0, split_sizes = tile_0, x = var_1085_cast_fp16)[name = string("op_1088_cast_fp16")]; tensor var_1095_begin_0 = const()[name = string("op_1095_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1095_end_0 = const()[name = string("op_1095_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1095_end_mask_0 = const()[name = string("op_1095_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1095_cast_fp16 = slice_by_index(begin = var_1095_begin_0, end = var_1095_end_0, end_mask = var_1095_end_mask_0, x = coreml_update_state_337)[name = string("op_1095_cast_fp16")]; tensor tile_1 = const()[name = string("tile_1"), val = tensor([1, 1])]; int32 var_1098_axis_0 = const()[name = string("op_1098_axis_0"), val = int32(1)]; tensor var_1098_cast_fp16_0, tensor var_1098_cast_fp16_1 = split(axis = var_1098_axis_0, split_sizes = tile_1, x = var_1095_cast_fp16)[name = string("op_1098_cast_fp16")]; tensor var_1101_split_sizes_0 = const()[name = string("op_1101_split_sizes_0"), val = tensor([8, 8])]; int32 var_1101_axis_0 = const()[name = string("op_1101_axis_0"), val = int32(1)]; tensor var_1101_0, tensor var_1101_1 = split(axis = var_1101_axis_0, split_sizes = var_1101_split_sizes_0, x = query_states_3_cast_fp16)[name = string("op_1101")]; bool attn_weights_1_transpose_x_0 = const()[name = string("attn_weights_1_transpose_x_0"), val = bool(false)]; bool attn_weights_1_transpose_y_0 = const()[name = string("attn_weights_1_transpose_y_0"), val = bool(false)]; tensor attn_weights_1_cast_fp16 = matmul(transpose_x = attn_weights_1_transpose_x_0, transpose_y = attn_weights_1_transpose_y_0, x = var_1088_cast_fp16_0, y = var_1101_0)[name = string("attn_weights_1_cast_fp16")]; fp16 var_1104_to_fp16 = const()[name = string("op_1104_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_3_cast_fp16 = mul(x = attn_weights_1_cast_fp16, y = var_1104_to_fp16)[name = string("attn_weights_3_cast_fp16")]; tensor attn_weights_5_cast_fp16 = add(x = attn_weights_3_cast_fp16, y = attn_mask_1)[name = string("attn_weights_5_cast_fp16")]; int32 var_1108 = const()[name = string("op_1108"), val = int32(-2)]; tensor attn_weights_7_cast_fp16 = softmax(axis = var_1108, x = attn_weights_5_cast_fp16)[name = string("attn_weights_7_cast_fp16")]; bool var_1114_transpose_x_1 = const()[name = string("op_1114_transpose_x_1"), val = bool(true)]; bool var_1114_transpose_y_1 = const()[name = string("op_1114_transpose_y_1"), val = bool(false)]; tensor var_1114_cast_fp16 = matmul(transpose_x = var_1114_transpose_x_1, transpose_y = var_1114_transpose_y_1, x = attn_weights_7_cast_fp16, y = var_1098_cast_fp16_0)[name = string("op_1114_cast_fp16")]; bool attn_weights_9_transpose_x_0 = const()[name = string("attn_weights_9_transpose_x_0"), val = bool(false)]; bool attn_weights_9_transpose_y_0 = const()[name = string("attn_weights_9_transpose_y_0"), val = bool(false)]; tensor attn_weights_9_cast_fp16 = matmul(transpose_x = attn_weights_9_transpose_x_0, transpose_y = attn_weights_9_transpose_y_0, x = var_1088_cast_fp16_1, y = var_1101_1)[name = string("attn_weights_9_cast_fp16")]; fp16 var_1116_to_fp16 = const()[name = string("op_1116_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_11_cast_fp16 = mul(x = attn_weights_9_cast_fp16, y = var_1116_to_fp16)[name = string("attn_weights_11_cast_fp16")]; tensor attn_weights_13_cast_fp16 = add(x = attn_weights_11_cast_fp16, y = attn_mask_1)[name = string("attn_weights_13_cast_fp16")]; int32 var_1120 = const()[name = string("op_1120"), val = int32(-2)]; tensor attn_weights_15_cast_fp16 = softmax(axis = var_1120, x = attn_weights_13_cast_fp16)[name = string("attn_weights_15_cast_fp16")]; bool attn_output_1_transpose_x_1 = const()[name = string("attn_output_1_transpose_x_1"), val = bool(true)]; bool attn_output_1_transpose_y_1 = const()[name = string("attn_output_1_transpose_y_1"), val = bool(false)]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_1, transpose_y = attn_output_1_transpose_y_1, x = attn_weights_15_cast_fp16, y = var_1098_cast_fp16_1)[name = string("attn_output_1_cast_fp16")]; int32 var_1128 = const()[name = string("op_1128"), val = int32(1)]; bool attn_output_3_interleave_0 = const()[name = string("attn_output_3_interleave_0"), val = bool(false)]; tensor attn_output_3_cast_fp16 = concat(axis = var_1128, interleave = attn_output_3_interleave_0, values = (var_1114_cast_fp16, attn_output_1_cast_fp16))[name = string("attn_output_3_cast_fp16")]; tensor var_1132_perm_0 = const()[name = string("op_1132_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_11x = const()[name = string("concat_11x"), val = tensor([1, 2048, 1, -1])]; tensor var_1132_cast_fp16 = transpose(perm = var_1132_perm_0, x = attn_output_3_cast_fp16)[name = string("transpose_597")]; tensor attn_output_7_cast_fp16 = reshape(shape = concat_11x, x = var_1132_cast_fp16)[name = string("attn_output_7_cast_fp16")]; tensor layers_0_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(993654976)))]; tensor hidden_states_3_strides_0 = const()[name = string("hidden_states_3_strides_0"), val = tensor([1, 1])]; string hidden_states_3_pad_type_0 = const()[name = string("hidden_states_3_pad_type_0"), val = string("valid")]; tensor hidden_states_3_pad_0 = const()[name = string("hidden_states_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_3_dilations_0 = const()[name = string("hidden_states_3_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_3_groups_0 = const()[name = string("hidden_states_3_groups_0"), val = int32(1)]; tensor hidden_states_3_cast_fp16 = conv(dilations = hidden_states_3_dilations_0, groups = hidden_states_3_groups_0, pad = hidden_states_3_pad_0, pad_type = hidden_states_3_pad_type_0, strides = hidden_states_3_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16, x = attn_output_7_cast_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor hidden_states_5_cast_fp16 = add(x = inputs_embeds_to_fp16, y = hidden_states_3_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1165_cast_fp16 = mul(x = hidden_states_5_cast_fp16, y = const_8_promoted_to_fp16)[name = string("op_1165_cast_fp16")]; int32 var_1163 = const()[name = string("op_1163"), val = int32(1)]; bool doubled_5_interleave_0 = const()[name = string("doubled_5_interleave_0"), val = bool(false)]; tensor doubled_5_cast_fp16 = concat(axis = var_1163, interleave = doubled_5_interleave_0, values = (hidden_states_5_cast_fp16, var_1165_cast_fp16))[name = string("doubled_5_cast_fp16")]; tensor out_3_axes_0 = const()[name = string("out_3_axes_0"), val = tensor([1])]; tensor out_3_gamma_0_to_fp16 = const()[name = string("out_3_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002043648)))]; fp16 var_1175_to_fp16 = const()[name = string("op_1175_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_3_cast_fp16 = layer_norm(axes = out_3_axes_0, epsilon = var_1175_to_fp16, gamma = out_3_gamma_0_to_fp16, x = doubled_5_cast_fp16)[name = string("out_3_cast_fp16")]; tensor var_1186_split_sizes_0 = const()[name = string("op_1186_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1186_axis_0 = const()[name = string("op_1186_axis_0"), val = int32(1)]; tensor var_1186_cast_fp16_0, tensor var_1186_cast_fp16_1 = split(axis = var_1186_axis_0, split_sizes = var_1186_split_sizes_0, x = out_3_cast_fp16)[name = string("op_1186_cast_fp16")]; tensor layers_0_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002051904)))]; tensor input_1_strides_0 = const()[name = string("input_1_strides_0"), val = tensor([1, 1])]; string input_1_pad_type_0 = const()[name = string("input_1_pad_type_0"), val = string("valid")]; tensor input_1_pad_0 = const()[name = string("input_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_1_dilations_0 = const()[name = string("input_1_dilations_0"), val = tensor([1, 1])]; int32 input_1_groups_0 = const()[name = string("input_1_groups_0"), val = int32(1)]; tensor input_1_cast_fp16 = conv(dilations = input_1_dilations_0, groups = input_1_groups_0, pad = input_1_pad_0, pad_type = input_1_pad_type_0, strides = input_1_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("input_1_cast_fp16")]; tensor var_1203_cast_fp16 = silu(x = input_1_cast_fp16)[name = string("op_1203_cast_fp16")]; tensor layers_0_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1027217792)))]; tensor var_1209_strides_0 = const()[name = string("op_1209_strides_0"), val = tensor([1, 1])]; string var_1209_pad_type_0 = const()[name = string("op_1209_pad_type_0"), val = string("valid")]; tensor var_1209_pad_0 = const()[name = string("op_1209_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1209_dilations_0 = const()[name = string("op_1209_dilations_0"), val = tensor([1, 1])]; int32 var_1209_groups_0 = const()[name = string("op_1209_groups_0"), val = int32(1)]; tensor var_1209_cast_fp16 = conv(dilations = var_1209_dilations_0, groups = var_1209_groups_0, pad = var_1209_pad_0, pad_type = var_1209_pad_type_0, strides = var_1209_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("op_1209_cast_fp16")]; tensor x_9_cast_fp16 = mul(x = var_1203_cast_fp16, y = var_1209_cast_fp16)[name = string("x_9_cast_fp16")]; tensor layers_0_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1052383680)))]; tensor hidden_states_7_strides_0 = const()[name = string("hidden_states_7_strides_0"), val = tensor([1, 1])]; string hidden_states_7_pad_type_0 = const()[name = string("hidden_states_7_pad_type_0"), val = string("valid")]; tensor hidden_states_7_pad_0 = const()[name = string("hidden_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_7_dilations_0 = const()[name = string("hidden_states_7_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_7_groups_0 = const()[name = string("hidden_states_7_groups_0"), val = int32(1)]; tensor hidden_states_7_cast_fp16 = conv(dilations = hidden_states_7_dilations_0, groups = hidden_states_7_groups_0, pad = hidden_states_7_pad_0, pad_type = hidden_states_7_pad_type_0, strides = hidden_states_7_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16, x = x_9_cast_fp16)[name = string("hidden_states_7_cast_fp16")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = hidden_states_7_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; fp16 const_10_promoted_to_fp16 = const()[name = string("const_10_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1227_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_10_promoted_to_fp16)[name = string("op_1227_cast_fp16")]; int32 var_1225 = const()[name = string("op_1225"), val = int32(1)]; bool doubled_9_interleave_0 = const()[name = string("doubled_9_interleave_0"), val = bool(false)]; tensor doubled_9_cast_fp16 = concat(axis = var_1225, interleave = doubled_9_interleave_0, values = (hidden_states_9_cast_fp16, var_1227_cast_fp16))[name = string("doubled_9_cast_fp16")]; tensor out_5_axes_0 = const()[name = string("out_5_axes_0"), val = tensor([1])]; tensor out_5_gamma_0_to_fp16 = const()[name = string("out_5_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077549568)))]; fp16 var_1237_to_fp16 = const()[name = string("op_1237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_5_cast_fp16 = layer_norm(axes = out_5_axes_0, epsilon = var_1237_to_fp16, gamma = out_5_gamma_0_to_fp16, x = doubled_9_cast_fp16)[name = string("out_5_cast_fp16")]; tensor var_1248_split_sizes_0 = const()[name = string("op_1248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1248_axis_0 = const()[name = string("op_1248_axis_0"), val = int32(1)]; tensor var_1248_cast_fp16_0, tensor var_1248_cast_fp16_1 = split(axis = var_1248_axis_0, split_sizes = var_1248_split_sizes_0, x = out_5_cast_fp16)[name = string("op_1248_cast_fp16")]; tensor layers_1_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077557824)))]; tensor query_states_7_strides_0 = const()[name = string("query_states_7_strides_0"), val = tensor([1, 1])]; string query_states_7_pad_type_0 = const()[name = string("query_states_7_pad_type_0"), val = string("valid")]; tensor query_states_7_pad_0 = const()[name = string("query_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_7_dilations_0 = const()[name = string("query_states_7_dilations_0"), val = tensor([1, 1])]; int32 query_states_7_groups_0 = const()[name = string("query_states_7_groups_0"), val = int32(1)]; tensor query_states_7_cast_fp16 = conv(dilations = query_states_7_dilations_0, groups = query_states_7_groups_0, pad = query_states_7_pad_0, pad_type = query_states_7_pad_type_0, strides = query_states_7_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("query_states_7_cast_fp16")]; tensor layers_1_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1085946496)))]; tensor key_states_11_strides_0 = const()[name = string("key_states_11_strides_0"), val = tensor([1, 1])]; string key_states_11_pad_type_0 = const()[name = string("key_states_11_pad_type_0"), val = string("valid")]; tensor key_states_11_pad_0 = const()[name = string("key_states_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_11_dilations_0 = const()[name = string("key_states_11_dilations_0"), val = tensor([1, 1])]; int32 key_states_11_groups_0 = const()[name = string("key_states_11_groups_0"), val = int32(1)]; tensor key_states_11_cast_fp16 = conv(dilations = key_states_11_dilations_0, groups = key_states_11_groups_0, pad = key_states_11_pad_0, pad_type = key_states_11_pad_type_0, strides = key_states_11_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("key_states_11_cast_fp16")]; tensor value_states_7_strides_0 = const()[name = string("value_states_7_strides_0"), val = tensor([1, 1])]; string value_states_7_pad_type_0 = const()[name = string("value_states_7_pad_type_0"), val = string("valid")]; tensor value_states_7_pad_0 = const()[name = string("value_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_7_dilations_0 = const()[name = string("value_states_7_dilations_0"), val = tensor([1, 1])]; int32 value_states_7_groups_0 = const()[name = string("value_states_7_groups_0"), val = int32(1)]; tensor value_states_7_cast_fp16 = conv(dilations = value_states_7_dilations_0, groups = value_states_7_groups_0, pad = value_states_7_pad_0, pad_type = value_states_7_pad_type_0, strides = value_states_7_strides_0, weight = layers_1_self_attn_v_proj_weight_cast_fp16, x = var_1248_cast_fp16_0)[name = string("value_states_7_cast_fp16")]; tensor concat_12x = const()[name = string("concat_12x"), val = tensor([1, 16, 128, -1])]; tensor x_11_cast_fp16 = reshape(shape = concat_12x, x = query_states_7_cast_fp16)[name = string("x_11_cast_fp16")]; tensor concat_13x = const()[name = string("concat_13x"), val = tensor([1, 2, 128, -1])]; tensor var_1305_cast_fp16 = reshape(shape = concat_13x, x = key_states_11_cast_fp16)[name = string("op_1305_cast_fp16")]; tensor concat_14x = const()[name = string("concat_14x"), val = tensor([1, 2, 128, -1])]; tensor var_1312_cast_fp16 = reshape(shape = concat_14x, x = value_states_7_cast_fp16)[name = string("op_1312_cast_fp16")]; tensor var_1316_cast_fp16 = mul(x = x_11_cast_fp16, y = var_869_cast_fp16)[name = string("op_1316_cast_fp16")]; tensor var_1317_split_sizes_0 = const()[name = string("op_1317_split_sizes_0"), val = tensor([64, 64])]; int32 var_1317_axis_0 = const()[name = string("op_1317_axis_0"), val = int32(-2)]; tensor var_1317_cast_fp16_0, tensor var_1317_cast_fp16_1 = split(axis = var_1317_axis_0, split_sizes = var_1317_split_sizes_0, x = x_11_cast_fp16)[name = string("op_1317_cast_fp16")]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1319_cast_fp16 = mul(x = var_1317_cast_fp16_1, y = const_12_promoted_to_fp16)[name = string("op_1319_cast_fp16")]; int32 var_1321 = const()[name = string("op_1321"), val = int32(-2)]; bool var_1322_interleave_0 = const()[name = string("op_1322_interleave_0"), val = bool(false)]; tensor var_1322_cast_fp16 = concat(axis = var_1321, interleave = var_1322_interleave_0, values = (var_1319_cast_fp16, var_1317_cast_fp16_0))[name = string("op_1322_cast_fp16")]; tensor var_1323_cast_fp16 = mul(x = var_1322_cast_fp16, y = var_878_cast_fp16)[name = string("op_1323_cast_fp16")]; tensor query_states_9_cast_fp16 = add(x = var_1316_cast_fp16, y = var_1323_cast_fp16)[name = string("query_states_9_cast_fp16")]; tensor var_1329_cast_fp16 = mul(x = var_1305_cast_fp16, y = var_869_cast_fp16)[name = string("op_1329_cast_fp16")]; tensor var_1330_split_sizes_0 = const()[name = string("op_1330_split_sizes_0"), val = tensor([64, 64])]; int32 var_1330_axis_0 = const()[name = string("op_1330_axis_0"), val = int32(-2)]; tensor var_1330_cast_fp16_0, tensor var_1330_cast_fp16_1 = split(axis = var_1330_axis_0, split_sizes = var_1330_split_sizes_0, x = var_1305_cast_fp16)[name = string("op_1330_cast_fp16")]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1332_cast_fp16 = mul(x = var_1330_cast_fp16_1, y = const_13_promoted_to_fp16)[name = string("op_1332_cast_fp16")]; int32 var_1334 = const()[name = string("op_1334"), val = int32(-2)]; bool var_1335_interleave_0 = const()[name = string("op_1335_interleave_0"), val = bool(false)]; tensor var_1335_cast_fp16 = concat(axis = var_1334, interleave = var_1335_interleave_0, values = (var_1332_cast_fp16, var_1330_cast_fp16_0))[name = string("op_1335_cast_fp16")]; tensor var_1336_cast_fp16 = mul(x = var_1335_cast_fp16, y = var_878_cast_fp16)[name = string("op_1336_cast_fp16")]; tensor key_states_15_cast_fp16 = add(x = var_1329_cast_fp16, y = var_1336_cast_fp16)[name = string("key_states_15_cast_fp16")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; int32 concat_17_axis_0 = const()[name = string("concat_17_axis_0"), val = int32(0)]; bool concat_17_interleave_0 = const()[name = string("concat_17_interleave_0"), val = bool(false)]; tensor concat_17 = concat(axis = concat_17_axis_0, interleave = concat_17_interleave_0, values = (expand_dims_12, expand_dims_13, position_id, expand_dims_15))[name = string("concat_17")]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; tensor concat_18_values1_0 = const()[name = string("concat_18_values1_0"), val = tensor([0])]; tensor concat_18_values3_0 = const()[name = string("concat_18_values3_0"), val = tensor([0])]; int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_16, concat_18_values1_0, cache_position_end, concat_18_values3_0))[name = string("concat_18")]; tensor key_states_17_perm_0 = const()[name = string("key_states_17_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_2_stride_0 = const()[name = string("key_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_17_cast_fp16 = transpose(perm = key_states_17_perm_0, x = key_states_15_cast_fp16)[name = string("transpose_596")]; tensor key_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = key_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = key_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_2_squeeze_mask_0, stride = key_cache_internal_tensor_assign_2_stride_0, update = key_states_17_cast_fp16, x = coreml_update_state_336)[name = string("key_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_2_cast_fp16, input = key_cache)[name = string("coreml_update_state_338_write_state")]; tensor coreml_update_state_338 = read_state(input = key_cache)[name = string("coreml_update_state_338")]; tensor value_states_9_perm_0 = const()[name = string("value_states_9_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_2_stride_0 = const()[name = string("value_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_9_cast_fp16 = transpose(perm = value_states_9_perm_0, x = var_1312_cast_fp16)[name = string("transpose_595")]; tensor value_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = value_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = value_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_2_squeeze_mask_0, stride = value_cache_internal_tensor_assign_2_stride_0, update = value_states_9_cast_fp16, x = coreml_update_state_337)[name = string("value_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_2_cast_fp16, input = value_cache)[name = string("coreml_update_state_339_write_state")]; tensor coreml_update_state_339 = read_state(input = value_cache)[name = string("coreml_update_state_339")]; tensor var_1406_begin_0 = const()[name = string("op_1406_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1406_end_0 = const()[name = string("op_1406_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1406_end_mask_0 = const()[name = string("op_1406_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1406_cast_fp16 = slice_by_index(begin = var_1406_begin_0, end = var_1406_end_0, end_mask = var_1406_end_mask_0, x = coreml_update_state_338)[name = string("op_1406_cast_fp16")]; tensor tile_2 = const()[name = string("tile_2"), val = tensor([1, 1])]; int32 var_1409_axis_0 = const()[name = string("op_1409_axis_0"), val = int32(1)]; tensor var_1409_cast_fp16_0, tensor var_1409_cast_fp16_1 = split(axis = var_1409_axis_0, split_sizes = tile_2, x = var_1406_cast_fp16)[name = string("op_1409_cast_fp16")]; tensor var_1416_begin_0 = const()[name = string("op_1416_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1416_end_0 = const()[name = string("op_1416_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1416_end_mask_0 = const()[name = string("op_1416_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1416_cast_fp16 = slice_by_index(begin = var_1416_begin_0, end = var_1416_end_0, end_mask = var_1416_end_mask_0, x = coreml_update_state_339)[name = string("op_1416_cast_fp16")]; tensor tile_3 = const()[name = string("tile_3"), val = tensor([1, 1])]; int32 var_1419_axis_0 = const()[name = string("op_1419_axis_0"), val = int32(1)]; tensor var_1419_cast_fp16_0, tensor var_1419_cast_fp16_1 = split(axis = var_1419_axis_0, split_sizes = tile_3, x = var_1416_cast_fp16)[name = string("op_1419_cast_fp16")]; tensor var_1422_split_sizes_0 = const()[name = string("op_1422_split_sizes_0"), val = tensor([8, 8])]; int32 var_1422_axis_0 = const()[name = string("op_1422_axis_0"), val = int32(1)]; tensor var_1422_0, tensor var_1422_1 = split(axis = var_1422_axis_0, split_sizes = var_1422_split_sizes_0, x = query_states_9_cast_fp16)[name = string("op_1422")]; bool attn_weights_17_transpose_x_0 = const()[name = string("attn_weights_17_transpose_x_0"), val = bool(false)]; bool attn_weights_17_transpose_y_0 = const()[name = string("attn_weights_17_transpose_y_0"), val = bool(false)]; tensor attn_weights_17_cast_fp16 = matmul(transpose_x = attn_weights_17_transpose_x_0, transpose_y = attn_weights_17_transpose_y_0, x = var_1409_cast_fp16_0, y = var_1422_0)[name = string("attn_weights_17_cast_fp16")]; fp16 var_1425_to_fp16 = const()[name = string("op_1425_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_19_cast_fp16 = mul(x = attn_weights_17_cast_fp16, y = var_1425_to_fp16)[name = string("attn_weights_19_cast_fp16")]; tensor attn_weights_21_cast_fp16 = add(x = attn_weights_19_cast_fp16, y = attn_mask_1)[name = string("attn_weights_21_cast_fp16")]; int32 var_1429 = const()[name = string("op_1429"), val = int32(-2)]; tensor attn_weights_23_cast_fp16 = softmax(axis = var_1429, x = attn_weights_21_cast_fp16)[name = string("attn_weights_23_cast_fp16")]; bool var_1435_transpose_x_1 = const()[name = string("op_1435_transpose_x_1"), val = bool(true)]; bool var_1435_transpose_y_1 = const()[name = string("op_1435_transpose_y_1"), val = bool(false)]; tensor var_1435_cast_fp16 = matmul(transpose_x = var_1435_transpose_x_1, transpose_y = var_1435_transpose_y_1, x = attn_weights_23_cast_fp16, y = var_1419_cast_fp16_0)[name = string("op_1435_cast_fp16")]; bool attn_weights_25_transpose_x_0 = const()[name = string("attn_weights_25_transpose_x_0"), val = bool(false)]; bool attn_weights_25_transpose_y_0 = const()[name = string("attn_weights_25_transpose_y_0"), val = bool(false)]; tensor attn_weights_25_cast_fp16 = matmul(transpose_x = attn_weights_25_transpose_x_0, transpose_y = attn_weights_25_transpose_y_0, x = var_1409_cast_fp16_1, y = var_1422_1)[name = string("attn_weights_25_cast_fp16")]; fp16 var_1437_to_fp16 = const()[name = string("op_1437_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_27_cast_fp16 = mul(x = attn_weights_25_cast_fp16, y = var_1437_to_fp16)[name = string("attn_weights_27_cast_fp16")]; tensor attn_weights_29_cast_fp16 = add(x = attn_weights_27_cast_fp16, y = attn_mask_1)[name = string("attn_weights_29_cast_fp16")]; int32 var_1441 = const()[name = string("op_1441"), val = int32(-2)]; tensor attn_weights_31_cast_fp16 = softmax(axis = var_1441, x = attn_weights_29_cast_fp16)[name = string("attn_weights_31_cast_fp16")]; bool attn_output_9_transpose_x_1 = const()[name = string("attn_output_9_transpose_x_1"), val = bool(true)]; bool attn_output_9_transpose_y_1 = const()[name = string("attn_output_9_transpose_y_1"), val = bool(false)]; tensor attn_output_9_cast_fp16 = matmul(transpose_x = attn_output_9_transpose_x_1, transpose_y = attn_output_9_transpose_y_1, x = attn_weights_31_cast_fp16, y = var_1419_cast_fp16_1)[name = string("attn_output_9_cast_fp16")]; int32 var_1449 = const()[name = string("op_1449"), val = int32(1)]; bool attn_output_11_interleave_0 = const()[name = string("attn_output_11_interleave_0"), val = bool(false)]; tensor attn_output_11_cast_fp16 = concat(axis = var_1449, interleave = attn_output_11_interleave_0, values = (var_1435_cast_fp16, attn_output_9_cast_fp16))[name = string("attn_output_11_cast_fp16")]; tensor var_1453_perm_0 = const()[name = string("op_1453_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_23x = const()[name = string("concat_23x"), val = tensor([1, 2048, 1, -1])]; tensor var_1453_cast_fp16 = transpose(perm = var_1453_perm_0, x = attn_output_11_cast_fp16)[name = string("transpose_594")]; tensor attn_output_15_cast_fp16 = reshape(shape = concat_23x, x = var_1453_cast_fp16)[name = string("attn_output_15_cast_fp16")]; tensor layers_1_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086995136)))]; tensor hidden_states_13_strides_0 = const()[name = string("hidden_states_13_strides_0"), val = tensor([1, 1])]; string hidden_states_13_pad_type_0 = const()[name = string("hidden_states_13_pad_type_0"), val = string("valid")]; tensor hidden_states_13_pad_0 = const()[name = string("hidden_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_13_dilations_0 = const()[name = string("hidden_states_13_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_13_groups_0 = const()[name = string("hidden_states_13_groups_0"), val = int32(1)]; tensor hidden_states_13_cast_fp16 = conv(dilations = hidden_states_13_dilations_0, groups = hidden_states_13_groups_0, pad = hidden_states_13_pad_0, pad_type = hidden_states_13_pad_type_0, strides = hidden_states_13_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16, x = attn_output_15_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; tensor hidden_states_15_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = hidden_states_13_cast_fp16)[name = string("hidden_states_15_cast_fp16")]; fp16 const_18_promoted_to_fp16 = const()[name = string("const_18_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1486_cast_fp16 = mul(x = hidden_states_15_cast_fp16, y = const_18_promoted_to_fp16)[name = string("op_1486_cast_fp16")]; int32 var_1484 = const()[name = string("op_1484"), val = int32(1)]; bool doubled_13_interleave_0 = const()[name = string("doubled_13_interleave_0"), val = bool(false)]; tensor doubled_13_cast_fp16 = concat(axis = var_1484, interleave = doubled_13_interleave_0, values = (hidden_states_15_cast_fp16, var_1486_cast_fp16))[name = string("doubled_13_cast_fp16")]; tensor out_7_axes_0 = const()[name = string("out_7_axes_0"), val = tensor([1])]; tensor out_7_gamma_0_to_fp16 = const()[name = string("out_7_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095383808)))]; fp16 var_1496_to_fp16 = const()[name = string("op_1496_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_7_cast_fp16 = layer_norm(axes = out_7_axes_0, epsilon = var_1496_to_fp16, gamma = out_7_gamma_0_to_fp16, x = doubled_13_cast_fp16)[name = string("out_7_cast_fp16")]; tensor var_1507_split_sizes_0 = const()[name = string("op_1507_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1507_axis_0 = const()[name = string("op_1507_axis_0"), val = int32(1)]; tensor var_1507_cast_fp16_0, tensor var_1507_cast_fp16_1 = split(axis = var_1507_axis_0, split_sizes = var_1507_split_sizes_0, x = out_7_cast_fp16)[name = string("op_1507_cast_fp16")]; tensor layers_1_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095392064)))]; tensor input_3_strides_0 = const()[name = string("input_3_strides_0"), val = tensor([1, 1])]; string input_3_pad_type_0 = const()[name = string("input_3_pad_type_0"), val = string("valid")]; tensor input_3_pad_0 = const()[name = string("input_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_3_dilations_0 = const()[name = string("input_3_dilations_0"), val = tensor([1, 1])]; int32 input_3_groups_0 = const()[name = string("input_3_groups_0"), val = int32(1)]; tensor input_3_cast_fp16 = conv(dilations = input_3_dilations_0, groups = input_3_groups_0, pad = input_3_pad_0, pad_type = input_3_pad_type_0, strides = input_3_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16, x = var_1507_cast_fp16_0)[name = string("input_3_cast_fp16")]; tensor var_1524_cast_fp16 = silu(x = input_3_cast_fp16)[name = string("op_1524_cast_fp16")]; tensor var_1530_strides_0 = const()[name = string("op_1530_strides_0"), val = tensor([1, 1])]; string var_1530_pad_type_0 = const()[name = string("op_1530_pad_type_0"), val = string("valid")]; tensor var_1530_pad_0 = const()[name = string("op_1530_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1530_dilations_0 = const()[name = string("op_1530_dilations_0"), val = tensor([1, 1])]; int32 var_1530_groups_0 = const()[name = string("op_1530_groups_0"), val = int32(1)]; tensor var_1530_cast_fp16 = conv(dilations = var_1530_dilations_0, groups = var_1530_groups_0, pad = var_1530_pad_0, pad_type = var_1530_pad_type_0, strides = var_1530_strides_0, weight = layers_1_mlp_up_proj_weight_cast_fp16, x = var_1507_cast_fp16_0)[name = string("op_1530_cast_fp16")]; tensor x_19_cast_fp16 = mul(x = var_1524_cast_fp16, y = var_1530_cast_fp16)[name = string("x_19_cast_fp16")]; tensor layers_1_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1120557952)))]; tensor hidden_states_17_strides_0 = const()[name = string("hidden_states_17_strides_0"), val = tensor([1, 1])]; string hidden_states_17_pad_type_0 = const()[name = string("hidden_states_17_pad_type_0"), val = string("valid")]; tensor hidden_states_17_pad_0 = const()[name = string("hidden_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_17_dilations_0 = const()[name = string("hidden_states_17_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_17_groups_0 = const()[name = string("hidden_states_17_groups_0"), val = int32(1)]; tensor hidden_states_17_cast_fp16 = conv(dilations = hidden_states_17_dilations_0, groups = hidden_states_17_groups_0, pad = hidden_states_17_pad_0, pad_type = hidden_states_17_pad_type_0, strides = hidden_states_17_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16, x = x_19_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; tensor hidden_states_19_cast_fp16 = add(x = hidden_states_15_cast_fp16, y = hidden_states_17_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1548_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_1548_cast_fp16")]; int32 var_1546 = const()[name = string("op_1546"), val = int32(1)]; bool doubled_17_interleave_0 = const()[name = string("doubled_17_interleave_0"), val = bool(false)]; tensor doubled_17_cast_fp16 = concat(axis = var_1546, interleave = doubled_17_interleave_0, values = (hidden_states_19_cast_fp16, var_1548_cast_fp16))[name = string("doubled_17_cast_fp16")]; tensor out_9_axes_0 = const()[name = string("out_9_axes_0"), val = tensor([1])]; tensor out_9_gamma_0_to_fp16 = const()[name = string("out_9_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145723840)))]; fp16 var_1558_to_fp16 = const()[name = string("op_1558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_9_cast_fp16 = layer_norm(axes = out_9_axes_0, epsilon = var_1558_to_fp16, gamma = out_9_gamma_0_to_fp16, x = doubled_17_cast_fp16)[name = string("out_9_cast_fp16")]; tensor var_1569_split_sizes_0 = const()[name = string("op_1569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1569_axis_0 = const()[name = string("op_1569_axis_0"), val = int32(1)]; tensor var_1569_cast_fp16_0, tensor var_1569_cast_fp16_1 = split(axis = var_1569_axis_0, split_sizes = var_1569_split_sizes_0, x = out_9_cast_fp16)[name = string("op_1569_cast_fp16")]; tensor layers_2_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145732096)))]; tensor query_states_13_strides_0 = const()[name = string("query_states_13_strides_0"), val = tensor([1, 1])]; string query_states_13_pad_type_0 = const()[name = string("query_states_13_pad_type_0"), val = string("valid")]; tensor query_states_13_pad_0 = const()[name = string("query_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_13_dilations_0 = const()[name = string("query_states_13_dilations_0"), val = tensor([1, 1])]; int32 query_states_13_groups_0 = const()[name = string("query_states_13_groups_0"), val = int32(1)]; tensor query_states_13_cast_fp16 = conv(dilations = query_states_13_dilations_0, groups = query_states_13_groups_0, pad = query_states_13_pad_0, pad_type = query_states_13_pad_type_0, strides = query_states_13_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("query_states_13_cast_fp16")]; tensor layers_2_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1154120768)))]; tensor key_states_21_strides_0 = const()[name = string("key_states_21_strides_0"), val = tensor([1, 1])]; string key_states_21_pad_type_0 = const()[name = string("key_states_21_pad_type_0"), val = string("valid")]; tensor key_states_21_pad_0 = const()[name = string("key_states_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_21_dilations_0 = const()[name = string("key_states_21_dilations_0"), val = tensor([1, 1])]; int32 key_states_21_groups_0 = const()[name = string("key_states_21_groups_0"), val = int32(1)]; tensor key_states_21_cast_fp16 = conv(dilations = key_states_21_dilations_0, groups = key_states_21_groups_0, pad = key_states_21_pad_0, pad_type = key_states_21_pad_type_0, strides = key_states_21_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("key_states_21_cast_fp16")]; tensor value_states_13_strides_0 = const()[name = string("value_states_13_strides_0"), val = tensor([1, 1])]; string value_states_13_pad_type_0 = const()[name = string("value_states_13_pad_type_0"), val = string("valid")]; tensor value_states_13_pad_0 = const()[name = string("value_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_13_dilations_0 = const()[name = string("value_states_13_dilations_0"), val = tensor([1, 1])]; int32 value_states_13_groups_0 = const()[name = string("value_states_13_groups_0"), val = int32(1)]; tensor value_states_13_cast_fp16 = conv(dilations = value_states_13_dilations_0, groups = value_states_13_groups_0, pad = value_states_13_pad_0, pad_type = value_states_13_pad_type_0, strides = value_states_13_strides_0, weight = layers_2_self_attn_v_proj_weight_cast_fp16, x = var_1569_cast_fp16_0)[name = string("value_states_13_cast_fp16")]; tensor concat_24x = const()[name = string("concat_24x"), val = tensor([1, 16, 128, -1])]; tensor x_21_cast_fp16 = reshape(shape = concat_24x, x = query_states_13_cast_fp16)[name = string("x_21_cast_fp16")]; tensor concat_25x = const()[name = string("concat_25x"), val = tensor([1, 2, 128, -1])]; tensor var_1626_cast_fp16 = reshape(shape = concat_25x, x = key_states_21_cast_fp16)[name = string("op_1626_cast_fp16")]; tensor concat_26x = const()[name = string("concat_26x"), val = tensor([1, 2, 128, -1])]; tensor var_1633_cast_fp16 = reshape(shape = concat_26x, x = value_states_13_cast_fp16)[name = string("op_1633_cast_fp16")]; tensor var_1637_cast_fp16 = mul(x = x_21_cast_fp16, y = var_869_cast_fp16)[name = string("op_1637_cast_fp16")]; tensor var_1638_split_sizes_0 = const()[name = string("op_1638_split_sizes_0"), val = tensor([64, 64])]; int32 var_1638_axis_0 = const()[name = string("op_1638_axis_0"), val = int32(-2)]; tensor var_1638_cast_fp16_0, tensor var_1638_cast_fp16_1 = split(axis = var_1638_axis_0, split_sizes = var_1638_split_sizes_0, x = x_21_cast_fp16)[name = string("op_1638_cast_fp16")]; fp16 const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1640_cast_fp16 = mul(x = var_1638_cast_fp16_1, y = const_22_promoted_to_fp16)[name = string("op_1640_cast_fp16")]; int32 var_1642 = const()[name = string("op_1642"), val = int32(-2)]; bool var_1643_interleave_0 = const()[name = string("op_1643_interleave_0"), val = bool(false)]; tensor var_1643_cast_fp16 = concat(axis = var_1642, interleave = var_1643_interleave_0, values = (var_1640_cast_fp16, var_1638_cast_fp16_0))[name = string("op_1643_cast_fp16")]; tensor var_1644_cast_fp16 = mul(x = var_1643_cast_fp16, y = var_878_cast_fp16)[name = string("op_1644_cast_fp16")]; tensor query_states_15_cast_fp16 = add(x = var_1637_cast_fp16, y = var_1644_cast_fp16)[name = string("query_states_15_cast_fp16")]; tensor var_1650_cast_fp16 = mul(x = var_1626_cast_fp16, y = var_869_cast_fp16)[name = string("op_1650_cast_fp16")]; tensor var_1651_split_sizes_0 = const()[name = string("op_1651_split_sizes_0"), val = tensor([64, 64])]; int32 var_1651_axis_0 = const()[name = string("op_1651_axis_0"), val = int32(-2)]; tensor var_1651_cast_fp16_0, tensor var_1651_cast_fp16_1 = split(axis = var_1651_axis_0, split_sizes = var_1651_split_sizes_0, x = var_1626_cast_fp16)[name = string("op_1651_cast_fp16")]; fp16 const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1653_cast_fp16 = mul(x = var_1651_cast_fp16_1, y = const_23_promoted_to_fp16)[name = string("op_1653_cast_fp16")]; int32 var_1655 = const()[name = string("op_1655"), val = int32(-2)]; bool var_1656_interleave_0 = const()[name = string("op_1656_interleave_0"), val = bool(false)]; tensor var_1656_cast_fp16 = concat(axis = var_1655, interleave = var_1656_interleave_0, values = (var_1653_cast_fp16, var_1651_cast_fp16_0))[name = string("op_1656_cast_fp16")]; tensor var_1657_cast_fp16 = mul(x = var_1656_cast_fp16, y = var_878_cast_fp16)[name = string("op_1657_cast_fp16")]; tensor key_states_25_cast_fp16 = add(x = var_1650_cast_fp16, y = var_1657_cast_fp16)[name = string("key_states_25_cast_fp16")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; int32 concat_29_axis_0 = const()[name = string("concat_29_axis_0"), val = int32(0)]; bool concat_29_interleave_0 = const()[name = string("concat_29_interleave_0"), val = bool(false)]; tensor concat_29 = concat(axis = concat_29_axis_0, interleave = concat_29_interleave_0, values = (expand_dims_24, expand_dims_25, position_id, expand_dims_27))[name = string("concat_29")]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; tensor concat_30_values1_0 = const()[name = string("concat_30_values1_0"), val = tensor([0])]; tensor concat_30_values3_0 = const()[name = string("concat_30_values3_0"), val = tensor([0])]; int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_28, concat_30_values1_0, cache_position_end, concat_30_values3_0))[name = string("concat_30")]; tensor key_states_27_perm_0 = const()[name = string("key_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_3_stride_0 = const()[name = string("key_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_27_cast_fp16 = transpose(perm = key_states_27_perm_0, x = key_states_25_cast_fp16)[name = string("transpose_593")]; tensor key_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = key_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = key_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_3_squeeze_mask_0, stride = key_cache_internal_tensor_assign_3_stride_0, update = key_states_27_cast_fp16, x = coreml_update_state_338)[name = string("key_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_3_cast_fp16, input = key_cache)[name = string("coreml_update_state_340_write_state")]; tensor coreml_update_state_340 = read_state(input = key_cache)[name = string("coreml_update_state_340")]; tensor value_states_15_perm_0 = const()[name = string("value_states_15_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_3_stride_0 = const()[name = string("value_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_15_cast_fp16 = transpose(perm = value_states_15_perm_0, x = var_1633_cast_fp16)[name = string("transpose_592")]; tensor value_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = value_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = value_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_3_squeeze_mask_0, stride = value_cache_internal_tensor_assign_3_stride_0, update = value_states_15_cast_fp16, x = coreml_update_state_339)[name = string("value_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_3_cast_fp16, input = value_cache)[name = string("coreml_update_state_341_write_state")]; tensor coreml_update_state_341 = read_state(input = value_cache)[name = string("coreml_update_state_341")]; tensor var_1727_begin_0 = const()[name = string("op_1727_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1727_end_0 = const()[name = string("op_1727_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1727_end_mask_0 = const()[name = string("op_1727_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1727_cast_fp16 = slice_by_index(begin = var_1727_begin_0, end = var_1727_end_0, end_mask = var_1727_end_mask_0, x = coreml_update_state_340)[name = string("op_1727_cast_fp16")]; tensor tile_4 = const()[name = string("tile_4"), val = tensor([1, 1])]; int32 var_1730_axis_0 = const()[name = string("op_1730_axis_0"), val = int32(1)]; tensor var_1730_cast_fp16_0, tensor var_1730_cast_fp16_1 = split(axis = var_1730_axis_0, split_sizes = tile_4, x = var_1727_cast_fp16)[name = string("op_1730_cast_fp16")]; tensor var_1737_begin_0 = const()[name = string("op_1737_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1737_end_0 = const()[name = string("op_1737_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1737_end_mask_0 = const()[name = string("op_1737_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1737_cast_fp16 = slice_by_index(begin = var_1737_begin_0, end = var_1737_end_0, end_mask = var_1737_end_mask_0, x = coreml_update_state_341)[name = string("op_1737_cast_fp16")]; tensor tile_5 = const()[name = string("tile_5"), val = tensor([1, 1])]; int32 var_1740_axis_0 = const()[name = string("op_1740_axis_0"), val = int32(1)]; tensor var_1740_cast_fp16_0, tensor var_1740_cast_fp16_1 = split(axis = var_1740_axis_0, split_sizes = tile_5, x = var_1737_cast_fp16)[name = string("op_1740_cast_fp16")]; tensor var_1743_split_sizes_0 = const()[name = string("op_1743_split_sizes_0"), val = tensor([8, 8])]; int32 var_1743_axis_0 = const()[name = string("op_1743_axis_0"), val = int32(1)]; tensor var_1743_0, tensor var_1743_1 = split(axis = var_1743_axis_0, split_sizes = var_1743_split_sizes_0, x = query_states_15_cast_fp16)[name = string("op_1743")]; bool attn_weights_33_transpose_x_0 = const()[name = string("attn_weights_33_transpose_x_0"), val = bool(false)]; bool attn_weights_33_transpose_y_0 = const()[name = string("attn_weights_33_transpose_y_0"), val = bool(false)]; tensor attn_weights_33_cast_fp16 = matmul(transpose_x = attn_weights_33_transpose_x_0, transpose_y = attn_weights_33_transpose_y_0, x = var_1730_cast_fp16_0, y = var_1743_0)[name = string("attn_weights_33_cast_fp16")]; fp16 var_1746_to_fp16 = const()[name = string("op_1746_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_35_cast_fp16 = mul(x = attn_weights_33_cast_fp16, y = var_1746_to_fp16)[name = string("attn_weights_35_cast_fp16")]; tensor attn_weights_37_cast_fp16 = add(x = attn_weights_35_cast_fp16, y = attn_mask_1)[name = string("attn_weights_37_cast_fp16")]; int32 var_1750 = const()[name = string("op_1750"), val = int32(-2)]; tensor attn_weights_39_cast_fp16 = softmax(axis = var_1750, x = attn_weights_37_cast_fp16)[name = string("attn_weights_39_cast_fp16")]; bool var_1756_transpose_x_1 = const()[name = string("op_1756_transpose_x_1"), val = bool(true)]; bool var_1756_transpose_y_1 = const()[name = string("op_1756_transpose_y_1"), val = bool(false)]; tensor var_1756_cast_fp16 = matmul(transpose_x = var_1756_transpose_x_1, transpose_y = var_1756_transpose_y_1, x = attn_weights_39_cast_fp16, y = var_1740_cast_fp16_0)[name = string("op_1756_cast_fp16")]; bool attn_weights_41_transpose_x_0 = const()[name = string("attn_weights_41_transpose_x_0"), val = bool(false)]; bool attn_weights_41_transpose_y_0 = const()[name = string("attn_weights_41_transpose_y_0"), val = bool(false)]; tensor attn_weights_41_cast_fp16 = matmul(transpose_x = attn_weights_41_transpose_x_0, transpose_y = attn_weights_41_transpose_y_0, x = var_1730_cast_fp16_1, y = var_1743_1)[name = string("attn_weights_41_cast_fp16")]; fp16 var_1758_to_fp16 = const()[name = string("op_1758_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_43_cast_fp16 = mul(x = attn_weights_41_cast_fp16, y = var_1758_to_fp16)[name = string("attn_weights_43_cast_fp16")]; tensor attn_weights_45_cast_fp16 = add(x = attn_weights_43_cast_fp16, y = attn_mask_1)[name = string("attn_weights_45_cast_fp16")]; int32 var_1762 = const()[name = string("op_1762"), val = int32(-2)]; tensor attn_weights_47_cast_fp16 = softmax(axis = var_1762, x = attn_weights_45_cast_fp16)[name = string("attn_weights_47_cast_fp16")]; bool attn_output_17_transpose_x_1 = const()[name = string("attn_output_17_transpose_x_1"), val = bool(true)]; bool attn_output_17_transpose_y_1 = const()[name = string("attn_output_17_transpose_y_1"), val = bool(false)]; tensor attn_output_17_cast_fp16 = matmul(transpose_x = attn_output_17_transpose_x_1, transpose_y = attn_output_17_transpose_y_1, x = attn_weights_47_cast_fp16, y = var_1740_cast_fp16_1)[name = string("attn_output_17_cast_fp16")]; int32 var_1770 = const()[name = string("op_1770"), val = int32(1)]; bool attn_output_19_interleave_0 = const()[name = string("attn_output_19_interleave_0"), val = bool(false)]; tensor attn_output_19_cast_fp16 = concat(axis = var_1770, interleave = attn_output_19_interleave_0, values = (var_1756_cast_fp16, attn_output_17_cast_fp16))[name = string("attn_output_19_cast_fp16")]; tensor var_1774_perm_0 = const()[name = string("op_1774_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_35x = const()[name = string("concat_35x"), val = tensor([1, 2048, 1, -1])]; tensor var_1774_cast_fp16 = transpose(perm = var_1774_perm_0, x = attn_output_19_cast_fp16)[name = string("transpose_591")]; tensor attn_output_23_cast_fp16 = reshape(shape = concat_35x, x = var_1774_cast_fp16)[name = string("attn_output_23_cast_fp16")]; tensor layers_2_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1155169408)))]; tensor hidden_states_23_strides_0 = const()[name = string("hidden_states_23_strides_0"), val = tensor([1, 1])]; string hidden_states_23_pad_type_0 = const()[name = string("hidden_states_23_pad_type_0"), val = string("valid")]; tensor hidden_states_23_pad_0 = const()[name = string("hidden_states_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_23_dilations_0 = const()[name = string("hidden_states_23_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_23_groups_0 = const()[name = string("hidden_states_23_groups_0"), val = int32(1)]; tensor hidden_states_23_cast_fp16 = conv(dilations = hidden_states_23_dilations_0, groups = hidden_states_23_groups_0, pad = hidden_states_23_pad_0, pad_type = hidden_states_23_pad_type_0, strides = hidden_states_23_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16, x = attn_output_23_cast_fp16)[name = string("hidden_states_23_cast_fp16")]; tensor hidden_states_25_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = hidden_states_23_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1807_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_1807_cast_fp16")]; int32 var_1805 = const()[name = string("op_1805"), val = int32(1)]; bool doubled_21_interleave_0 = const()[name = string("doubled_21_interleave_0"), val = bool(false)]; tensor doubled_21_cast_fp16 = concat(axis = var_1805, interleave = doubled_21_interleave_0, values = (hidden_states_25_cast_fp16, var_1807_cast_fp16))[name = string("doubled_21_cast_fp16")]; tensor out_11_axes_0 = const()[name = string("out_11_axes_0"), val = tensor([1])]; tensor out_11_gamma_0_to_fp16 = const()[name = string("out_11_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163558080)))]; fp16 var_1817_to_fp16 = const()[name = string("op_1817_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_11_cast_fp16 = layer_norm(axes = out_11_axes_0, epsilon = var_1817_to_fp16, gamma = out_11_gamma_0_to_fp16, x = doubled_21_cast_fp16)[name = string("out_11_cast_fp16")]; tensor var_1828_split_sizes_0 = const()[name = string("op_1828_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1828_axis_0 = const()[name = string("op_1828_axis_0"), val = int32(1)]; tensor var_1828_cast_fp16_0, tensor var_1828_cast_fp16_1 = split(axis = var_1828_axis_0, split_sizes = var_1828_split_sizes_0, x = out_11_cast_fp16)[name = string("op_1828_cast_fp16")]; tensor layers_2_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163566336)))]; tensor input_5_strides_0 = const()[name = string("input_5_strides_0"), val = tensor([1, 1])]; string input_5_pad_type_0 = const()[name = string("input_5_pad_type_0"), val = string("valid")]; tensor input_5_pad_0 = const()[name = string("input_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_5_dilations_0 = const()[name = string("input_5_dilations_0"), val = tensor([1, 1])]; int32 input_5_groups_0 = const()[name = string("input_5_groups_0"), val = int32(1)]; tensor input_5_cast_fp16 = conv(dilations = input_5_dilations_0, groups = input_5_groups_0, pad = input_5_pad_0, pad_type = input_5_pad_type_0, strides = input_5_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16, x = var_1828_cast_fp16_0)[name = string("input_5_cast_fp16")]; tensor var_1845_cast_fp16 = silu(x = input_5_cast_fp16)[name = string("op_1845_cast_fp16")]; tensor var_1851_strides_0 = const()[name = string("op_1851_strides_0"), val = tensor([1, 1])]; string var_1851_pad_type_0 = const()[name = string("op_1851_pad_type_0"), val = string("valid")]; tensor var_1851_pad_0 = const()[name = string("op_1851_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1851_dilations_0 = const()[name = string("op_1851_dilations_0"), val = tensor([1, 1])]; int32 var_1851_groups_0 = const()[name = string("op_1851_groups_0"), val = int32(1)]; tensor var_1851_cast_fp16 = conv(dilations = var_1851_dilations_0, groups = var_1851_groups_0, pad = var_1851_pad_0, pad_type = var_1851_pad_type_0, strides = var_1851_strides_0, weight = layers_2_mlp_up_proj_weight_cast_fp16, x = var_1828_cast_fp16_0)[name = string("op_1851_cast_fp16")]; tensor x_29_cast_fp16 = mul(x = var_1845_cast_fp16, y = var_1851_cast_fp16)[name = string("x_29_cast_fp16")]; tensor layers_2_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1188732224)))]; tensor hidden_states_27_strides_0 = const()[name = string("hidden_states_27_strides_0"), val = tensor([1, 1])]; string hidden_states_27_pad_type_0 = const()[name = string("hidden_states_27_pad_type_0"), val = string("valid")]; tensor hidden_states_27_pad_0 = const()[name = string("hidden_states_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_27_dilations_0 = const()[name = string("hidden_states_27_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_27_groups_0 = const()[name = string("hidden_states_27_groups_0"), val = int32(1)]; tensor hidden_states_27_cast_fp16 = conv(dilations = hidden_states_27_dilations_0, groups = hidden_states_27_groups_0, pad = hidden_states_27_pad_0, pad_type = hidden_states_27_pad_type_0, strides = hidden_states_27_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16, x = x_29_cast_fp16)[name = string("hidden_states_27_cast_fp16")]; tensor hidden_states_29_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = hidden_states_27_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; fp16 const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1869_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_30_promoted_to_fp16)[name = string("op_1869_cast_fp16")]; int32 var_1867 = const()[name = string("op_1867"), val = int32(1)]; bool doubled_25_interleave_0 = const()[name = string("doubled_25_interleave_0"), val = bool(false)]; tensor doubled_25_cast_fp16 = concat(axis = var_1867, interleave = doubled_25_interleave_0, values = (hidden_states_29_cast_fp16, var_1869_cast_fp16))[name = string("doubled_25_cast_fp16")]; tensor out_13_axes_0 = const()[name = string("out_13_axes_0"), val = tensor([1])]; tensor out_13_gamma_0_to_fp16 = const()[name = string("out_13_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213898112)))]; fp16 var_1879_to_fp16 = const()[name = string("op_1879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_13_cast_fp16 = layer_norm(axes = out_13_axes_0, epsilon = var_1879_to_fp16, gamma = out_13_gamma_0_to_fp16, x = doubled_25_cast_fp16)[name = string("out_13_cast_fp16")]; tensor var_1890_split_sizes_0 = const()[name = string("op_1890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1890_axis_0 = const()[name = string("op_1890_axis_0"), val = int32(1)]; tensor var_1890_cast_fp16_0, tensor var_1890_cast_fp16_1 = split(axis = var_1890_axis_0, split_sizes = var_1890_split_sizes_0, x = out_13_cast_fp16)[name = string("op_1890_cast_fp16")]; tensor layers_3_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213906368)))]; tensor query_states_19_strides_0 = const()[name = string("query_states_19_strides_0"), val = tensor([1, 1])]; string query_states_19_pad_type_0 = const()[name = string("query_states_19_pad_type_0"), val = string("valid")]; tensor query_states_19_pad_0 = const()[name = string("query_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_19_dilations_0 = const()[name = string("query_states_19_dilations_0"), val = tensor([1, 1])]; int32 query_states_19_groups_0 = const()[name = string("query_states_19_groups_0"), val = int32(1)]; tensor query_states_19_cast_fp16 = conv(dilations = query_states_19_dilations_0, groups = query_states_19_groups_0, pad = query_states_19_pad_0, pad_type = query_states_19_pad_type_0, strides = query_states_19_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("query_states_19_cast_fp16")]; tensor layers_3_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1222295040)))]; tensor key_states_31_strides_0 = const()[name = string("key_states_31_strides_0"), val = tensor([1, 1])]; string key_states_31_pad_type_0 = const()[name = string("key_states_31_pad_type_0"), val = string("valid")]; tensor key_states_31_pad_0 = const()[name = string("key_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_31_dilations_0 = const()[name = string("key_states_31_dilations_0"), val = tensor([1, 1])]; int32 key_states_31_groups_0 = const()[name = string("key_states_31_groups_0"), val = int32(1)]; tensor key_states_31_cast_fp16 = conv(dilations = key_states_31_dilations_0, groups = key_states_31_groups_0, pad = key_states_31_pad_0, pad_type = key_states_31_pad_type_0, strides = key_states_31_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("key_states_31_cast_fp16")]; tensor value_states_19_strides_0 = const()[name = string("value_states_19_strides_0"), val = tensor([1, 1])]; string value_states_19_pad_type_0 = const()[name = string("value_states_19_pad_type_0"), val = string("valid")]; tensor value_states_19_pad_0 = const()[name = string("value_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_19_dilations_0 = const()[name = string("value_states_19_dilations_0"), val = tensor([1, 1])]; int32 value_states_19_groups_0 = const()[name = string("value_states_19_groups_0"), val = int32(1)]; tensor value_states_19_cast_fp16 = conv(dilations = value_states_19_dilations_0, groups = value_states_19_groups_0, pad = value_states_19_pad_0, pad_type = value_states_19_pad_type_0, strides = value_states_19_strides_0, weight = layers_3_self_attn_v_proj_weight_cast_fp16, x = var_1890_cast_fp16_0)[name = string("value_states_19_cast_fp16")]; tensor concat_36x = const()[name = string("concat_36x"), val = tensor([1, 16, 128, -1])]; tensor x_31_cast_fp16 = reshape(shape = concat_36x, x = query_states_19_cast_fp16)[name = string("x_31_cast_fp16")]; tensor concat_37x = const()[name = string("concat_37x"), val = tensor([1, 2, 128, -1])]; tensor var_1947_cast_fp16 = reshape(shape = concat_37x, x = key_states_31_cast_fp16)[name = string("op_1947_cast_fp16")]; tensor concat_38x = const()[name = string("concat_38x"), val = tensor([1, 2, 128, -1])]; tensor var_1954_cast_fp16 = reshape(shape = concat_38x, x = value_states_19_cast_fp16)[name = string("op_1954_cast_fp16")]; tensor var_1958_cast_fp16 = mul(x = x_31_cast_fp16, y = var_869_cast_fp16)[name = string("op_1958_cast_fp16")]; tensor var_1959_split_sizes_0 = const()[name = string("op_1959_split_sizes_0"), val = tensor([64, 64])]; int32 var_1959_axis_0 = const()[name = string("op_1959_axis_0"), val = int32(-2)]; tensor var_1959_cast_fp16_0, tensor var_1959_cast_fp16_1 = split(axis = var_1959_axis_0, split_sizes = var_1959_split_sizes_0, x = x_31_cast_fp16)[name = string("op_1959_cast_fp16")]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1961_cast_fp16 = mul(x = var_1959_cast_fp16_1, y = const_32_promoted_to_fp16)[name = string("op_1961_cast_fp16")]; int32 var_1963 = const()[name = string("op_1963"), val = int32(-2)]; bool var_1964_interleave_0 = const()[name = string("op_1964_interleave_0"), val = bool(false)]; tensor var_1964_cast_fp16 = concat(axis = var_1963, interleave = var_1964_interleave_0, values = (var_1961_cast_fp16, var_1959_cast_fp16_0))[name = string("op_1964_cast_fp16")]; tensor var_1965_cast_fp16 = mul(x = var_1964_cast_fp16, y = var_878_cast_fp16)[name = string("op_1965_cast_fp16")]; tensor query_states_21_cast_fp16 = add(x = var_1958_cast_fp16, y = var_1965_cast_fp16)[name = string("query_states_21_cast_fp16")]; tensor var_1971_cast_fp16 = mul(x = var_1947_cast_fp16, y = var_869_cast_fp16)[name = string("op_1971_cast_fp16")]; tensor var_1972_split_sizes_0 = const()[name = string("op_1972_split_sizes_0"), val = tensor([64, 64])]; int32 var_1972_axis_0 = const()[name = string("op_1972_axis_0"), val = int32(-2)]; tensor var_1972_cast_fp16_0, tensor var_1972_cast_fp16_1 = split(axis = var_1972_axis_0, split_sizes = var_1972_split_sizes_0, x = var_1947_cast_fp16)[name = string("op_1972_cast_fp16")]; fp16 const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1974_cast_fp16 = mul(x = var_1972_cast_fp16_1, y = const_33_promoted_to_fp16)[name = string("op_1974_cast_fp16")]; int32 var_1976 = const()[name = string("op_1976"), val = int32(-2)]; bool var_1977_interleave_0 = const()[name = string("op_1977_interleave_0"), val = bool(false)]; tensor var_1977_cast_fp16 = concat(axis = var_1976, interleave = var_1977_interleave_0, values = (var_1974_cast_fp16, var_1972_cast_fp16_0))[name = string("op_1977_cast_fp16")]; tensor var_1978_cast_fp16 = mul(x = var_1977_cast_fp16, y = var_878_cast_fp16)[name = string("op_1978_cast_fp16")]; tensor key_states_35_cast_fp16 = add(x = var_1971_cast_fp16, y = var_1978_cast_fp16)[name = string("key_states_35_cast_fp16")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; int32 concat_41_axis_0 = const()[name = string("concat_41_axis_0"), val = int32(0)]; bool concat_41_interleave_0 = const()[name = string("concat_41_interleave_0"), val = bool(false)]; tensor concat_41 = concat(axis = concat_41_axis_0, interleave = concat_41_interleave_0, values = (expand_dims_36, expand_dims_37, position_id, expand_dims_39))[name = string("concat_41")]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; tensor concat_42_values1_0 = const()[name = string("concat_42_values1_0"), val = tensor([0])]; tensor concat_42_values3_0 = const()[name = string("concat_42_values3_0"), val = tensor([0])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_40, concat_42_values1_0, cache_position_end, concat_42_values3_0))[name = string("concat_42")]; tensor key_states_37_perm_0 = const()[name = string("key_states_37_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_4_stride_0 = const()[name = string("key_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_37_cast_fp16 = transpose(perm = key_states_37_perm_0, x = key_states_35_cast_fp16)[name = string("transpose_590")]; tensor key_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = key_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = key_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_4_squeeze_mask_0, stride = key_cache_internal_tensor_assign_4_stride_0, update = key_states_37_cast_fp16, x = coreml_update_state_340)[name = string("key_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_4_cast_fp16, input = key_cache)[name = string("coreml_update_state_342_write_state")]; tensor coreml_update_state_342 = read_state(input = key_cache)[name = string("coreml_update_state_342")]; tensor value_states_21_perm_0 = const()[name = string("value_states_21_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_4_stride_0 = const()[name = string("value_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_21_cast_fp16 = transpose(perm = value_states_21_perm_0, x = var_1954_cast_fp16)[name = string("transpose_589")]; tensor value_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = value_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = value_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_4_squeeze_mask_0, stride = value_cache_internal_tensor_assign_4_stride_0, update = value_states_21_cast_fp16, x = coreml_update_state_341)[name = string("value_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_4_cast_fp16, input = value_cache)[name = string("coreml_update_state_343_write_state")]; tensor coreml_update_state_343 = read_state(input = value_cache)[name = string("coreml_update_state_343")]; tensor var_2048_begin_0 = const()[name = string("op_2048_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2048_end_0 = const()[name = string("op_2048_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2048_end_mask_0 = const()[name = string("op_2048_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2048_cast_fp16 = slice_by_index(begin = var_2048_begin_0, end = var_2048_end_0, end_mask = var_2048_end_mask_0, x = coreml_update_state_342)[name = string("op_2048_cast_fp16")]; tensor tile_6 = const()[name = string("tile_6"), val = tensor([1, 1])]; int32 var_2051_axis_0 = const()[name = string("op_2051_axis_0"), val = int32(1)]; tensor var_2051_cast_fp16_0, tensor var_2051_cast_fp16_1 = split(axis = var_2051_axis_0, split_sizes = tile_6, x = var_2048_cast_fp16)[name = string("op_2051_cast_fp16")]; tensor var_2058_begin_0 = const()[name = string("op_2058_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2058_end_0 = const()[name = string("op_2058_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2058_end_mask_0 = const()[name = string("op_2058_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2058_cast_fp16 = slice_by_index(begin = var_2058_begin_0, end = var_2058_end_0, end_mask = var_2058_end_mask_0, x = coreml_update_state_343)[name = string("op_2058_cast_fp16")]; tensor tile_7 = const()[name = string("tile_7"), val = tensor([1, 1])]; int32 var_2061_axis_0 = const()[name = string("op_2061_axis_0"), val = int32(1)]; tensor var_2061_cast_fp16_0, tensor var_2061_cast_fp16_1 = split(axis = var_2061_axis_0, split_sizes = tile_7, x = var_2058_cast_fp16)[name = string("op_2061_cast_fp16")]; tensor var_2064_split_sizes_0 = const()[name = string("op_2064_split_sizes_0"), val = tensor([8, 8])]; int32 var_2064_axis_0 = const()[name = string("op_2064_axis_0"), val = int32(1)]; tensor var_2064_0, tensor var_2064_1 = split(axis = var_2064_axis_0, split_sizes = var_2064_split_sizes_0, x = query_states_21_cast_fp16)[name = string("op_2064")]; bool attn_weights_49_transpose_x_0 = const()[name = string("attn_weights_49_transpose_x_0"), val = bool(false)]; bool attn_weights_49_transpose_y_0 = const()[name = string("attn_weights_49_transpose_y_0"), val = bool(false)]; tensor attn_weights_49_cast_fp16 = matmul(transpose_x = attn_weights_49_transpose_x_0, transpose_y = attn_weights_49_transpose_y_0, x = var_2051_cast_fp16_0, y = var_2064_0)[name = string("attn_weights_49_cast_fp16")]; fp16 var_2067_to_fp16 = const()[name = string("op_2067_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_51_cast_fp16 = mul(x = attn_weights_49_cast_fp16, y = var_2067_to_fp16)[name = string("attn_weights_51_cast_fp16")]; tensor attn_weights_53_cast_fp16 = add(x = attn_weights_51_cast_fp16, y = attn_mask_1)[name = string("attn_weights_53_cast_fp16")]; int32 var_2071 = const()[name = string("op_2071"), val = int32(-2)]; tensor attn_weights_55_cast_fp16 = softmax(axis = var_2071, x = attn_weights_53_cast_fp16)[name = string("attn_weights_55_cast_fp16")]; bool var_2077_transpose_x_1 = const()[name = string("op_2077_transpose_x_1"), val = bool(true)]; bool var_2077_transpose_y_1 = const()[name = string("op_2077_transpose_y_1"), val = bool(false)]; tensor var_2077_cast_fp16 = matmul(transpose_x = var_2077_transpose_x_1, transpose_y = var_2077_transpose_y_1, x = attn_weights_55_cast_fp16, y = var_2061_cast_fp16_0)[name = string("op_2077_cast_fp16")]; bool attn_weights_57_transpose_x_0 = const()[name = string("attn_weights_57_transpose_x_0"), val = bool(false)]; bool attn_weights_57_transpose_y_0 = const()[name = string("attn_weights_57_transpose_y_0"), val = bool(false)]; tensor attn_weights_57_cast_fp16 = matmul(transpose_x = attn_weights_57_transpose_x_0, transpose_y = attn_weights_57_transpose_y_0, x = var_2051_cast_fp16_1, y = var_2064_1)[name = string("attn_weights_57_cast_fp16")]; fp16 var_2079_to_fp16 = const()[name = string("op_2079_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_59_cast_fp16 = mul(x = attn_weights_57_cast_fp16, y = var_2079_to_fp16)[name = string("attn_weights_59_cast_fp16")]; tensor attn_weights_61_cast_fp16 = add(x = attn_weights_59_cast_fp16, y = attn_mask_1)[name = string("attn_weights_61_cast_fp16")]; int32 var_2083 = const()[name = string("op_2083"), val = int32(-2)]; tensor attn_weights_63_cast_fp16 = softmax(axis = var_2083, x = attn_weights_61_cast_fp16)[name = string("attn_weights_63_cast_fp16")]; bool attn_output_25_transpose_x_1 = const()[name = string("attn_output_25_transpose_x_1"), val = bool(true)]; bool attn_output_25_transpose_y_1 = const()[name = string("attn_output_25_transpose_y_1"), val = bool(false)]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_1, transpose_y = attn_output_25_transpose_y_1, x = attn_weights_63_cast_fp16, y = var_2061_cast_fp16_1)[name = string("attn_output_25_cast_fp16")]; int32 var_2091 = const()[name = string("op_2091"), val = int32(1)]; bool attn_output_27_interleave_0 = const()[name = string("attn_output_27_interleave_0"), val = bool(false)]; tensor attn_output_27_cast_fp16 = concat(axis = var_2091, interleave = attn_output_27_interleave_0, values = (var_2077_cast_fp16, attn_output_25_cast_fp16))[name = string("attn_output_27_cast_fp16")]; tensor var_2095_perm_0 = const()[name = string("op_2095_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_47x = const()[name = string("concat_47x"), val = tensor([1, 2048, 1, -1])]; tensor var_2095_cast_fp16 = transpose(perm = var_2095_perm_0, x = attn_output_27_cast_fp16)[name = string("transpose_588")]; tensor attn_output_31_cast_fp16 = reshape(shape = concat_47x, x = var_2095_cast_fp16)[name = string("attn_output_31_cast_fp16")]; tensor hidden_states_33_strides_0 = const()[name = string("hidden_states_33_strides_0"), val = tensor([1, 1])]; string hidden_states_33_pad_type_0 = const()[name = string("hidden_states_33_pad_type_0"), val = string("valid")]; tensor hidden_states_33_pad_0 = const()[name = string("hidden_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_33_dilations_0 = const()[name = string("hidden_states_33_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_33_groups_0 = const()[name = string("hidden_states_33_groups_0"), val = int32(1)]; tensor hidden_states_33_cast_fp16 = conv(dilations = hidden_states_33_dilations_0, groups = hidden_states_33_groups_0, pad = hidden_states_33_pad_0, pad_type = hidden_states_33_pad_type_0, strides = hidden_states_33_strides_0, weight = layers_3_self_attn_o_proj_weight_cast_fp16, x = attn_output_31_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; tensor hidden_states_35_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = hidden_states_33_cast_fp16)[name = string("hidden_states_35_cast_fp16")]; fp16 const_38_promoted_to_fp16 = const()[name = string("const_38_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2128_cast_fp16 = mul(x = hidden_states_35_cast_fp16, y = const_38_promoted_to_fp16)[name = string("op_2128_cast_fp16")]; int32 var_2126 = const()[name = string("op_2126"), val = int32(1)]; bool doubled_29_interleave_0 = const()[name = string("doubled_29_interleave_0"), val = bool(false)]; tensor doubled_29_cast_fp16 = concat(axis = var_2126, interleave = doubled_29_interleave_0, values = (hidden_states_35_cast_fp16, var_2128_cast_fp16))[name = string("doubled_29_cast_fp16")]; tensor out_15_axes_0 = const()[name = string("out_15_axes_0"), val = tensor([1])]; tensor out_15_gamma_0_to_fp16 = const()[name = string("out_15_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223343680)))]; fp16 var_2138_to_fp16 = const()[name = string("op_2138_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_15_cast_fp16 = layer_norm(axes = out_15_axes_0, epsilon = var_2138_to_fp16, gamma = out_15_gamma_0_to_fp16, x = doubled_29_cast_fp16)[name = string("out_15_cast_fp16")]; tensor var_2149_split_sizes_0 = const()[name = string("op_2149_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2149_axis_0 = const()[name = string("op_2149_axis_0"), val = int32(1)]; tensor var_2149_cast_fp16_0, tensor var_2149_cast_fp16_1 = split(axis = var_2149_axis_0, split_sizes = var_2149_split_sizes_0, x = out_15_cast_fp16)[name = string("op_2149_cast_fp16")]; tensor layers_3_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223351936)))]; tensor input_7_strides_0 = const()[name = string("input_7_strides_0"), val = tensor([1, 1])]; string input_7_pad_type_0 = const()[name = string("input_7_pad_type_0"), val = string("valid")]; tensor input_7_pad_0 = const()[name = string("input_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_7_dilations_0 = const()[name = string("input_7_dilations_0"), val = tensor([1, 1])]; int32 input_7_groups_0 = const()[name = string("input_7_groups_0"), val = int32(1)]; tensor input_7_cast_fp16 = conv(dilations = input_7_dilations_0, groups = input_7_groups_0, pad = input_7_pad_0, pad_type = input_7_pad_type_0, strides = input_7_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("input_7_cast_fp16")]; tensor var_2166_cast_fp16 = silu(x = input_7_cast_fp16)[name = string("op_2166_cast_fp16")]; tensor layers_3_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1248517824)))]; tensor var_2172_strides_0 = const()[name = string("op_2172_strides_0"), val = tensor([1, 1])]; string var_2172_pad_type_0 = const()[name = string("op_2172_pad_type_0"), val = string("valid")]; tensor var_2172_pad_0 = const()[name = string("op_2172_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2172_dilations_0 = const()[name = string("op_2172_dilations_0"), val = tensor([1, 1])]; int32 var_2172_groups_0 = const()[name = string("op_2172_groups_0"), val = int32(1)]; tensor var_2172_cast_fp16 = conv(dilations = var_2172_dilations_0, groups = var_2172_groups_0, pad = var_2172_pad_0, pad_type = var_2172_pad_type_0, strides = var_2172_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("op_2172_cast_fp16")]; tensor x_39_cast_fp16 = mul(x = var_2166_cast_fp16, y = var_2172_cast_fp16)[name = string("x_39_cast_fp16")]; tensor hidden_states_37_strides_0 = const()[name = string("hidden_states_37_strides_0"), val = tensor([1, 1])]; string hidden_states_37_pad_type_0 = const()[name = string("hidden_states_37_pad_type_0"), val = string("valid")]; tensor hidden_states_37_pad_0 = const()[name = string("hidden_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_37_dilations_0 = const()[name = string("hidden_states_37_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_37_groups_0 = const()[name = string("hidden_states_37_groups_0"), val = int32(1)]; tensor hidden_states_37_cast_fp16 = conv(dilations = hidden_states_37_dilations_0, groups = hidden_states_37_groups_0, pad = hidden_states_37_pad_0, pad_type = hidden_states_37_pad_type_0, strides = hidden_states_37_strides_0, weight = layers_3_mlp_down_proj_weight_cast_fp16, x = x_39_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; tensor hidden_states_39_cast_fp16 = add(x = hidden_states_35_cast_fp16, y = hidden_states_37_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; fp16 const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2190_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_40_promoted_to_fp16)[name = string("op_2190_cast_fp16")]; int32 var_2188 = const()[name = string("op_2188"), val = int32(1)]; bool doubled_33_interleave_0 = const()[name = string("doubled_33_interleave_0"), val = bool(false)]; tensor doubled_33_cast_fp16 = concat(axis = var_2188, interleave = doubled_33_interleave_0, values = (hidden_states_39_cast_fp16, var_2190_cast_fp16))[name = string("doubled_33_cast_fp16")]; tensor out_17_axes_0 = const()[name = string("out_17_axes_0"), val = tensor([1])]; tensor out_17_gamma_0_to_fp16 = const()[name = string("out_17_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273683712)))]; fp16 var_2200_to_fp16 = const()[name = string("op_2200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_17_cast_fp16 = layer_norm(axes = out_17_axes_0, epsilon = var_2200_to_fp16, gamma = out_17_gamma_0_to_fp16, x = doubled_33_cast_fp16)[name = string("out_17_cast_fp16")]; tensor var_2211_split_sizes_0 = const()[name = string("op_2211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2211_axis_0 = const()[name = string("op_2211_axis_0"), val = int32(1)]; tensor var_2211_cast_fp16_0, tensor var_2211_cast_fp16_1 = split(axis = var_2211_axis_0, split_sizes = var_2211_split_sizes_0, x = out_17_cast_fp16)[name = string("op_2211_cast_fp16")]; tensor layers_4_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273691968)))]; tensor query_states_25_strides_0 = const()[name = string("query_states_25_strides_0"), val = tensor([1, 1])]; string query_states_25_pad_type_0 = const()[name = string("query_states_25_pad_type_0"), val = string("valid")]; tensor query_states_25_pad_0 = const()[name = string("query_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_25_dilations_0 = const()[name = string("query_states_25_dilations_0"), val = tensor([1, 1])]; int32 query_states_25_groups_0 = const()[name = string("query_states_25_groups_0"), val = int32(1)]; tensor query_states_25_cast_fp16 = conv(dilations = query_states_25_dilations_0, groups = query_states_25_groups_0, pad = query_states_25_pad_0, pad_type = query_states_25_pad_type_0, strides = query_states_25_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("query_states_25_cast_fp16")]; tensor layers_4_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1282080640)))]; tensor key_states_41_strides_0 = const()[name = string("key_states_41_strides_0"), val = tensor([1, 1])]; string key_states_41_pad_type_0 = const()[name = string("key_states_41_pad_type_0"), val = string("valid")]; tensor key_states_41_pad_0 = const()[name = string("key_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_41_dilations_0 = const()[name = string("key_states_41_dilations_0"), val = tensor([1, 1])]; int32 key_states_41_groups_0 = const()[name = string("key_states_41_groups_0"), val = int32(1)]; tensor key_states_41_cast_fp16 = conv(dilations = key_states_41_dilations_0, groups = key_states_41_groups_0, pad = key_states_41_pad_0, pad_type = key_states_41_pad_type_0, strides = key_states_41_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("key_states_41_cast_fp16")]; tensor value_states_25_strides_0 = const()[name = string("value_states_25_strides_0"), val = tensor([1, 1])]; string value_states_25_pad_type_0 = const()[name = string("value_states_25_pad_type_0"), val = string("valid")]; tensor value_states_25_pad_0 = const()[name = string("value_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_25_dilations_0 = const()[name = string("value_states_25_dilations_0"), val = tensor([1, 1])]; int32 value_states_25_groups_0 = const()[name = string("value_states_25_groups_0"), val = int32(1)]; tensor value_states_25_cast_fp16 = conv(dilations = value_states_25_dilations_0, groups = value_states_25_groups_0, pad = value_states_25_pad_0, pad_type = value_states_25_pad_type_0, strides = value_states_25_strides_0, weight = layers_4_self_attn_v_proj_weight_cast_fp16, x = var_2211_cast_fp16_0)[name = string("value_states_25_cast_fp16")]; tensor concat_48x = const()[name = string("concat_48x"), val = tensor([1, 16, 128, -1])]; tensor x_41_cast_fp16 = reshape(shape = concat_48x, x = query_states_25_cast_fp16)[name = string("x_41_cast_fp16")]; tensor concat_49x = const()[name = string("concat_49x"), val = tensor([1, 2, 128, -1])]; tensor var_2268_cast_fp16 = reshape(shape = concat_49x, x = key_states_41_cast_fp16)[name = string("op_2268_cast_fp16")]; tensor concat_50x = const()[name = string("concat_50x"), val = tensor([1, 2, 128, -1])]; tensor var_2275_cast_fp16 = reshape(shape = concat_50x, x = value_states_25_cast_fp16)[name = string("op_2275_cast_fp16")]; tensor var_2279_cast_fp16 = mul(x = x_41_cast_fp16, y = var_869_cast_fp16)[name = string("op_2279_cast_fp16")]; tensor var_2280_split_sizes_0 = const()[name = string("op_2280_split_sizes_0"), val = tensor([64, 64])]; int32 var_2280_axis_0 = const()[name = string("op_2280_axis_0"), val = int32(-2)]; tensor var_2280_cast_fp16_0, tensor var_2280_cast_fp16_1 = split(axis = var_2280_axis_0, split_sizes = var_2280_split_sizes_0, x = x_41_cast_fp16)[name = string("op_2280_cast_fp16")]; fp16 const_42_promoted_to_fp16 = const()[name = string("const_42_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2282_cast_fp16 = mul(x = var_2280_cast_fp16_1, y = const_42_promoted_to_fp16)[name = string("op_2282_cast_fp16")]; int32 var_2284 = const()[name = string("op_2284"), val = int32(-2)]; bool var_2285_interleave_0 = const()[name = string("op_2285_interleave_0"), val = bool(false)]; tensor var_2285_cast_fp16 = concat(axis = var_2284, interleave = var_2285_interleave_0, values = (var_2282_cast_fp16, var_2280_cast_fp16_0))[name = string("op_2285_cast_fp16")]; tensor var_2286_cast_fp16 = mul(x = var_2285_cast_fp16, y = var_878_cast_fp16)[name = string("op_2286_cast_fp16")]; tensor query_states_27_cast_fp16 = add(x = var_2279_cast_fp16, y = var_2286_cast_fp16)[name = string("query_states_27_cast_fp16")]; tensor var_2292_cast_fp16 = mul(x = var_2268_cast_fp16, y = var_869_cast_fp16)[name = string("op_2292_cast_fp16")]; tensor var_2293_split_sizes_0 = const()[name = string("op_2293_split_sizes_0"), val = tensor([64, 64])]; int32 var_2293_axis_0 = const()[name = string("op_2293_axis_0"), val = int32(-2)]; tensor var_2293_cast_fp16_0, tensor var_2293_cast_fp16_1 = split(axis = var_2293_axis_0, split_sizes = var_2293_split_sizes_0, x = var_2268_cast_fp16)[name = string("op_2293_cast_fp16")]; fp16 const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2295_cast_fp16 = mul(x = var_2293_cast_fp16_1, y = const_43_promoted_to_fp16)[name = string("op_2295_cast_fp16")]; int32 var_2297 = const()[name = string("op_2297"), val = int32(-2)]; bool var_2298_interleave_0 = const()[name = string("op_2298_interleave_0"), val = bool(false)]; tensor var_2298_cast_fp16 = concat(axis = var_2297, interleave = var_2298_interleave_0, values = (var_2295_cast_fp16, var_2293_cast_fp16_0))[name = string("op_2298_cast_fp16")]; tensor var_2299_cast_fp16 = mul(x = var_2298_cast_fp16, y = var_878_cast_fp16)[name = string("op_2299_cast_fp16")]; tensor key_states_45_cast_fp16 = add(x = var_2292_cast_fp16, y = var_2299_cast_fp16)[name = string("key_states_45_cast_fp16")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; int32 concat_53_axis_0 = const()[name = string("concat_53_axis_0"), val = int32(0)]; bool concat_53_interleave_0 = const()[name = string("concat_53_interleave_0"), val = bool(false)]; tensor concat_53 = concat(axis = concat_53_axis_0, interleave = concat_53_interleave_0, values = (expand_dims_48, expand_dims_49, position_id, expand_dims_51))[name = string("concat_53")]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; tensor concat_54_values1_0 = const()[name = string("concat_54_values1_0"), val = tensor([0])]; tensor concat_54_values3_0 = const()[name = string("concat_54_values3_0"), val = tensor([0])]; int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_52, concat_54_values1_0, cache_position_end, concat_54_values3_0))[name = string("concat_54")]; tensor key_states_47_perm_0 = const()[name = string("key_states_47_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_5_stride_0 = const()[name = string("key_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_47_cast_fp16 = transpose(perm = key_states_47_perm_0, x = key_states_45_cast_fp16)[name = string("transpose_587")]; tensor key_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = key_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = key_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_5_squeeze_mask_0, stride = key_cache_internal_tensor_assign_5_stride_0, update = key_states_47_cast_fp16, x = coreml_update_state_342)[name = string("key_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_5_cast_fp16, input = key_cache)[name = string("coreml_update_state_344_write_state")]; tensor coreml_update_state_344 = read_state(input = key_cache)[name = string("coreml_update_state_344")]; tensor value_states_27_perm_0 = const()[name = string("value_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_5_stride_0 = const()[name = string("value_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_27_cast_fp16 = transpose(perm = value_states_27_perm_0, x = var_2275_cast_fp16)[name = string("transpose_586")]; tensor value_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = value_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = value_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_5_squeeze_mask_0, stride = value_cache_internal_tensor_assign_5_stride_0, update = value_states_27_cast_fp16, x = coreml_update_state_343)[name = string("value_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_5_cast_fp16, input = value_cache)[name = string("coreml_update_state_345_write_state")]; tensor coreml_update_state_345 = read_state(input = value_cache)[name = string("coreml_update_state_345")]; tensor var_2369_begin_0 = const()[name = string("op_2369_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2369_end_0 = const()[name = string("op_2369_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2369_end_mask_0 = const()[name = string("op_2369_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2369_cast_fp16 = slice_by_index(begin = var_2369_begin_0, end = var_2369_end_0, end_mask = var_2369_end_mask_0, x = coreml_update_state_344)[name = string("op_2369_cast_fp16")]; tensor tile_8 = const()[name = string("tile_8"), val = tensor([1, 1])]; int32 var_2372_axis_0 = const()[name = string("op_2372_axis_0"), val = int32(1)]; tensor var_2372_cast_fp16_0, tensor var_2372_cast_fp16_1 = split(axis = var_2372_axis_0, split_sizes = tile_8, x = var_2369_cast_fp16)[name = string("op_2372_cast_fp16")]; tensor var_2379_begin_0 = const()[name = string("op_2379_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2379_end_0 = const()[name = string("op_2379_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2379_end_mask_0 = const()[name = string("op_2379_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2379_cast_fp16 = slice_by_index(begin = var_2379_begin_0, end = var_2379_end_0, end_mask = var_2379_end_mask_0, x = coreml_update_state_345)[name = string("op_2379_cast_fp16")]; tensor tile_9 = const()[name = string("tile_9"), val = tensor([1, 1])]; int32 var_2382_axis_0 = const()[name = string("op_2382_axis_0"), val = int32(1)]; tensor var_2382_cast_fp16_0, tensor var_2382_cast_fp16_1 = split(axis = var_2382_axis_0, split_sizes = tile_9, x = var_2379_cast_fp16)[name = string("op_2382_cast_fp16")]; tensor var_2385_split_sizes_0 = const()[name = string("op_2385_split_sizes_0"), val = tensor([8, 8])]; int32 var_2385_axis_0 = const()[name = string("op_2385_axis_0"), val = int32(1)]; tensor var_2385_0, tensor var_2385_1 = split(axis = var_2385_axis_0, split_sizes = var_2385_split_sizes_0, x = query_states_27_cast_fp16)[name = string("op_2385")]; bool attn_weights_65_transpose_x_0 = const()[name = string("attn_weights_65_transpose_x_0"), val = bool(false)]; bool attn_weights_65_transpose_y_0 = const()[name = string("attn_weights_65_transpose_y_0"), val = bool(false)]; tensor attn_weights_65_cast_fp16 = matmul(transpose_x = attn_weights_65_transpose_x_0, transpose_y = attn_weights_65_transpose_y_0, x = var_2372_cast_fp16_0, y = var_2385_0)[name = string("attn_weights_65_cast_fp16")]; fp16 var_2388_to_fp16 = const()[name = string("op_2388_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_67_cast_fp16 = mul(x = attn_weights_65_cast_fp16, y = var_2388_to_fp16)[name = string("attn_weights_67_cast_fp16")]; tensor attn_weights_69_cast_fp16 = add(x = attn_weights_67_cast_fp16, y = attn_mask_1)[name = string("attn_weights_69_cast_fp16")]; int32 var_2392 = const()[name = string("op_2392"), val = int32(-2)]; tensor attn_weights_71_cast_fp16 = softmax(axis = var_2392, x = attn_weights_69_cast_fp16)[name = string("attn_weights_71_cast_fp16")]; bool var_2398_transpose_x_1 = const()[name = string("op_2398_transpose_x_1"), val = bool(true)]; bool var_2398_transpose_y_1 = const()[name = string("op_2398_transpose_y_1"), val = bool(false)]; tensor var_2398_cast_fp16 = matmul(transpose_x = var_2398_transpose_x_1, transpose_y = var_2398_transpose_y_1, x = attn_weights_71_cast_fp16, y = var_2382_cast_fp16_0)[name = string("op_2398_cast_fp16")]; bool attn_weights_73_transpose_x_0 = const()[name = string("attn_weights_73_transpose_x_0"), val = bool(false)]; bool attn_weights_73_transpose_y_0 = const()[name = string("attn_weights_73_transpose_y_0"), val = bool(false)]; tensor attn_weights_73_cast_fp16 = matmul(transpose_x = attn_weights_73_transpose_x_0, transpose_y = attn_weights_73_transpose_y_0, x = var_2372_cast_fp16_1, y = var_2385_1)[name = string("attn_weights_73_cast_fp16")]; fp16 var_2400_to_fp16 = const()[name = string("op_2400_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_75_cast_fp16 = mul(x = attn_weights_73_cast_fp16, y = var_2400_to_fp16)[name = string("attn_weights_75_cast_fp16")]; tensor attn_weights_77_cast_fp16 = add(x = attn_weights_75_cast_fp16, y = attn_mask_1)[name = string("attn_weights_77_cast_fp16")]; int32 var_2404 = const()[name = string("op_2404"), val = int32(-2)]; tensor attn_weights_79_cast_fp16 = softmax(axis = var_2404, x = attn_weights_77_cast_fp16)[name = string("attn_weights_79_cast_fp16")]; bool attn_output_33_transpose_x_1 = const()[name = string("attn_output_33_transpose_x_1"), val = bool(true)]; bool attn_output_33_transpose_y_1 = const()[name = string("attn_output_33_transpose_y_1"), val = bool(false)]; tensor attn_output_33_cast_fp16 = matmul(transpose_x = attn_output_33_transpose_x_1, transpose_y = attn_output_33_transpose_y_1, x = attn_weights_79_cast_fp16, y = var_2382_cast_fp16_1)[name = string("attn_output_33_cast_fp16")]; int32 var_2412 = const()[name = string("op_2412"), val = int32(1)]; bool attn_output_35_interleave_0 = const()[name = string("attn_output_35_interleave_0"), val = bool(false)]; tensor attn_output_35_cast_fp16 = concat(axis = var_2412, interleave = attn_output_35_interleave_0, values = (var_2398_cast_fp16, attn_output_33_cast_fp16))[name = string("attn_output_35_cast_fp16")]; tensor var_2416_perm_0 = const()[name = string("op_2416_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_59x = const()[name = string("concat_59x"), val = tensor([1, 2048, 1, -1])]; tensor var_2416_cast_fp16 = transpose(perm = var_2416_perm_0, x = attn_output_35_cast_fp16)[name = string("transpose_585")]; tensor attn_output_39_cast_fp16 = reshape(shape = concat_59x, x = var_2416_cast_fp16)[name = string("attn_output_39_cast_fp16")]; tensor hidden_states_43_strides_0 = const()[name = string("hidden_states_43_strides_0"), val = tensor([1, 1])]; string hidden_states_43_pad_type_0 = const()[name = string("hidden_states_43_pad_type_0"), val = string("valid")]; tensor hidden_states_43_pad_0 = const()[name = string("hidden_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_43_dilations_0 = const()[name = string("hidden_states_43_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_43_groups_0 = const()[name = string("hidden_states_43_groups_0"), val = int32(1)]; tensor hidden_states_43_cast_fp16 = conv(dilations = hidden_states_43_dilations_0, groups = hidden_states_43_groups_0, pad = hidden_states_43_pad_0, pad_type = hidden_states_43_pad_type_0, strides = hidden_states_43_strides_0, weight = layers_4_self_attn_o_proj_weight_cast_fp16, x = attn_output_39_cast_fp16)[name = string("hidden_states_43_cast_fp16")]; tensor hidden_states_45_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = hidden_states_43_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; fp16 const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2449_cast_fp16 = mul(x = hidden_states_45_cast_fp16, y = const_48_promoted_to_fp16)[name = string("op_2449_cast_fp16")]; int32 var_2447 = const()[name = string("op_2447"), val = int32(1)]; bool doubled_37_interleave_0 = const()[name = string("doubled_37_interleave_0"), val = bool(false)]; tensor doubled_37_cast_fp16 = concat(axis = var_2447, interleave = doubled_37_interleave_0, values = (hidden_states_45_cast_fp16, var_2449_cast_fp16))[name = string("doubled_37_cast_fp16")]; tensor out_19_axes_0 = const()[name = string("out_19_axes_0"), val = tensor([1])]; tensor out_19_gamma_0_to_fp16 = const()[name = string("out_19_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283129280)))]; fp16 var_2459_to_fp16 = const()[name = string("op_2459_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_19_cast_fp16 = layer_norm(axes = out_19_axes_0, epsilon = var_2459_to_fp16, gamma = out_19_gamma_0_to_fp16, x = doubled_37_cast_fp16)[name = string("out_19_cast_fp16")]; tensor var_2470_split_sizes_0 = const()[name = string("op_2470_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2470_axis_0 = const()[name = string("op_2470_axis_0"), val = int32(1)]; tensor var_2470_cast_fp16_0, tensor var_2470_cast_fp16_1 = split(axis = var_2470_axis_0, split_sizes = var_2470_split_sizes_0, x = out_19_cast_fp16)[name = string("op_2470_cast_fp16")]; tensor input_9_strides_0 = const()[name = string("input_9_strides_0"), val = tensor([1, 1])]; string input_9_pad_type_0 = const()[name = string("input_9_pad_type_0"), val = string("valid")]; tensor input_9_pad_0 = const()[name = string("input_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_9_dilations_0 = const()[name = string("input_9_dilations_0"), val = tensor([1, 1])]; int32 input_9_groups_0 = const()[name = string("input_9_groups_0"), val = int32(1)]; tensor input_9_cast_fp16 = conv(dilations = input_9_dilations_0, groups = input_9_groups_0, pad = input_9_pad_0, pad_type = input_9_pad_type_0, strides = input_9_strides_0, weight = layers_4_mlp_gate_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("input_9_cast_fp16")]; tensor var_2487_cast_fp16 = silu(x = input_9_cast_fp16)[name = string("op_2487_cast_fp16")]; tensor var_2493_strides_0 = const()[name = string("op_2493_strides_0"), val = tensor([1, 1])]; string var_2493_pad_type_0 = const()[name = string("op_2493_pad_type_0"), val = string("valid")]; tensor var_2493_pad_0 = const()[name = string("op_2493_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2493_dilations_0 = const()[name = string("op_2493_dilations_0"), val = tensor([1, 1])]; int32 var_2493_groups_0 = const()[name = string("op_2493_groups_0"), val = int32(1)]; tensor var_2493_cast_fp16 = conv(dilations = var_2493_dilations_0, groups = var_2493_groups_0, pad = var_2493_pad_0, pad_type = var_2493_pad_type_0, strides = var_2493_strides_0, weight = layers_4_mlp_up_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("op_2493_cast_fp16")]; tensor x_49_cast_fp16 = mul(x = var_2487_cast_fp16, y = var_2493_cast_fp16)[name = string("x_49_cast_fp16")]; tensor hidden_states_47_strides_0 = const()[name = string("hidden_states_47_strides_0"), val = tensor([1, 1])]; string hidden_states_47_pad_type_0 = const()[name = string("hidden_states_47_pad_type_0"), val = string("valid")]; tensor hidden_states_47_pad_0 = const()[name = string("hidden_states_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_47_dilations_0 = const()[name = string("hidden_states_47_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_47_groups_0 = const()[name = string("hidden_states_47_groups_0"), val = int32(1)]; tensor hidden_states_47_cast_fp16 = conv(dilations = hidden_states_47_dilations_0, groups = hidden_states_47_groups_0, pad = hidden_states_47_pad_0, pad_type = hidden_states_47_pad_type_0, strides = hidden_states_47_strides_0, weight = layers_4_mlp_down_proj_weight_cast_fp16, x = x_49_cast_fp16)[name = string("hidden_states_47_cast_fp16")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = hidden_states_47_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; fp16 const_50_promoted_to_fp16 = const()[name = string("const_50_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2511_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_50_promoted_to_fp16)[name = string("op_2511_cast_fp16")]; int32 var_2509 = const()[name = string("op_2509"), val = int32(1)]; bool doubled_41_interleave_0 = const()[name = string("doubled_41_interleave_0"), val = bool(false)]; tensor doubled_41_cast_fp16 = concat(axis = var_2509, interleave = doubled_41_interleave_0, values = (hidden_states_49_cast_fp16, var_2511_cast_fp16))[name = string("doubled_41_cast_fp16")]; tensor out_21_axes_0 = const()[name = string("out_21_axes_0"), val = tensor([1])]; tensor out_21_gamma_0_to_fp16 = const()[name = string("out_21_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283137536)))]; fp16 var_2521_to_fp16 = const()[name = string("op_2521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_21_cast_fp16 = layer_norm(axes = out_21_axes_0, epsilon = var_2521_to_fp16, gamma = out_21_gamma_0_to_fp16, x = doubled_41_cast_fp16)[name = string("out_21_cast_fp16")]; tensor var_2532_split_sizes_0 = const()[name = string("op_2532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2532_axis_0 = const()[name = string("op_2532_axis_0"), val = int32(1)]; tensor var_2532_cast_fp16_0, tensor var_2532_cast_fp16_1 = split(axis = var_2532_axis_0, split_sizes = var_2532_split_sizes_0, x = out_21_cast_fp16)[name = string("op_2532_cast_fp16")]; tensor layers_5_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283145792)))]; tensor query_states_31_strides_0 = const()[name = string("query_states_31_strides_0"), val = tensor([1, 1])]; string query_states_31_pad_type_0 = const()[name = string("query_states_31_pad_type_0"), val = string("valid")]; tensor query_states_31_pad_0 = const()[name = string("query_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_31_dilations_0 = const()[name = string("query_states_31_dilations_0"), val = tensor([1, 1])]; int32 query_states_31_groups_0 = const()[name = string("query_states_31_groups_0"), val = int32(1)]; tensor query_states_31_cast_fp16 = conv(dilations = query_states_31_dilations_0, groups = query_states_31_groups_0, pad = query_states_31_pad_0, pad_type = query_states_31_pad_type_0, strides = query_states_31_strides_0, weight = layers_5_self_attn_q_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("query_states_31_cast_fp16")]; tensor layers_5_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1291534464)))]; tensor key_states_51_strides_0 = const()[name = string("key_states_51_strides_0"), val = tensor([1, 1])]; string key_states_51_pad_type_0 = const()[name = string("key_states_51_pad_type_0"), val = string("valid")]; tensor key_states_51_pad_0 = const()[name = string("key_states_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_51_dilations_0 = const()[name = string("key_states_51_dilations_0"), val = tensor([1, 1])]; int32 key_states_51_groups_0 = const()[name = string("key_states_51_groups_0"), val = int32(1)]; tensor key_states_51_cast_fp16 = conv(dilations = key_states_51_dilations_0, groups = key_states_51_groups_0, pad = key_states_51_pad_0, pad_type = key_states_51_pad_type_0, strides = key_states_51_strides_0, weight = layers_5_self_attn_k_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("key_states_51_cast_fp16")]; tensor value_states_31_strides_0 = const()[name = string("value_states_31_strides_0"), val = tensor([1, 1])]; string value_states_31_pad_type_0 = const()[name = string("value_states_31_pad_type_0"), val = string("valid")]; tensor value_states_31_pad_0 = const()[name = string("value_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_31_dilations_0 = const()[name = string("value_states_31_dilations_0"), val = tensor([1, 1])]; int32 value_states_31_groups_0 = const()[name = string("value_states_31_groups_0"), val = int32(1)]; tensor value_states_31_cast_fp16 = conv(dilations = value_states_31_dilations_0, groups = value_states_31_groups_0, pad = value_states_31_pad_0, pad_type = value_states_31_pad_type_0, strides = value_states_31_strides_0, weight = layers_5_self_attn_v_proj_weight_cast_fp16, x = var_2532_cast_fp16_0)[name = string("value_states_31_cast_fp16")]; tensor concat_60x = const()[name = string("concat_60x"), val = tensor([1, 16, 128, -1])]; tensor x_51_cast_fp16 = reshape(shape = concat_60x, x = query_states_31_cast_fp16)[name = string("x_51_cast_fp16")]; tensor concat_61x = const()[name = string("concat_61x"), val = tensor([1, 2, 128, -1])]; tensor var_2589_cast_fp16 = reshape(shape = concat_61x, x = key_states_51_cast_fp16)[name = string("op_2589_cast_fp16")]; tensor concat_62x = const()[name = string("concat_62x"), val = tensor([1, 2, 128, -1])]; tensor var_2596_cast_fp16 = reshape(shape = concat_62x, x = value_states_31_cast_fp16)[name = string("op_2596_cast_fp16")]; tensor var_2600_cast_fp16 = mul(x = x_51_cast_fp16, y = var_869_cast_fp16)[name = string("op_2600_cast_fp16")]; tensor var_2601_split_sizes_0 = const()[name = string("op_2601_split_sizes_0"), val = tensor([64, 64])]; int32 var_2601_axis_0 = const()[name = string("op_2601_axis_0"), val = int32(-2)]; tensor var_2601_cast_fp16_0, tensor var_2601_cast_fp16_1 = split(axis = var_2601_axis_0, split_sizes = var_2601_split_sizes_0, x = x_51_cast_fp16)[name = string("op_2601_cast_fp16")]; fp16 const_52_promoted_to_fp16 = const()[name = string("const_52_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2603_cast_fp16 = mul(x = var_2601_cast_fp16_1, y = const_52_promoted_to_fp16)[name = string("op_2603_cast_fp16")]; int32 var_2605 = const()[name = string("op_2605"), val = int32(-2)]; bool var_2606_interleave_0 = const()[name = string("op_2606_interleave_0"), val = bool(false)]; tensor var_2606_cast_fp16 = concat(axis = var_2605, interleave = var_2606_interleave_0, values = (var_2603_cast_fp16, var_2601_cast_fp16_0))[name = string("op_2606_cast_fp16")]; tensor var_2607_cast_fp16 = mul(x = var_2606_cast_fp16, y = var_878_cast_fp16)[name = string("op_2607_cast_fp16")]; tensor query_states_33_cast_fp16 = add(x = var_2600_cast_fp16, y = var_2607_cast_fp16)[name = string("query_states_33_cast_fp16")]; tensor var_2613_cast_fp16 = mul(x = var_2589_cast_fp16, y = var_869_cast_fp16)[name = string("op_2613_cast_fp16")]; tensor var_2614_split_sizes_0 = const()[name = string("op_2614_split_sizes_0"), val = tensor([64, 64])]; int32 var_2614_axis_0 = const()[name = string("op_2614_axis_0"), val = int32(-2)]; tensor var_2614_cast_fp16_0, tensor var_2614_cast_fp16_1 = split(axis = var_2614_axis_0, split_sizes = var_2614_split_sizes_0, x = var_2589_cast_fp16)[name = string("op_2614_cast_fp16")]; fp16 const_53_promoted_to_fp16 = const()[name = string("const_53_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2616_cast_fp16 = mul(x = var_2614_cast_fp16_1, y = const_53_promoted_to_fp16)[name = string("op_2616_cast_fp16")]; int32 var_2618 = const()[name = string("op_2618"), val = int32(-2)]; bool var_2619_interleave_0 = const()[name = string("op_2619_interleave_0"), val = bool(false)]; tensor var_2619_cast_fp16 = concat(axis = var_2618, interleave = var_2619_interleave_0, values = (var_2616_cast_fp16, var_2614_cast_fp16_0))[name = string("op_2619_cast_fp16")]; tensor var_2620_cast_fp16 = mul(x = var_2619_cast_fp16, y = var_878_cast_fp16)[name = string("op_2620_cast_fp16")]; tensor key_states_55_cast_fp16 = add(x = var_2613_cast_fp16, y = var_2620_cast_fp16)[name = string("key_states_55_cast_fp16")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; int32 concat_65_axis_0 = const()[name = string("concat_65_axis_0"), val = int32(0)]; bool concat_65_interleave_0 = const()[name = string("concat_65_interleave_0"), val = bool(false)]; tensor concat_65 = concat(axis = concat_65_axis_0, interleave = concat_65_interleave_0, values = (expand_dims_60, expand_dims_61, position_id, expand_dims_63))[name = string("concat_65")]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; tensor concat_66_values1_0 = const()[name = string("concat_66_values1_0"), val = tensor([0])]; tensor concat_66_values3_0 = const()[name = string("concat_66_values3_0"), val = tensor([0])]; int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_64, concat_66_values1_0, cache_position_end, concat_66_values3_0))[name = string("concat_66")]; tensor key_states_57_perm_0 = const()[name = string("key_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_6_stride_0 = const()[name = string("key_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_57_cast_fp16 = transpose(perm = key_states_57_perm_0, x = key_states_55_cast_fp16)[name = string("transpose_584")]; tensor key_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = key_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = key_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_6_squeeze_mask_0, stride = key_cache_internal_tensor_assign_6_stride_0, update = key_states_57_cast_fp16, x = coreml_update_state_344)[name = string("key_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_6_cast_fp16, input = key_cache)[name = string("coreml_update_state_346_write_state")]; tensor coreml_update_state_346 = read_state(input = key_cache)[name = string("coreml_update_state_346")]; tensor value_states_33_perm_0 = const()[name = string("value_states_33_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_6_stride_0 = const()[name = string("value_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_33_cast_fp16 = transpose(perm = value_states_33_perm_0, x = var_2596_cast_fp16)[name = string("transpose_583")]; tensor value_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = value_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = value_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_6_squeeze_mask_0, stride = value_cache_internal_tensor_assign_6_stride_0, update = value_states_33_cast_fp16, x = coreml_update_state_345)[name = string("value_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_6_cast_fp16, input = value_cache)[name = string("coreml_update_state_347_write_state")]; tensor coreml_update_state_347 = read_state(input = value_cache)[name = string("coreml_update_state_347")]; tensor var_2690_begin_0 = const()[name = string("op_2690_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2690_end_0 = const()[name = string("op_2690_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2690_end_mask_0 = const()[name = string("op_2690_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2690_cast_fp16 = slice_by_index(begin = var_2690_begin_0, end = var_2690_end_0, end_mask = var_2690_end_mask_0, x = coreml_update_state_346)[name = string("op_2690_cast_fp16")]; tensor tile_10 = const()[name = string("tile_10"), val = tensor([1, 1])]; int32 var_2693_axis_0 = const()[name = string("op_2693_axis_0"), val = int32(1)]; tensor var_2693_cast_fp16_0, tensor var_2693_cast_fp16_1 = split(axis = var_2693_axis_0, split_sizes = tile_10, x = var_2690_cast_fp16)[name = string("op_2693_cast_fp16")]; tensor var_2700_begin_0 = const()[name = string("op_2700_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2700_end_0 = const()[name = string("op_2700_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2700_end_mask_0 = const()[name = string("op_2700_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2700_cast_fp16 = slice_by_index(begin = var_2700_begin_0, end = var_2700_end_0, end_mask = var_2700_end_mask_0, x = coreml_update_state_347)[name = string("op_2700_cast_fp16")]; tensor tile_11 = const()[name = string("tile_11"), val = tensor([1, 1])]; int32 var_2703_axis_0 = const()[name = string("op_2703_axis_0"), val = int32(1)]; tensor var_2703_cast_fp16_0, tensor var_2703_cast_fp16_1 = split(axis = var_2703_axis_0, split_sizes = tile_11, x = var_2700_cast_fp16)[name = string("op_2703_cast_fp16")]; tensor var_2706_split_sizes_0 = const()[name = string("op_2706_split_sizes_0"), val = tensor([8, 8])]; int32 var_2706_axis_0 = const()[name = string("op_2706_axis_0"), val = int32(1)]; tensor var_2706_0, tensor var_2706_1 = split(axis = var_2706_axis_0, split_sizes = var_2706_split_sizes_0, x = query_states_33_cast_fp16)[name = string("op_2706")]; bool attn_weights_81_transpose_x_0 = const()[name = string("attn_weights_81_transpose_x_0"), val = bool(false)]; bool attn_weights_81_transpose_y_0 = const()[name = string("attn_weights_81_transpose_y_0"), val = bool(false)]; tensor attn_weights_81_cast_fp16 = matmul(transpose_x = attn_weights_81_transpose_x_0, transpose_y = attn_weights_81_transpose_y_0, x = var_2693_cast_fp16_0, y = var_2706_0)[name = string("attn_weights_81_cast_fp16")]; fp16 var_2709_to_fp16 = const()[name = string("op_2709_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_83_cast_fp16 = mul(x = attn_weights_81_cast_fp16, y = var_2709_to_fp16)[name = string("attn_weights_83_cast_fp16")]; tensor attn_weights_85_cast_fp16 = add(x = attn_weights_83_cast_fp16, y = attn_mask_1)[name = string("attn_weights_85_cast_fp16")]; int32 var_2713 = const()[name = string("op_2713"), val = int32(-2)]; tensor attn_weights_87_cast_fp16 = softmax(axis = var_2713, x = attn_weights_85_cast_fp16)[name = string("attn_weights_87_cast_fp16")]; bool var_2719_transpose_x_1 = const()[name = string("op_2719_transpose_x_1"), val = bool(true)]; bool var_2719_transpose_y_1 = const()[name = string("op_2719_transpose_y_1"), val = bool(false)]; tensor var_2719_cast_fp16 = matmul(transpose_x = var_2719_transpose_x_1, transpose_y = var_2719_transpose_y_1, x = attn_weights_87_cast_fp16, y = var_2703_cast_fp16_0)[name = string("op_2719_cast_fp16")]; bool attn_weights_89_transpose_x_0 = const()[name = string("attn_weights_89_transpose_x_0"), val = bool(false)]; bool attn_weights_89_transpose_y_0 = const()[name = string("attn_weights_89_transpose_y_0"), val = bool(false)]; tensor attn_weights_89_cast_fp16 = matmul(transpose_x = attn_weights_89_transpose_x_0, transpose_y = attn_weights_89_transpose_y_0, x = var_2693_cast_fp16_1, y = var_2706_1)[name = string("attn_weights_89_cast_fp16")]; fp16 var_2721_to_fp16 = const()[name = string("op_2721_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_91_cast_fp16 = mul(x = attn_weights_89_cast_fp16, y = var_2721_to_fp16)[name = string("attn_weights_91_cast_fp16")]; tensor attn_weights_93_cast_fp16 = add(x = attn_weights_91_cast_fp16, y = attn_mask_1)[name = string("attn_weights_93_cast_fp16")]; int32 var_2725 = const()[name = string("op_2725"), val = int32(-2)]; tensor attn_weights_95_cast_fp16 = softmax(axis = var_2725, x = attn_weights_93_cast_fp16)[name = string("attn_weights_95_cast_fp16")]; bool attn_output_41_transpose_x_1 = const()[name = string("attn_output_41_transpose_x_1"), val = bool(true)]; bool attn_output_41_transpose_y_1 = const()[name = string("attn_output_41_transpose_y_1"), val = bool(false)]; tensor attn_output_41_cast_fp16 = matmul(transpose_x = attn_output_41_transpose_x_1, transpose_y = attn_output_41_transpose_y_1, x = attn_weights_95_cast_fp16, y = var_2703_cast_fp16_1)[name = string("attn_output_41_cast_fp16")]; int32 var_2733 = const()[name = string("op_2733"), val = int32(1)]; bool attn_output_43_interleave_0 = const()[name = string("attn_output_43_interleave_0"), val = bool(false)]; tensor attn_output_43_cast_fp16 = concat(axis = var_2733, interleave = attn_output_43_interleave_0, values = (var_2719_cast_fp16, attn_output_41_cast_fp16))[name = string("attn_output_43_cast_fp16")]; tensor var_2737_perm_0 = const()[name = string("op_2737_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_71x = const()[name = string("concat_71x"), val = tensor([1, 2048, 1, -1])]; tensor var_2737_cast_fp16 = transpose(perm = var_2737_perm_0, x = attn_output_43_cast_fp16)[name = string("transpose_582")]; tensor attn_output_47_cast_fp16 = reshape(shape = concat_71x, x = var_2737_cast_fp16)[name = string("attn_output_47_cast_fp16")]; tensor hidden_states_53_strides_0 = const()[name = string("hidden_states_53_strides_0"), val = tensor([1, 1])]; string hidden_states_53_pad_type_0 = const()[name = string("hidden_states_53_pad_type_0"), val = string("valid")]; tensor hidden_states_53_pad_0 = const()[name = string("hidden_states_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_53_dilations_0 = const()[name = string("hidden_states_53_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_53_groups_0 = const()[name = string("hidden_states_53_groups_0"), val = int32(1)]; tensor hidden_states_53_cast_fp16 = conv(dilations = hidden_states_53_dilations_0, groups = hidden_states_53_groups_0, pad = hidden_states_53_pad_0, pad_type = hidden_states_53_pad_type_0, strides = hidden_states_53_strides_0, weight = layers_5_self_attn_o_proj_weight_cast_fp16, x = attn_output_47_cast_fp16)[name = string("hidden_states_53_cast_fp16")]; tensor hidden_states_55_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = hidden_states_53_cast_fp16)[name = string("hidden_states_55_cast_fp16")]; fp16 const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2770_cast_fp16 = mul(x = hidden_states_55_cast_fp16, y = const_58_promoted_to_fp16)[name = string("op_2770_cast_fp16")]; int32 var_2768 = const()[name = string("op_2768"), val = int32(1)]; bool doubled_45_interleave_0 = const()[name = string("doubled_45_interleave_0"), val = bool(false)]; tensor doubled_45_cast_fp16 = concat(axis = var_2768, interleave = doubled_45_interleave_0, values = (hidden_states_55_cast_fp16, var_2770_cast_fp16))[name = string("doubled_45_cast_fp16")]; tensor out_23_axes_0 = const()[name = string("out_23_axes_0"), val = tensor([1])]; tensor out_23_gamma_0_to_fp16 = const()[name = string("out_23_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292583104)))]; fp16 var_2780_to_fp16 = const()[name = string("op_2780_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_23_cast_fp16 = layer_norm(axes = out_23_axes_0, epsilon = var_2780_to_fp16, gamma = out_23_gamma_0_to_fp16, x = doubled_45_cast_fp16)[name = string("out_23_cast_fp16")]; tensor var_2791_split_sizes_0 = const()[name = string("op_2791_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2791_axis_0 = const()[name = string("op_2791_axis_0"), val = int32(1)]; tensor var_2791_cast_fp16_0, tensor var_2791_cast_fp16_1 = split(axis = var_2791_axis_0, split_sizes = var_2791_split_sizes_0, x = out_23_cast_fp16)[name = string("op_2791_cast_fp16")]; tensor layers_5_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_5_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292591360)))]; tensor input_11_strides_0 = const()[name = string("input_11_strides_0"), val = tensor([1, 1])]; string input_11_pad_type_0 = const()[name = string("input_11_pad_type_0"), val = string("valid")]; tensor input_11_pad_0 = const()[name = string("input_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_11_dilations_0 = const()[name = string("input_11_dilations_0"), val = tensor([1, 1])]; int32 input_11_groups_0 = const()[name = string("input_11_groups_0"), val = int32(1)]; tensor input_11_cast_fp16 = conv(dilations = input_11_dilations_0, groups = input_11_groups_0, pad = input_11_pad_0, pad_type = input_11_pad_type_0, strides = input_11_strides_0, weight = layers_5_mlp_gate_proj_weight_to_fp16, x = var_2791_cast_fp16_0)[name = string("input_11_cast_fp16")]; tensor var_2808_cast_fp16 = silu(x = input_11_cast_fp16)[name = string("op_2808_cast_fp16")]; tensor var_2814_strides_0 = const()[name = string("op_2814_strides_0"), val = tensor([1, 1])]; string var_2814_pad_type_0 = const()[name = string("op_2814_pad_type_0"), val = string("valid")]; tensor var_2814_pad_0 = const()[name = string("op_2814_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2814_dilations_0 = const()[name = string("op_2814_dilations_0"), val = tensor([1, 1])]; int32 var_2814_groups_0 = const()[name = string("op_2814_groups_0"), val = int32(1)]; tensor var_2814_cast_fp16 = conv(dilations = var_2814_dilations_0, groups = var_2814_groups_0, pad = var_2814_pad_0, pad_type = var_2814_pad_type_0, strides = var_2814_strides_0, weight = layers_5_mlp_up_proj_weight_cast_fp16, x = var_2791_cast_fp16_0)[name = string("op_2814_cast_fp16")]; tensor x_59_cast_fp16 = mul(x = var_2808_cast_fp16, y = var_2814_cast_fp16)[name = string("x_59_cast_fp16")]; tensor hidden_states_57_strides_0 = const()[name = string("hidden_states_57_strides_0"), val = tensor([1, 1])]; string hidden_states_57_pad_type_0 = const()[name = string("hidden_states_57_pad_type_0"), val = string("valid")]; tensor hidden_states_57_pad_0 = const()[name = string("hidden_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_57_dilations_0 = const()[name = string("hidden_states_57_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_57_groups_0 = const()[name = string("hidden_states_57_groups_0"), val = int32(1)]; tensor hidden_states_57_cast_fp16 = conv(dilations = hidden_states_57_dilations_0, groups = hidden_states_57_groups_0, pad = hidden_states_57_pad_0, pad_type = hidden_states_57_pad_type_0, strides = hidden_states_57_strides_0, weight = layers_5_mlp_down_proj_weight_cast_fp16, x = x_59_cast_fp16)[name = string("hidden_states_57_cast_fp16")]; tensor hidden_states_59_cast_fp16 = add(x = hidden_states_55_cast_fp16, y = hidden_states_57_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2832_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_2832_cast_fp16")]; int32 var_2830 = const()[name = string("op_2830"), val = int32(1)]; bool doubled_49_interleave_0 = const()[name = string("doubled_49_interleave_0"), val = bool(false)]; tensor doubled_49_cast_fp16 = concat(axis = var_2830, interleave = doubled_49_interleave_0, values = (hidden_states_59_cast_fp16, var_2832_cast_fp16))[name = string("doubled_49_cast_fp16")]; tensor out_25_axes_0 = const()[name = string("out_25_axes_0"), val = tensor([1])]; tensor out_25_gamma_0_to_fp16 = const()[name = string("out_25_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317757248)))]; fp16 var_2842_to_fp16 = const()[name = string("op_2842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_25_cast_fp16 = layer_norm(axes = out_25_axes_0, epsilon = var_2842_to_fp16, gamma = out_25_gamma_0_to_fp16, x = doubled_49_cast_fp16)[name = string("out_25_cast_fp16")]; tensor var_2853_split_sizes_0 = const()[name = string("op_2853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2853_axis_0 = const()[name = string("op_2853_axis_0"), val = int32(1)]; tensor var_2853_cast_fp16_0, tensor var_2853_cast_fp16_1 = split(axis = var_2853_axis_0, split_sizes = var_2853_split_sizes_0, x = out_25_cast_fp16)[name = string("op_2853_cast_fp16")]; tensor layers_6_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317765504)))]; tensor query_states_37_strides_0 = const()[name = string("query_states_37_strides_0"), val = tensor([1, 1])]; string query_states_37_pad_type_0 = const()[name = string("query_states_37_pad_type_0"), val = string("valid")]; tensor query_states_37_pad_0 = const()[name = string("query_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_37_dilations_0 = const()[name = string("query_states_37_dilations_0"), val = tensor([1, 1])]; int32 query_states_37_groups_0 = const()[name = string("query_states_37_groups_0"), val = int32(1)]; tensor query_states_37_cast_fp16 = conv(dilations = query_states_37_dilations_0, groups = query_states_37_groups_0, pad = query_states_37_pad_0, pad_type = query_states_37_pad_type_0, strides = query_states_37_strides_0, weight = layers_6_self_attn_q_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("query_states_37_cast_fp16")]; tensor layers_6_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1326154176)))]; tensor key_states_61_strides_0 = const()[name = string("key_states_61_strides_0"), val = tensor([1, 1])]; string key_states_61_pad_type_0 = const()[name = string("key_states_61_pad_type_0"), val = string("valid")]; tensor key_states_61_pad_0 = const()[name = string("key_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_61_dilations_0 = const()[name = string("key_states_61_dilations_0"), val = tensor([1, 1])]; int32 key_states_61_groups_0 = const()[name = string("key_states_61_groups_0"), val = int32(1)]; tensor key_states_61_cast_fp16 = conv(dilations = key_states_61_dilations_0, groups = key_states_61_groups_0, pad = key_states_61_pad_0, pad_type = key_states_61_pad_type_0, strides = key_states_61_strides_0, weight = layers_6_self_attn_k_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("key_states_61_cast_fp16")]; tensor value_states_37_strides_0 = const()[name = string("value_states_37_strides_0"), val = tensor([1, 1])]; string value_states_37_pad_type_0 = const()[name = string("value_states_37_pad_type_0"), val = string("valid")]; tensor value_states_37_pad_0 = const()[name = string("value_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_37_dilations_0 = const()[name = string("value_states_37_dilations_0"), val = tensor([1, 1])]; int32 value_states_37_groups_0 = const()[name = string("value_states_37_groups_0"), val = int32(1)]; tensor value_states_37_cast_fp16 = conv(dilations = value_states_37_dilations_0, groups = value_states_37_groups_0, pad = value_states_37_pad_0, pad_type = value_states_37_pad_type_0, strides = value_states_37_strides_0, weight = layers_6_self_attn_v_proj_weight_cast_fp16, x = var_2853_cast_fp16_0)[name = string("value_states_37_cast_fp16")]; tensor concat_72x = const()[name = string("concat_72x"), val = tensor([1, 16, 128, -1])]; tensor x_61_cast_fp16 = reshape(shape = concat_72x, x = query_states_37_cast_fp16)[name = string("x_61_cast_fp16")]; tensor concat_73x = const()[name = string("concat_73x"), val = tensor([1, 2, 128, -1])]; tensor var_2910_cast_fp16 = reshape(shape = concat_73x, x = key_states_61_cast_fp16)[name = string("op_2910_cast_fp16")]; tensor concat_74x = const()[name = string("concat_74x"), val = tensor([1, 2, 128, -1])]; tensor var_2917_cast_fp16 = reshape(shape = concat_74x, x = value_states_37_cast_fp16)[name = string("op_2917_cast_fp16")]; tensor var_2921_cast_fp16 = mul(x = x_61_cast_fp16, y = var_869_cast_fp16)[name = string("op_2921_cast_fp16")]; tensor var_2922_split_sizes_0 = const()[name = string("op_2922_split_sizes_0"), val = tensor([64, 64])]; int32 var_2922_axis_0 = const()[name = string("op_2922_axis_0"), val = int32(-2)]; tensor var_2922_cast_fp16_0, tensor var_2922_cast_fp16_1 = split(axis = var_2922_axis_0, split_sizes = var_2922_split_sizes_0, x = x_61_cast_fp16)[name = string("op_2922_cast_fp16")]; fp16 const_62_promoted_to_fp16 = const()[name = string("const_62_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2924_cast_fp16 = mul(x = var_2922_cast_fp16_1, y = const_62_promoted_to_fp16)[name = string("op_2924_cast_fp16")]; int32 var_2926 = const()[name = string("op_2926"), val = int32(-2)]; bool var_2927_interleave_0 = const()[name = string("op_2927_interleave_0"), val = bool(false)]; tensor var_2927_cast_fp16 = concat(axis = var_2926, interleave = var_2927_interleave_0, values = (var_2924_cast_fp16, var_2922_cast_fp16_0))[name = string("op_2927_cast_fp16")]; tensor var_2928_cast_fp16 = mul(x = var_2927_cast_fp16, y = var_878_cast_fp16)[name = string("op_2928_cast_fp16")]; tensor query_states_39_cast_fp16 = add(x = var_2921_cast_fp16, y = var_2928_cast_fp16)[name = string("query_states_39_cast_fp16")]; tensor var_2934_cast_fp16 = mul(x = var_2910_cast_fp16, y = var_869_cast_fp16)[name = string("op_2934_cast_fp16")]; tensor var_2935_split_sizes_0 = const()[name = string("op_2935_split_sizes_0"), val = tensor([64, 64])]; int32 var_2935_axis_0 = const()[name = string("op_2935_axis_0"), val = int32(-2)]; tensor var_2935_cast_fp16_0, tensor var_2935_cast_fp16_1 = split(axis = var_2935_axis_0, split_sizes = var_2935_split_sizes_0, x = var_2910_cast_fp16)[name = string("op_2935_cast_fp16")]; fp16 const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2937_cast_fp16 = mul(x = var_2935_cast_fp16_1, y = const_63_promoted_to_fp16)[name = string("op_2937_cast_fp16")]; int32 var_2939 = const()[name = string("op_2939"), val = int32(-2)]; bool var_2940_interleave_0 = const()[name = string("op_2940_interleave_0"), val = bool(false)]; tensor var_2940_cast_fp16 = concat(axis = var_2939, interleave = var_2940_interleave_0, values = (var_2937_cast_fp16, var_2935_cast_fp16_0))[name = string("op_2940_cast_fp16")]; tensor var_2941_cast_fp16 = mul(x = var_2940_cast_fp16, y = var_878_cast_fp16)[name = string("op_2941_cast_fp16")]; tensor key_states_65_cast_fp16 = add(x = var_2934_cast_fp16, y = var_2941_cast_fp16)[name = string("key_states_65_cast_fp16")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; int32 concat_77_axis_0 = const()[name = string("concat_77_axis_0"), val = int32(0)]; bool concat_77_interleave_0 = const()[name = string("concat_77_interleave_0"), val = bool(false)]; tensor concat_77 = concat(axis = concat_77_axis_0, interleave = concat_77_interleave_0, values = (expand_dims_72, expand_dims_73, position_id, expand_dims_75))[name = string("concat_77")]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; tensor concat_78_values1_0 = const()[name = string("concat_78_values1_0"), val = tensor([0])]; tensor concat_78_values3_0 = const()[name = string("concat_78_values3_0"), val = tensor([0])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_76, concat_78_values1_0, cache_position_end, concat_78_values3_0))[name = string("concat_78")]; tensor key_states_67_perm_0 = const()[name = string("key_states_67_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_7_stride_0 = const()[name = string("key_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_67_cast_fp16 = transpose(perm = key_states_67_perm_0, x = key_states_65_cast_fp16)[name = string("transpose_581")]; tensor key_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = key_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = key_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_7_squeeze_mask_0, stride = key_cache_internal_tensor_assign_7_stride_0, update = key_states_67_cast_fp16, x = coreml_update_state_346)[name = string("key_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_7_cast_fp16, input = key_cache)[name = string("coreml_update_state_348_write_state")]; tensor coreml_update_state_348 = read_state(input = key_cache)[name = string("coreml_update_state_348")]; tensor value_states_39_perm_0 = const()[name = string("value_states_39_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_7_stride_0 = const()[name = string("value_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_39_cast_fp16 = transpose(perm = value_states_39_perm_0, x = var_2917_cast_fp16)[name = string("transpose_580")]; tensor value_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = value_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = value_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_7_squeeze_mask_0, stride = value_cache_internal_tensor_assign_7_stride_0, update = value_states_39_cast_fp16, x = coreml_update_state_347)[name = string("value_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_7_cast_fp16, input = value_cache)[name = string("coreml_update_state_349_write_state")]; tensor coreml_update_state_349 = read_state(input = value_cache)[name = string("coreml_update_state_349")]; tensor var_3011_begin_0 = const()[name = string("op_3011_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3011_end_0 = const()[name = string("op_3011_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3011_end_mask_0 = const()[name = string("op_3011_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3011_cast_fp16 = slice_by_index(begin = var_3011_begin_0, end = var_3011_end_0, end_mask = var_3011_end_mask_0, x = coreml_update_state_348)[name = string("op_3011_cast_fp16")]; tensor tile_12 = const()[name = string("tile_12"), val = tensor([1, 1])]; int32 var_3014_axis_0 = const()[name = string("op_3014_axis_0"), val = int32(1)]; tensor var_3014_cast_fp16_0, tensor var_3014_cast_fp16_1 = split(axis = var_3014_axis_0, split_sizes = tile_12, x = var_3011_cast_fp16)[name = string("op_3014_cast_fp16")]; tensor var_3021_begin_0 = const()[name = string("op_3021_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3021_end_0 = const()[name = string("op_3021_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3021_end_mask_0 = const()[name = string("op_3021_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3021_cast_fp16 = slice_by_index(begin = var_3021_begin_0, end = var_3021_end_0, end_mask = var_3021_end_mask_0, x = coreml_update_state_349)[name = string("op_3021_cast_fp16")]; tensor tile_13 = const()[name = string("tile_13"), val = tensor([1, 1])]; int32 var_3024_axis_0 = const()[name = string("op_3024_axis_0"), val = int32(1)]; tensor var_3024_cast_fp16_0, tensor var_3024_cast_fp16_1 = split(axis = var_3024_axis_0, split_sizes = tile_13, x = var_3021_cast_fp16)[name = string("op_3024_cast_fp16")]; tensor var_3027_split_sizes_0 = const()[name = string("op_3027_split_sizes_0"), val = tensor([8, 8])]; int32 var_3027_axis_0 = const()[name = string("op_3027_axis_0"), val = int32(1)]; tensor var_3027_0, tensor var_3027_1 = split(axis = var_3027_axis_0, split_sizes = var_3027_split_sizes_0, x = query_states_39_cast_fp16)[name = string("op_3027")]; bool attn_weights_97_transpose_x_0 = const()[name = string("attn_weights_97_transpose_x_0"), val = bool(false)]; bool attn_weights_97_transpose_y_0 = const()[name = string("attn_weights_97_transpose_y_0"), val = bool(false)]; tensor attn_weights_97_cast_fp16 = matmul(transpose_x = attn_weights_97_transpose_x_0, transpose_y = attn_weights_97_transpose_y_0, x = var_3014_cast_fp16_0, y = var_3027_0)[name = string("attn_weights_97_cast_fp16")]; fp16 var_3030_to_fp16 = const()[name = string("op_3030_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_99_cast_fp16 = mul(x = attn_weights_97_cast_fp16, y = var_3030_to_fp16)[name = string("attn_weights_99_cast_fp16")]; tensor attn_weights_101_cast_fp16 = add(x = attn_weights_99_cast_fp16, y = attn_mask_1)[name = string("attn_weights_101_cast_fp16")]; int32 var_3034 = const()[name = string("op_3034"), val = int32(-2)]; tensor attn_weights_103_cast_fp16 = softmax(axis = var_3034, x = attn_weights_101_cast_fp16)[name = string("attn_weights_103_cast_fp16")]; bool var_3040_transpose_x_1 = const()[name = string("op_3040_transpose_x_1"), val = bool(true)]; bool var_3040_transpose_y_1 = const()[name = string("op_3040_transpose_y_1"), val = bool(false)]; tensor var_3040_cast_fp16 = matmul(transpose_x = var_3040_transpose_x_1, transpose_y = var_3040_transpose_y_1, x = attn_weights_103_cast_fp16, y = var_3024_cast_fp16_0)[name = string("op_3040_cast_fp16")]; bool attn_weights_105_transpose_x_0 = const()[name = string("attn_weights_105_transpose_x_0"), val = bool(false)]; bool attn_weights_105_transpose_y_0 = const()[name = string("attn_weights_105_transpose_y_0"), val = bool(false)]; tensor attn_weights_105_cast_fp16 = matmul(transpose_x = attn_weights_105_transpose_x_0, transpose_y = attn_weights_105_transpose_y_0, x = var_3014_cast_fp16_1, y = var_3027_1)[name = string("attn_weights_105_cast_fp16")]; fp16 var_3042_to_fp16 = const()[name = string("op_3042_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_107_cast_fp16 = mul(x = attn_weights_105_cast_fp16, y = var_3042_to_fp16)[name = string("attn_weights_107_cast_fp16")]; tensor attn_weights_109_cast_fp16 = add(x = attn_weights_107_cast_fp16, y = attn_mask_1)[name = string("attn_weights_109_cast_fp16")]; int32 var_3046 = const()[name = string("op_3046"), val = int32(-2)]; tensor attn_weights_111_cast_fp16 = softmax(axis = var_3046, x = attn_weights_109_cast_fp16)[name = string("attn_weights_111_cast_fp16")]; bool attn_output_49_transpose_x_1 = const()[name = string("attn_output_49_transpose_x_1"), val = bool(true)]; bool attn_output_49_transpose_y_1 = const()[name = string("attn_output_49_transpose_y_1"), val = bool(false)]; tensor attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_1, transpose_y = attn_output_49_transpose_y_1, x = attn_weights_111_cast_fp16, y = var_3024_cast_fp16_1)[name = string("attn_output_49_cast_fp16")]; int32 var_3054 = const()[name = string("op_3054"), val = int32(1)]; bool attn_output_51_interleave_0 = const()[name = string("attn_output_51_interleave_0"), val = bool(false)]; tensor attn_output_51_cast_fp16 = concat(axis = var_3054, interleave = attn_output_51_interleave_0, values = (var_3040_cast_fp16, attn_output_49_cast_fp16))[name = string("attn_output_51_cast_fp16")]; tensor var_3058_perm_0 = const()[name = string("op_3058_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_83x = const()[name = string("concat_83x"), val = tensor([1, 2048, 1, -1])]; tensor var_3058_cast_fp16 = transpose(perm = var_3058_perm_0, x = attn_output_51_cast_fp16)[name = string("transpose_579")]; tensor attn_output_55_cast_fp16 = reshape(shape = concat_83x, x = var_3058_cast_fp16)[name = string("attn_output_55_cast_fp16")]; tensor hidden_states_63_strides_0 = const()[name = string("hidden_states_63_strides_0"), val = tensor([1, 1])]; string hidden_states_63_pad_type_0 = const()[name = string("hidden_states_63_pad_type_0"), val = string("valid")]; tensor hidden_states_63_pad_0 = const()[name = string("hidden_states_63_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_63_dilations_0 = const()[name = string("hidden_states_63_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_63_groups_0 = const()[name = string("hidden_states_63_groups_0"), val = int32(1)]; tensor hidden_states_63_cast_fp16 = conv(dilations = hidden_states_63_dilations_0, groups = hidden_states_63_groups_0, pad = hidden_states_63_pad_0, pad_type = hidden_states_63_pad_type_0, strides = hidden_states_63_strides_0, weight = layers_6_self_attn_o_proj_weight_cast_fp16, x = attn_output_55_cast_fp16)[name = string("hidden_states_63_cast_fp16")]; tensor hidden_states_65_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = hidden_states_63_cast_fp16)[name = string("hidden_states_65_cast_fp16")]; fp16 const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3091_cast_fp16 = mul(x = hidden_states_65_cast_fp16, y = const_68_promoted_to_fp16)[name = string("op_3091_cast_fp16")]; int32 var_3089 = const()[name = string("op_3089"), val = int32(1)]; bool doubled_53_interleave_0 = const()[name = string("doubled_53_interleave_0"), val = bool(false)]; tensor doubled_53_cast_fp16 = concat(axis = var_3089, interleave = doubled_53_interleave_0, values = (hidden_states_65_cast_fp16, var_3091_cast_fp16))[name = string("doubled_53_cast_fp16")]; tensor out_27_axes_0 = const()[name = string("out_27_axes_0"), val = tensor([1])]; tensor out_27_gamma_0_to_fp16 = const()[name = string("out_27_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327202816)))]; fp16 var_3101_to_fp16 = const()[name = string("op_3101_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_27_cast_fp16 = layer_norm(axes = out_27_axes_0, epsilon = var_3101_to_fp16, gamma = out_27_gamma_0_to_fp16, x = doubled_53_cast_fp16)[name = string("out_27_cast_fp16")]; tensor var_3112_split_sizes_0 = const()[name = string("op_3112_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3112_axis_0 = const()[name = string("op_3112_axis_0"), val = int32(1)]; tensor var_3112_cast_fp16_0, tensor var_3112_cast_fp16_1 = split(axis = var_3112_axis_0, split_sizes = var_3112_split_sizes_0, x = out_27_cast_fp16)[name = string("op_3112_cast_fp16")]; tensor input_13_strides_0 = const()[name = string("input_13_strides_0"), val = tensor([1, 1])]; string input_13_pad_type_0 = const()[name = string("input_13_pad_type_0"), val = string("valid")]; tensor input_13_pad_0 = const()[name = string("input_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_13_dilations_0 = const()[name = string("input_13_dilations_0"), val = tensor([1, 1])]; int32 input_13_groups_0 = const()[name = string("input_13_groups_0"), val = int32(1)]; tensor input_13_cast_fp16 = conv(dilations = input_13_dilations_0, groups = input_13_groups_0, pad = input_13_pad_0, pad_type = input_13_pad_type_0, strides = input_13_strides_0, weight = layers_6_mlp_gate_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("input_13_cast_fp16")]; tensor var_3129_cast_fp16 = silu(x = input_13_cast_fp16)[name = string("op_3129_cast_fp16")]; tensor var_3135_strides_0 = const()[name = string("op_3135_strides_0"), val = tensor([1, 1])]; string var_3135_pad_type_0 = const()[name = string("op_3135_pad_type_0"), val = string("valid")]; tensor var_3135_pad_0 = const()[name = string("op_3135_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3135_dilations_0 = const()[name = string("op_3135_dilations_0"), val = tensor([1, 1])]; int32 var_3135_groups_0 = const()[name = string("op_3135_groups_0"), val = int32(1)]; tensor var_3135_cast_fp16 = conv(dilations = var_3135_dilations_0, groups = var_3135_groups_0, pad = var_3135_pad_0, pad_type = var_3135_pad_type_0, strides = var_3135_strides_0, weight = layers_6_mlp_up_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("op_3135_cast_fp16")]; tensor x_69_cast_fp16 = mul(x = var_3129_cast_fp16, y = var_3135_cast_fp16)[name = string("x_69_cast_fp16")]; tensor hidden_states_67_strides_0 = const()[name = string("hidden_states_67_strides_0"), val = tensor([1, 1])]; string hidden_states_67_pad_type_0 = const()[name = string("hidden_states_67_pad_type_0"), val = string("valid")]; tensor hidden_states_67_pad_0 = const()[name = string("hidden_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_67_dilations_0 = const()[name = string("hidden_states_67_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_67_groups_0 = const()[name = string("hidden_states_67_groups_0"), val = int32(1)]; tensor hidden_states_67_cast_fp16 = conv(dilations = hidden_states_67_dilations_0, groups = hidden_states_67_groups_0, pad = hidden_states_67_pad_0, pad_type = hidden_states_67_pad_type_0, strides = hidden_states_67_strides_0, weight = layers_6_mlp_down_proj_weight_cast_fp16, x = x_69_cast_fp16)[name = string("hidden_states_67_cast_fp16")]; tensor hidden_states_69_cast_fp16 = add(x = hidden_states_65_cast_fp16, y = hidden_states_67_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3153_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_70_promoted_to_fp16)[name = string("op_3153_cast_fp16")]; int32 var_3151 = const()[name = string("op_3151"), val = int32(1)]; bool doubled_57_interleave_0 = const()[name = string("doubled_57_interleave_0"), val = bool(false)]; tensor doubled_57_cast_fp16 = concat(axis = var_3151, interleave = doubled_57_interleave_0, values = (hidden_states_69_cast_fp16, var_3153_cast_fp16))[name = string("doubled_57_cast_fp16")]; tensor out_29_axes_0 = const()[name = string("out_29_axes_0"), val = tensor([1])]; tensor out_29_gamma_0_to_fp16 = const()[name = string("out_29_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327211072)))]; fp16 var_3163_to_fp16 = const()[name = string("op_3163_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_29_cast_fp16 = layer_norm(axes = out_29_axes_0, epsilon = var_3163_to_fp16, gamma = out_29_gamma_0_to_fp16, x = doubled_57_cast_fp16)[name = string("out_29_cast_fp16")]; tensor var_3174_split_sizes_0 = const()[name = string("op_3174_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3174_axis_0 = const()[name = string("op_3174_axis_0"), val = int32(1)]; tensor var_3174_cast_fp16_0, tensor var_3174_cast_fp16_1 = split(axis = var_3174_axis_0, split_sizes = var_3174_split_sizes_0, x = out_29_cast_fp16)[name = string("op_3174_cast_fp16")]; tensor layers_7_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327219328)))]; tensor query_states_43_strides_0 = const()[name = string("query_states_43_strides_0"), val = tensor([1, 1])]; string query_states_43_pad_type_0 = const()[name = string("query_states_43_pad_type_0"), val = string("valid")]; tensor query_states_43_pad_0 = const()[name = string("query_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_43_dilations_0 = const()[name = string("query_states_43_dilations_0"), val = tensor([1, 1])]; int32 query_states_43_groups_0 = const()[name = string("query_states_43_groups_0"), val = int32(1)]; tensor query_states_43_cast_fp16 = conv(dilations = query_states_43_dilations_0, groups = query_states_43_groups_0, pad = query_states_43_pad_0, pad_type = query_states_43_pad_type_0, strides = query_states_43_strides_0, weight = layers_7_self_attn_q_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("query_states_43_cast_fp16")]; tensor layers_7_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1335608000)))]; tensor key_states_71_strides_0 = const()[name = string("key_states_71_strides_0"), val = tensor([1, 1])]; string key_states_71_pad_type_0 = const()[name = string("key_states_71_pad_type_0"), val = string("valid")]; tensor key_states_71_pad_0 = const()[name = string("key_states_71_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_71_dilations_0 = const()[name = string("key_states_71_dilations_0"), val = tensor([1, 1])]; int32 key_states_71_groups_0 = const()[name = string("key_states_71_groups_0"), val = int32(1)]; tensor key_states_71_cast_fp16 = conv(dilations = key_states_71_dilations_0, groups = key_states_71_groups_0, pad = key_states_71_pad_0, pad_type = key_states_71_pad_type_0, strides = key_states_71_strides_0, weight = layers_7_self_attn_k_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("key_states_71_cast_fp16")]; tensor value_states_43_strides_0 = const()[name = string("value_states_43_strides_0"), val = tensor([1, 1])]; string value_states_43_pad_type_0 = const()[name = string("value_states_43_pad_type_0"), val = string("valid")]; tensor value_states_43_pad_0 = const()[name = string("value_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_43_dilations_0 = const()[name = string("value_states_43_dilations_0"), val = tensor([1, 1])]; int32 value_states_43_groups_0 = const()[name = string("value_states_43_groups_0"), val = int32(1)]; tensor value_states_43_cast_fp16 = conv(dilations = value_states_43_dilations_0, groups = value_states_43_groups_0, pad = value_states_43_pad_0, pad_type = value_states_43_pad_type_0, strides = value_states_43_strides_0, weight = layers_7_self_attn_v_proj_weight_cast_fp16, x = var_3174_cast_fp16_0)[name = string("value_states_43_cast_fp16")]; tensor concat_84x = const()[name = string("concat_84x"), val = tensor([1, 16, 128, -1])]; tensor x_71_cast_fp16 = reshape(shape = concat_84x, x = query_states_43_cast_fp16)[name = string("x_71_cast_fp16")]; tensor concat_85x = const()[name = string("concat_85x"), val = tensor([1, 2, 128, -1])]; tensor var_3231_cast_fp16 = reshape(shape = concat_85x, x = key_states_71_cast_fp16)[name = string("op_3231_cast_fp16")]; tensor concat_86x = const()[name = string("concat_86x"), val = tensor([1, 2, 128, -1])]; tensor var_3238_cast_fp16 = reshape(shape = concat_86x, x = value_states_43_cast_fp16)[name = string("op_3238_cast_fp16")]; tensor var_3242_cast_fp16 = mul(x = x_71_cast_fp16, y = var_869_cast_fp16)[name = string("op_3242_cast_fp16")]; tensor var_3243_split_sizes_0 = const()[name = string("op_3243_split_sizes_0"), val = tensor([64, 64])]; int32 var_3243_axis_0 = const()[name = string("op_3243_axis_0"), val = int32(-2)]; tensor var_3243_cast_fp16_0, tensor var_3243_cast_fp16_1 = split(axis = var_3243_axis_0, split_sizes = var_3243_split_sizes_0, x = x_71_cast_fp16)[name = string("op_3243_cast_fp16")]; fp16 const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3245_cast_fp16 = mul(x = var_3243_cast_fp16_1, y = const_72_promoted_to_fp16)[name = string("op_3245_cast_fp16")]; int32 var_3247 = const()[name = string("op_3247"), val = int32(-2)]; bool var_3248_interleave_0 = const()[name = string("op_3248_interleave_0"), val = bool(false)]; tensor var_3248_cast_fp16 = concat(axis = var_3247, interleave = var_3248_interleave_0, values = (var_3245_cast_fp16, var_3243_cast_fp16_0))[name = string("op_3248_cast_fp16")]; tensor var_3249_cast_fp16 = mul(x = var_3248_cast_fp16, y = var_878_cast_fp16)[name = string("op_3249_cast_fp16")]; tensor query_states_45_cast_fp16 = add(x = var_3242_cast_fp16, y = var_3249_cast_fp16)[name = string("query_states_45_cast_fp16")]; tensor var_3255_cast_fp16 = mul(x = var_3231_cast_fp16, y = var_869_cast_fp16)[name = string("op_3255_cast_fp16")]; tensor var_3256_split_sizes_0 = const()[name = string("op_3256_split_sizes_0"), val = tensor([64, 64])]; int32 var_3256_axis_0 = const()[name = string("op_3256_axis_0"), val = int32(-2)]; tensor var_3256_cast_fp16_0, tensor var_3256_cast_fp16_1 = split(axis = var_3256_axis_0, split_sizes = var_3256_split_sizes_0, x = var_3231_cast_fp16)[name = string("op_3256_cast_fp16")]; fp16 const_73_promoted_to_fp16 = const()[name = string("const_73_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3258_cast_fp16 = mul(x = var_3256_cast_fp16_1, y = const_73_promoted_to_fp16)[name = string("op_3258_cast_fp16")]; int32 var_3260 = const()[name = string("op_3260"), val = int32(-2)]; bool var_3261_interleave_0 = const()[name = string("op_3261_interleave_0"), val = bool(false)]; tensor var_3261_cast_fp16 = concat(axis = var_3260, interleave = var_3261_interleave_0, values = (var_3258_cast_fp16, var_3256_cast_fp16_0))[name = string("op_3261_cast_fp16")]; tensor var_3262_cast_fp16 = mul(x = var_3261_cast_fp16, y = var_878_cast_fp16)[name = string("op_3262_cast_fp16")]; tensor key_states_75_cast_fp16 = add(x = var_3255_cast_fp16, y = var_3262_cast_fp16)[name = string("key_states_75_cast_fp16")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; int32 concat_89_axis_0 = const()[name = string("concat_89_axis_0"), val = int32(0)]; bool concat_89_interleave_0 = const()[name = string("concat_89_interleave_0"), val = bool(false)]; tensor concat_89 = concat(axis = concat_89_axis_0, interleave = concat_89_interleave_0, values = (expand_dims_84, expand_dims_85, position_id, expand_dims_87))[name = string("concat_89")]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; tensor concat_90_values1_0 = const()[name = string("concat_90_values1_0"), val = tensor([0])]; tensor concat_90_values3_0 = const()[name = string("concat_90_values3_0"), val = tensor([0])]; int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_88, concat_90_values1_0, cache_position_end, concat_90_values3_0))[name = string("concat_90")]; tensor key_states_77_perm_0 = const()[name = string("key_states_77_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_8_stride_0 = const()[name = string("key_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_77_cast_fp16 = transpose(perm = key_states_77_perm_0, x = key_states_75_cast_fp16)[name = string("transpose_578")]; tensor key_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = key_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = key_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_8_squeeze_mask_0, stride = key_cache_internal_tensor_assign_8_stride_0, update = key_states_77_cast_fp16, x = coreml_update_state_348)[name = string("key_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_8_cast_fp16, input = key_cache)[name = string("coreml_update_state_350_write_state")]; tensor coreml_update_state_350 = read_state(input = key_cache)[name = string("coreml_update_state_350")]; tensor value_states_45_perm_0 = const()[name = string("value_states_45_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_8_stride_0 = const()[name = string("value_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_45_cast_fp16 = transpose(perm = value_states_45_perm_0, x = var_3238_cast_fp16)[name = string("transpose_577")]; tensor value_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = value_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = value_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_8_squeeze_mask_0, stride = value_cache_internal_tensor_assign_8_stride_0, update = value_states_45_cast_fp16, x = coreml_update_state_349)[name = string("value_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_8_cast_fp16, input = value_cache)[name = string("coreml_update_state_351_write_state")]; tensor coreml_update_state_351 = read_state(input = value_cache)[name = string("coreml_update_state_351")]; tensor var_3332_begin_0 = const()[name = string("op_3332_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3332_end_0 = const()[name = string("op_3332_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3332_end_mask_0 = const()[name = string("op_3332_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3332_cast_fp16 = slice_by_index(begin = var_3332_begin_0, end = var_3332_end_0, end_mask = var_3332_end_mask_0, x = coreml_update_state_350)[name = string("op_3332_cast_fp16")]; tensor tile_14 = const()[name = string("tile_14"), val = tensor([1, 1])]; int32 var_3335_axis_0 = const()[name = string("op_3335_axis_0"), val = int32(1)]; tensor var_3335_cast_fp16_0, tensor var_3335_cast_fp16_1 = split(axis = var_3335_axis_0, split_sizes = tile_14, x = var_3332_cast_fp16)[name = string("op_3335_cast_fp16")]; tensor var_3342_begin_0 = const()[name = string("op_3342_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3342_end_0 = const()[name = string("op_3342_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3342_end_mask_0 = const()[name = string("op_3342_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3342_cast_fp16 = slice_by_index(begin = var_3342_begin_0, end = var_3342_end_0, end_mask = var_3342_end_mask_0, x = coreml_update_state_351)[name = string("op_3342_cast_fp16")]; tensor tile_15 = const()[name = string("tile_15"), val = tensor([1, 1])]; int32 var_3345_axis_0 = const()[name = string("op_3345_axis_0"), val = int32(1)]; tensor var_3345_cast_fp16_0, tensor var_3345_cast_fp16_1 = split(axis = var_3345_axis_0, split_sizes = tile_15, x = var_3342_cast_fp16)[name = string("op_3345_cast_fp16")]; tensor var_3348_split_sizes_0 = const()[name = string("op_3348_split_sizes_0"), val = tensor([8, 8])]; int32 var_3348_axis_0 = const()[name = string("op_3348_axis_0"), val = int32(1)]; tensor var_3348_0, tensor var_3348_1 = split(axis = var_3348_axis_0, split_sizes = var_3348_split_sizes_0, x = query_states_45_cast_fp16)[name = string("op_3348")]; bool attn_weights_113_transpose_x_0 = const()[name = string("attn_weights_113_transpose_x_0"), val = bool(false)]; bool attn_weights_113_transpose_y_0 = const()[name = string("attn_weights_113_transpose_y_0"), val = bool(false)]; tensor attn_weights_113_cast_fp16 = matmul(transpose_x = attn_weights_113_transpose_x_0, transpose_y = attn_weights_113_transpose_y_0, x = var_3335_cast_fp16_0, y = var_3348_0)[name = string("attn_weights_113_cast_fp16")]; fp16 var_3351_to_fp16 = const()[name = string("op_3351_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_115_cast_fp16 = mul(x = attn_weights_113_cast_fp16, y = var_3351_to_fp16)[name = string("attn_weights_115_cast_fp16")]; tensor attn_weights_117_cast_fp16 = add(x = attn_weights_115_cast_fp16, y = attn_mask_1)[name = string("attn_weights_117_cast_fp16")]; int32 var_3355 = const()[name = string("op_3355"), val = int32(-2)]; tensor attn_weights_119_cast_fp16 = softmax(axis = var_3355, x = attn_weights_117_cast_fp16)[name = string("attn_weights_119_cast_fp16")]; bool var_3361_transpose_x_1 = const()[name = string("op_3361_transpose_x_1"), val = bool(true)]; bool var_3361_transpose_y_1 = const()[name = string("op_3361_transpose_y_1"), val = bool(false)]; tensor var_3361_cast_fp16 = matmul(transpose_x = var_3361_transpose_x_1, transpose_y = var_3361_transpose_y_1, x = attn_weights_119_cast_fp16, y = var_3345_cast_fp16_0)[name = string("op_3361_cast_fp16")]; bool attn_weights_121_transpose_x_0 = const()[name = string("attn_weights_121_transpose_x_0"), val = bool(false)]; bool attn_weights_121_transpose_y_0 = const()[name = string("attn_weights_121_transpose_y_0"), val = bool(false)]; tensor attn_weights_121_cast_fp16 = matmul(transpose_x = attn_weights_121_transpose_x_0, transpose_y = attn_weights_121_transpose_y_0, x = var_3335_cast_fp16_1, y = var_3348_1)[name = string("attn_weights_121_cast_fp16")]; fp16 var_3363_to_fp16 = const()[name = string("op_3363_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_123_cast_fp16 = mul(x = attn_weights_121_cast_fp16, y = var_3363_to_fp16)[name = string("attn_weights_123_cast_fp16")]; tensor attn_weights_125_cast_fp16 = add(x = attn_weights_123_cast_fp16, y = attn_mask_1)[name = string("attn_weights_125_cast_fp16")]; int32 var_3367 = const()[name = string("op_3367"), val = int32(-2)]; tensor attn_weights_127_cast_fp16 = softmax(axis = var_3367, x = attn_weights_125_cast_fp16)[name = string("attn_weights_127_cast_fp16")]; bool attn_output_57_transpose_x_1 = const()[name = string("attn_output_57_transpose_x_1"), val = bool(true)]; bool attn_output_57_transpose_y_1 = const()[name = string("attn_output_57_transpose_y_1"), val = bool(false)]; tensor attn_output_57_cast_fp16 = matmul(transpose_x = attn_output_57_transpose_x_1, transpose_y = attn_output_57_transpose_y_1, x = attn_weights_127_cast_fp16, y = var_3345_cast_fp16_1)[name = string("attn_output_57_cast_fp16")]; int32 var_3375 = const()[name = string("op_3375"), val = int32(1)]; bool attn_output_59_interleave_0 = const()[name = string("attn_output_59_interleave_0"), val = bool(false)]; tensor attn_output_59_cast_fp16 = concat(axis = var_3375, interleave = attn_output_59_interleave_0, values = (var_3361_cast_fp16, attn_output_57_cast_fp16))[name = string("attn_output_59_cast_fp16")]; tensor var_3379_perm_0 = const()[name = string("op_3379_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_95x = const()[name = string("concat_95x"), val = tensor([1, 2048, 1, -1])]; tensor var_3379_cast_fp16 = transpose(perm = var_3379_perm_0, x = attn_output_59_cast_fp16)[name = string("transpose_576")]; tensor attn_output_63_cast_fp16 = reshape(shape = concat_95x, x = var_3379_cast_fp16)[name = string("attn_output_63_cast_fp16")]; tensor hidden_states_73_strides_0 = const()[name = string("hidden_states_73_strides_0"), val = tensor([1, 1])]; string hidden_states_73_pad_type_0 = const()[name = string("hidden_states_73_pad_type_0"), val = string("valid")]; tensor hidden_states_73_pad_0 = const()[name = string("hidden_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_73_dilations_0 = const()[name = string("hidden_states_73_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_73_groups_0 = const()[name = string("hidden_states_73_groups_0"), val = int32(1)]; tensor hidden_states_73_cast_fp16 = conv(dilations = hidden_states_73_dilations_0, groups = hidden_states_73_groups_0, pad = hidden_states_73_pad_0, pad_type = hidden_states_73_pad_type_0, strides = hidden_states_73_strides_0, weight = layers_7_self_attn_o_proj_weight_cast_fp16, x = attn_output_63_cast_fp16)[name = string("hidden_states_73_cast_fp16")]; tensor hidden_states_75_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = hidden_states_73_cast_fp16)[name = string("hidden_states_75_cast_fp16")]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3412_cast_fp16 = mul(x = hidden_states_75_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_3412_cast_fp16")]; int32 var_3410 = const()[name = string("op_3410"), val = int32(1)]; bool doubled_61_interleave_0 = const()[name = string("doubled_61_interleave_0"), val = bool(false)]; tensor doubled_61_cast_fp16 = concat(axis = var_3410, interleave = doubled_61_interleave_0, values = (hidden_states_75_cast_fp16, var_3412_cast_fp16))[name = string("doubled_61_cast_fp16")]; tensor out_31_axes_0 = const()[name = string("out_31_axes_0"), val = tensor([1])]; tensor out_31_gamma_0_to_fp16 = const()[name = string("out_31_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336656640)))]; fp16 var_3422_to_fp16 = const()[name = string("op_3422_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_31_cast_fp16 = layer_norm(axes = out_31_axes_0, epsilon = var_3422_to_fp16, gamma = out_31_gamma_0_to_fp16, x = doubled_61_cast_fp16)[name = string("out_31_cast_fp16")]; tensor var_3433_split_sizes_0 = const()[name = string("op_3433_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3433_axis_0 = const()[name = string("op_3433_axis_0"), val = int32(1)]; tensor var_3433_cast_fp16_0, tensor var_3433_cast_fp16_1 = split(axis = var_3433_axis_0, split_sizes = var_3433_split_sizes_0, x = out_31_cast_fp16)[name = string("op_3433_cast_fp16")]; tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; tensor input_15_cast_fp16 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = layers_7_mlp_gate_proj_weight_cast_fp16, x = var_3433_cast_fp16_0)[name = string("input_15_cast_fp16")]; tensor var_3450_cast_fp16 = silu(x = input_15_cast_fp16)[name = string("op_3450_cast_fp16")]; tensor layers_7_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336664896)))]; tensor var_3456_strides_0 = const()[name = string("op_3456_strides_0"), val = tensor([1, 1])]; string var_3456_pad_type_0 = const()[name = string("op_3456_pad_type_0"), val = string("valid")]; tensor var_3456_pad_0 = const()[name = string("op_3456_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3456_dilations_0 = const()[name = string("op_3456_dilations_0"), val = tensor([1, 1])]; int32 var_3456_groups_0 = const()[name = string("op_3456_groups_0"), val = int32(1)]; tensor var_3456_cast_fp16 = conv(dilations = var_3456_dilations_0, groups = var_3456_groups_0, pad = var_3456_pad_0, pad_type = var_3456_pad_type_0, strides = var_3456_strides_0, weight = layers_7_mlp_up_proj_weight_to_fp16, x = var_3433_cast_fp16_0)[name = string("op_3456_cast_fp16")]; tensor x_79_cast_fp16 = mul(x = var_3450_cast_fp16, y = var_3456_cast_fp16)[name = string("x_79_cast_fp16")]; tensor layers_7_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1361830784)))]; tensor hidden_states_77_strides_0 = const()[name = string("hidden_states_77_strides_0"), val = tensor([1, 1])]; string hidden_states_77_pad_type_0 = const()[name = string("hidden_states_77_pad_type_0"), val = string("valid")]; tensor hidden_states_77_pad_0 = const()[name = string("hidden_states_77_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_77_dilations_0 = const()[name = string("hidden_states_77_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_77_groups_0 = const()[name = string("hidden_states_77_groups_0"), val = int32(1)]; tensor hidden_states_77_cast_fp16 = conv(dilations = hidden_states_77_dilations_0, groups = hidden_states_77_groups_0, pad = hidden_states_77_pad_0, pad_type = hidden_states_77_pad_type_0, strides = hidden_states_77_strides_0, weight = layers_7_mlp_down_proj_weight_to_fp16, x = x_79_cast_fp16)[name = string("hidden_states_77_cast_fp16")]; tensor hidden_states_79_cast_fp16 = add(x = hidden_states_75_cast_fp16, y = hidden_states_77_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3474_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_80_promoted_to_fp16)[name = string("op_3474_cast_fp16")]; int32 var_3472 = const()[name = string("op_3472"), val = int32(1)]; bool doubled_65_interleave_0 = const()[name = string("doubled_65_interleave_0"), val = bool(false)]; tensor doubled_65_cast_fp16 = concat(axis = var_3472, interleave = doubled_65_interleave_0, values = (hidden_states_79_cast_fp16, var_3474_cast_fp16))[name = string("doubled_65_cast_fp16")]; tensor out_33_axes_0 = const()[name = string("out_33_axes_0"), val = tensor([1])]; tensor out_33_gamma_0_to_fp16 = const()[name = string("out_33_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1386996672)))]; fp16 var_3484_to_fp16 = const()[name = string("op_3484_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_33_cast_fp16 = layer_norm(axes = out_33_axes_0, epsilon = var_3484_to_fp16, gamma = out_33_gamma_0_to_fp16, x = doubled_65_cast_fp16)[name = string("out_33_cast_fp16")]; tensor var_3495_split_sizes_0 = const()[name = string("op_3495_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3495_axis_0 = const()[name = string("op_3495_axis_0"), val = int32(1)]; tensor var_3495_cast_fp16_0, tensor var_3495_cast_fp16_1 = split(axis = var_3495_axis_0, split_sizes = var_3495_split_sizes_0, x = out_33_cast_fp16)[name = string("op_3495_cast_fp16")]; tensor layers_8_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1387004928)))]; tensor query_states_49_strides_0 = const()[name = string("query_states_49_strides_0"), val = tensor([1, 1])]; string query_states_49_pad_type_0 = const()[name = string("query_states_49_pad_type_0"), val = string("valid")]; tensor query_states_49_pad_0 = const()[name = string("query_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_49_dilations_0 = const()[name = string("query_states_49_dilations_0"), val = tensor([1, 1])]; int32 query_states_49_groups_0 = const()[name = string("query_states_49_groups_0"), val = int32(1)]; tensor query_states_49_cast_fp16 = conv(dilations = query_states_49_dilations_0, groups = query_states_49_groups_0, pad = query_states_49_pad_0, pad_type = query_states_49_pad_type_0, strides = query_states_49_strides_0, weight = layers_8_self_attn_q_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("query_states_49_cast_fp16")]; tensor layers_8_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1395393600)))]; tensor key_states_81_strides_0 = const()[name = string("key_states_81_strides_0"), val = tensor([1, 1])]; string key_states_81_pad_type_0 = const()[name = string("key_states_81_pad_type_0"), val = string("valid")]; tensor key_states_81_pad_0 = const()[name = string("key_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_81_dilations_0 = const()[name = string("key_states_81_dilations_0"), val = tensor([1, 1])]; int32 key_states_81_groups_0 = const()[name = string("key_states_81_groups_0"), val = int32(1)]; tensor key_states_81_cast_fp16 = conv(dilations = key_states_81_dilations_0, groups = key_states_81_groups_0, pad = key_states_81_pad_0, pad_type = key_states_81_pad_type_0, strides = key_states_81_strides_0, weight = layers_8_self_attn_k_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("key_states_81_cast_fp16")]; tensor value_states_49_strides_0 = const()[name = string("value_states_49_strides_0"), val = tensor([1, 1])]; string value_states_49_pad_type_0 = const()[name = string("value_states_49_pad_type_0"), val = string("valid")]; tensor value_states_49_pad_0 = const()[name = string("value_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_49_dilations_0 = const()[name = string("value_states_49_dilations_0"), val = tensor([1, 1])]; int32 value_states_49_groups_0 = const()[name = string("value_states_49_groups_0"), val = int32(1)]; tensor value_states_49_cast_fp16 = conv(dilations = value_states_49_dilations_0, groups = value_states_49_groups_0, pad = value_states_49_pad_0, pad_type = value_states_49_pad_type_0, strides = value_states_49_strides_0, weight = layers_8_self_attn_v_proj_weight_cast_fp16, x = var_3495_cast_fp16_0)[name = string("value_states_49_cast_fp16")]; tensor concat_96x = const()[name = string("concat_96x"), val = tensor([1, 16, 128, -1])]; tensor x_81_cast_fp16 = reshape(shape = concat_96x, x = query_states_49_cast_fp16)[name = string("x_81_cast_fp16")]; tensor concat_97x = const()[name = string("concat_97x"), val = tensor([1, 2, 128, -1])]; tensor var_3552_cast_fp16 = reshape(shape = concat_97x, x = key_states_81_cast_fp16)[name = string("op_3552_cast_fp16")]; tensor concat_98x = const()[name = string("concat_98x"), val = tensor([1, 2, 128, -1])]; tensor var_3559_cast_fp16 = reshape(shape = concat_98x, x = value_states_49_cast_fp16)[name = string("op_3559_cast_fp16")]; tensor var_3563_cast_fp16 = mul(x = x_81_cast_fp16, y = var_869_cast_fp16)[name = string("op_3563_cast_fp16")]; tensor var_3564_split_sizes_0 = const()[name = string("op_3564_split_sizes_0"), val = tensor([64, 64])]; int32 var_3564_axis_0 = const()[name = string("op_3564_axis_0"), val = int32(-2)]; tensor var_3564_cast_fp16_0, tensor var_3564_cast_fp16_1 = split(axis = var_3564_axis_0, split_sizes = var_3564_split_sizes_0, x = x_81_cast_fp16)[name = string("op_3564_cast_fp16")]; fp16 const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3566_cast_fp16 = mul(x = var_3564_cast_fp16_1, y = const_82_promoted_to_fp16)[name = string("op_3566_cast_fp16")]; int32 var_3568 = const()[name = string("op_3568"), val = int32(-2)]; bool var_3569_interleave_0 = const()[name = string("op_3569_interleave_0"), val = bool(false)]; tensor var_3569_cast_fp16 = concat(axis = var_3568, interleave = var_3569_interleave_0, values = (var_3566_cast_fp16, var_3564_cast_fp16_0))[name = string("op_3569_cast_fp16")]; tensor var_3570_cast_fp16 = mul(x = var_3569_cast_fp16, y = var_878_cast_fp16)[name = string("op_3570_cast_fp16")]; tensor query_states_51_cast_fp16 = add(x = var_3563_cast_fp16, y = var_3570_cast_fp16)[name = string("query_states_51_cast_fp16")]; tensor var_3576_cast_fp16 = mul(x = var_3552_cast_fp16, y = var_869_cast_fp16)[name = string("op_3576_cast_fp16")]; tensor var_3577_split_sizes_0 = const()[name = string("op_3577_split_sizes_0"), val = tensor([64, 64])]; int32 var_3577_axis_0 = const()[name = string("op_3577_axis_0"), val = int32(-2)]; tensor var_3577_cast_fp16_0, tensor var_3577_cast_fp16_1 = split(axis = var_3577_axis_0, split_sizes = var_3577_split_sizes_0, x = var_3552_cast_fp16)[name = string("op_3577_cast_fp16")]; fp16 const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3579_cast_fp16 = mul(x = var_3577_cast_fp16_1, y = const_83_promoted_to_fp16)[name = string("op_3579_cast_fp16")]; int32 var_3581 = const()[name = string("op_3581"), val = int32(-2)]; bool var_3582_interleave_0 = const()[name = string("op_3582_interleave_0"), val = bool(false)]; tensor var_3582_cast_fp16 = concat(axis = var_3581, interleave = var_3582_interleave_0, values = (var_3579_cast_fp16, var_3577_cast_fp16_0))[name = string("op_3582_cast_fp16")]; tensor var_3583_cast_fp16 = mul(x = var_3582_cast_fp16, y = var_878_cast_fp16)[name = string("op_3583_cast_fp16")]; tensor key_states_85_cast_fp16 = add(x = var_3576_cast_fp16, y = var_3583_cast_fp16)[name = string("key_states_85_cast_fp16")]; tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; int32 concat_101_axis_0 = const()[name = string("concat_101_axis_0"), val = int32(0)]; bool concat_101_interleave_0 = const()[name = string("concat_101_interleave_0"), val = bool(false)]; tensor concat_101 = concat(axis = concat_101_axis_0, interleave = concat_101_interleave_0, values = (expand_dims_96, expand_dims_97, position_id, expand_dims_99))[name = string("concat_101")]; tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; tensor concat_102_values1_0 = const()[name = string("concat_102_values1_0"), val = tensor([0])]; tensor concat_102_values3_0 = const()[name = string("concat_102_values3_0"), val = tensor([0])]; int32 concat_102_axis_0 = const()[name = string("concat_102_axis_0"), val = int32(0)]; bool concat_102_interleave_0 = const()[name = string("concat_102_interleave_0"), val = bool(false)]; tensor concat_102 = concat(axis = concat_102_axis_0, interleave = concat_102_interleave_0, values = (expand_dims_100, concat_102_values1_0, cache_position_end, concat_102_values3_0))[name = string("concat_102")]; tensor key_states_87_perm_0 = const()[name = string("key_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_9_stride_0 = const()[name = string("key_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_87_cast_fp16 = transpose(perm = key_states_87_perm_0, x = key_states_85_cast_fp16)[name = string("transpose_575")]; tensor key_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = key_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = key_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_9_squeeze_mask_0, stride = key_cache_internal_tensor_assign_9_stride_0, update = key_states_87_cast_fp16, x = coreml_update_state_350)[name = string("key_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_9_cast_fp16, input = key_cache)[name = string("coreml_update_state_352_write_state")]; tensor coreml_update_state_352 = read_state(input = key_cache)[name = string("coreml_update_state_352")]; tensor value_states_51_perm_0 = const()[name = string("value_states_51_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_9_stride_0 = const()[name = string("value_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_51_cast_fp16 = transpose(perm = value_states_51_perm_0, x = var_3559_cast_fp16)[name = string("transpose_574")]; tensor value_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = value_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = value_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_9_squeeze_mask_0, stride = value_cache_internal_tensor_assign_9_stride_0, update = value_states_51_cast_fp16, x = coreml_update_state_351)[name = string("value_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_9_cast_fp16, input = value_cache)[name = string("coreml_update_state_353_write_state")]; tensor coreml_update_state_353 = read_state(input = value_cache)[name = string("coreml_update_state_353")]; tensor var_3653_begin_0 = const()[name = string("op_3653_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3653_end_0 = const()[name = string("op_3653_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3653_end_mask_0 = const()[name = string("op_3653_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3653_cast_fp16 = slice_by_index(begin = var_3653_begin_0, end = var_3653_end_0, end_mask = var_3653_end_mask_0, x = coreml_update_state_352)[name = string("op_3653_cast_fp16")]; tensor tile_16 = const()[name = string("tile_16"), val = tensor([1, 1])]; int32 var_3656_axis_0 = const()[name = string("op_3656_axis_0"), val = int32(1)]; tensor var_3656_cast_fp16_0, tensor var_3656_cast_fp16_1 = split(axis = var_3656_axis_0, split_sizes = tile_16, x = var_3653_cast_fp16)[name = string("op_3656_cast_fp16")]; tensor var_3663_begin_0 = const()[name = string("op_3663_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3663_end_0 = const()[name = string("op_3663_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3663_end_mask_0 = const()[name = string("op_3663_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3663_cast_fp16 = slice_by_index(begin = var_3663_begin_0, end = var_3663_end_0, end_mask = var_3663_end_mask_0, x = coreml_update_state_353)[name = string("op_3663_cast_fp16")]; tensor tile_17 = const()[name = string("tile_17"), val = tensor([1, 1])]; int32 var_3666_axis_0 = const()[name = string("op_3666_axis_0"), val = int32(1)]; tensor var_3666_cast_fp16_0, tensor var_3666_cast_fp16_1 = split(axis = var_3666_axis_0, split_sizes = tile_17, x = var_3663_cast_fp16)[name = string("op_3666_cast_fp16")]; tensor var_3669_split_sizes_0 = const()[name = string("op_3669_split_sizes_0"), val = tensor([8, 8])]; int32 var_3669_axis_0 = const()[name = string("op_3669_axis_0"), val = int32(1)]; tensor var_3669_0, tensor var_3669_1 = split(axis = var_3669_axis_0, split_sizes = var_3669_split_sizes_0, x = query_states_51_cast_fp16)[name = string("op_3669")]; bool attn_weights_129_transpose_x_0 = const()[name = string("attn_weights_129_transpose_x_0"), val = bool(false)]; bool attn_weights_129_transpose_y_0 = const()[name = string("attn_weights_129_transpose_y_0"), val = bool(false)]; tensor attn_weights_129_cast_fp16 = matmul(transpose_x = attn_weights_129_transpose_x_0, transpose_y = attn_weights_129_transpose_y_0, x = var_3656_cast_fp16_0, y = var_3669_0)[name = string("attn_weights_129_cast_fp16")]; fp16 var_3672_to_fp16 = const()[name = string("op_3672_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_131_cast_fp16 = mul(x = attn_weights_129_cast_fp16, y = var_3672_to_fp16)[name = string("attn_weights_131_cast_fp16")]; tensor attn_weights_133_cast_fp16 = add(x = attn_weights_131_cast_fp16, y = attn_mask_1)[name = string("attn_weights_133_cast_fp16")]; int32 var_3676 = const()[name = string("op_3676"), val = int32(-2)]; tensor attn_weights_135_cast_fp16 = softmax(axis = var_3676, x = attn_weights_133_cast_fp16)[name = string("attn_weights_135_cast_fp16")]; bool var_3682_transpose_x_1 = const()[name = string("op_3682_transpose_x_1"), val = bool(true)]; bool var_3682_transpose_y_1 = const()[name = string("op_3682_transpose_y_1"), val = bool(false)]; tensor var_3682_cast_fp16 = matmul(transpose_x = var_3682_transpose_x_1, transpose_y = var_3682_transpose_y_1, x = attn_weights_135_cast_fp16, y = var_3666_cast_fp16_0)[name = string("op_3682_cast_fp16")]; bool attn_weights_137_transpose_x_0 = const()[name = string("attn_weights_137_transpose_x_0"), val = bool(false)]; bool attn_weights_137_transpose_y_0 = const()[name = string("attn_weights_137_transpose_y_0"), val = bool(false)]; tensor attn_weights_137_cast_fp16 = matmul(transpose_x = attn_weights_137_transpose_x_0, transpose_y = attn_weights_137_transpose_y_0, x = var_3656_cast_fp16_1, y = var_3669_1)[name = string("attn_weights_137_cast_fp16")]; fp16 var_3684_to_fp16 = const()[name = string("op_3684_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_139_cast_fp16 = mul(x = attn_weights_137_cast_fp16, y = var_3684_to_fp16)[name = string("attn_weights_139_cast_fp16")]; tensor attn_weights_141_cast_fp16 = add(x = attn_weights_139_cast_fp16, y = attn_mask_1)[name = string("attn_weights_141_cast_fp16")]; int32 var_3688 = const()[name = string("op_3688"), val = int32(-2)]; tensor attn_weights_143_cast_fp16 = softmax(axis = var_3688, x = attn_weights_141_cast_fp16)[name = string("attn_weights_143_cast_fp16")]; bool attn_output_65_transpose_x_1 = const()[name = string("attn_output_65_transpose_x_1"), val = bool(true)]; bool attn_output_65_transpose_y_1 = const()[name = string("attn_output_65_transpose_y_1"), val = bool(false)]; tensor attn_output_65_cast_fp16 = matmul(transpose_x = attn_output_65_transpose_x_1, transpose_y = attn_output_65_transpose_y_1, x = attn_weights_143_cast_fp16, y = var_3666_cast_fp16_1)[name = string("attn_output_65_cast_fp16")]; int32 var_3696 = const()[name = string("op_3696"), val = int32(1)]; bool attn_output_67_interleave_0 = const()[name = string("attn_output_67_interleave_0"), val = bool(false)]; tensor attn_output_67_cast_fp16 = concat(axis = var_3696, interleave = attn_output_67_interleave_0, values = (var_3682_cast_fp16, attn_output_65_cast_fp16))[name = string("attn_output_67_cast_fp16")]; tensor var_3700_perm_0 = const()[name = string("op_3700_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_107x = const()[name = string("concat_107x"), val = tensor([1, 2048, 1, -1])]; tensor var_3700_cast_fp16 = transpose(perm = var_3700_perm_0, x = attn_output_67_cast_fp16)[name = string("transpose_573")]; tensor attn_output_71_cast_fp16 = reshape(shape = concat_107x, x = var_3700_cast_fp16)[name = string("attn_output_71_cast_fp16")]; tensor hidden_states_83_strides_0 = const()[name = string("hidden_states_83_strides_0"), val = tensor([1, 1])]; string hidden_states_83_pad_type_0 = const()[name = string("hidden_states_83_pad_type_0"), val = string("valid")]; tensor hidden_states_83_pad_0 = const()[name = string("hidden_states_83_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_83_dilations_0 = const()[name = string("hidden_states_83_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_83_groups_0 = const()[name = string("hidden_states_83_groups_0"), val = int32(1)]; tensor hidden_states_83_cast_fp16 = conv(dilations = hidden_states_83_dilations_0, groups = hidden_states_83_groups_0, pad = hidden_states_83_pad_0, pad_type = hidden_states_83_pad_type_0, strides = hidden_states_83_strides_0, weight = layers_8_self_attn_o_proj_weight_cast_fp16, x = attn_output_71_cast_fp16)[name = string("hidden_states_83_cast_fp16")]; tensor hidden_states_85_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = hidden_states_83_cast_fp16)[name = string("hidden_states_85_cast_fp16")]; fp16 const_88_promoted_to_fp16 = const()[name = string("const_88_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3733_cast_fp16 = mul(x = hidden_states_85_cast_fp16, y = const_88_promoted_to_fp16)[name = string("op_3733_cast_fp16")]; int32 var_3731 = const()[name = string("op_3731"), val = int32(1)]; bool doubled_69_interleave_0 = const()[name = string("doubled_69_interleave_0"), val = bool(false)]; tensor doubled_69_cast_fp16 = concat(axis = var_3731, interleave = doubled_69_interleave_0, values = (hidden_states_85_cast_fp16, var_3733_cast_fp16))[name = string("doubled_69_cast_fp16")]; tensor out_35_axes_0 = const()[name = string("out_35_axes_0"), val = tensor([1])]; tensor out_35_gamma_0_to_fp16 = const()[name = string("out_35_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396442240)))]; fp16 var_3743_to_fp16 = const()[name = string("op_3743_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_35_cast_fp16 = layer_norm(axes = out_35_axes_0, epsilon = var_3743_to_fp16, gamma = out_35_gamma_0_to_fp16, x = doubled_69_cast_fp16)[name = string("out_35_cast_fp16")]; tensor var_3754_split_sizes_0 = const()[name = string("op_3754_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3754_axis_0 = const()[name = string("op_3754_axis_0"), val = int32(1)]; tensor var_3754_cast_fp16_0, tensor var_3754_cast_fp16_1 = split(axis = var_3754_axis_0, split_sizes = var_3754_split_sizes_0, x = out_35_cast_fp16)[name = string("op_3754_cast_fp16")]; tensor input_17_strides_0 = const()[name = string("input_17_strides_0"), val = tensor([1, 1])]; string input_17_pad_type_0 = const()[name = string("input_17_pad_type_0"), val = string("valid")]; tensor input_17_pad_0 = const()[name = string("input_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_17_dilations_0 = const()[name = string("input_17_dilations_0"), val = tensor([1, 1])]; int32 input_17_groups_0 = const()[name = string("input_17_groups_0"), val = int32(1)]; tensor input_17_cast_fp16 = conv(dilations = input_17_dilations_0, groups = input_17_groups_0, pad = input_17_pad_0, pad_type = input_17_pad_type_0, strides = input_17_strides_0, weight = layers_8_mlp_gate_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("input_17_cast_fp16")]; tensor var_3771_cast_fp16 = silu(x = input_17_cast_fp16)[name = string("op_3771_cast_fp16")]; tensor var_3777_strides_0 = const()[name = string("op_3777_strides_0"), val = tensor([1, 1])]; string var_3777_pad_type_0 = const()[name = string("op_3777_pad_type_0"), val = string("valid")]; tensor var_3777_pad_0 = const()[name = string("op_3777_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3777_dilations_0 = const()[name = string("op_3777_dilations_0"), val = tensor([1, 1])]; int32 var_3777_groups_0 = const()[name = string("op_3777_groups_0"), val = int32(1)]; tensor var_3777_cast_fp16 = conv(dilations = var_3777_dilations_0, groups = var_3777_groups_0, pad = var_3777_pad_0, pad_type = var_3777_pad_type_0, strides = var_3777_strides_0, weight = layers_8_mlp_up_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("op_3777_cast_fp16")]; tensor x_89_cast_fp16 = mul(x = var_3771_cast_fp16, y = var_3777_cast_fp16)[name = string("x_89_cast_fp16")]; tensor hidden_states_87_strides_0 = const()[name = string("hidden_states_87_strides_0"), val = tensor([1, 1])]; string hidden_states_87_pad_type_0 = const()[name = string("hidden_states_87_pad_type_0"), val = string("valid")]; tensor hidden_states_87_pad_0 = const()[name = string("hidden_states_87_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_87_dilations_0 = const()[name = string("hidden_states_87_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_87_groups_0 = const()[name = string("hidden_states_87_groups_0"), val = int32(1)]; tensor hidden_states_87_cast_fp16 = conv(dilations = hidden_states_87_dilations_0, groups = hidden_states_87_groups_0, pad = hidden_states_87_pad_0, pad_type = hidden_states_87_pad_type_0, strides = hidden_states_87_strides_0, weight = layers_8_mlp_down_proj_weight_cast_fp16, x = x_89_cast_fp16)[name = string("hidden_states_87_cast_fp16")]; tensor hidden_states_89_cast_fp16 = add(x = hidden_states_85_cast_fp16, y = hidden_states_87_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3795_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_3795_cast_fp16")]; int32 var_3793 = const()[name = string("op_3793"), val = int32(1)]; bool doubled_73_interleave_0 = const()[name = string("doubled_73_interleave_0"), val = bool(false)]; tensor doubled_73_cast_fp16 = concat(axis = var_3793, interleave = doubled_73_interleave_0, values = (hidden_states_89_cast_fp16, var_3795_cast_fp16))[name = string("doubled_73_cast_fp16")]; tensor out_37_axes_0 = const()[name = string("out_37_axes_0"), val = tensor([1])]; tensor out_37_gamma_0_to_fp16 = const()[name = string("out_37_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396450496)))]; fp16 var_3805_to_fp16 = const()[name = string("op_3805_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_37_cast_fp16 = layer_norm(axes = out_37_axes_0, epsilon = var_3805_to_fp16, gamma = out_37_gamma_0_to_fp16, x = doubled_73_cast_fp16)[name = string("out_37_cast_fp16")]; tensor var_3816_split_sizes_0 = const()[name = string("op_3816_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3816_axis_0 = const()[name = string("op_3816_axis_0"), val = int32(1)]; tensor var_3816_cast_fp16_0, tensor var_3816_cast_fp16_1 = split(axis = var_3816_axis_0, split_sizes = var_3816_split_sizes_0, x = out_37_cast_fp16)[name = string("op_3816_cast_fp16")]; tensor layers_9_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396458752)))]; tensor query_states_55_strides_0 = const()[name = string("query_states_55_strides_0"), val = tensor([1, 1])]; string query_states_55_pad_type_0 = const()[name = string("query_states_55_pad_type_0"), val = string("valid")]; tensor query_states_55_pad_0 = const()[name = string("query_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_55_dilations_0 = const()[name = string("query_states_55_dilations_0"), val = tensor([1, 1])]; int32 query_states_55_groups_0 = const()[name = string("query_states_55_groups_0"), val = int32(1)]; tensor query_states_55_cast_fp16 = conv(dilations = query_states_55_dilations_0, groups = query_states_55_groups_0, pad = query_states_55_pad_0, pad_type = query_states_55_pad_type_0, strides = query_states_55_strides_0, weight = layers_9_self_attn_q_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("query_states_55_cast_fp16")]; tensor layers_9_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1404847424)))]; tensor key_states_91_strides_0 = const()[name = string("key_states_91_strides_0"), val = tensor([1, 1])]; string key_states_91_pad_type_0 = const()[name = string("key_states_91_pad_type_0"), val = string("valid")]; tensor key_states_91_pad_0 = const()[name = string("key_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_91_dilations_0 = const()[name = string("key_states_91_dilations_0"), val = tensor([1, 1])]; int32 key_states_91_groups_0 = const()[name = string("key_states_91_groups_0"), val = int32(1)]; tensor key_states_91_cast_fp16 = conv(dilations = key_states_91_dilations_0, groups = key_states_91_groups_0, pad = key_states_91_pad_0, pad_type = key_states_91_pad_type_0, strides = key_states_91_strides_0, weight = layers_9_self_attn_k_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("key_states_91_cast_fp16")]; tensor value_states_55_strides_0 = const()[name = string("value_states_55_strides_0"), val = tensor([1, 1])]; string value_states_55_pad_type_0 = const()[name = string("value_states_55_pad_type_0"), val = string("valid")]; tensor value_states_55_pad_0 = const()[name = string("value_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_55_dilations_0 = const()[name = string("value_states_55_dilations_0"), val = tensor([1, 1])]; int32 value_states_55_groups_0 = const()[name = string("value_states_55_groups_0"), val = int32(1)]; tensor value_states_55_cast_fp16 = conv(dilations = value_states_55_dilations_0, groups = value_states_55_groups_0, pad = value_states_55_pad_0, pad_type = value_states_55_pad_type_0, strides = value_states_55_strides_0, weight = layers_9_self_attn_v_proj_weight_cast_fp16, x = var_3816_cast_fp16_0)[name = string("value_states_55_cast_fp16")]; tensor concat_108x = const()[name = string("concat_108x"), val = tensor([1, 16, 128, -1])]; tensor x_91_cast_fp16 = reshape(shape = concat_108x, x = query_states_55_cast_fp16)[name = string("x_91_cast_fp16")]; tensor concat_109x = const()[name = string("concat_109x"), val = tensor([1, 2, 128, -1])]; tensor var_3873_cast_fp16 = reshape(shape = concat_109x, x = key_states_91_cast_fp16)[name = string("op_3873_cast_fp16")]; tensor concat_110x = const()[name = string("concat_110x"), val = tensor([1, 2, 128, -1])]; tensor var_3880_cast_fp16 = reshape(shape = concat_110x, x = value_states_55_cast_fp16)[name = string("op_3880_cast_fp16")]; tensor var_3884_cast_fp16 = mul(x = x_91_cast_fp16, y = var_869_cast_fp16)[name = string("op_3884_cast_fp16")]; tensor var_3885_split_sizes_0 = const()[name = string("op_3885_split_sizes_0"), val = tensor([64, 64])]; int32 var_3885_axis_0 = const()[name = string("op_3885_axis_0"), val = int32(-2)]; tensor var_3885_cast_fp16_0, tensor var_3885_cast_fp16_1 = split(axis = var_3885_axis_0, split_sizes = var_3885_split_sizes_0, x = x_91_cast_fp16)[name = string("op_3885_cast_fp16")]; fp16 const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3887_cast_fp16 = mul(x = var_3885_cast_fp16_1, y = const_92_promoted_to_fp16)[name = string("op_3887_cast_fp16")]; int32 var_3889 = const()[name = string("op_3889"), val = int32(-2)]; bool var_3890_interleave_0 = const()[name = string("op_3890_interleave_0"), val = bool(false)]; tensor var_3890_cast_fp16 = concat(axis = var_3889, interleave = var_3890_interleave_0, values = (var_3887_cast_fp16, var_3885_cast_fp16_0))[name = string("op_3890_cast_fp16")]; tensor var_3891_cast_fp16 = mul(x = var_3890_cast_fp16, y = var_878_cast_fp16)[name = string("op_3891_cast_fp16")]; tensor query_states_57_cast_fp16 = add(x = var_3884_cast_fp16, y = var_3891_cast_fp16)[name = string("query_states_57_cast_fp16")]; tensor var_3897_cast_fp16 = mul(x = var_3873_cast_fp16, y = var_869_cast_fp16)[name = string("op_3897_cast_fp16")]; tensor var_3898_split_sizes_0 = const()[name = string("op_3898_split_sizes_0"), val = tensor([64, 64])]; int32 var_3898_axis_0 = const()[name = string("op_3898_axis_0"), val = int32(-2)]; tensor var_3898_cast_fp16_0, tensor var_3898_cast_fp16_1 = split(axis = var_3898_axis_0, split_sizes = var_3898_split_sizes_0, x = var_3873_cast_fp16)[name = string("op_3898_cast_fp16")]; fp16 const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3900_cast_fp16 = mul(x = var_3898_cast_fp16_1, y = const_93_promoted_to_fp16)[name = string("op_3900_cast_fp16")]; int32 var_3902 = const()[name = string("op_3902"), val = int32(-2)]; bool var_3903_interleave_0 = const()[name = string("op_3903_interleave_0"), val = bool(false)]; tensor var_3903_cast_fp16 = concat(axis = var_3902, interleave = var_3903_interleave_0, values = (var_3900_cast_fp16, var_3898_cast_fp16_0))[name = string("op_3903_cast_fp16")]; tensor var_3904_cast_fp16 = mul(x = var_3903_cast_fp16, y = var_878_cast_fp16)[name = string("op_3904_cast_fp16")]; tensor key_states_95_cast_fp16 = add(x = var_3897_cast_fp16, y = var_3904_cast_fp16)[name = string("key_states_95_cast_fp16")]; tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; int32 concat_113_axis_0 = const()[name = string("concat_113_axis_0"), val = int32(0)]; bool concat_113_interleave_0 = const()[name = string("concat_113_interleave_0"), val = bool(false)]; tensor concat_113 = concat(axis = concat_113_axis_0, interleave = concat_113_interleave_0, values = (expand_dims_108, expand_dims_109, position_id, expand_dims_111))[name = string("concat_113")]; tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; tensor concat_114_values1_0 = const()[name = string("concat_114_values1_0"), val = tensor([0])]; tensor concat_114_values3_0 = const()[name = string("concat_114_values3_0"), val = tensor([0])]; int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_112, concat_114_values1_0, cache_position_end, concat_114_values3_0))[name = string("concat_114")]; tensor key_states_97_perm_0 = const()[name = string("key_states_97_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_10_stride_0 = const()[name = string("key_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_97_cast_fp16 = transpose(perm = key_states_97_perm_0, x = key_states_95_cast_fp16)[name = string("transpose_572")]; tensor key_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = key_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = key_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_10_squeeze_mask_0, stride = key_cache_internal_tensor_assign_10_stride_0, update = key_states_97_cast_fp16, x = coreml_update_state_352)[name = string("key_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_10_cast_fp16, input = key_cache)[name = string("coreml_update_state_354_write_state")]; tensor coreml_update_state_354 = read_state(input = key_cache)[name = string("coreml_update_state_354")]; tensor value_states_57_perm_0 = const()[name = string("value_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_10_stride_0 = const()[name = string("value_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_57_cast_fp16 = transpose(perm = value_states_57_perm_0, x = var_3880_cast_fp16)[name = string("transpose_571")]; tensor value_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = value_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = value_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_10_squeeze_mask_0, stride = value_cache_internal_tensor_assign_10_stride_0, update = value_states_57_cast_fp16, x = coreml_update_state_353)[name = string("value_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_10_cast_fp16, input = value_cache)[name = string("coreml_update_state_355_write_state")]; tensor coreml_update_state_355 = read_state(input = value_cache)[name = string("coreml_update_state_355")]; tensor var_3974_begin_0 = const()[name = string("op_3974_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3974_end_0 = const()[name = string("op_3974_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3974_end_mask_0 = const()[name = string("op_3974_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3974_cast_fp16 = slice_by_index(begin = var_3974_begin_0, end = var_3974_end_0, end_mask = var_3974_end_mask_0, x = coreml_update_state_354)[name = string("op_3974_cast_fp16")]; tensor tile_18 = const()[name = string("tile_18"), val = tensor([1, 1])]; int32 var_3977_axis_0 = const()[name = string("op_3977_axis_0"), val = int32(1)]; tensor var_3977_cast_fp16_0, tensor var_3977_cast_fp16_1 = split(axis = var_3977_axis_0, split_sizes = tile_18, x = var_3974_cast_fp16)[name = string("op_3977_cast_fp16")]; tensor var_3984_begin_0 = const()[name = string("op_3984_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3984_end_0 = const()[name = string("op_3984_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3984_end_mask_0 = const()[name = string("op_3984_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3984_cast_fp16 = slice_by_index(begin = var_3984_begin_0, end = var_3984_end_0, end_mask = var_3984_end_mask_0, x = coreml_update_state_355)[name = string("op_3984_cast_fp16")]; tensor tile_19 = const()[name = string("tile_19"), val = tensor([1, 1])]; int32 var_3987_axis_0 = const()[name = string("op_3987_axis_0"), val = int32(1)]; tensor var_3987_cast_fp16_0, tensor var_3987_cast_fp16_1 = split(axis = var_3987_axis_0, split_sizes = tile_19, x = var_3984_cast_fp16)[name = string("op_3987_cast_fp16")]; tensor var_3990_split_sizes_0 = const()[name = string("op_3990_split_sizes_0"), val = tensor([8, 8])]; int32 var_3990_axis_0 = const()[name = string("op_3990_axis_0"), val = int32(1)]; tensor var_3990_0, tensor var_3990_1 = split(axis = var_3990_axis_0, split_sizes = var_3990_split_sizes_0, x = query_states_57_cast_fp16)[name = string("op_3990")]; bool attn_weights_145_transpose_x_0 = const()[name = string("attn_weights_145_transpose_x_0"), val = bool(false)]; bool attn_weights_145_transpose_y_0 = const()[name = string("attn_weights_145_transpose_y_0"), val = bool(false)]; tensor attn_weights_145_cast_fp16 = matmul(transpose_x = attn_weights_145_transpose_x_0, transpose_y = attn_weights_145_transpose_y_0, x = var_3977_cast_fp16_0, y = var_3990_0)[name = string("attn_weights_145_cast_fp16")]; fp16 var_3993_to_fp16 = const()[name = string("op_3993_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_147_cast_fp16 = mul(x = attn_weights_145_cast_fp16, y = var_3993_to_fp16)[name = string("attn_weights_147_cast_fp16")]; tensor attn_weights_149_cast_fp16 = add(x = attn_weights_147_cast_fp16, y = attn_mask_1)[name = string("attn_weights_149_cast_fp16")]; int32 var_3997 = const()[name = string("op_3997"), val = int32(-2)]; tensor attn_weights_151_cast_fp16 = softmax(axis = var_3997, x = attn_weights_149_cast_fp16)[name = string("attn_weights_151_cast_fp16")]; bool var_4003_transpose_x_1 = const()[name = string("op_4003_transpose_x_1"), val = bool(true)]; bool var_4003_transpose_y_1 = const()[name = string("op_4003_transpose_y_1"), val = bool(false)]; tensor var_4003_cast_fp16 = matmul(transpose_x = var_4003_transpose_x_1, transpose_y = var_4003_transpose_y_1, x = attn_weights_151_cast_fp16, y = var_3987_cast_fp16_0)[name = string("op_4003_cast_fp16")]; bool attn_weights_153_transpose_x_0 = const()[name = string("attn_weights_153_transpose_x_0"), val = bool(false)]; bool attn_weights_153_transpose_y_0 = const()[name = string("attn_weights_153_transpose_y_0"), val = bool(false)]; tensor attn_weights_153_cast_fp16 = matmul(transpose_x = attn_weights_153_transpose_x_0, transpose_y = attn_weights_153_transpose_y_0, x = var_3977_cast_fp16_1, y = var_3990_1)[name = string("attn_weights_153_cast_fp16")]; fp16 var_4005_to_fp16 = const()[name = string("op_4005_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_155_cast_fp16 = mul(x = attn_weights_153_cast_fp16, y = var_4005_to_fp16)[name = string("attn_weights_155_cast_fp16")]; tensor attn_weights_157_cast_fp16 = add(x = attn_weights_155_cast_fp16, y = attn_mask_1)[name = string("attn_weights_157_cast_fp16")]; int32 var_4009 = const()[name = string("op_4009"), val = int32(-2)]; tensor attn_weights_159_cast_fp16 = softmax(axis = var_4009, x = attn_weights_157_cast_fp16)[name = string("attn_weights_159_cast_fp16")]; bool attn_output_73_transpose_x_1 = const()[name = string("attn_output_73_transpose_x_1"), val = bool(true)]; bool attn_output_73_transpose_y_1 = const()[name = string("attn_output_73_transpose_y_1"), val = bool(false)]; tensor attn_output_73_cast_fp16 = matmul(transpose_x = attn_output_73_transpose_x_1, transpose_y = attn_output_73_transpose_y_1, x = attn_weights_159_cast_fp16, y = var_3987_cast_fp16_1)[name = string("attn_output_73_cast_fp16")]; int32 var_4017 = const()[name = string("op_4017"), val = int32(1)]; bool attn_output_75_interleave_0 = const()[name = string("attn_output_75_interleave_0"), val = bool(false)]; tensor attn_output_75_cast_fp16 = concat(axis = var_4017, interleave = attn_output_75_interleave_0, values = (var_4003_cast_fp16, attn_output_73_cast_fp16))[name = string("attn_output_75_cast_fp16")]; tensor var_4021_perm_0 = const()[name = string("op_4021_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_119x = const()[name = string("concat_119x"), val = tensor([1, 2048, 1, -1])]; tensor var_4021_cast_fp16 = transpose(perm = var_4021_perm_0, x = attn_output_75_cast_fp16)[name = string("transpose_570")]; tensor attn_output_79_cast_fp16 = reshape(shape = concat_119x, x = var_4021_cast_fp16)[name = string("attn_output_79_cast_fp16")]; tensor hidden_states_93_strides_0 = const()[name = string("hidden_states_93_strides_0"), val = tensor([1, 1])]; string hidden_states_93_pad_type_0 = const()[name = string("hidden_states_93_pad_type_0"), val = string("valid")]; tensor hidden_states_93_pad_0 = const()[name = string("hidden_states_93_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_93_dilations_0 = const()[name = string("hidden_states_93_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_93_groups_0 = const()[name = string("hidden_states_93_groups_0"), val = int32(1)]; tensor hidden_states_93_cast_fp16 = conv(dilations = hidden_states_93_dilations_0, groups = hidden_states_93_groups_0, pad = hidden_states_93_pad_0, pad_type = hidden_states_93_pad_type_0, strides = hidden_states_93_strides_0, weight = layers_9_self_attn_o_proj_weight_cast_fp16, x = attn_output_79_cast_fp16)[name = string("hidden_states_93_cast_fp16")]; tensor hidden_states_95_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = hidden_states_93_cast_fp16)[name = string("hidden_states_95_cast_fp16")]; fp16 const_98_promoted_to_fp16 = const()[name = string("const_98_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4054_cast_fp16 = mul(x = hidden_states_95_cast_fp16, y = const_98_promoted_to_fp16)[name = string("op_4054_cast_fp16")]; int32 var_4052 = const()[name = string("op_4052"), val = int32(1)]; bool doubled_77_interleave_0 = const()[name = string("doubled_77_interleave_0"), val = bool(false)]; tensor doubled_77_cast_fp16 = concat(axis = var_4052, interleave = doubled_77_interleave_0, values = (hidden_states_95_cast_fp16, var_4054_cast_fp16))[name = string("doubled_77_cast_fp16")]; tensor out_39_axes_0 = const()[name = string("out_39_axes_0"), val = tensor([1])]; tensor out_39_gamma_0_to_fp16 = const()[name = string("out_39_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405896064)))]; fp16 var_4064_to_fp16 = const()[name = string("op_4064_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_39_cast_fp16 = layer_norm(axes = out_39_axes_0, epsilon = var_4064_to_fp16, gamma = out_39_gamma_0_to_fp16, x = doubled_77_cast_fp16)[name = string("out_39_cast_fp16")]; tensor var_4075_split_sizes_0 = const()[name = string("op_4075_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4075_axis_0 = const()[name = string("op_4075_axis_0"), val = int32(1)]; tensor var_4075_cast_fp16_0, tensor var_4075_cast_fp16_1 = split(axis = var_4075_axis_0, split_sizes = var_4075_split_sizes_0, x = out_39_cast_fp16)[name = string("op_4075_cast_fp16")]; tensor input_19_strides_0 = const()[name = string("input_19_strides_0"), val = tensor([1, 1])]; string input_19_pad_type_0 = const()[name = string("input_19_pad_type_0"), val = string("valid")]; tensor input_19_pad_0 = const()[name = string("input_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_19_dilations_0 = const()[name = string("input_19_dilations_0"), val = tensor([1, 1])]; int32 input_19_groups_0 = const()[name = string("input_19_groups_0"), val = int32(1)]; tensor input_19_cast_fp16 = conv(dilations = input_19_dilations_0, groups = input_19_groups_0, pad = input_19_pad_0, pad_type = input_19_pad_type_0, strides = input_19_strides_0, weight = layers_9_mlp_gate_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("input_19_cast_fp16")]; tensor var_4092_cast_fp16 = silu(x = input_19_cast_fp16)[name = string("op_4092_cast_fp16")]; tensor var_4098_strides_0 = const()[name = string("op_4098_strides_0"), val = tensor([1, 1])]; string var_4098_pad_type_0 = const()[name = string("op_4098_pad_type_0"), val = string("valid")]; tensor var_4098_pad_0 = const()[name = string("op_4098_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4098_dilations_0 = const()[name = string("op_4098_dilations_0"), val = tensor([1, 1])]; int32 var_4098_groups_0 = const()[name = string("op_4098_groups_0"), val = int32(1)]; tensor var_4098_cast_fp16 = conv(dilations = var_4098_dilations_0, groups = var_4098_groups_0, pad = var_4098_pad_0, pad_type = var_4098_pad_type_0, strides = var_4098_strides_0, weight = layers_9_mlp_up_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("op_4098_cast_fp16")]; tensor x_99_cast_fp16 = mul(x = var_4092_cast_fp16, y = var_4098_cast_fp16)[name = string("x_99_cast_fp16")]; tensor hidden_states_97_strides_0 = const()[name = string("hidden_states_97_strides_0"), val = tensor([1, 1])]; string hidden_states_97_pad_type_0 = const()[name = string("hidden_states_97_pad_type_0"), val = string("valid")]; tensor hidden_states_97_pad_0 = const()[name = string("hidden_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_97_dilations_0 = const()[name = string("hidden_states_97_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_97_groups_0 = const()[name = string("hidden_states_97_groups_0"), val = int32(1)]; tensor hidden_states_97_cast_fp16 = conv(dilations = hidden_states_97_dilations_0, groups = hidden_states_97_groups_0, pad = hidden_states_97_pad_0, pad_type = hidden_states_97_pad_type_0, strides = hidden_states_97_strides_0, weight = layers_9_mlp_down_proj_weight_cast_fp16, x = x_99_cast_fp16)[name = string("hidden_states_97_cast_fp16")]; tensor hidden_states_99_cast_fp16 = add(x = hidden_states_95_cast_fp16, y = hidden_states_97_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; fp16 const_100_promoted_to_fp16 = const()[name = string("const_100_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4116_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_100_promoted_to_fp16)[name = string("op_4116_cast_fp16")]; int32 var_4114 = const()[name = string("op_4114"), val = int32(1)]; bool doubled_81_interleave_0 = const()[name = string("doubled_81_interleave_0"), val = bool(false)]; tensor doubled_81_cast_fp16 = concat(axis = var_4114, interleave = doubled_81_interleave_0, values = (hidden_states_99_cast_fp16, var_4116_cast_fp16))[name = string("doubled_81_cast_fp16")]; tensor out_41_axes_0 = const()[name = string("out_41_axes_0"), val = tensor([1])]; tensor out_41_gamma_0_to_fp16 = const()[name = string("out_41_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405904320)))]; fp16 var_4126_to_fp16 = const()[name = string("op_4126_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_41_cast_fp16 = layer_norm(axes = out_41_axes_0, epsilon = var_4126_to_fp16, gamma = out_41_gamma_0_to_fp16, x = doubled_81_cast_fp16)[name = string("out_41_cast_fp16")]; tensor var_4137_split_sizes_0 = const()[name = string("op_4137_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4137_axis_0 = const()[name = string("op_4137_axis_0"), val = int32(1)]; tensor var_4137_cast_fp16_0, tensor var_4137_cast_fp16_1 = split(axis = var_4137_axis_0, split_sizes = var_4137_split_sizes_0, x = out_41_cast_fp16)[name = string("op_4137_cast_fp16")]; tensor layers_10_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405912576)))]; tensor query_states_61_strides_0 = const()[name = string("query_states_61_strides_0"), val = tensor([1, 1])]; string query_states_61_pad_type_0 = const()[name = string("query_states_61_pad_type_0"), val = string("valid")]; tensor query_states_61_pad_0 = const()[name = string("query_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_61_dilations_0 = const()[name = string("query_states_61_dilations_0"), val = tensor([1, 1])]; int32 query_states_61_groups_0 = const()[name = string("query_states_61_groups_0"), val = int32(1)]; tensor query_states_61_cast_fp16 = conv(dilations = query_states_61_dilations_0, groups = query_states_61_groups_0, pad = query_states_61_pad_0, pad_type = query_states_61_pad_type_0, strides = query_states_61_strides_0, weight = layers_10_self_attn_q_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("query_states_61_cast_fp16")]; tensor layers_10_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1414301248)))]; tensor key_states_101_strides_0 = const()[name = string("key_states_101_strides_0"), val = tensor([1, 1])]; string key_states_101_pad_type_0 = const()[name = string("key_states_101_pad_type_0"), val = string("valid")]; tensor key_states_101_pad_0 = const()[name = string("key_states_101_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_101_dilations_0 = const()[name = string("key_states_101_dilations_0"), val = tensor([1, 1])]; int32 key_states_101_groups_0 = const()[name = string("key_states_101_groups_0"), val = int32(1)]; tensor key_states_101_cast_fp16 = conv(dilations = key_states_101_dilations_0, groups = key_states_101_groups_0, pad = key_states_101_pad_0, pad_type = key_states_101_pad_type_0, strides = key_states_101_strides_0, weight = layers_10_self_attn_k_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("key_states_101_cast_fp16")]; tensor value_states_61_strides_0 = const()[name = string("value_states_61_strides_0"), val = tensor([1, 1])]; string value_states_61_pad_type_0 = const()[name = string("value_states_61_pad_type_0"), val = string("valid")]; tensor value_states_61_pad_0 = const()[name = string("value_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_61_dilations_0 = const()[name = string("value_states_61_dilations_0"), val = tensor([1, 1])]; int32 value_states_61_groups_0 = const()[name = string("value_states_61_groups_0"), val = int32(1)]; tensor value_states_61_cast_fp16 = conv(dilations = value_states_61_dilations_0, groups = value_states_61_groups_0, pad = value_states_61_pad_0, pad_type = value_states_61_pad_type_0, strides = value_states_61_strides_0, weight = layers_10_self_attn_v_proj_weight_cast_fp16, x = var_4137_cast_fp16_0)[name = string("value_states_61_cast_fp16")]; tensor concat_120x = const()[name = string("concat_120x"), val = tensor([1, 16, 128, -1])]; tensor x_101_cast_fp16 = reshape(shape = concat_120x, x = query_states_61_cast_fp16)[name = string("x_101_cast_fp16")]; tensor concat_121x = const()[name = string("concat_121x"), val = tensor([1, 2, 128, -1])]; tensor var_4194_cast_fp16 = reshape(shape = concat_121x, x = key_states_101_cast_fp16)[name = string("op_4194_cast_fp16")]; tensor concat_122x = const()[name = string("concat_122x"), val = tensor([1, 2, 128, -1])]; tensor var_4201_cast_fp16 = reshape(shape = concat_122x, x = value_states_61_cast_fp16)[name = string("op_4201_cast_fp16")]; tensor var_4205_cast_fp16 = mul(x = x_101_cast_fp16, y = var_869_cast_fp16)[name = string("op_4205_cast_fp16")]; tensor var_4206_split_sizes_0 = const()[name = string("op_4206_split_sizes_0"), val = tensor([64, 64])]; int32 var_4206_axis_0 = const()[name = string("op_4206_axis_0"), val = int32(-2)]; tensor var_4206_cast_fp16_0, tensor var_4206_cast_fp16_1 = split(axis = var_4206_axis_0, split_sizes = var_4206_split_sizes_0, x = x_101_cast_fp16)[name = string("op_4206_cast_fp16")]; fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4208_cast_fp16 = mul(x = var_4206_cast_fp16_1, y = const_102_promoted_to_fp16)[name = string("op_4208_cast_fp16")]; int32 var_4210 = const()[name = string("op_4210"), val = int32(-2)]; bool var_4211_interleave_0 = const()[name = string("op_4211_interleave_0"), val = bool(false)]; tensor var_4211_cast_fp16 = concat(axis = var_4210, interleave = var_4211_interleave_0, values = (var_4208_cast_fp16, var_4206_cast_fp16_0))[name = string("op_4211_cast_fp16")]; tensor var_4212_cast_fp16 = mul(x = var_4211_cast_fp16, y = var_878_cast_fp16)[name = string("op_4212_cast_fp16")]; tensor query_states_63_cast_fp16 = add(x = var_4205_cast_fp16, y = var_4212_cast_fp16)[name = string("query_states_63_cast_fp16")]; tensor var_4218_cast_fp16 = mul(x = var_4194_cast_fp16, y = var_869_cast_fp16)[name = string("op_4218_cast_fp16")]; tensor var_4219_split_sizes_0 = const()[name = string("op_4219_split_sizes_0"), val = tensor([64, 64])]; int32 var_4219_axis_0 = const()[name = string("op_4219_axis_0"), val = int32(-2)]; tensor var_4219_cast_fp16_0, tensor var_4219_cast_fp16_1 = split(axis = var_4219_axis_0, split_sizes = var_4219_split_sizes_0, x = var_4194_cast_fp16)[name = string("op_4219_cast_fp16")]; fp16 const_103_promoted_to_fp16 = const()[name = string("const_103_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4221_cast_fp16 = mul(x = var_4219_cast_fp16_1, y = const_103_promoted_to_fp16)[name = string("op_4221_cast_fp16")]; int32 var_4223 = const()[name = string("op_4223"), val = int32(-2)]; bool var_4224_interleave_0 = const()[name = string("op_4224_interleave_0"), val = bool(false)]; tensor var_4224_cast_fp16 = concat(axis = var_4223, interleave = var_4224_interleave_0, values = (var_4221_cast_fp16, var_4219_cast_fp16_0))[name = string("op_4224_cast_fp16")]; tensor var_4225_cast_fp16 = mul(x = var_4224_cast_fp16, y = var_878_cast_fp16)[name = string("op_4225_cast_fp16")]; tensor key_states_105_cast_fp16 = add(x = var_4218_cast_fp16, y = var_4225_cast_fp16)[name = string("key_states_105_cast_fp16")]; tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; int32 concat_125_axis_0 = const()[name = string("concat_125_axis_0"), val = int32(0)]; bool concat_125_interleave_0 = const()[name = string("concat_125_interleave_0"), val = bool(false)]; tensor concat_125 = concat(axis = concat_125_axis_0, interleave = concat_125_interleave_0, values = (expand_dims_120, expand_dims_121, position_id, expand_dims_123))[name = string("concat_125")]; tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; tensor concat_126_values1_0 = const()[name = string("concat_126_values1_0"), val = tensor([0])]; tensor concat_126_values3_0 = const()[name = string("concat_126_values3_0"), val = tensor([0])]; int32 concat_126_axis_0 = const()[name = string("concat_126_axis_0"), val = int32(0)]; bool concat_126_interleave_0 = const()[name = string("concat_126_interleave_0"), val = bool(false)]; tensor concat_126 = concat(axis = concat_126_axis_0, interleave = concat_126_interleave_0, values = (expand_dims_124, concat_126_values1_0, cache_position_end, concat_126_values3_0))[name = string("concat_126")]; tensor key_states_107_perm_0 = const()[name = string("key_states_107_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_11_stride_0 = const()[name = string("key_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_107_cast_fp16 = transpose(perm = key_states_107_perm_0, x = key_states_105_cast_fp16)[name = string("transpose_569")]; tensor key_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = key_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = key_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_11_squeeze_mask_0, stride = key_cache_internal_tensor_assign_11_stride_0, update = key_states_107_cast_fp16, x = coreml_update_state_354)[name = string("key_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_11_cast_fp16, input = key_cache)[name = string("coreml_update_state_356_write_state")]; tensor coreml_update_state_356 = read_state(input = key_cache)[name = string("coreml_update_state_356")]; tensor value_states_63_perm_0 = const()[name = string("value_states_63_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_11_stride_0 = const()[name = string("value_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_63_cast_fp16 = transpose(perm = value_states_63_perm_0, x = var_4201_cast_fp16)[name = string("transpose_568")]; tensor value_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = value_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = value_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_11_squeeze_mask_0, stride = value_cache_internal_tensor_assign_11_stride_0, update = value_states_63_cast_fp16, x = coreml_update_state_355)[name = string("value_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_11_cast_fp16, input = value_cache)[name = string("coreml_update_state_357_write_state")]; tensor coreml_update_state_357 = read_state(input = value_cache)[name = string("coreml_update_state_357")]; tensor var_4295_begin_0 = const()[name = string("op_4295_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4295_end_0 = const()[name = string("op_4295_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4295_end_mask_0 = const()[name = string("op_4295_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4295_cast_fp16 = slice_by_index(begin = var_4295_begin_0, end = var_4295_end_0, end_mask = var_4295_end_mask_0, x = coreml_update_state_356)[name = string("op_4295_cast_fp16")]; tensor tile_20 = const()[name = string("tile_20"), val = tensor([1, 1])]; int32 var_4298_axis_0 = const()[name = string("op_4298_axis_0"), val = int32(1)]; tensor var_4298_cast_fp16_0, tensor var_4298_cast_fp16_1 = split(axis = var_4298_axis_0, split_sizes = tile_20, x = var_4295_cast_fp16)[name = string("op_4298_cast_fp16")]; tensor var_4305_begin_0 = const()[name = string("op_4305_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4305_end_0 = const()[name = string("op_4305_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4305_end_mask_0 = const()[name = string("op_4305_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4305_cast_fp16 = slice_by_index(begin = var_4305_begin_0, end = var_4305_end_0, end_mask = var_4305_end_mask_0, x = coreml_update_state_357)[name = string("op_4305_cast_fp16")]; tensor tile_21 = const()[name = string("tile_21"), val = tensor([1, 1])]; int32 var_4308_axis_0 = const()[name = string("op_4308_axis_0"), val = int32(1)]; tensor var_4308_cast_fp16_0, tensor var_4308_cast_fp16_1 = split(axis = var_4308_axis_0, split_sizes = tile_21, x = var_4305_cast_fp16)[name = string("op_4308_cast_fp16")]; tensor var_4311_split_sizes_0 = const()[name = string("op_4311_split_sizes_0"), val = tensor([8, 8])]; int32 var_4311_axis_0 = const()[name = string("op_4311_axis_0"), val = int32(1)]; tensor var_4311_0, tensor var_4311_1 = split(axis = var_4311_axis_0, split_sizes = var_4311_split_sizes_0, x = query_states_63_cast_fp16)[name = string("op_4311")]; bool attn_weights_161_transpose_x_0 = const()[name = string("attn_weights_161_transpose_x_0"), val = bool(false)]; bool attn_weights_161_transpose_y_0 = const()[name = string("attn_weights_161_transpose_y_0"), val = bool(false)]; tensor attn_weights_161_cast_fp16 = matmul(transpose_x = attn_weights_161_transpose_x_0, transpose_y = attn_weights_161_transpose_y_0, x = var_4298_cast_fp16_0, y = var_4311_0)[name = string("attn_weights_161_cast_fp16")]; fp16 var_4314_to_fp16 = const()[name = string("op_4314_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_163_cast_fp16 = mul(x = attn_weights_161_cast_fp16, y = var_4314_to_fp16)[name = string("attn_weights_163_cast_fp16")]; tensor attn_weights_165_cast_fp16 = add(x = attn_weights_163_cast_fp16, y = attn_mask_1)[name = string("attn_weights_165_cast_fp16")]; int32 var_4318 = const()[name = string("op_4318"), val = int32(-2)]; tensor attn_weights_167_cast_fp16 = softmax(axis = var_4318, x = attn_weights_165_cast_fp16)[name = string("attn_weights_167_cast_fp16")]; bool var_4324_transpose_x_1 = const()[name = string("op_4324_transpose_x_1"), val = bool(true)]; bool var_4324_transpose_y_1 = const()[name = string("op_4324_transpose_y_1"), val = bool(false)]; tensor var_4324_cast_fp16 = matmul(transpose_x = var_4324_transpose_x_1, transpose_y = var_4324_transpose_y_1, x = attn_weights_167_cast_fp16, y = var_4308_cast_fp16_0)[name = string("op_4324_cast_fp16")]; bool attn_weights_169_transpose_x_0 = const()[name = string("attn_weights_169_transpose_x_0"), val = bool(false)]; bool attn_weights_169_transpose_y_0 = const()[name = string("attn_weights_169_transpose_y_0"), val = bool(false)]; tensor attn_weights_169_cast_fp16 = matmul(transpose_x = attn_weights_169_transpose_x_0, transpose_y = attn_weights_169_transpose_y_0, x = var_4298_cast_fp16_1, y = var_4311_1)[name = string("attn_weights_169_cast_fp16")]; fp16 var_4326_to_fp16 = const()[name = string("op_4326_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_171_cast_fp16 = mul(x = attn_weights_169_cast_fp16, y = var_4326_to_fp16)[name = string("attn_weights_171_cast_fp16")]; tensor attn_weights_173_cast_fp16 = add(x = attn_weights_171_cast_fp16, y = attn_mask_1)[name = string("attn_weights_173_cast_fp16")]; int32 var_4330 = const()[name = string("op_4330"), val = int32(-2)]; tensor attn_weights_175_cast_fp16 = softmax(axis = var_4330, x = attn_weights_173_cast_fp16)[name = string("attn_weights_175_cast_fp16")]; bool attn_output_81_transpose_x_1 = const()[name = string("attn_output_81_transpose_x_1"), val = bool(true)]; bool attn_output_81_transpose_y_1 = const()[name = string("attn_output_81_transpose_y_1"), val = bool(false)]; tensor attn_output_81_cast_fp16 = matmul(transpose_x = attn_output_81_transpose_x_1, transpose_y = attn_output_81_transpose_y_1, x = attn_weights_175_cast_fp16, y = var_4308_cast_fp16_1)[name = string("attn_output_81_cast_fp16")]; int32 var_4338 = const()[name = string("op_4338"), val = int32(1)]; bool attn_output_83_interleave_0 = const()[name = string("attn_output_83_interleave_0"), val = bool(false)]; tensor attn_output_83_cast_fp16 = concat(axis = var_4338, interleave = attn_output_83_interleave_0, values = (var_4324_cast_fp16, attn_output_81_cast_fp16))[name = string("attn_output_83_cast_fp16")]; tensor var_4342_perm_0 = const()[name = string("op_4342_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_131x = const()[name = string("concat_131x"), val = tensor([1, 2048, 1, -1])]; tensor var_4342_cast_fp16 = transpose(perm = var_4342_perm_0, x = attn_output_83_cast_fp16)[name = string("transpose_567")]; tensor attn_output_87_cast_fp16 = reshape(shape = concat_131x, x = var_4342_cast_fp16)[name = string("attn_output_87_cast_fp16")]; tensor hidden_states_103_strides_0 = const()[name = string("hidden_states_103_strides_0"), val = tensor([1, 1])]; string hidden_states_103_pad_type_0 = const()[name = string("hidden_states_103_pad_type_0"), val = string("valid")]; tensor hidden_states_103_pad_0 = const()[name = string("hidden_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_103_dilations_0 = const()[name = string("hidden_states_103_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_103_groups_0 = const()[name = string("hidden_states_103_groups_0"), val = int32(1)]; tensor hidden_states_103_cast_fp16 = conv(dilations = hidden_states_103_dilations_0, groups = hidden_states_103_groups_0, pad = hidden_states_103_pad_0, pad_type = hidden_states_103_pad_type_0, strides = hidden_states_103_strides_0, weight = layers_10_self_attn_o_proj_weight_cast_fp16, x = attn_output_87_cast_fp16)[name = string("hidden_states_103_cast_fp16")]; tensor hidden_states_105_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = hidden_states_103_cast_fp16)[name = string("hidden_states_105_cast_fp16")]; fp16 const_108_promoted_to_fp16 = const()[name = string("const_108_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4375_cast_fp16 = mul(x = hidden_states_105_cast_fp16, y = const_108_promoted_to_fp16)[name = string("op_4375_cast_fp16")]; int32 var_4373 = const()[name = string("op_4373"), val = int32(1)]; bool doubled_85_interleave_0 = const()[name = string("doubled_85_interleave_0"), val = bool(false)]; tensor doubled_85_cast_fp16 = concat(axis = var_4373, interleave = doubled_85_interleave_0, values = (hidden_states_105_cast_fp16, var_4375_cast_fp16))[name = string("doubled_85_cast_fp16")]; tensor out_43_axes_0 = const()[name = string("out_43_axes_0"), val = tensor([1])]; tensor out_43_gamma_0_to_fp16 = const()[name = string("out_43_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415349888)))]; fp16 var_4385_to_fp16 = const()[name = string("op_4385_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_43_cast_fp16 = layer_norm(axes = out_43_axes_0, epsilon = var_4385_to_fp16, gamma = out_43_gamma_0_to_fp16, x = doubled_85_cast_fp16)[name = string("out_43_cast_fp16")]; tensor var_4396_split_sizes_0 = const()[name = string("op_4396_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4396_axis_0 = const()[name = string("op_4396_axis_0"), val = int32(1)]; tensor var_4396_cast_fp16_0, tensor var_4396_cast_fp16_1 = split(axis = var_4396_axis_0, split_sizes = var_4396_split_sizes_0, x = out_43_cast_fp16)[name = string("op_4396_cast_fp16")]; tensor input_21_strides_0 = const()[name = string("input_21_strides_0"), val = tensor([1, 1])]; string input_21_pad_type_0 = const()[name = string("input_21_pad_type_0"), val = string("valid")]; tensor input_21_pad_0 = const()[name = string("input_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_21_dilations_0 = const()[name = string("input_21_dilations_0"), val = tensor([1, 1])]; int32 input_21_groups_0 = const()[name = string("input_21_groups_0"), val = int32(1)]; tensor input_21_cast_fp16 = conv(dilations = input_21_dilations_0, groups = input_21_groups_0, pad = input_21_pad_0, pad_type = input_21_pad_type_0, strides = input_21_strides_0, weight = layers_10_mlp_gate_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("input_21_cast_fp16")]; tensor var_4413_cast_fp16 = silu(x = input_21_cast_fp16)[name = string("op_4413_cast_fp16")]; tensor var_4419_strides_0 = const()[name = string("op_4419_strides_0"), val = tensor([1, 1])]; string var_4419_pad_type_0 = const()[name = string("op_4419_pad_type_0"), val = string("valid")]; tensor var_4419_pad_0 = const()[name = string("op_4419_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4419_dilations_0 = const()[name = string("op_4419_dilations_0"), val = tensor([1, 1])]; int32 var_4419_groups_0 = const()[name = string("op_4419_groups_0"), val = int32(1)]; tensor var_4419_cast_fp16 = conv(dilations = var_4419_dilations_0, groups = var_4419_groups_0, pad = var_4419_pad_0, pad_type = var_4419_pad_type_0, strides = var_4419_strides_0, weight = layers_10_mlp_up_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("op_4419_cast_fp16")]; tensor x_109_cast_fp16 = mul(x = var_4413_cast_fp16, y = var_4419_cast_fp16)[name = string("x_109_cast_fp16")]; tensor hidden_states_107_strides_0 = const()[name = string("hidden_states_107_strides_0"), val = tensor([1, 1])]; string hidden_states_107_pad_type_0 = const()[name = string("hidden_states_107_pad_type_0"), val = string("valid")]; tensor hidden_states_107_pad_0 = const()[name = string("hidden_states_107_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_107_dilations_0 = const()[name = string("hidden_states_107_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_107_groups_0 = const()[name = string("hidden_states_107_groups_0"), val = int32(1)]; tensor hidden_states_107_cast_fp16 = conv(dilations = hidden_states_107_dilations_0, groups = hidden_states_107_groups_0, pad = hidden_states_107_pad_0, pad_type = hidden_states_107_pad_type_0, strides = hidden_states_107_strides_0, weight = layers_10_mlp_down_proj_weight_cast_fp16, x = x_109_cast_fp16)[name = string("hidden_states_107_cast_fp16")]; tensor hidden_states_109_cast_fp16 = add(x = hidden_states_105_cast_fp16, y = hidden_states_107_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; fp16 const_110_promoted_to_fp16 = const()[name = string("const_110_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4437_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_110_promoted_to_fp16)[name = string("op_4437_cast_fp16")]; int32 var_4435 = const()[name = string("op_4435"), val = int32(1)]; bool doubled_89_interleave_0 = const()[name = string("doubled_89_interleave_0"), val = bool(false)]; tensor doubled_89_cast_fp16 = concat(axis = var_4435, interleave = doubled_89_interleave_0, values = (hidden_states_109_cast_fp16, var_4437_cast_fp16))[name = string("doubled_89_cast_fp16")]; tensor out_45_axes_0 = const()[name = string("out_45_axes_0"), val = tensor([1])]; tensor out_45_gamma_0_to_fp16 = const()[name = string("out_45_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415358144)))]; fp16 var_4447_to_fp16 = const()[name = string("op_4447_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_45_cast_fp16 = layer_norm(axes = out_45_axes_0, epsilon = var_4447_to_fp16, gamma = out_45_gamma_0_to_fp16, x = doubled_89_cast_fp16)[name = string("out_45_cast_fp16")]; tensor var_4458_split_sizes_0 = const()[name = string("op_4458_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4458_axis_0 = const()[name = string("op_4458_axis_0"), val = int32(1)]; tensor var_4458_cast_fp16_0, tensor var_4458_cast_fp16_1 = split(axis = var_4458_axis_0, split_sizes = var_4458_split_sizes_0, x = out_45_cast_fp16)[name = string("op_4458_cast_fp16")]; tensor query_states_67_strides_0 = const()[name = string("query_states_67_strides_0"), val = tensor([1, 1])]; string query_states_67_pad_type_0 = const()[name = string("query_states_67_pad_type_0"), val = string("valid")]; tensor query_states_67_pad_0 = const()[name = string("query_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_67_dilations_0 = const()[name = string("query_states_67_dilations_0"), val = tensor([1, 1])]; int32 query_states_67_groups_0 = const()[name = string("query_states_67_groups_0"), val = int32(1)]; tensor query_states_67_cast_fp16 = conv(dilations = query_states_67_dilations_0, groups = query_states_67_groups_0, pad = query_states_67_pad_0, pad_type = query_states_67_pad_type_0, strides = query_states_67_strides_0, weight = layers_11_self_attn_q_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("query_states_67_cast_fp16")]; tensor key_states_111_strides_0 = const()[name = string("key_states_111_strides_0"), val = tensor([1, 1])]; string key_states_111_pad_type_0 = const()[name = string("key_states_111_pad_type_0"), val = string("valid")]; tensor key_states_111_pad_0 = const()[name = string("key_states_111_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_111_dilations_0 = const()[name = string("key_states_111_dilations_0"), val = tensor([1, 1])]; int32 key_states_111_groups_0 = const()[name = string("key_states_111_groups_0"), val = int32(1)]; tensor key_states_111_cast_fp16 = conv(dilations = key_states_111_dilations_0, groups = key_states_111_groups_0, pad = key_states_111_pad_0, pad_type = key_states_111_pad_type_0, strides = key_states_111_strides_0, weight = layers_11_self_attn_k_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("key_states_111_cast_fp16")]; tensor value_states_67_strides_0 = const()[name = string("value_states_67_strides_0"), val = tensor([1, 1])]; string value_states_67_pad_type_0 = const()[name = string("value_states_67_pad_type_0"), val = string("valid")]; tensor value_states_67_pad_0 = const()[name = string("value_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_67_dilations_0 = const()[name = string("value_states_67_dilations_0"), val = tensor([1, 1])]; int32 value_states_67_groups_0 = const()[name = string("value_states_67_groups_0"), val = int32(1)]; tensor value_states_67_cast_fp16 = conv(dilations = value_states_67_dilations_0, groups = value_states_67_groups_0, pad = value_states_67_pad_0, pad_type = value_states_67_pad_type_0, strides = value_states_67_strides_0, weight = layers_11_self_attn_v_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("value_states_67_cast_fp16")]; tensor concat_132x = const()[name = string("concat_132x"), val = tensor([1, 16, 128, -1])]; tensor x_111_cast_fp16 = reshape(shape = concat_132x, x = query_states_67_cast_fp16)[name = string("x_111_cast_fp16")]; tensor concat_133x = const()[name = string("concat_133x"), val = tensor([1, 2, 128, -1])]; tensor var_4515_cast_fp16 = reshape(shape = concat_133x, x = key_states_111_cast_fp16)[name = string("op_4515_cast_fp16")]; tensor concat_134x = const()[name = string("concat_134x"), val = tensor([1, 2, 128, -1])]; tensor var_4522_cast_fp16 = reshape(shape = concat_134x, x = value_states_67_cast_fp16)[name = string("op_4522_cast_fp16")]; tensor var_4526_cast_fp16 = mul(x = x_111_cast_fp16, y = var_869_cast_fp16)[name = string("op_4526_cast_fp16")]; tensor var_4527_split_sizes_0 = const()[name = string("op_4527_split_sizes_0"), val = tensor([64, 64])]; int32 var_4527_axis_0 = const()[name = string("op_4527_axis_0"), val = int32(-2)]; tensor var_4527_cast_fp16_0, tensor var_4527_cast_fp16_1 = split(axis = var_4527_axis_0, split_sizes = var_4527_split_sizes_0, x = x_111_cast_fp16)[name = string("op_4527_cast_fp16")]; fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4529_cast_fp16 = mul(x = var_4527_cast_fp16_1, y = const_112_promoted_to_fp16)[name = string("op_4529_cast_fp16")]; int32 var_4531 = const()[name = string("op_4531"), val = int32(-2)]; bool var_4532_interleave_0 = const()[name = string("op_4532_interleave_0"), val = bool(false)]; tensor var_4532_cast_fp16 = concat(axis = var_4531, interleave = var_4532_interleave_0, values = (var_4529_cast_fp16, var_4527_cast_fp16_0))[name = string("op_4532_cast_fp16")]; tensor var_4533_cast_fp16 = mul(x = var_4532_cast_fp16, y = var_878_cast_fp16)[name = string("op_4533_cast_fp16")]; tensor query_states_69_cast_fp16 = add(x = var_4526_cast_fp16, y = var_4533_cast_fp16)[name = string("query_states_69_cast_fp16")]; tensor var_4539_cast_fp16 = mul(x = var_4515_cast_fp16, y = var_869_cast_fp16)[name = string("op_4539_cast_fp16")]; tensor var_4540_split_sizes_0 = const()[name = string("op_4540_split_sizes_0"), val = tensor([64, 64])]; int32 var_4540_axis_0 = const()[name = string("op_4540_axis_0"), val = int32(-2)]; tensor var_4540_cast_fp16_0, tensor var_4540_cast_fp16_1 = split(axis = var_4540_axis_0, split_sizes = var_4540_split_sizes_0, x = var_4515_cast_fp16)[name = string("op_4540_cast_fp16")]; fp16 const_113_promoted_to_fp16 = const()[name = string("const_113_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4542_cast_fp16 = mul(x = var_4540_cast_fp16_1, y = const_113_promoted_to_fp16)[name = string("op_4542_cast_fp16")]; int32 var_4544 = const()[name = string("op_4544"), val = int32(-2)]; bool var_4545_interleave_0 = const()[name = string("op_4545_interleave_0"), val = bool(false)]; tensor var_4545_cast_fp16 = concat(axis = var_4544, interleave = var_4545_interleave_0, values = (var_4542_cast_fp16, var_4540_cast_fp16_0))[name = string("op_4545_cast_fp16")]; tensor var_4546_cast_fp16 = mul(x = var_4545_cast_fp16, y = var_878_cast_fp16)[name = string("op_4546_cast_fp16")]; tensor key_states_115_cast_fp16 = add(x = var_4539_cast_fp16, y = var_4546_cast_fp16)[name = string("key_states_115_cast_fp16")]; tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; int32 concat_137_axis_0 = const()[name = string("concat_137_axis_0"), val = int32(0)]; bool concat_137_interleave_0 = const()[name = string("concat_137_interleave_0"), val = bool(false)]; tensor concat_137 = concat(axis = concat_137_axis_0, interleave = concat_137_interleave_0, values = (expand_dims_132, expand_dims_133, position_id, expand_dims_135))[name = string("concat_137")]; tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; tensor concat_138_values1_0 = const()[name = string("concat_138_values1_0"), val = tensor([0])]; tensor concat_138_values3_0 = const()[name = string("concat_138_values3_0"), val = tensor([0])]; int32 concat_138_axis_0 = const()[name = string("concat_138_axis_0"), val = int32(0)]; bool concat_138_interleave_0 = const()[name = string("concat_138_interleave_0"), val = bool(false)]; tensor concat_138 = concat(axis = concat_138_axis_0, interleave = concat_138_interleave_0, values = (expand_dims_136, concat_138_values1_0, cache_position_end, concat_138_values3_0))[name = string("concat_138")]; tensor key_states_117_perm_0 = const()[name = string("key_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_12_stride_0 = const()[name = string("key_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_117_cast_fp16 = transpose(perm = key_states_117_perm_0, x = key_states_115_cast_fp16)[name = string("transpose_566")]; tensor key_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = key_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = key_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_12_squeeze_mask_0, stride = key_cache_internal_tensor_assign_12_stride_0, update = key_states_117_cast_fp16, x = coreml_update_state_356)[name = string("key_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_12_cast_fp16, input = key_cache)[name = string("coreml_update_state_358_write_state")]; tensor coreml_update_state_358 = read_state(input = key_cache)[name = string("coreml_update_state_358")]; tensor value_states_69_perm_0 = const()[name = string("value_states_69_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_12_stride_0 = const()[name = string("value_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_69_cast_fp16 = transpose(perm = value_states_69_perm_0, x = var_4522_cast_fp16)[name = string("transpose_565")]; tensor value_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = value_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = value_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_12_squeeze_mask_0, stride = value_cache_internal_tensor_assign_12_stride_0, update = value_states_69_cast_fp16, x = coreml_update_state_357)[name = string("value_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_12_cast_fp16, input = value_cache)[name = string("coreml_update_state_359_write_state")]; tensor coreml_update_state_359 = read_state(input = value_cache)[name = string("coreml_update_state_359")]; tensor var_4616_begin_0 = const()[name = string("op_4616_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4616_end_0 = const()[name = string("op_4616_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4616_end_mask_0 = const()[name = string("op_4616_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4616_cast_fp16 = slice_by_index(begin = var_4616_begin_0, end = var_4616_end_0, end_mask = var_4616_end_mask_0, x = coreml_update_state_358)[name = string("op_4616_cast_fp16")]; tensor tile_22 = const()[name = string("tile_22"), val = tensor([1, 1])]; int32 var_4619_axis_0 = const()[name = string("op_4619_axis_0"), val = int32(1)]; tensor var_4619_cast_fp16_0, tensor var_4619_cast_fp16_1 = split(axis = var_4619_axis_0, split_sizes = tile_22, x = var_4616_cast_fp16)[name = string("op_4619_cast_fp16")]; tensor var_4626_begin_0 = const()[name = string("op_4626_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4626_end_0 = const()[name = string("op_4626_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4626_end_mask_0 = const()[name = string("op_4626_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4626_cast_fp16 = slice_by_index(begin = var_4626_begin_0, end = var_4626_end_0, end_mask = var_4626_end_mask_0, x = coreml_update_state_359)[name = string("op_4626_cast_fp16")]; tensor tile_23 = const()[name = string("tile_23"), val = tensor([1, 1])]; int32 var_4629_axis_0 = const()[name = string("op_4629_axis_0"), val = int32(1)]; tensor var_4629_cast_fp16_0, tensor var_4629_cast_fp16_1 = split(axis = var_4629_axis_0, split_sizes = tile_23, x = var_4626_cast_fp16)[name = string("op_4629_cast_fp16")]; tensor var_4632_split_sizes_0 = const()[name = string("op_4632_split_sizes_0"), val = tensor([8, 8])]; int32 var_4632_axis_0 = const()[name = string("op_4632_axis_0"), val = int32(1)]; tensor var_4632_0, tensor var_4632_1 = split(axis = var_4632_axis_0, split_sizes = var_4632_split_sizes_0, x = query_states_69_cast_fp16)[name = string("op_4632")]; bool attn_weights_177_transpose_x_0 = const()[name = string("attn_weights_177_transpose_x_0"), val = bool(false)]; bool attn_weights_177_transpose_y_0 = const()[name = string("attn_weights_177_transpose_y_0"), val = bool(false)]; tensor attn_weights_177_cast_fp16 = matmul(transpose_x = attn_weights_177_transpose_x_0, transpose_y = attn_weights_177_transpose_y_0, x = var_4619_cast_fp16_0, y = var_4632_0)[name = string("attn_weights_177_cast_fp16")]; fp16 var_4635_to_fp16 = const()[name = string("op_4635_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_179_cast_fp16 = mul(x = attn_weights_177_cast_fp16, y = var_4635_to_fp16)[name = string("attn_weights_179_cast_fp16")]; tensor attn_weights_181_cast_fp16 = add(x = attn_weights_179_cast_fp16, y = attn_mask_1)[name = string("attn_weights_181_cast_fp16")]; int32 var_4639 = const()[name = string("op_4639"), val = int32(-2)]; tensor attn_weights_183_cast_fp16 = softmax(axis = var_4639, x = attn_weights_181_cast_fp16)[name = string("attn_weights_183_cast_fp16")]; bool var_4645_transpose_x_1 = const()[name = string("op_4645_transpose_x_1"), val = bool(true)]; bool var_4645_transpose_y_1 = const()[name = string("op_4645_transpose_y_1"), val = bool(false)]; tensor var_4645_cast_fp16 = matmul(transpose_x = var_4645_transpose_x_1, transpose_y = var_4645_transpose_y_1, x = attn_weights_183_cast_fp16, y = var_4629_cast_fp16_0)[name = string("op_4645_cast_fp16")]; bool attn_weights_185_transpose_x_0 = const()[name = string("attn_weights_185_transpose_x_0"), val = bool(false)]; bool attn_weights_185_transpose_y_0 = const()[name = string("attn_weights_185_transpose_y_0"), val = bool(false)]; tensor attn_weights_185_cast_fp16 = matmul(transpose_x = attn_weights_185_transpose_x_0, transpose_y = attn_weights_185_transpose_y_0, x = var_4619_cast_fp16_1, y = var_4632_1)[name = string("attn_weights_185_cast_fp16")]; fp16 var_4647_to_fp16 = const()[name = string("op_4647_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_187_cast_fp16 = mul(x = attn_weights_185_cast_fp16, y = var_4647_to_fp16)[name = string("attn_weights_187_cast_fp16")]; tensor attn_weights_189_cast_fp16 = add(x = attn_weights_187_cast_fp16, y = attn_mask_1)[name = string("attn_weights_189_cast_fp16")]; int32 var_4651 = const()[name = string("op_4651"), val = int32(-2)]; tensor attn_weights_191_cast_fp16 = softmax(axis = var_4651, x = attn_weights_189_cast_fp16)[name = string("attn_weights_191_cast_fp16")]; bool attn_output_89_transpose_x_1 = const()[name = string("attn_output_89_transpose_x_1"), val = bool(true)]; bool attn_output_89_transpose_y_1 = const()[name = string("attn_output_89_transpose_y_1"), val = bool(false)]; tensor attn_output_89_cast_fp16 = matmul(transpose_x = attn_output_89_transpose_x_1, transpose_y = attn_output_89_transpose_y_1, x = attn_weights_191_cast_fp16, y = var_4629_cast_fp16_1)[name = string("attn_output_89_cast_fp16")]; int32 var_4659 = const()[name = string("op_4659"), val = int32(1)]; bool attn_output_91_interleave_0 = const()[name = string("attn_output_91_interleave_0"), val = bool(false)]; tensor attn_output_91_cast_fp16 = concat(axis = var_4659, interleave = attn_output_91_interleave_0, values = (var_4645_cast_fp16, attn_output_89_cast_fp16))[name = string("attn_output_91_cast_fp16")]; tensor var_4663_perm_0 = const()[name = string("op_4663_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_143x = const()[name = string("concat_143x"), val = tensor([1, 2048, 1, -1])]; tensor var_4663_cast_fp16 = transpose(perm = var_4663_perm_0, x = attn_output_91_cast_fp16)[name = string("transpose_564")]; tensor attn_output_95_cast_fp16 = reshape(shape = concat_143x, x = var_4663_cast_fp16)[name = string("attn_output_95_cast_fp16")]; tensor hidden_states_113_strides_0 = const()[name = string("hidden_states_113_strides_0"), val = tensor([1, 1])]; string hidden_states_113_pad_type_0 = const()[name = string("hidden_states_113_pad_type_0"), val = string("valid")]; tensor hidden_states_113_pad_0 = const()[name = string("hidden_states_113_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_113_dilations_0 = const()[name = string("hidden_states_113_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_113_groups_0 = const()[name = string("hidden_states_113_groups_0"), val = int32(1)]; tensor hidden_states_113_cast_fp16 = conv(dilations = hidden_states_113_dilations_0, groups = hidden_states_113_groups_0, pad = hidden_states_113_pad_0, pad_type = hidden_states_113_pad_type_0, strides = hidden_states_113_strides_0, weight = layers_11_self_attn_o_proj_weight_cast_fp16, x = attn_output_95_cast_fp16)[name = string("hidden_states_113_cast_fp16")]; tensor hidden_states_115_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = hidden_states_113_cast_fp16)[name = string("hidden_states_115_cast_fp16")]; fp16 const_118_promoted_to_fp16 = const()[name = string("const_118_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4696_cast_fp16 = mul(x = hidden_states_115_cast_fp16, y = const_118_promoted_to_fp16)[name = string("op_4696_cast_fp16")]; int32 var_4694 = const()[name = string("op_4694"), val = int32(1)]; bool doubled_93_interleave_0 = const()[name = string("doubled_93_interleave_0"), val = bool(false)]; tensor doubled_93_cast_fp16 = concat(axis = var_4694, interleave = doubled_93_interleave_0, values = (hidden_states_115_cast_fp16, var_4696_cast_fp16))[name = string("doubled_93_cast_fp16")]; tensor out_47_axes_0 = const()[name = string("out_47_axes_0"), val = tensor([1])]; tensor out_47_gamma_0_to_fp16 = const()[name = string("out_47_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415366400)))]; fp16 var_4706_to_fp16 = const()[name = string("op_4706_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_47_cast_fp16 = layer_norm(axes = out_47_axes_0, epsilon = var_4706_to_fp16, gamma = out_47_gamma_0_to_fp16, x = doubled_93_cast_fp16)[name = string("out_47_cast_fp16")]; tensor var_4717_split_sizes_0 = const()[name = string("op_4717_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4717_axis_0 = const()[name = string("op_4717_axis_0"), val = int32(1)]; tensor var_4717_cast_fp16_0, tensor var_4717_cast_fp16_1 = split(axis = var_4717_axis_0, split_sizes = var_4717_split_sizes_0, x = out_47_cast_fp16)[name = string("op_4717_cast_fp16")]; tensor input_23_strides_0 = const()[name = string("input_23_strides_0"), val = tensor([1, 1])]; string input_23_pad_type_0 = const()[name = string("input_23_pad_type_0"), val = string("valid")]; tensor input_23_pad_0 = const()[name = string("input_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_23_dilations_0 = const()[name = string("input_23_dilations_0"), val = tensor([1, 1])]; int32 input_23_groups_0 = const()[name = string("input_23_groups_0"), val = int32(1)]; tensor input_23_cast_fp16 = conv(dilations = input_23_dilations_0, groups = input_23_groups_0, pad = input_23_pad_0, pad_type = input_23_pad_type_0, strides = input_23_strides_0, weight = layers_11_mlp_gate_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("input_23_cast_fp16")]; tensor var_4734_cast_fp16 = silu(x = input_23_cast_fp16)[name = string("op_4734_cast_fp16")]; tensor var_4740_strides_0 = const()[name = string("op_4740_strides_0"), val = tensor([1, 1])]; string var_4740_pad_type_0 = const()[name = string("op_4740_pad_type_0"), val = string("valid")]; tensor var_4740_pad_0 = const()[name = string("op_4740_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4740_dilations_0 = const()[name = string("op_4740_dilations_0"), val = tensor([1, 1])]; int32 var_4740_groups_0 = const()[name = string("op_4740_groups_0"), val = int32(1)]; tensor var_4740_cast_fp16 = conv(dilations = var_4740_dilations_0, groups = var_4740_groups_0, pad = var_4740_pad_0, pad_type = var_4740_pad_type_0, strides = var_4740_strides_0, weight = layers_11_mlp_up_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("op_4740_cast_fp16")]; tensor x_119_cast_fp16 = mul(x = var_4734_cast_fp16, y = var_4740_cast_fp16)[name = string("x_119_cast_fp16")]; tensor hidden_states_117_strides_0 = const()[name = string("hidden_states_117_strides_0"), val = tensor([1, 1])]; string hidden_states_117_pad_type_0 = const()[name = string("hidden_states_117_pad_type_0"), val = string("valid")]; tensor hidden_states_117_pad_0 = const()[name = string("hidden_states_117_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_117_dilations_0 = const()[name = string("hidden_states_117_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_117_groups_0 = const()[name = string("hidden_states_117_groups_0"), val = int32(1)]; tensor hidden_states_117_cast_fp16 = conv(dilations = hidden_states_117_dilations_0, groups = hidden_states_117_groups_0, pad = hidden_states_117_pad_0, pad_type = hidden_states_117_pad_type_0, strides = hidden_states_117_strides_0, weight = layers_11_mlp_down_proj_weight_cast_fp16, x = x_119_cast_fp16)[name = string("hidden_states_117_cast_fp16")]; tensor hidden_states_119_cast_fp16 = add(x = hidden_states_115_cast_fp16, y = hidden_states_117_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; fp16 const_120_promoted_to_fp16 = const()[name = string("const_120_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4758_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_120_promoted_to_fp16)[name = string("op_4758_cast_fp16")]; int32 var_4756 = const()[name = string("op_4756"), val = int32(1)]; bool doubled_97_interleave_0 = const()[name = string("doubled_97_interleave_0"), val = bool(false)]; tensor doubled_97_cast_fp16 = concat(axis = var_4756, interleave = doubled_97_interleave_0, values = (hidden_states_119_cast_fp16, var_4758_cast_fp16))[name = string("doubled_97_cast_fp16")]; tensor out_49_axes_0 = const()[name = string("out_49_axes_0"), val = tensor([1])]; tensor out_49_gamma_0_to_fp16 = const()[name = string("out_49_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415374656)))]; fp16 var_4768_to_fp16 = const()[name = string("op_4768_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_49_cast_fp16 = layer_norm(axes = out_49_axes_0, epsilon = var_4768_to_fp16, gamma = out_49_gamma_0_to_fp16, x = doubled_97_cast_fp16)[name = string("out_49_cast_fp16")]; tensor var_4779_split_sizes_0 = const()[name = string("op_4779_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4779_axis_0 = const()[name = string("op_4779_axis_0"), val = int32(1)]; tensor var_4779_cast_fp16_0, tensor var_4779_cast_fp16_1 = split(axis = var_4779_axis_0, split_sizes = var_4779_split_sizes_0, x = out_49_cast_fp16)[name = string("op_4779_cast_fp16")]; tensor query_states_73_strides_0 = const()[name = string("query_states_73_strides_0"), val = tensor([1, 1])]; string query_states_73_pad_type_0 = const()[name = string("query_states_73_pad_type_0"), val = string("valid")]; tensor query_states_73_pad_0 = const()[name = string("query_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_73_dilations_0 = const()[name = string("query_states_73_dilations_0"), val = tensor([1, 1])]; int32 query_states_73_groups_0 = const()[name = string("query_states_73_groups_0"), val = int32(1)]; tensor query_states_73_cast_fp16 = conv(dilations = query_states_73_dilations_0, groups = query_states_73_groups_0, pad = query_states_73_pad_0, pad_type = query_states_73_pad_type_0, strides = query_states_73_strides_0, weight = layers_12_self_attn_q_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("query_states_73_cast_fp16")]; tensor key_states_121_strides_0 = const()[name = string("key_states_121_strides_0"), val = tensor([1, 1])]; string key_states_121_pad_type_0 = const()[name = string("key_states_121_pad_type_0"), val = string("valid")]; tensor key_states_121_pad_0 = const()[name = string("key_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_121_dilations_0 = const()[name = string("key_states_121_dilations_0"), val = tensor([1, 1])]; int32 key_states_121_groups_0 = const()[name = string("key_states_121_groups_0"), val = int32(1)]; tensor key_states_121_cast_fp16 = conv(dilations = key_states_121_dilations_0, groups = key_states_121_groups_0, pad = key_states_121_pad_0, pad_type = key_states_121_pad_type_0, strides = key_states_121_strides_0, weight = layers_12_self_attn_k_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("key_states_121_cast_fp16")]; tensor value_states_73_strides_0 = const()[name = string("value_states_73_strides_0"), val = tensor([1, 1])]; string value_states_73_pad_type_0 = const()[name = string("value_states_73_pad_type_0"), val = string("valid")]; tensor value_states_73_pad_0 = const()[name = string("value_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_73_dilations_0 = const()[name = string("value_states_73_dilations_0"), val = tensor([1, 1])]; int32 value_states_73_groups_0 = const()[name = string("value_states_73_groups_0"), val = int32(1)]; tensor value_states_73_cast_fp16 = conv(dilations = value_states_73_dilations_0, groups = value_states_73_groups_0, pad = value_states_73_pad_0, pad_type = value_states_73_pad_type_0, strides = value_states_73_strides_0, weight = layers_12_self_attn_v_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("value_states_73_cast_fp16")]; tensor concat_144x = const()[name = string("concat_144x"), val = tensor([1, 16, 128, -1])]; tensor x_121_cast_fp16 = reshape(shape = concat_144x, x = query_states_73_cast_fp16)[name = string("x_121_cast_fp16")]; tensor concat_145x = const()[name = string("concat_145x"), val = tensor([1, 2, 128, -1])]; tensor var_4836_cast_fp16 = reshape(shape = concat_145x, x = key_states_121_cast_fp16)[name = string("op_4836_cast_fp16")]; tensor concat_146x = const()[name = string("concat_146x"), val = tensor([1, 2, 128, -1])]; tensor var_4843_cast_fp16 = reshape(shape = concat_146x, x = value_states_73_cast_fp16)[name = string("op_4843_cast_fp16")]; tensor var_4847_cast_fp16 = mul(x = x_121_cast_fp16, y = var_869_cast_fp16)[name = string("op_4847_cast_fp16")]; tensor var_4848_split_sizes_0 = const()[name = string("op_4848_split_sizes_0"), val = tensor([64, 64])]; int32 var_4848_axis_0 = const()[name = string("op_4848_axis_0"), val = int32(-2)]; tensor var_4848_cast_fp16_0, tensor var_4848_cast_fp16_1 = split(axis = var_4848_axis_0, split_sizes = var_4848_split_sizes_0, x = x_121_cast_fp16)[name = string("op_4848_cast_fp16")]; fp16 const_122_promoted_to_fp16 = const()[name = string("const_122_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4850_cast_fp16 = mul(x = var_4848_cast_fp16_1, y = const_122_promoted_to_fp16)[name = string("op_4850_cast_fp16")]; int32 var_4852 = const()[name = string("op_4852"), val = int32(-2)]; bool var_4853_interleave_0 = const()[name = string("op_4853_interleave_0"), val = bool(false)]; tensor var_4853_cast_fp16 = concat(axis = var_4852, interleave = var_4853_interleave_0, values = (var_4850_cast_fp16, var_4848_cast_fp16_0))[name = string("op_4853_cast_fp16")]; tensor var_4854_cast_fp16 = mul(x = var_4853_cast_fp16, y = var_878_cast_fp16)[name = string("op_4854_cast_fp16")]; tensor query_states_75_cast_fp16 = add(x = var_4847_cast_fp16, y = var_4854_cast_fp16)[name = string("query_states_75_cast_fp16")]; tensor var_4860_cast_fp16 = mul(x = var_4836_cast_fp16, y = var_869_cast_fp16)[name = string("op_4860_cast_fp16")]; tensor var_4861_split_sizes_0 = const()[name = string("op_4861_split_sizes_0"), val = tensor([64, 64])]; int32 var_4861_axis_0 = const()[name = string("op_4861_axis_0"), val = int32(-2)]; tensor var_4861_cast_fp16_0, tensor var_4861_cast_fp16_1 = split(axis = var_4861_axis_0, split_sizes = var_4861_split_sizes_0, x = var_4836_cast_fp16)[name = string("op_4861_cast_fp16")]; fp16 const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4863_cast_fp16 = mul(x = var_4861_cast_fp16_1, y = const_123_promoted_to_fp16)[name = string("op_4863_cast_fp16")]; int32 var_4865 = const()[name = string("op_4865"), val = int32(-2)]; bool var_4866_interleave_0 = const()[name = string("op_4866_interleave_0"), val = bool(false)]; tensor var_4866_cast_fp16 = concat(axis = var_4865, interleave = var_4866_interleave_0, values = (var_4863_cast_fp16, var_4861_cast_fp16_0))[name = string("op_4866_cast_fp16")]; tensor var_4867_cast_fp16 = mul(x = var_4866_cast_fp16, y = var_878_cast_fp16)[name = string("op_4867_cast_fp16")]; tensor key_states_125_cast_fp16 = add(x = var_4860_cast_fp16, y = var_4867_cast_fp16)[name = string("key_states_125_cast_fp16")]; tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; int32 concat_149_axis_0 = const()[name = string("concat_149_axis_0"), val = int32(0)]; bool concat_149_interleave_0 = const()[name = string("concat_149_interleave_0"), val = bool(false)]; tensor concat_149 = concat(axis = concat_149_axis_0, interleave = concat_149_interleave_0, values = (expand_dims_144, expand_dims_145, position_id, expand_dims_147))[name = string("concat_149")]; tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; tensor concat_150_values1_0 = const()[name = string("concat_150_values1_0"), val = tensor([0])]; tensor concat_150_values3_0 = const()[name = string("concat_150_values3_0"), val = tensor([0])]; int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_148, concat_150_values1_0, cache_position_end, concat_150_values3_0))[name = string("concat_150")]; tensor key_states_127_perm_0 = const()[name = string("key_states_127_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_13_stride_0 = const()[name = string("key_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_127_cast_fp16 = transpose(perm = key_states_127_perm_0, x = key_states_125_cast_fp16)[name = string("transpose_563")]; tensor key_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = key_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = key_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_13_squeeze_mask_0, stride = key_cache_internal_tensor_assign_13_stride_0, update = key_states_127_cast_fp16, x = coreml_update_state_358)[name = string("key_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_13_cast_fp16, input = key_cache)[name = string("coreml_update_state_360_write_state")]; tensor coreml_update_state_360 = read_state(input = key_cache)[name = string("coreml_update_state_360")]; tensor value_states_75_perm_0 = const()[name = string("value_states_75_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_13_stride_0 = const()[name = string("value_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_75_cast_fp16 = transpose(perm = value_states_75_perm_0, x = var_4843_cast_fp16)[name = string("transpose_562")]; tensor value_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = value_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = value_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_13_squeeze_mask_0, stride = value_cache_internal_tensor_assign_13_stride_0, update = value_states_75_cast_fp16, x = coreml_update_state_359)[name = string("value_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_13_cast_fp16, input = value_cache)[name = string("coreml_update_state_361_write_state")]; tensor coreml_update_state_361 = read_state(input = value_cache)[name = string("coreml_update_state_361")]; tensor var_4937_begin_0 = const()[name = string("op_4937_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4937_end_0 = const()[name = string("op_4937_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4937_end_mask_0 = const()[name = string("op_4937_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4937_cast_fp16 = slice_by_index(begin = var_4937_begin_0, end = var_4937_end_0, end_mask = var_4937_end_mask_0, x = coreml_update_state_360)[name = string("op_4937_cast_fp16")]; tensor tile_24 = const()[name = string("tile_24"), val = tensor([1, 1])]; int32 var_4940_axis_0 = const()[name = string("op_4940_axis_0"), val = int32(1)]; tensor var_4940_cast_fp16_0, tensor var_4940_cast_fp16_1 = split(axis = var_4940_axis_0, split_sizes = tile_24, x = var_4937_cast_fp16)[name = string("op_4940_cast_fp16")]; tensor var_4947_begin_0 = const()[name = string("op_4947_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4947_end_0 = const()[name = string("op_4947_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4947_end_mask_0 = const()[name = string("op_4947_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4947_cast_fp16 = slice_by_index(begin = var_4947_begin_0, end = var_4947_end_0, end_mask = var_4947_end_mask_0, x = coreml_update_state_361)[name = string("op_4947_cast_fp16")]; tensor tile_25 = const()[name = string("tile_25"), val = tensor([1, 1])]; int32 var_4950_axis_0 = const()[name = string("op_4950_axis_0"), val = int32(1)]; tensor var_4950_cast_fp16_0, tensor var_4950_cast_fp16_1 = split(axis = var_4950_axis_0, split_sizes = tile_25, x = var_4947_cast_fp16)[name = string("op_4950_cast_fp16")]; tensor var_4953_split_sizes_0 = const()[name = string("op_4953_split_sizes_0"), val = tensor([8, 8])]; int32 var_4953_axis_0 = const()[name = string("op_4953_axis_0"), val = int32(1)]; tensor var_4953_0, tensor var_4953_1 = split(axis = var_4953_axis_0, split_sizes = var_4953_split_sizes_0, x = query_states_75_cast_fp16)[name = string("op_4953")]; bool attn_weights_193_transpose_x_0 = const()[name = string("attn_weights_193_transpose_x_0"), val = bool(false)]; bool attn_weights_193_transpose_y_0 = const()[name = string("attn_weights_193_transpose_y_0"), val = bool(false)]; tensor attn_weights_193_cast_fp16 = matmul(transpose_x = attn_weights_193_transpose_x_0, transpose_y = attn_weights_193_transpose_y_0, x = var_4940_cast_fp16_0, y = var_4953_0)[name = string("attn_weights_193_cast_fp16")]; fp16 var_4956_to_fp16 = const()[name = string("op_4956_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_195_cast_fp16 = mul(x = attn_weights_193_cast_fp16, y = var_4956_to_fp16)[name = string("attn_weights_195_cast_fp16")]; tensor attn_weights_197_cast_fp16 = add(x = attn_weights_195_cast_fp16, y = attn_mask_1)[name = string("attn_weights_197_cast_fp16")]; int32 var_4960 = const()[name = string("op_4960"), val = int32(-2)]; tensor attn_weights_199_cast_fp16 = softmax(axis = var_4960, x = attn_weights_197_cast_fp16)[name = string("attn_weights_199_cast_fp16")]; bool var_4966_transpose_x_1 = const()[name = string("op_4966_transpose_x_1"), val = bool(true)]; bool var_4966_transpose_y_1 = const()[name = string("op_4966_transpose_y_1"), val = bool(false)]; tensor var_4966_cast_fp16 = matmul(transpose_x = var_4966_transpose_x_1, transpose_y = var_4966_transpose_y_1, x = attn_weights_199_cast_fp16, y = var_4950_cast_fp16_0)[name = string("op_4966_cast_fp16")]; bool attn_weights_201_transpose_x_0 = const()[name = string("attn_weights_201_transpose_x_0"), val = bool(false)]; bool attn_weights_201_transpose_y_0 = const()[name = string("attn_weights_201_transpose_y_0"), val = bool(false)]; tensor attn_weights_201_cast_fp16 = matmul(transpose_x = attn_weights_201_transpose_x_0, transpose_y = attn_weights_201_transpose_y_0, x = var_4940_cast_fp16_1, y = var_4953_1)[name = string("attn_weights_201_cast_fp16")]; fp16 var_4968_to_fp16 = const()[name = string("op_4968_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_203_cast_fp16 = mul(x = attn_weights_201_cast_fp16, y = var_4968_to_fp16)[name = string("attn_weights_203_cast_fp16")]; tensor attn_weights_205_cast_fp16 = add(x = attn_weights_203_cast_fp16, y = attn_mask_1)[name = string("attn_weights_205_cast_fp16")]; int32 var_4972 = const()[name = string("op_4972"), val = int32(-2)]; tensor attn_weights_207_cast_fp16 = softmax(axis = var_4972, x = attn_weights_205_cast_fp16)[name = string("attn_weights_207_cast_fp16")]; bool attn_output_97_transpose_x_1 = const()[name = string("attn_output_97_transpose_x_1"), val = bool(true)]; bool attn_output_97_transpose_y_1 = const()[name = string("attn_output_97_transpose_y_1"), val = bool(false)]; tensor attn_output_97_cast_fp16 = matmul(transpose_x = attn_output_97_transpose_x_1, transpose_y = attn_output_97_transpose_y_1, x = attn_weights_207_cast_fp16, y = var_4950_cast_fp16_1)[name = string("attn_output_97_cast_fp16")]; int32 var_4980 = const()[name = string("op_4980"), val = int32(1)]; bool attn_output_99_interleave_0 = const()[name = string("attn_output_99_interleave_0"), val = bool(false)]; tensor attn_output_99_cast_fp16 = concat(axis = var_4980, interleave = attn_output_99_interleave_0, values = (var_4966_cast_fp16, attn_output_97_cast_fp16))[name = string("attn_output_99_cast_fp16")]; tensor var_4984_perm_0 = const()[name = string("op_4984_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_155x = const()[name = string("concat_155x"), val = tensor([1, 2048, 1, -1])]; tensor var_4984_cast_fp16 = transpose(perm = var_4984_perm_0, x = attn_output_99_cast_fp16)[name = string("transpose_561")]; tensor attn_output_103_cast_fp16 = reshape(shape = concat_155x, x = var_4984_cast_fp16)[name = string("attn_output_103_cast_fp16")]; tensor hidden_states_123_strides_0 = const()[name = string("hidden_states_123_strides_0"), val = tensor([1, 1])]; string hidden_states_123_pad_type_0 = const()[name = string("hidden_states_123_pad_type_0"), val = string("valid")]; tensor hidden_states_123_pad_0 = const()[name = string("hidden_states_123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_123_dilations_0 = const()[name = string("hidden_states_123_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_123_groups_0 = const()[name = string("hidden_states_123_groups_0"), val = int32(1)]; tensor hidden_states_123_cast_fp16 = conv(dilations = hidden_states_123_dilations_0, groups = hidden_states_123_groups_0, pad = hidden_states_123_pad_0, pad_type = hidden_states_123_pad_type_0, strides = hidden_states_123_strides_0, weight = layers_12_self_attn_o_proj_weight_cast_fp16, x = attn_output_103_cast_fp16)[name = string("hidden_states_123_cast_fp16")]; tensor hidden_states_125_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = hidden_states_123_cast_fp16)[name = string("hidden_states_125_cast_fp16")]; fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5017_cast_fp16 = mul(x = hidden_states_125_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_5017_cast_fp16")]; int32 var_5015 = const()[name = string("op_5015"), val = int32(1)]; bool doubled_101_interleave_0 = const()[name = string("doubled_101_interleave_0"), val = bool(false)]; tensor doubled_101_cast_fp16 = concat(axis = var_5015, interleave = doubled_101_interleave_0, values = (hidden_states_125_cast_fp16, var_5017_cast_fp16))[name = string("doubled_101_cast_fp16")]; tensor out_51_axes_0 = const()[name = string("out_51_axes_0"), val = tensor([1])]; tensor out_51_gamma_0_to_fp16 = const()[name = string("out_51_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415382912)))]; fp16 var_5027_to_fp16 = const()[name = string("op_5027_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_51_cast_fp16 = layer_norm(axes = out_51_axes_0, epsilon = var_5027_to_fp16, gamma = out_51_gamma_0_to_fp16, x = doubled_101_cast_fp16)[name = string("out_51_cast_fp16")]; tensor var_5038_split_sizes_0 = const()[name = string("op_5038_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5038_axis_0 = const()[name = string("op_5038_axis_0"), val = int32(1)]; tensor var_5038_cast_fp16_0, tensor var_5038_cast_fp16_1 = split(axis = var_5038_axis_0, split_sizes = var_5038_split_sizes_0, x = out_51_cast_fp16)[name = string("op_5038_cast_fp16")]; tensor input_25_strides_0 = const()[name = string("input_25_strides_0"), val = tensor([1, 1])]; string input_25_pad_type_0 = const()[name = string("input_25_pad_type_0"), val = string("valid")]; tensor input_25_pad_0 = const()[name = string("input_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_25_dilations_0 = const()[name = string("input_25_dilations_0"), val = tensor([1, 1])]; int32 input_25_groups_0 = const()[name = string("input_25_groups_0"), val = int32(1)]; tensor input_25_cast_fp16 = conv(dilations = input_25_dilations_0, groups = input_25_groups_0, pad = input_25_pad_0, pad_type = input_25_pad_type_0, strides = input_25_strides_0, weight = layers_12_mlp_gate_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("input_25_cast_fp16")]; tensor var_5055_cast_fp16 = silu(x = input_25_cast_fp16)[name = string("op_5055_cast_fp16")]; tensor var_5061_strides_0 = const()[name = string("op_5061_strides_0"), val = tensor([1, 1])]; string var_5061_pad_type_0 = const()[name = string("op_5061_pad_type_0"), val = string("valid")]; tensor var_5061_pad_0 = const()[name = string("op_5061_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5061_dilations_0 = const()[name = string("op_5061_dilations_0"), val = tensor([1, 1])]; int32 var_5061_groups_0 = const()[name = string("op_5061_groups_0"), val = int32(1)]; tensor var_5061_cast_fp16 = conv(dilations = var_5061_dilations_0, groups = var_5061_groups_0, pad = var_5061_pad_0, pad_type = var_5061_pad_type_0, strides = var_5061_strides_0, weight = layers_12_mlp_up_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("op_5061_cast_fp16")]; tensor x_129_cast_fp16 = mul(x = var_5055_cast_fp16, y = var_5061_cast_fp16)[name = string("x_129_cast_fp16")]; tensor hidden_states_127_strides_0 = const()[name = string("hidden_states_127_strides_0"), val = tensor([1, 1])]; string hidden_states_127_pad_type_0 = const()[name = string("hidden_states_127_pad_type_0"), val = string("valid")]; tensor hidden_states_127_pad_0 = const()[name = string("hidden_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_127_dilations_0 = const()[name = string("hidden_states_127_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_127_groups_0 = const()[name = string("hidden_states_127_groups_0"), val = int32(1)]; tensor hidden_states_127_cast_fp16 = conv(dilations = hidden_states_127_dilations_0, groups = hidden_states_127_groups_0, pad = hidden_states_127_pad_0, pad_type = hidden_states_127_pad_type_0, strides = hidden_states_127_strides_0, weight = layers_12_mlp_down_proj_weight_cast_fp16, x = x_129_cast_fp16)[name = string("hidden_states_127_cast_fp16")]; tensor hidden_states_129_cast_fp16 = add(x = hidden_states_125_cast_fp16, y = hidden_states_127_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5079_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_130_promoted_to_fp16)[name = string("op_5079_cast_fp16")]; int32 var_5077 = const()[name = string("op_5077"), val = int32(1)]; bool doubled_105_interleave_0 = const()[name = string("doubled_105_interleave_0"), val = bool(false)]; tensor doubled_105_cast_fp16 = concat(axis = var_5077, interleave = doubled_105_interleave_0, values = (hidden_states_129_cast_fp16, var_5079_cast_fp16))[name = string("doubled_105_cast_fp16")]; tensor out_53_axes_0 = const()[name = string("out_53_axes_0"), val = tensor([1])]; tensor out_53_gamma_0_to_fp16 = const()[name = string("out_53_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415391168)))]; fp16 var_5089_to_fp16 = const()[name = string("op_5089_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_53_cast_fp16 = layer_norm(axes = out_53_axes_0, epsilon = var_5089_to_fp16, gamma = out_53_gamma_0_to_fp16, x = doubled_105_cast_fp16)[name = string("out_53_cast_fp16")]; tensor var_5100_split_sizes_0 = const()[name = string("op_5100_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5100_axis_0 = const()[name = string("op_5100_axis_0"), val = int32(1)]; tensor var_5100_cast_fp16_0, tensor var_5100_cast_fp16_1 = split(axis = var_5100_axis_0, split_sizes = var_5100_split_sizes_0, x = out_53_cast_fp16)[name = string("op_5100_cast_fp16")]; tensor query_states_79_strides_0 = const()[name = string("query_states_79_strides_0"), val = tensor([1, 1])]; string query_states_79_pad_type_0 = const()[name = string("query_states_79_pad_type_0"), val = string("valid")]; tensor query_states_79_pad_0 = const()[name = string("query_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_79_dilations_0 = const()[name = string("query_states_79_dilations_0"), val = tensor([1, 1])]; int32 query_states_79_groups_0 = const()[name = string("query_states_79_groups_0"), val = int32(1)]; tensor query_states_79_cast_fp16 = conv(dilations = query_states_79_dilations_0, groups = query_states_79_groups_0, pad = query_states_79_pad_0, pad_type = query_states_79_pad_type_0, strides = query_states_79_strides_0, weight = layers_13_self_attn_q_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("query_states_79_cast_fp16")]; tensor key_states_131_strides_0 = const()[name = string("key_states_131_strides_0"), val = tensor([1, 1])]; string key_states_131_pad_type_0 = const()[name = string("key_states_131_pad_type_0"), val = string("valid")]; tensor key_states_131_pad_0 = const()[name = string("key_states_131_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_131_dilations_0 = const()[name = string("key_states_131_dilations_0"), val = tensor([1, 1])]; int32 key_states_131_groups_0 = const()[name = string("key_states_131_groups_0"), val = int32(1)]; tensor key_states_131_cast_fp16 = conv(dilations = key_states_131_dilations_0, groups = key_states_131_groups_0, pad = key_states_131_pad_0, pad_type = key_states_131_pad_type_0, strides = key_states_131_strides_0, weight = layers_13_self_attn_k_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("key_states_131_cast_fp16")]; tensor value_states_79_strides_0 = const()[name = string("value_states_79_strides_0"), val = tensor([1, 1])]; string value_states_79_pad_type_0 = const()[name = string("value_states_79_pad_type_0"), val = string("valid")]; tensor value_states_79_pad_0 = const()[name = string("value_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_79_dilations_0 = const()[name = string("value_states_79_dilations_0"), val = tensor([1, 1])]; int32 value_states_79_groups_0 = const()[name = string("value_states_79_groups_0"), val = int32(1)]; tensor value_states_79_cast_fp16 = conv(dilations = value_states_79_dilations_0, groups = value_states_79_groups_0, pad = value_states_79_pad_0, pad_type = value_states_79_pad_type_0, strides = value_states_79_strides_0, weight = layers_13_self_attn_v_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("value_states_79_cast_fp16")]; tensor concat_156x = const()[name = string("concat_156x"), val = tensor([1, 16, 128, -1])]; tensor x_131_cast_fp16 = reshape(shape = concat_156x, x = query_states_79_cast_fp16)[name = string("x_131_cast_fp16")]; tensor concat_157x = const()[name = string("concat_157x"), val = tensor([1, 2, 128, -1])]; tensor var_5157_cast_fp16 = reshape(shape = concat_157x, x = key_states_131_cast_fp16)[name = string("op_5157_cast_fp16")]; tensor concat_158x = const()[name = string("concat_158x"), val = tensor([1, 2, 128, -1])]; tensor var_5164_cast_fp16 = reshape(shape = concat_158x, x = value_states_79_cast_fp16)[name = string("op_5164_cast_fp16")]; tensor var_5168_cast_fp16 = mul(x = x_131_cast_fp16, y = var_869_cast_fp16)[name = string("op_5168_cast_fp16")]; tensor var_5169_split_sizes_0 = const()[name = string("op_5169_split_sizes_0"), val = tensor([64, 64])]; int32 var_5169_axis_0 = const()[name = string("op_5169_axis_0"), val = int32(-2)]; tensor var_5169_cast_fp16_0, tensor var_5169_cast_fp16_1 = split(axis = var_5169_axis_0, split_sizes = var_5169_split_sizes_0, x = x_131_cast_fp16)[name = string("op_5169_cast_fp16")]; fp16 const_132_promoted_to_fp16 = const()[name = string("const_132_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5171_cast_fp16 = mul(x = var_5169_cast_fp16_1, y = const_132_promoted_to_fp16)[name = string("op_5171_cast_fp16")]; int32 var_5173 = const()[name = string("op_5173"), val = int32(-2)]; bool var_5174_interleave_0 = const()[name = string("op_5174_interleave_0"), val = bool(false)]; tensor var_5174_cast_fp16 = concat(axis = var_5173, interleave = var_5174_interleave_0, values = (var_5171_cast_fp16, var_5169_cast_fp16_0))[name = string("op_5174_cast_fp16")]; tensor var_5175_cast_fp16 = mul(x = var_5174_cast_fp16, y = var_878_cast_fp16)[name = string("op_5175_cast_fp16")]; tensor query_states_81_cast_fp16 = add(x = var_5168_cast_fp16, y = var_5175_cast_fp16)[name = string("query_states_81_cast_fp16")]; tensor var_5181_cast_fp16 = mul(x = var_5157_cast_fp16, y = var_869_cast_fp16)[name = string("op_5181_cast_fp16")]; tensor var_5182_split_sizes_0 = const()[name = string("op_5182_split_sizes_0"), val = tensor([64, 64])]; int32 var_5182_axis_0 = const()[name = string("op_5182_axis_0"), val = int32(-2)]; tensor var_5182_cast_fp16_0, tensor var_5182_cast_fp16_1 = split(axis = var_5182_axis_0, split_sizes = var_5182_split_sizes_0, x = var_5157_cast_fp16)[name = string("op_5182_cast_fp16")]; fp16 const_133_promoted_to_fp16 = const()[name = string("const_133_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5184_cast_fp16 = mul(x = var_5182_cast_fp16_1, y = const_133_promoted_to_fp16)[name = string("op_5184_cast_fp16")]; int32 var_5186 = const()[name = string("op_5186"), val = int32(-2)]; bool var_5187_interleave_0 = const()[name = string("op_5187_interleave_0"), val = bool(false)]; tensor var_5187_cast_fp16 = concat(axis = var_5186, interleave = var_5187_interleave_0, values = (var_5184_cast_fp16, var_5182_cast_fp16_0))[name = string("op_5187_cast_fp16")]; tensor var_5188_cast_fp16 = mul(x = var_5187_cast_fp16, y = var_878_cast_fp16)[name = string("op_5188_cast_fp16")]; tensor key_states_135_cast_fp16 = add(x = var_5181_cast_fp16, y = var_5188_cast_fp16)[name = string("key_states_135_cast_fp16")]; tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; int32 concat_161_axis_0 = const()[name = string("concat_161_axis_0"), val = int32(0)]; bool concat_161_interleave_0 = const()[name = string("concat_161_interleave_0"), val = bool(false)]; tensor concat_161 = concat(axis = concat_161_axis_0, interleave = concat_161_interleave_0, values = (expand_dims_156, expand_dims_157, position_id, expand_dims_159))[name = string("concat_161")]; tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; tensor concat_162_values1_0 = const()[name = string("concat_162_values1_0"), val = tensor([0])]; tensor concat_162_values3_0 = const()[name = string("concat_162_values3_0"), val = tensor([0])]; int32 concat_162_axis_0 = const()[name = string("concat_162_axis_0"), val = int32(0)]; bool concat_162_interleave_0 = const()[name = string("concat_162_interleave_0"), val = bool(false)]; tensor concat_162 = concat(axis = concat_162_axis_0, interleave = concat_162_interleave_0, values = (expand_dims_160, concat_162_values1_0, cache_position_end, concat_162_values3_0))[name = string("concat_162")]; tensor key_states_137_perm_0 = const()[name = string("key_states_137_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_14_stride_0 = const()[name = string("key_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_137_cast_fp16 = transpose(perm = key_states_137_perm_0, x = key_states_135_cast_fp16)[name = string("transpose_560")]; tensor key_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = key_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = key_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_14_squeeze_mask_0, stride = key_cache_internal_tensor_assign_14_stride_0, update = key_states_137_cast_fp16, x = coreml_update_state_360)[name = string("key_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_14_cast_fp16, input = key_cache)[name = string("coreml_update_state_362_write_state")]; tensor coreml_update_state_362 = read_state(input = key_cache)[name = string("coreml_update_state_362")]; tensor value_states_81_perm_0 = const()[name = string("value_states_81_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_14_stride_0 = const()[name = string("value_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_81_cast_fp16 = transpose(perm = value_states_81_perm_0, x = var_5164_cast_fp16)[name = string("transpose_559")]; tensor value_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = value_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = value_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_14_squeeze_mask_0, stride = value_cache_internal_tensor_assign_14_stride_0, update = value_states_81_cast_fp16, x = coreml_update_state_361)[name = string("value_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_14_cast_fp16, input = value_cache)[name = string("coreml_update_state_363_write_state")]; tensor coreml_update_state_363 = read_state(input = value_cache)[name = string("coreml_update_state_363")]; tensor var_5258_begin_0 = const()[name = string("op_5258_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5258_end_0 = const()[name = string("op_5258_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5258_end_mask_0 = const()[name = string("op_5258_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5258_cast_fp16 = slice_by_index(begin = var_5258_begin_0, end = var_5258_end_0, end_mask = var_5258_end_mask_0, x = coreml_update_state_362)[name = string("op_5258_cast_fp16")]; tensor tile_26 = const()[name = string("tile_26"), val = tensor([1, 1])]; int32 var_5261_axis_0 = const()[name = string("op_5261_axis_0"), val = int32(1)]; tensor var_5261_cast_fp16_0, tensor var_5261_cast_fp16_1 = split(axis = var_5261_axis_0, split_sizes = tile_26, x = var_5258_cast_fp16)[name = string("op_5261_cast_fp16")]; tensor var_5268_begin_0 = const()[name = string("op_5268_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5268_end_0 = const()[name = string("op_5268_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5268_end_mask_0 = const()[name = string("op_5268_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5268_cast_fp16 = slice_by_index(begin = var_5268_begin_0, end = var_5268_end_0, end_mask = var_5268_end_mask_0, x = coreml_update_state_363)[name = string("op_5268_cast_fp16")]; tensor tile_27 = const()[name = string("tile_27"), val = tensor([1, 1])]; int32 var_5271_axis_0 = const()[name = string("op_5271_axis_0"), val = int32(1)]; tensor var_5271_cast_fp16_0, tensor var_5271_cast_fp16_1 = split(axis = var_5271_axis_0, split_sizes = tile_27, x = var_5268_cast_fp16)[name = string("op_5271_cast_fp16")]; tensor var_5274_split_sizes_0 = const()[name = string("op_5274_split_sizes_0"), val = tensor([8, 8])]; int32 var_5274_axis_0 = const()[name = string("op_5274_axis_0"), val = int32(1)]; tensor var_5274_0, tensor var_5274_1 = split(axis = var_5274_axis_0, split_sizes = var_5274_split_sizes_0, x = query_states_81_cast_fp16)[name = string("op_5274")]; bool attn_weights_209_transpose_x_0 = const()[name = string("attn_weights_209_transpose_x_0"), val = bool(false)]; bool attn_weights_209_transpose_y_0 = const()[name = string("attn_weights_209_transpose_y_0"), val = bool(false)]; tensor attn_weights_209_cast_fp16 = matmul(transpose_x = attn_weights_209_transpose_x_0, transpose_y = attn_weights_209_transpose_y_0, x = var_5261_cast_fp16_0, y = var_5274_0)[name = string("attn_weights_209_cast_fp16")]; fp16 var_5277_to_fp16 = const()[name = string("op_5277_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_211_cast_fp16 = mul(x = attn_weights_209_cast_fp16, y = var_5277_to_fp16)[name = string("attn_weights_211_cast_fp16")]; tensor attn_weights_213_cast_fp16 = add(x = attn_weights_211_cast_fp16, y = attn_mask_1)[name = string("attn_weights_213_cast_fp16")]; int32 var_5281 = const()[name = string("op_5281"), val = int32(-2)]; tensor attn_weights_215_cast_fp16 = softmax(axis = var_5281, x = attn_weights_213_cast_fp16)[name = string("attn_weights_215_cast_fp16")]; bool var_5287_transpose_x_1 = const()[name = string("op_5287_transpose_x_1"), val = bool(true)]; bool var_5287_transpose_y_1 = const()[name = string("op_5287_transpose_y_1"), val = bool(false)]; tensor var_5287_cast_fp16 = matmul(transpose_x = var_5287_transpose_x_1, transpose_y = var_5287_transpose_y_1, x = attn_weights_215_cast_fp16, y = var_5271_cast_fp16_0)[name = string("op_5287_cast_fp16")]; bool attn_weights_217_transpose_x_0 = const()[name = string("attn_weights_217_transpose_x_0"), val = bool(false)]; bool attn_weights_217_transpose_y_0 = const()[name = string("attn_weights_217_transpose_y_0"), val = bool(false)]; tensor attn_weights_217_cast_fp16 = matmul(transpose_x = attn_weights_217_transpose_x_0, transpose_y = attn_weights_217_transpose_y_0, x = var_5261_cast_fp16_1, y = var_5274_1)[name = string("attn_weights_217_cast_fp16")]; fp16 var_5289_to_fp16 = const()[name = string("op_5289_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_219_cast_fp16 = mul(x = attn_weights_217_cast_fp16, y = var_5289_to_fp16)[name = string("attn_weights_219_cast_fp16")]; tensor attn_weights_221_cast_fp16 = add(x = attn_weights_219_cast_fp16, y = attn_mask_1)[name = string("attn_weights_221_cast_fp16")]; int32 var_5293 = const()[name = string("op_5293"), val = int32(-2)]; tensor attn_weights_223_cast_fp16 = softmax(axis = var_5293, x = attn_weights_221_cast_fp16)[name = string("attn_weights_223_cast_fp16")]; bool attn_output_105_transpose_x_1 = const()[name = string("attn_output_105_transpose_x_1"), val = bool(true)]; bool attn_output_105_transpose_y_1 = const()[name = string("attn_output_105_transpose_y_1"), val = bool(false)]; tensor attn_output_105_cast_fp16 = matmul(transpose_x = attn_output_105_transpose_x_1, transpose_y = attn_output_105_transpose_y_1, x = attn_weights_223_cast_fp16, y = var_5271_cast_fp16_1)[name = string("attn_output_105_cast_fp16")]; int32 var_5301 = const()[name = string("op_5301"), val = int32(1)]; bool attn_output_107_interleave_0 = const()[name = string("attn_output_107_interleave_0"), val = bool(false)]; tensor attn_output_107_cast_fp16 = concat(axis = var_5301, interleave = attn_output_107_interleave_0, values = (var_5287_cast_fp16, attn_output_105_cast_fp16))[name = string("attn_output_107_cast_fp16")]; tensor var_5305_perm_0 = const()[name = string("op_5305_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_167x = const()[name = string("concat_167x"), val = tensor([1, 2048, 1, -1])]; tensor var_5305_cast_fp16 = transpose(perm = var_5305_perm_0, x = attn_output_107_cast_fp16)[name = string("transpose_558")]; tensor attn_output_111_cast_fp16 = reshape(shape = concat_167x, x = var_5305_cast_fp16)[name = string("attn_output_111_cast_fp16")]; tensor hidden_states_133_strides_0 = const()[name = string("hidden_states_133_strides_0"), val = tensor([1, 1])]; string hidden_states_133_pad_type_0 = const()[name = string("hidden_states_133_pad_type_0"), val = string("valid")]; tensor hidden_states_133_pad_0 = const()[name = string("hidden_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_133_dilations_0 = const()[name = string("hidden_states_133_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_133_groups_0 = const()[name = string("hidden_states_133_groups_0"), val = int32(1)]; tensor hidden_states_133_cast_fp16 = conv(dilations = hidden_states_133_dilations_0, groups = hidden_states_133_groups_0, pad = hidden_states_133_pad_0, pad_type = hidden_states_133_pad_type_0, strides = hidden_states_133_strides_0, weight = layers_13_self_attn_o_proj_weight_cast_fp16, x = attn_output_111_cast_fp16)[name = string("hidden_states_133_cast_fp16")]; tensor hidden_states_135_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = hidden_states_133_cast_fp16)[name = string("hidden_states_135_cast_fp16")]; fp16 const_138_promoted_to_fp16 = const()[name = string("const_138_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5338_cast_fp16 = mul(x = hidden_states_135_cast_fp16, y = const_138_promoted_to_fp16)[name = string("op_5338_cast_fp16")]; int32 var_5336 = const()[name = string("op_5336"), val = int32(1)]; bool doubled_109_interleave_0 = const()[name = string("doubled_109_interleave_0"), val = bool(false)]; tensor doubled_109_cast_fp16 = concat(axis = var_5336, interleave = doubled_109_interleave_0, values = (hidden_states_135_cast_fp16, var_5338_cast_fp16))[name = string("doubled_109_cast_fp16")]; tensor out_55_axes_0 = const()[name = string("out_55_axes_0"), val = tensor([1])]; tensor out_55_gamma_0_to_fp16 = const()[name = string("out_55_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415399424)))]; fp16 var_5348_to_fp16 = const()[name = string("op_5348_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_55_cast_fp16 = layer_norm(axes = out_55_axes_0, epsilon = var_5348_to_fp16, gamma = out_55_gamma_0_to_fp16, x = doubled_109_cast_fp16)[name = string("out_55_cast_fp16")]; tensor var_5359_split_sizes_0 = const()[name = string("op_5359_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5359_axis_0 = const()[name = string("op_5359_axis_0"), val = int32(1)]; tensor var_5359_cast_fp16_0, tensor var_5359_cast_fp16_1 = split(axis = var_5359_axis_0, split_sizes = var_5359_split_sizes_0, x = out_55_cast_fp16)[name = string("op_5359_cast_fp16")]; tensor input_27_strides_0 = const()[name = string("input_27_strides_0"), val = tensor([1, 1])]; string input_27_pad_type_0 = const()[name = string("input_27_pad_type_0"), val = string("valid")]; tensor input_27_pad_0 = const()[name = string("input_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_27_dilations_0 = const()[name = string("input_27_dilations_0"), val = tensor([1, 1])]; int32 input_27_groups_0 = const()[name = string("input_27_groups_0"), val = int32(1)]; tensor input_27_cast_fp16 = conv(dilations = input_27_dilations_0, groups = input_27_groups_0, pad = input_27_pad_0, pad_type = input_27_pad_type_0, strides = input_27_strides_0, weight = layers_13_mlp_gate_proj_weight_cast_fp16, x = var_5359_cast_fp16_0)[name = string("input_27_cast_fp16")]; tensor var_5376_cast_fp16 = silu(x = input_27_cast_fp16)[name = string("op_5376_cast_fp16")]; tensor layers_13_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_13_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415407680)))]; tensor var_5382_strides_0 = const()[name = string("op_5382_strides_0"), val = tensor([1, 1])]; string var_5382_pad_type_0 = const()[name = string("op_5382_pad_type_0"), val = string("valid")]; tensor var_5382_pad_0 = const()[name = string("op_5382_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5382_dilations_0 = const()[name = string("op_5382_dilations_0"), val = tensor([1, 1])]; int32 var_5382_groups_0 = const()[name = string("op_5382_groups_0"), val = int32(1)]; tensor var_5382_cast_fp16 = conv(dilations = var_5382_dilations_0, groups = var_5382_groups_0, pad = var_5382_pad_0, pad_type = var_5382_pad_type_0, strides = var_5382_strides_0, weight = layers_13_mlp_up_proj_weight_to_fp16, x = var_5359_cast_fp16_0)[name = string("op_5382_cast_fp16")]; tensor x_139_cast_fp16 = mul(x = var_5376_cast_fp16, y = var_5382_cast_fp16)[name = string("x_139_cast_fp16")]; tensor hidden_states_137_strides_0 = const()[name = string("hidden_states_137_strides_0"), val = tensor([1, 1])]; string hidden_states_137_pad_type_0 = const()[name = string("hidden_states_137_pad_type_0"), val = string("valid")]; tensor hidden_states_137_pad_0 = const()[name = string("hidden_states_137_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_137_dilations_0 = const()[name = string("hidden_states_137_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_137_groups_0 = const()[name = string("hidden_states_137_groups_0"), val = int32(1)]; tensor hidden_states_137_cast_fp16 = conv(dilations = hidden_states_137_dilations_0, groups = hidden_states_137_groups_0, pad = hidden_states_137_pad_0, pad_type = hidden_states_137_pad_type_0, strides = hidden_states_137_strides_0, weight = layers_13_mlp_down_proj_weight_cast_fp16, x = x_139_cast_fp16)[name = string("hidden_states_137_cast_fp16")]; tensor hidden_states_139_cast_fp16 = add(x = hidden_states_135_cast_fp16, y = hidden_states_137_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; fp16 const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5400_cast_fp16 = mul(x = hidden_states_139_cast_fp16, y = const_140_promoted_to_fp16)[name = string("op_5400_cast_fp16")]; int32 var_5398 = const()[name = string("op_5398"), val = int32(1)]; bool doubled_113_interleave_0 = const()[name = string("doubled_113_interleave_0"), val = bool(false)]; tensor doubled_113_cast_fp16 = concat(axis = var_5398, interleave = doubled_113_interleave_0, values = (hidden_states_139_cast_fp16, var_5400_cast_fp16))[name = string("doubled_113_cast_fp16")]; tensor out_57_axes_0 = const()[name = string("out_57_axes_0"), val = tensor([1])]; tensor out_57_gamma_0_to_fp16 = const()[name = string("out_57_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440573568)))]; fp16 var_5410_to_fp16 = const()[name = string("op_5410_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_57_cast_fp16 = layer_norm(axes = out_57_axes_0, epsilon = var_5410_to_fp16, gamma = out_57_gamma_0_to_fp16, x = doubled_113_cast_fp16)[name = string("out_57_cast_fp16")]; tensor var_5421_split_sizes_0 = const()[name = string("op_5421_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5421_axis_0 = const()[name = string("op_5421_axis_0"), val = int32(1)]; tensor var_5421_cast_fp16_0, tensor var_5421_cast_fp16_1 = split(axis = var_5421_axis_0, split_sizes = var_5421_split_sizes_0, x = out_57_cast_fp16)[name = string("op_5421_cast_fp16")]; tensor query_states_85_strides_0 = const()[name = string("query_states_85_strides_0"), val = tensor([1, 1])]; string query_states_85_pad_type_0 = const()[name = string("query_states_85_pad_type_0"), val = string("valid")]; tensor query_states_85_pad_0 = const()[name = string("query_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_85_dilations_0 = const()[name = string("query_states_85_dilations_0"), val = tensor([1, 1])]; int32 query_states_85_groups_0 = const()[name = string("query_states_85_groups_0"), val = int32(1)]; tensor query_states_85_cast_fp16 = conv(dilations = query_states_85_dilations_0, groups = query_states_85_groups_0, pad = query_states_85_pad_0, pad_type = query_states_85_pad_type_0, strides = query_states_85_strides_0, weight = layers_14_self_attn_q_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("query_states_85_cast_fp16")]; tensor layers_14_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_14_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440581824)))]; tensor key_states_141_strides_0 = const()[name = string("key_states_141_strides_0"), val = tensor([1, 1])]; string key_states_141_pad_type_0 = const()[name = string("key_states_141_pad_type_0"), val = string("valid")]; tensor key_states_141_pad_0 = const()[name = string("key_states_141_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_141_dilations_0 = const()[name = string("key_states_141_dilations_0"), val = tensor([1, 1])]; int32 key_states_141_groups_0 = const()[name = string("key_states_141_groups_0"), val = int32(1)]; tensor key_states_141_cast_fp16 = conv(dilations = key_states_141_dilations_0, groups = key_states_141_groups_0, pad = key_states_141_pad_0, pad_type = key_states_141_pad_type_0, strides = key_states_141_strides_0, weight = layers_14_self_attn_k_proj_weight_to_fp16, x = var_5421_cast_fp16_0)[name = string("key_states_141_cast_fp16")]; tensor value_states_85_strides_0 = const()[name = string("value_states_85_strides_0"), val = tensor([1, 1])]; string value_states_85_pad_type_0 = const()[name = string("value_states_85_pad_type_0"), val = string("valid")]; tensor value_states_85_pad_0 = const()[name = string("value_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_85_dilations_0 = const()[name = string("value_states_85_dilations_0"), val = tensor([1, 1])]; int32 value_states_85_groups_0 = const()[name = string("value_states_85_groups_0"), val = int32(1)]; tensor value_states_85_cast_fp16 = conv(dilations = value_states_85_dilations_0, groups = value_states_85_groups_0, pad = value_states_85_pad_0, pad_type = value_states_85_pad_type_0, strides = value_states_85_strides_0, weight = layers_14_self_attn_v_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("value_states_85_cast_fp16")]; tensor concat_168x = const()[name = string("concat_168x"), val = tensor([1, 16, 128, -1])]; tensor x_141_cast_fp16 = reshape(shape = concat_168x, x = query_states_85_cast_fp16)[name = string("x_141_cast_fp16")]; tensor concat_169x = const()[name = string("concat_169x"), val = tensor([1, 2, 128, -1])]; tensor var_5478_cast_fp16 = reshape(shape = concat_169x, x = key_states_141_cast_fp16)[name = string("op_5478_cast_fp16")]; tensor concat_170x = const()[name = string("concat_170x"), val = tensor([1, 2, 128, -1])]; tensor var_5485_cast_fp16 = reshape(shape = concat_170x, x = value_states_85_cast_fp16)[name = string("op_5485_cast_fp16")]; tensor var_5489_cast_fp16 = mul(x = x_141_cast_fp16, y = var_869_cast_fp16)[name = string("op_5489_cast_fp16")]; tensor var_5490_split_sizes_0 = const()[name = string("op_5490_split_sizes_0"), val = tensor([64, 64])]; int32 var_5490_axis_0 = const()[name = string("op_5490_axis_0"), val = int32(-2)]; tensor var_5490_cast_fp16_0, tensor var_5490_cast_fp16_1 = split(axis = var_5490_axis_0, split_sizes = var_5490_split_sizes_0, x = x_141_cast_fp16)[name = string("op_5490_cast_fp16")]; fp16 const_142_promoted_to_fp16 = const()[name = string("const_142_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5492_cast_fp16 = mul(x = var_5490_cast_fp16_1, y = const_142_promoted_to_fp16)[name = string("op_5492_cast_fp16")]; int32 var_5494 = const()[name = string("op_5494"), val = int32(-2)]; bool var_5495_interleave_0 = const()[name = string("op_5495_interleave_0"), val = bool(false)]; tensor var_5495_cast_fp16 = concat(axis = var_5494, interleave = var_5495_interleave_0, values = (var_5492_cast_fp16, var_5490_cast_fp16_0))[name = string("op_5495_cast_fp16")]; tensor var_5496_cast_fp16 = mul(x = var_5495_cast_fp16, y = var_878_cast_fp16)[name = string("op_5496_cast_fp16")]; tensor query_states_87_cast_fp16 = add(x = var_5489_cast_fp16, y = var_5496_cast_fp16)[name = string("query_states_87_cast_fp16")]; tensor var_5502_cast_fp16 = mul(x = var_5478_cast_fp16, y = var_869_cast_fp16)[name = string("op_5502_cast_fp16")]; tensor var_5503_split_sizes_0 = const()[name = string("op_5503_split_sizes_0"), val = tensor([64, 64])]; int32 var_5503_axis_0 = const()[name = string("op_5503_axis_0"), val = int32(-2)]; tensor var_5503_cast_fp16_0, tensor var_5503_cast_fp16_1 = split(axis = var_5503_axis_0, split_sizes = var_5503_split_sizes_0, x = var_5478_cast_fp16)[name = string("op_5503_cast_fp16")]; fp16 const_143_promoted_to_fp16 = const()[name = string("const_143_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5505_cast_fp16 = mul(x = var_5503_cast_fp16_1, y = const_143_promoted_to_fp16)[name = string("op_5505_cast_fp16")]; int32 var_5507 = const()[name = string("op_5507"), val = int32(-2)]; bool var_5508_interleave_0 = const()[name = string("op_5508_interleave_0"), val = bool(false)]; tensor var_5508_cast_fp16 = concat(axis = var_5507, interleave = var_5508_interleave_0, values = (var_5505_cast_fp16, var_5503_cast_fp16_0))[name = string("op_5508_cast_fp16")]; tensor var_5509_cast_fp16 = mul(x = var_5508_cast_fp16, y = var_878_cast_fp16)[name = string("op_5509_cast_fp16")]; tensor key_states_145_cast_fp16 = add(x = var_5502_cast_fp16, y = var_5509_cast_fp16)[name = string("key_states_145_cast_fp16")]; tensor expand_dims_168 = const()[name = string("expand_dims_168"), val = tensor([14])]; tensor expand_dims_169 = const()[name = string("expand_dims_169"), val = tensor([0])]; tensor expand_dims_171 = const()[name = string("expand_dims_171"), val = tensor([0])]; int32 concat_173_axis_0 = const()[name = string("concat_173_axis_0"), val = int32(0)]; bool concat_173_interleave_0 = const()[name = string("concat_173_interleave_0"), val = bool(false)]; tensor concat_173 = concat(axis = concat_173_axis_0, interleave = concat_173_interleave_0, values = (expand_dims_168, expand_dims_169, position_id, expand_dims_171))[name = string("concat_173")]; tensor expand_dims_172 = const()[name = string("expand_dims_172"), val = tensor([15])]; tensor concat_174_values1_0 = const()[name = string("concat_174_values1_0"), val = tensor([0])]; tensor concat_174_values3_0 = const()[name = string("concat_174_values3_0"), val = tensor([0])]; int32 concat_174_axis_0 = const()[name = string("concat_174_axis_0"), val = int32(0)]; bool concat_174_interleave_0 = const()[name = string("concat_174_interleave_0"), val = bool(false)]; tensor concat_174 = concat(axis = concat_174_axis_0, interleave = concat_174_interleave_0, values = (expand_dims_172, concat_174_values1_0, cache_position_end, concat_174_values3_0))[name = string("concat_174")]; tensor key_states_147_perm_0 = const()[name = string("key_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_15_stride_0 = const()[name = string("key_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_147_cast_fp16 = transpose(perm = key_states_147_perm_0, x = key_states_145_cast_fp16)[name = string("transpose_557")]; tensor key_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = key_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = key_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_15_squeeze_mask_0, stride = key_cache_internal_tensor_assign_15_stride_0, update = key_states_147_cast_fp16, x = coreml_update_state_362)[name = string("key_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_15_cast_fp16, input = key_cache)[name = string("coreml_update_state_364_write_state")]; tensor coreml_update_state_364 = read_state(input = key_cache)[name = string("coreml_update_state_364")]; tensor value_states_87_perm_0 = const()[name = string("value_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_15_stride_0 = const()[name = string("value_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_87_cast_fp16 = transpose(perm = value_states_87_perm_0, x = var_5485_cast_fp16)[name = string("transpose_556")]; tensor value_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = value_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = value_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_15_squeeze_mask_0, stride = value_cache_internal_tensor_assign_15_stride_0, update = value_states_87_cast_fp16, x = coreml_update_state_363)[name = string("value_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_15_cast_fp16, input = value_cache)[name = string("coreml_update_state_365_write_state")]; tensor coreml_update_state_365 = read_state(input = value_cache)[name = string("coreml_update_state_365")]; tensor var_5579_begin_0 = const()[name = string("op_5579_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5579_end_0 = const()[name = string("op_5579_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5579_end_mask_0 = const()[name = string("op_5579_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5579_cast_fp16 = slice_by_index(begin = var_5579_begin_0, end = var_5579_end_0, end_mask = var_5579_end_mask_0, x = coreml_update_state_364)[name = string("op_5579_cast_fp16")]; tensor tile_28 = const()[name = string("tile_28"), val = tensor([1, 1])]; int32 var_5582_axis_0 = const()[name = string("op_5582_axis_0"), val = int32(1)]; tensor var_5582_cast_fp16_0, tensor var_5582_cast_fp16_1 = split(axis = var_5582_axis_0, split_sizes = tile_28, x = var_5579_cast_fp16)[name = string("op_5582_cast_fp16")]; tensor var_5589_begin_0 = const()[name = string("op_5589_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5589_end_0 = const()[name = string("op_5589_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5589_end_mask_0 = const()[name = string("op_5589_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5589_cast_fp16 = slice_by_index(begin = var_5589_begin_0, end = var_5589_end_0, end_mask = var_5589_end_mask_0, x = coreml_update_state_365)[name = string("op_5589_cast_fp16")]; tensor tile_29 = const()[name = string("tile_29"), val = tensor([1, 1])]; int32 var_5592_axis_0 = const()[name = string("op_5592_axis_0"), val = int32(1)]; tensor var_5592_cast_fp16_0, tensor var_5592_cast_fp16_1 = split(axis = var_5592_axis_0, split_sizes = tile_29, x = var_5589_cast_fp16)[name = string("op_5592_cast_fp16")]; tensor var_5595_split_sizes_0 = const()[name = string("op_5595_split_sizes_0"), val = tensor([8, 8])]; int32 var_5595_axis_0 = const()[name = string("op_5595_axis_0"), val = int32(1)]; tensor var_5595_0, tensor var_5595_1 = split(axis = var_5595_axis_0, split_sizes = var_5595_split_sizes_0, x = query_states_87_cast_fp16)[name = string("op_5595")]; bool attn_weights_225_transpose_x_0 = const()[name = string("attn_weights_225_transpose_x_0"), val = bool(false)]; bool attn_weights_225_transpose_y_0 = const()[name = string("attn_weights_225_transpose_y_0"), val = bool(false)]; tensor attn_weights_225_cast_fp16 = matmul(transpose_x = attn_weights_225_transpose_x_0, transpose_y = attn_weights_225_transpose_y_0, x = var_5582_cast_fp16_0, y = var_5595_0)[name = string("attn_weights_225_cast_fp16")]; fp16 var_5598_to_fp16 = const()[name = string("op_5598_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_227_cast_fp16 = mul(x = attn_weights_225_cast_fp16, y = var_5598_to_fp16)[name = string("attn_weights_227_cast_fp16")]; tensor attn_weights_229_cast_fp16 = add(x = attn_weights_227_cast_fp16, y = attn_mask_1)[name = string("attn_weights_229_cast_fp16")]; int32 var_5602 = const()[name = string("op_5602"), val = int32(-2)]; tensor attn_weights_231_cast_fp16 = softmax(axis = var_5602, x = attn_weights_229_cast_fp16)[name = string("attn_weights_231_cast_fp16")]; bool var_5608_transpose_x_1 = const()[name = string("op_5608_transpose_x_1"), val = bool(true)]; bool var_5608_transpose_y_1 = const()[name = string("op_5608_transpose_y_1"), val = bool(false)]; tensor var_5608_cast_fp16 = matmul(transpose_x = var_5608_transpose_x_1, transpose_y = var_5608_transpose_y_1, x = attn_weights_231_cast_fp16, y = var_5592_cast_fp16_0)[name = string("op_5608_cast_fp16")]; bool attn_weights_233_transpose_x_0 = const()[name = string("attn_weights_233_transpose_x_0"), val = bool(false)]; bool attn_weights_233_transpose_y_0 = const()[name = string("attn_weights_233_transpose_y_0"), val = bool(false)]; tensor attn_weights_233_cast_fp16 = matmul(transpose_x = attn_weights_233_transpose_x_0, transpose_y = attn_weights_233_transpose_y_0, x = var_5582_cast_fp16_1, y = var_5595_1)[name = string("attn_weights_233_cast_fp16")]; fp16 var_5610_to_fp16 = const()[name = string("op_5610_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_235_cast_fp16 = mul(x = attn_weights_233_cast_fp16, y = var_5610_to_fp16)[name = string("attn_weights_235_cast_fp16")]; tensor attn_weights_237_cast_fp16 = add(x = attn_weights_235_cast_fp16, y = attn_mask_1)[name = string("attn_weights_237_cast_fp16")]; int32 var_5614 = const()[name = string("op_5614"), val = int32(-2)]; tensor attn_weights_239_cast_fp16 = softmax(axis = var_5614, x = attn_weights_237_cast_fp16)[name = string("attn_weights_239_cast_fp16")]; bool attn_output_113_transpose_x_1 = const()[name = string("attn_output_113_transpose_x_1"), val = bool(true)]; bool attn_output_113_transpose_y_1 = const()[name = string("attn_output_113_transpose_y_1"), val = bool(false)]; tensor attn_output_113_cast_fp16 = matmul(transpose_x = attn_output_113_transpose_x_1, transpose_y = attn_output_113_transpose_y_1, x = attn_weights_239_cast_fp16, y = var_5592_cast_fp16_1)[name = string("attn_output_113_cast_fp16")]; int32 var_5622 = const()[name = string("op_5622"), val = int32(1)]; bool attn_output_115_interleave_0 = const()[name = string("attn_output_115_interleave_0"), val = bool(false)]; tensor attn_output_115_cast_fp16 = concat(axis = var_5622, interleave = attn_output_115_interleave_0, values = (var_5608_cast_fp16, attn_output_113_cast_fp16))[name = string("attn_output_115_cast_fp16")]; tensor var_5626_perm_0 = const()[name = string("op_5626_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_179x = const()[name = string("concat_179x"), val = tensor([1, 2048, 1, -1])]; tensor var_5626_cast_fp16 = transpose(perm = var_5626_perm_0, x = attn_output_115_cast_fp16)[name = string("transpose_555")]; tensor attn_output_119_cast_fp16 = reshape(shape = concat_179x, x = var_5626_cast_fp16)[name = string("attn_output_119_cast_fp16")]; tensor hidden_states_143_strides_0 = const()[name = string("hidden_states_143_strides_0"), val = tensor([1, 1])]; string hidden_states_143_pad_type_0 = const()[name = string("hidden_states_143_pad_type_0"), val = string("valid")]; tensor hidden_states_143_pad_0 = const()[name = string("hidden_states_143_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_143_dilations_0 = const()[name = string("hidden_states_143_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_143_groups_0 = const()[name = string("hidden_states_143_groups_0"), val = int32(1)]; tensor hidden_states_143_cast_fp16 = conv(dilations = hidden_states_143_dilations_0, groups = hidden_states_143_groups_0, pad = hidden_states_143_pad_0, pad_type = hidden_states_143_pad_type_0, strides = hidden_states_143_strides_0, weight = layers_14_self_attn_o_proj_weight_cast_fp16, x = attn_output_119_cast_fp16)[name = string("hidden_states_143_cast_fp16")]; tensor hidden_states_145_cast_fp16 = add(x = hidden_states_139_cast_fp16, y = hidden_states_143_cast_fp16)[name = string("hidden_states_145_cast_fp16")]; fp16 const_148_promoted_to_fp16 = const()[name = string("const_148_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5659_cast_fp16 = mul(x = hidden_states_145_cast_fp16, y = const_148_promoted_to_fp16)[name = string("op_5659_cast_fp16")]; int32 var_5657 = const()[name = string("op_5657"), val = int32(1)]; bool doubled_117_interleave_0 = const()[name = string("doubled_117_interleave_0"), val = bool(false)]; tensor doubled_117_cast_fp16 = concat(axis = var_5657, interleave = doubled_117_interleave_0, values = (hidden_states_145_cast_fp16, var_5659_cast_fp16))[name = string("doubled_117_cast_fp16")]; tensor out_59_axes_0 = const()[name = string("out_59_axes_0"), val = tensor([1])]; tensor out_59_gamma_0_to_fp16 = const()[name = string("out_59_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441630464)))]; fp16 var_5669_to_fp16 = const()[name = string("op_5669_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_59_cast_fp16 = layer_norm(axes = out_59_axes_0, epsilon = var_5669_to_fp16, gamma = out_59_gamma_0_to_fp16, x = doubled_117_cast_fp16)[name = string("out_59_cast_fp16")]; tensor var_5680_split_sizes_0 = const()[name = string("op_5680_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5680_axis_0 = const()[name = string("op_5680_axis_0"), val = int32(1)]; tensor var_5680_cast_fp16_0, tensor var_5680_cast_fp16_1 = split(axis = var_5680_axis_0, split_sizes = var_5680_split_sizes_0, x = out_59_cast_fp16)[name = string("op_5680_cast_fp16")]; tensor input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor([1, 1])]; string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; tensor input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor([1, 1])]; int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; tensor input_29_cast_fp16 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_14_mlp_gate_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("input_29_cast_fp16")]; tensor var_5697_cast_fp16 = silu(x = input_29_cast_fp16)[name = string("op_5697_cast_fp16")]; tensor var_5703_strides_0 = const()[name = string("op_5703_strides_0"), val = tensor([1, 1])]; string var_5703_pad_type_0 = const()[name = string("op_5703_pad_type_0"), val = string("valid")]; tensor var_5703_pad_0 = const()[name = string("op_5703_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5703_dilations_0 = const()[name = string("op_5703_dilations_0"), val = tensor([1, 1])]; int32 var_5703_groups_0 = const()[name = string("op_5703_groups_0"), val = int32(1)]; tensor var_5703_cast_fp16 = conv(dilations = var_5703_dilations_0, groups = var_5703_groups_0, pad = var_5703_pad_0, pad_type = var_5703_pad_type_0, strides = var_5703_strides_0, weight = layers_14_mlp_up_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("op_5703_cast_fp16")]; tensor x_149_cast_fp16 = mul(x = var_5697_cast_fp16, y = var_5703_cast_fp16)[name = string("x_149_cast_fp16")]; tensor hidden_states_147_strides_0 = const()[name = string("hidden_states_147_strides_0"), val = tensor([1, 1])]; string hidden_states_147_pad_type_0 = const()[name = string("hidden_states_147_pad_type_0"), val = string("valid")]; tensor hidden_states_147_pad_0 = const()[name = string("hidden_states_147_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_147_dilations_0 = const()[name = string("hidden_states_147_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_147_groups_0 = const()[name = string("hidden_states_147_groups_0"), val = int32(1)]; tensor hidden_states_147_cast_fp16 = conv(dilations = hidden_states_147_dilations_0, groups = hidden_states_147_groups_0, pad = hidden_states_147_pad_0, pad_type = hidden_states_147_pad_type_0, strides = hidden_states_147_strides_0, weight = layers_14_mlp_down_proj_weight_cast_fp16, x = x_149_cast_fp16)[name = string("hidden_states_147_cast_fp16")]; tensor hidden_states_149_cast_fp16 = add(x = hidden_states_145_cast_fp16, y = hidden_states_147_cast_fp16)[name = string("hidden_states_149_cast_fp16")]; fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5721_cast_fp16 = mul(x = hidden_states_149_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_5721_cast_fp16")]; int32 var_5719 = const()[name = string("op_5719"), val = int32(1)]; bool doubled_121_interleave_0 = const()[name = string("doubled_121_interleave_0"), val = bool(false)]; tensor doubled_121_cast_fp16 = concat(axis = var_5719, interleave = doubled_121_interleave_0, values = (hidden_states_149_cast_fp16, var_5721_cast_fp16))[name = string("doubled_121_cast_fp16")]; tensor out_61_axes_0 = const()[name = string("out_61_axes_0"), val = tensor([1])]; tensor out_61_gamma_0_to_fp16 = const()[name = string("out_61_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441638720)))]; fp16 var_5731_to_fp16 = const()[name = string("op_5731_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_61_cast_fp16 = layer_norm(axes = out_61_axes_0, epsilon = var_5731_to_fp16, gamma = out_61_gamma_0_to_fp16, x = doubled_121_cast_fp16)[name = string("out_61_cast_fp16")]; tensor var_5742_split_sizes_0 = const()[name = string("op_5742_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5742_axis_0 = const()[name = string("op_5742_axis_0"), val = int32(1)]; tensor var_5742_cast_fp16_0, tensor var_5742_cast_fp16_1 = split(axis = var_5742_axis_0, split_sizes = var_5742_split_sizes_0, x = out_61_cast_fp16)[name = string("op_5742_cast_fp16")]; tensor query_states_91_strides_0 = const()[name = string("query_states_91_strides_0"), val = tensor([1, 1])]; string query_states_91_pad_type_0 = const()[name = string("query_states_91_pad_type_0"), val = string("valid")]; tensor query_states_91_pad_0 = const()[name = string("query_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_91_dilations_0 = const()[name = string("query_states_91_dilations_0"), val = tensor([1, 1])]; int32 query_states_91_groups_0 = const()[name = string("query_states_91_groups_0"), val = int32(1)]; tensor query_states_91_cast_fp16 = conv(dilations = query_states_91_dilations_0, groups = query_states_91_groups_0, pad = query_states_91_pad_0, pad_type = query_states_91_pad_type_0, strides = query_states_91_strides_0, weight = layers_15_self_attn_q_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("query_states_91_cast_fp16")]; tensor key_states_151_strides_0 = const()[name = string("key_states_151_strides_0"), val = tensor([1, 1])]; string key_states_151_pad_type_0 = const()[name = string("key_states_151_pad_type_0"), val = string("valid")]; tensor key_states_151_pad_0 = const()[name = string("key_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_151_dilations_0 = const()[name = string("key_states_151_dilations_0"), val = tensor([1, 1])]; int32 key_states_151_groups_0 = const()[name = string("key_states_151_groups_0"), val = int32(1)]; tensor key_states_151_cast_fp16 = conv(dilations = key_states_151_dilations_0, groups = key_states_151_groups_0, pad = key_states_151_pad_0, pad_type = key_states_151_pad_type_0, strides = key_states_151_strides_0, weight = layers_15_self_attn_k_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("key_states_151_cast_fp16")]; tensor value_states_91_strides_0 = const()[name = string("value_states_91_strides_0"), val = tensor([1, 1])]; string value_states_91_pad_type_0 = const()[name = string("value_states_91_pad_type_0"), val = string("valid")]; tensor value_states_91_pad_0 = const()[name = string("value_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_91_dilations_0 = const()[name = string("value_states_91_dilations_0"), val = tensor([1, 1])]; int32 value_states_91_groups_0 = const()[name = string("value_states_91_groups_0"), val = int32(1)]; tensor value_states_91_cast_fp16 = conv(dilations = value_states_91_dilations_0, groups = value_states_91_groups_0, pad = value_states_91_pad_0, pad_type = value_states_91_pad_type_0, strides = value_states_91_strides_0, weight = layers_15_self_attn_v_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("value_states_91_cast_fp16")]; tensor concat_180x = const()[name = string("concat_180x"), val = tensor([1, 16, 128, -1])]; tensor x_151_cast_fp16 = reshape(shape = concat_180x, x = query_states_91_cast_fp16)[name = string("x_151_cast_fp16")]; tensor concat_181x = const()[name = string("concat_181x"), val = tensor([1, 2, 128, -1])]; tensor var_5799_cast_fp16 = reshape(shape = concat_181x, x = key_states_151_cast_fp16)[name = string("op_5799_cast_fp16")]; tensor concat_182x = const()[name = string("concat_182x"), val = tensor([1, 2, 128, -1])]; tensor var_5806_cast_fp16 = reshape(shape = concat_182x, x = value_states_91_cast_fp16)[name = string("op_5806_cast_fp16")]; tensor var_5810_cast_fp16 = mul(x = x_151_cast_fp16, y = var_869_cast_fp16)[name = string("op_5810_cast_fp16")]; tensor var_5811_split_sizes_0 = const()[name = string("op_5811_split_sizes_0"), val = tensor([64, 64])]; int32 var_5811_axis_0 = const()[name = string("op_5811_axis_0"), val = int32(-2)]; tensor var_5811_cast_fp16_0, tensor var_5811_cast_fp16_1 = split(axis = var_5811_axis_0, split_sizes = var_5811_split_sizes_0, x = x_151_cast_fp16)[name = string("op_5811_cast_fp16")]; fp16 const_152_promoted_to_fp16 = const()[name = string("const_152_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5813_cast_fp16 = mul(x = var_5811_cast_fp16_1, y = const_152_promoted_to_fp16)[name = string("op_5813_cast_fp16")]; int32 var_5815 = const()[name = string("op_5815"), val = int32(-2)]; bool var_5816_interleave_0 = const()[name = string("op_5816_interleave_0"), val = bool(false)]; tensor var_5816_cast_fp16 = concat(axis = var_5815, interleave = var_5816_interleave_0, values = (var_5813_cast_fp16, var_5811_cast_fp16_0))[name = string("op_5816_cast_fp16")]; tensor var_5817_cast_fp16 = mul(x = var_5816_cast_fp16, y = var_878_cast_fp16)[name = string("op_5817_cast_fp16")]; tensor query_states_93_cast_fp16 = add(x = var_5810_cast_fp16, y = var_5817_cast_fp16)[name = string("query_states_93_cast_fp16")]; tensor var_5823_cast_fp16 = mul(x = var_5799_cast_fp16, y = var_869_cast_fp16)[name = string("op_5823_cast_fp16")]; tensor var_5824_split_sizes_0 = const()[name = string("op_5824_split_sizes_0"), val = tensor([64, 64])]; int32 var_5824_axis_0 = const()[name = string("op_5824_axis_0"), val = int32(-2)]; tensor var_5824_cast_fp16_0, tensor var_5824_cast_fp16_1 = split(axis = var_5824_axis_0, split_sizes = var_5824_split_sizes_0, x = var_5799_cast_fp16)[name = string("op_5824_cast_fp16")]; fp16 const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5826_cast_fp16 = mul(x = var_5824_cast_fp16_1, y = const_153_promoted_to_fp16)[name = string("op_5826_cast_fp16")]; int32 var_5828 = const()[name = string("op_5828"), val = int32(-2)]; bool var_5829_interleave_0 = const()[name = string("op_5829_interleave_0"), val = bool(false)]; tensor var_5829_cast_fp16 = concat(axis = var_5828, interleave = var_5829_interleave_0, values = (var_5826_cast_fp16, var_5824_cast_fp16_0))[name = string("op_5829_cast_fp16")]; tensor var_5830_cast_fp16 = mul(x = var_5829_cast_fp16, y = var_878_cast_fp16)[name = string("op_5830_cast_fp16")]; tensor key_states_155_cast_fp16 = add(x = var_5823_cast_fp16, y = var_5830_cast_fp16)[name = string("key_states_155_cast_fp16")]; tensor expand_dims_180 = const()[name = string("expand_dims_180"), val = tensor([15])]; tensor expand_dims_181 = const()[name = string("expand_dims_181"), val = tensor([0])]; tensor expand_dims_183 = const()[name = string("expand_dims_183"), val = tensor([0])]; int32 concat_185_axis_0 = const()[name = string("concat_185_axis_0"), val = int32(0)]; bool concat_185_interleave_0 = const()[name = string("concat_185_interleave_0"), val = bool(false)]; tensor concat_185 = concat(axis = concat_185_axis_0, interleave = concat_185_interleave_0, values = (expand_dims_180, expand_dims_181, position_id, expand_dims_183))[name = string("concat_185")]; tensor expand_dims_184 = const()[name = string("expand_dims_184"), val = tensor([16])]; tensor concat_186_values1_0 = const()[name = string("concat_186_values1_0"), val = tensor([0])]; tensor concat_186_values3_0 = const()[name = string("concat_186_values3_0"), val = tensor([0])]; int32 concat_186_axis_0 = const()[name = string("concat_186_axis_0"), val = int32(0)]; bool concat_186_interleave_0 = const()[name = string("concat_186_interleave_0"), val = bool(false)]; tensor concat_186 = concat(axis = concat_186_axis_0, interleave = concat_186_interleave_0, values = (expand_dims_184, concat_186_values1_0, cache_position_end, concat_186_values3_0))[name = string("concat_186")]; tensor key_states_157_perm_0 = const()[name = string("key_states_157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_16_stride_0 = const()[name = string("key_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_157_cast_fp16 = transpose(perm = key_states_157_perm_0, x = key_states_155_cast_fp16)[name = string("transpose_554")]; tensor key_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = key_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = key_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_16_squeeze_mask_0, stride = key_cache_internal_tensor_assign_16_stride_0, update = key_states_157_cast_fp16, x = coreml_update_state_364)[name = string("key_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_16_cast_fp16, input = key_cache)[name = string("coreml_update_state_366_write_state")]; tensor coreml_update_state_366 = read_state(input = key_cache)[name = string("coreml_update_state_366")]; tensor value_states_93_perm_0 = const()[name = string("value_states_93_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_16_stride_0 = const()[name = string("value_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_93_cast_fp16 = transpose(perm = value_states_93_perm_0, x = var_5806_cast_fp16)[name = string("transpose_553")]; tensor value_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = value_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = value_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_16_squeeze_mask_0, stride = value_cache_internal_tensor_assign_16_stride_0, update = value_states_93_cast_fp16, x = coreml_update_state_365)[name = string("value_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_16_cast_fp16, input = value_cache)[name = string("coreml_update_state_367_write_state")]; tensor coreml_update_state_367 = read_state(input = value_cache)[name = string("coreml_update_state_367")]; tensor var_5900_begin_0 = const()[name = string("op_5900_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5900_end_0 = const()[name = string("op_5900_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5900_end_mask_0 = const()[name = string("op_5900_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5900_cast_fp16 = slice_by_index(begin = var_5900_begin_0, end = var_5900_end_0, end_mask = var_5900_end_mask_0, x = coreml_update_state_366)[name = string("op_5900_cast_fp16")]; tensor tile_30 = const()[name = string("tile_30"), val = tensor([1, 1])]; int32 var_5903_axis_0 = const()[name = string("op_5903_axis_0"), val = int32(1)]; tensor var_5903_cast_fp16_0, tensor var_5903_cast_fp16_1 = split(axis = var_5903_axis_0, split_sizes = tile_30, x = var_5900_cast_fp16)[name = string("op_5903_cast_fp16")]; tensor var_5910_begin_0 = const()[name = string("op_5910_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5910_end_0 = const()[name = string("op_5910_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5910_end_mask_0 = const()[name = string("op_5910_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5910_cast_fp16 = slice_by_index(begin = var_5910_begin_0, end = var_5910_end_0, end_mask = var_5910_end_mask_0, x = coreml_update_state_367)[name = string("op_5910_cast_fp16")]; tensor tile_31 = const()[name = string("tile_31"), val = tensor([1, 1])]; int32 var_5913_axis_0 = const()[name = string("op_5913_axis_0"), val = int32(1)]; tensor var_5913_cast_fp16_0, tensor var_5913_cast_fp16_1 = split(axis = var_5913_axis_0, split_sizes = tile_31, x = var_5910_cast_fp16)[name = string("op_5913_cast_fp16")]; tensor var_5916_split_sizes_0 = const()[name = string("op_5916_split_sizes_0"), val = tensor([8, 8])]; int32 var_5916_axis_0 = const()[name = string("op_5916_axis_0"), val = int32(1)]; tensor var_5916_0, tensor var_5916_1 = split(axis = var_5916_axis_0, split_sizes = var_5916_split_sizes_0, x = query_states_93_cast_fp16)[name = string("op_5916")]; bool attn_weights_241_transpose_x_0 = const()[name = string("attn_weights_241_transpose_x_0"), val = bool(false)]; bool attn_weights_241_transpose_y_0 = const()[name = string("attn_weights_241_transpose_y_0"), val = bool(false)]; tensor attn_weights_241_cast_fp16 = matmul(transpose_x = attn_weights_241_transpose_x_0, transpose_y = attn_weights_241_transpose_y_0, x = var_5903_cast_fp16_0, y = var_5916_0)[name = string("attn_weights_241_cast_fp16")]; fp16 var_5919_to_fp16 = const()[name = string("op_5919_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_243_cast_fp16 = mul(x = attn_weights_241_cast_fp16, y = var_5919_to_fp16)[name = string("attn_weights_243_cast_fp16")]; tensor attn_weights_245_cast_fp16 = add(x = attn_weights_243_cast_fp16, y = attn_mask_1)[name = string("attn_weights_245_cast_fp16")]; int32 var_5923 = const()[name = string("op_5923"), val = int32(-2)]; tensor attn_weights_247_cast_fp16 = softmax(axis = var_5923, x = attn_weights_245_cast_fp16)[name = string("attn_weights_247_cast_fp16")]; bool var_5929_transpose_x_1 = const()[name = string("op_5929_transpose_x_1"), val = bool(true)]; bool var_5929_transpose_y_1 = const()[name = string("op_5929_transpose_y_1"), val = bool(false)]; tensor var_5929_cast_fp16 = matmul(transpose_x = var_5929_transpose_x_1, transpose_y = var_5929_transpose_y_1, x = attn_weights_247_cast_fp16, y = var_5913_cast_fp16_0)[name = string("op_5929_cast_fp16")]; bool attn_weights_249_transpose_x_0 = const()[name = string("attn_weights_249_transpose_x_0"), val = bool(false)]; bool attn_weights_249_transpose_y_0 = const()[name = string("attn_weights_249_transpose_y_0"), val = bool(false)]; tensor attn_weights_249_cast_fp16 = matmul(transpose_x = attn_weights_249_transpose_x_0, transpose_y = attn_weights_249_transpose_y_0, x = var_5903_cast_fp16_1, y = var_5916_1)[name = string("attn_weights_249_cast_fp16")]; fp16 var_5931_to_fp16 = const()[name = string("op_5931_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_251_cast_fp16 = mul(x = attn_weights_249_cast_fp16, y = var_5931_to_fp16)[name = string("attn_weights_251_cast_fp16")]; tensor attn_weights_253_cast_fp16 = add(x = attn_weights_251_cast_fp16, y = attn_mask_1)[name = string("attn_weights_253_cast_fp16")]; int32 var_5935 = const()[name = string("op_5935"), val = int32(-2)]; tensor attn_weights_255_cast_fp16 = softmax(axis = var_5935, x = attn_weights_253_cast_fp16)[name = string("attn_weights_255_cast_fp16")]; bool attn_output_121_transpose_x_1 = const()[name = string("attn_output_121_transpose_x_1"), val = bool(true)]; bool attn_output_121_transpose_y_1 = const()[name = string("attn_output_121_transpose_y_1"), val = bool(false)]; tensor attn_output_121_cast_fp16 = matmul(transpose_x = attn_output_121_transpose_x_1, transpose_y = attn_output_121_transpose_y_1, x = attn_weights_255_cast_fp16, y = var_5913_cast_fp16_1)[name = string("attn_output_121_cast_fp16")]; int32 var_5943 = const()[name = string("op_5943"), val = int32(1)]; bool attn_output_123_interleave_0 = const()[name = string("attn_output_123_interleave_0"), val = bool(false)]; tensor attn_output_123_cast_fp16 = concat(axis = var_5943, interleave = attn_output_123_interleave_0, values = (var_5929_cast_fp16, attn_output_121_cast_fp16))[name = string("attn_output_123_cast_fp16")]; tensor var_5947_perm_0 = const()[name = string("op_5947_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_191x = const()[name = string("concat_191x"), val = tensor([1, 2048, 1, -1])]; tensor var_5947_cast_fp16 = transpose(perm = var_5947_perm_0, x = attn_output_123_cast_fp16)[name = string("transpose_552")]; tensor attn_output_127_cast_fp16 = reshape(shape = concat_191x, x = var_5947_cast_fp16)[name = string("attn_output_127_cast_fp16")]; tensor hidden_states_153_strides_0 = const()[name = string("hidden_states_153_strides_0"), val = tensor([1, 1])]; string hidden_states_153_pad_type_0 = const()[name = string("hidden_states_153_pad_type_0"), val = string("valid")]; tensor hidden_states_153_pad_0 = const()[name = string("hidden_states_153_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_153_dilations_0 = const()[name = string("hidden_states_153_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_153_groups_0 = const()[name = string("hidden_states_153_groups_0"), val = int32(1)]; tensor hidden_states_153_cast_fp16 = conv(dilations = hidden_states_153_dilations_0, groups = hidden_states_153_groups_0, pad = hidden_states_153_pad_0, pad_type = hidden_states_153_pad_type_0, strides = hidden_states_153_strides_0, weight = layers_15_self_attn_o_proj_weight_cast_fp16, x = attn_output_127_cast_fp16)[name = string("hidden_states_153_cast_fp16")]; tensor hidden_states_155_cast_fp16 = add(x = hidden_states_149_cast_fp16, y = hidden_states_153_cast_fp16)[name = string("hidden_states_155_cast_fp16")]; fp16 const_158_promoted_to_fp16 = const()[name = string("const_158_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5980_cast_fp16 = mul(x = hidden_states_155_cast_fp16, y = const_158_promoted_to_fp16)[name = string("op_5980_cast_fp16")]; int32 var_5978 = const()[name = string("op_5978"), val = int32(1)]; bool doubled_125_interleave_0 = const()[name = string("doubled_125_interleave_0"), val = bool(false)]; tensor doubled_125_cast_fp16 = concat(axis = var_5978, interleave = doubled_125_interleave_0, values = (hidden_states_155_cast_fp16, var_5980_cast_fp16))[name = string("doubled_125_cast_fp16")]; tensor out_63_axes_0 = const()[name = string("out_63_axes_0"), val = tensor([1])]; tensor out_63_gamma_0_to_fp16 = const()[name = string("out_63_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441646976)))]; fp16 var_5990_to_fp16 = const()[name = string("op_5990_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_63_cast_fp16 = layer_norm(axes = out_63_axes_0, epsilon = var_5990_to_fp16, gamma = out_63_gamma_0_to_fp16, x = doubled_125_cast_fp16)[name = string("out_63_cast_fp16")]; tensor var_6001_split_sizes_0 = const()[name = string("op_6001_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6001_axis_0 = const()[name = string("op_6001_axis_0"), val = int32(1)]; tensor var_6001_cast_fp16_0, tensor var_6001_cast_fp16_1 = split(axis = var_6001_axis_0, split_sizes = var_6001_split_sizes_0, x = out_63_cast_fp16)[name = string("op_6001_cast_fp16")]; tensor input_31_strides_0 = const()[name = string("input_31_strides_0"), val = tensor([1, 1])]; string input_31_pad_type_0 = const()[name = string("input_31_pad_type_0"), val = string("valid")]; tensor input_31_pad_0 = const()[name = string("input_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_31_dilations_0 = const()[name = string("input_31_dilations_0"), val = tensor([1, 1])]; int32 input_31_groups_0 = const()[name = string("input_31_groups_0"), val = int32(1)]; tensor input_31_cast_fp16 = conv(dilations = input_31_dilations_0, groups = input_31_groups_0, pad = input_31_pad_0, pad_type = input_31_pad_type_0, strides = input_31_strides_0, weight = layers_15_mlp_gate_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("input_31_cast_fp16")]; tensor var_6018_cast_fp16 = silu(x = input_31_cast_fp16)[name = string("op_6018_cast_fp16")]; tensor var_6024_strides_0 = const()[name = string("op_6024_strides_0"), val = tensor([1, 1])]; string var_6024_pad_type_0 = const()[name = string("op_6024_pad_type_0"), val = string("valid")]; tensor var_6024_pad_0 = const()[name = string("op_6024_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6024_dilations_0 = const()[name = string("op_6024_dilations_0"), val = tensor([1, 1])]; int32 var_6024_groups_0 = const()[name = string("op_6024_groups_0"), val = int32(1)]; tensor var_6024_cast_fp16 = conv(dilations = var_6024_dilations_0, groups = var_6024_groups_0, pad = var_6024_pad_0, pad_type = var_6024_pad_type_0, strides = var_6024_strides_0, weight = layers_15_mlp_up_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("op_6024_cast_fp16")]; tensor x_159_cast_fp16 = mul(x = var_6018_cast_fp16, y = var_6024_cast_fp16)[name = string("x_159_cast_fp16")]; tensor hidden_states_157_strides_0 = const()[name = string("hidden_states_157_strides_0"), val = tensor([1, 1])]; string hidden_states_157_pad_type_0 = const()[name = string("hidden_states_157_pad_type_0"), val = string("valid")]; tensor hidden_states_157_pad_0 = const()[name = string("hidden_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_157_dilations_0 = const()[name = string("hidden_states_157_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_157_groups_0 = const()[name = string("hidden_states_157_groups_0"), val = int32(1)]; tensor hidden_states_157_cast_fp16 = conv(dilations = hidden_states_157_dilations_0, groups = hidden_states_157_groups_0, pad = hidden_states_157_pad_0, pad_type = hidden_states_157_pad_type_0, strides = hidden_states_157_strides_0, weight = layers_15_mlp_down_proj_weight_cast_fp16, x = x_159_cast_fp16)[name = string("hidden_states_157_cast_fp16")]; tensor hidden_states_159_cast_fp16 = add(x = hidden_states_155_cast_fp16, y = hidden_states_157_cast_fp16)[name = string("hidden_states_159_cast_fp16")]; fp16 const_160_promoted_to_fp16 = const()[name = string("const_160_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6042_cast_fp16 = mul(x = hidden_states_159_cast_fp16, y = const_160_promoted_to_fp16)[name = string("op_6042_cast_fp16")]; int32 var_6040 = const()[name = string("op_6040"), val = int32(1)]; bool doubled_129_interleave_0 = const()[name = string("doubled_129_interleave_0"), val = bool(false)]; tensor doubled_129_cast_fp16 = concat(axis = var_6040, interleave = doubled_129_interleave_0, values = (hidden_states_159_cast_fp16, var_6042_cast_fp16))[name = string("doubled_129_cast_fp16")]; tensor out_65_axes_0 = const()[name = string("out_65_axes_0"), val = tensor([1])]; tensor out_65_gamma_0_to_fp16 = const()[name = string("out_65_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441655232)))]; fp16 var_6052_to_fp16 = const()[name = string("op_6052_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_65_cast_fp16 = layer_norm(axes = out_65_axes_0, epsilon = var_6052_to_fp16, gamma = out_65_gamma_0_to_fp16, x = doubled_129_cast_fp16)[name = string("out_65_cast_fp16")]; tensor var_6063_split_sizes_0 = const()[name = string("op_6063_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6063_axis_0 = const()[name = string("op_6063_axis_0"), val = int32(1)]; tensor var_6063_cast_fp16_0, tensor var_6063_cast_fp16_1 = split(axis = var_6063_axis_0, split_sizes = var_6063_split_sizes_0, x = out_65_cast_fp16)[name = string("op_6063_cast_fp16")]; tensor query_states_97_strides_0 = const()[name = string("query_states_97_strides_0"), val = tensor([1, 1])]; string query_states_97_pad_type_0 = const()[name = string("query_states_97_pad_type_0"), val = string("valid")]; tensor query_states_97_pad_0 = const()[name = string("query_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_97_dilations_0 = const()[name = string("query_states_97_dilations_0"), val = tensor([1, 1])]; int32 query_states_97_groups_0 = const()[name = string("query_states_97_groups_0"), val = int32(1)]; tensor query_states_97_cast_fp16 = conv(dilations = query_states_97_dilations_0, groups = query_states_97_groups_0, pad = query_states_97_pad_0, pad_type = query_states_97_pad_type_0, strides = query_states_97_strides_0, weight = layers_16_self_attn_q_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("query_states_97_cast_fp16")]; tensor key_states_161_strides_0 = const()[name = string("key_states_161_strides_0"), val = tensor([1, 1])]; string key_states_161_pad_type_0 = const()[name = string("key_states_161_pad_type_0"), val = string("valid")]; tensor key_states_161_pad_0 = const()[name = string("key_states_161_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_161_dilations_0 = const()[name = string("key_states_161_dilations_0"), val = tensor([1, 1])]; int32 key_states_161_groups_0 = const()[name = string("key_states_161_groups_0"), val = int32(1)]; tensor key_states_161_cast_fp16 = conv(dilations = key_states_161_dilations_0, groups = key_states_161_groups_0, pad = key_states_161_pad_0, pad_type = key_states_161_pad_type_0, strides = key_states_161_strides_0, weight = layers_16_self_attn_k_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("key_states_161_cast_fp16")]; tensor value_states_97_strides_0 = const()[name = string("value_states_97_strides_0"), val = tensor([1, 1])]; string value_states_97_pad_type_0 = const()[name = string("value_states_97_pad_type_0"), val = string("valid")]; tensor value_states_97_pad_0 = const()[name = string("value_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_97_dilations_0 = const()[name = string("value_states_97_dilations_0"), val = tensor([1, 1])]; int32 value_states_97_groups_0 = const()[name = string("value_states_97_groups_0"), val = int32(1)]; tensor value_states_97_cast_fp16 = conv(dilations = value_states_97_dilations_0, groups = value_states_97_groups_0, pad = value_states_97_pad_0, pad_type = value_states_97_pad_type_0, strides = value_states_97_strides_0, weight = layers_16_self_attn_v_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("value_states_97_cast_fp16")]; tensor concat_192x = const()[name = string("concat_192x"), val = tensor([1, 16, 128, -1])]; tensor x_161_cast_fp16 = reshape(shape = concat_192x, x = query_states_97_cast_fp16)[name = string("x_161_cast_fp16")]; tensor concat_193x = const()[name = string("concat_193x"), val = tensor([1, 2, 128, -1])]; tensor var_6120_cast_fp16 = reshape(shape = concat_193x, x = key_states_161_cast_fp16)[name = string("op_6120_cast_fp16")]; tensor concat_194x = const()[name = string("concat_194x"), val = tensor([1, 2, 128, -1])]; tensor var_6127_cast_fp16 = reshape(shape = concat_194x, x = value_states_97_cast_fp16)[name = string("op_6127_cast_fp16")]; tensor var_6131_cast_fp16 = mul(x = x_161_cast_fp16, y = var_869_cast_fp16)[name = string("op_6131_cast_fp16")]; tensor var_6132_split_sizes_0 = const()[name = string("op_6132_split_sizes_0"), val = tensor([64, 64])]; int32 var_6132_axis_0 = const()[name = string("op_6132_axis_0"), val = int32(-2)]; tensor var_6132_cast_fp16_0, tensor var_6132_cast_fp16_1 = split(axis = var_6132_axis_0, split_sizes = var_6132_split_sizes_0, x = x_161_cast_fp16)[name = string("op_6132_cast_fp16")]; fp16 const_162_promoted_to_fp16 = const()[name = string("const_162_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6134_cast_fp16 = mul(x = var_6132_cast_fp16_1, y = const_162_promoted_to_fp16)[name = string("op_6134_cast_fp16")]; int32 var_6136 = const()[name = string("op_6136"), val = int32(-2)]; bool var_6137_interleave_0 = const()[name = string("op_6137_interleave_0"), val = bool(false)]; tensor var_6137_cast_fp16 = concat(axis = var_6136, interleave = var_6137_interleave_0, values = (var_6134_cast_fp16, var_6132_cast_fp16_0))[name = string("op_6137_cast_fp16")]; tensor var_6138_cast_fp16 = mul(x = var_6137_cast_fp16, y = var_878_cast_fp16)[name = string("op_6138_cast_fp16")]; tensor query_states_99_cast_fp16 = add(x = var_6131_cast_fp16, y = var_6138_cast_fp16)[name = string("query_states_99_cast_fp16")]; tensor var_6144_cast_fp16 = mul(x = var_6120_cast_fp16, y = var_869_cast_fp16)[name = string("op_6144_cast_fp16")]; tensor var_6145_split_sizes_0 = const()[name = string("op_6145_split_sizes_0"), val = tensor([64, 64])]; int32 var_6145_axis_0 = const()[name = string("op_6145_axis_0"), val = int32(-2)]; tensor var_6145_cast_fp16_0, tensor var_6145_cast_fp16_1 = split(axis = var_6145_axis_0, split_sizes = var_6145_split_sizes_0, x = var_6120_cast_fp16)[name = string("op_6145_cast_fp16")]; fp16 const_163_promoted_to_fp16 = const()[name = string("const_163_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6147_cast_fp16 = mul(x = var_6145_cast_fp16_1, y = const_163_promoted_to_fp16)[name = string("op_6147_cast_fp16")]; int32 var_6149 = const()[name = string("op_6149"), val = int32(-2)]; bool var_6150_interleave_0 = const()[name = string("op_6150_interleave_0"), val = bool(false)]; tensor var_6150_cast_fp16 = concat(axis = var_6149, interleave = var_6150_interleave_0, values = (var_6147_cast_fp16, var_6145_cast_fp16_0))[name = string("op_6150_cast_fp16")]; tensor var_6151_cast_fp16 = mul(x = var_6150_cast_fp16, y = var_878_cast_fp16)[name = string("op_6151_cast_fp16")]; tensor key_states_165_cast_fp16 = add(x = var_6144_cast_fp16, y = var_6151_cast_fp16)[name = string("key_states_165_cast_fp16")]; tensor expand_dims_192 = const()[name = string("expand_dims_192"), val = tensor([16])]; tensor expand_dims_193 = const()[name = string("expand_dims_193"), val = tensor([0])]; tensor expand_dims_195 = const()[name = string("expand_dims_195"), val = tensor([0])]; int32 concat_197_axis_0 = const()[name = string("concat_197_axis_0"), val = int32(0)]; bool concat_197_interleave_0 = const()[name = string("concat_197_interleave_0"), val = bool(false)]; tensor concat_197 = concat(axis = concat_197_axis_0, interleave = concat_197_interleave_0, values = (expand_dims_192, expand_dims_193, position_id, expand_dims_195))[name = string("concat_197")]; tensor expand_dims_196 = const()[name = string("expand_dims_196"), val = tensor([17])]; tensor concat_198_values1_0 = const()[name = string("concat_198_values1_0"), val = tensor([0])]; tensor concat_198_values3_0 = const()[name = string("concat_198_values3_0"), val = tensor([0])]; int32 concat_198_axis_0 = const()[name = string("concat_198_axis_0"), val = int32(0)]; bool concat_198_interleave_0 = const()[name = string("concat_198_interleave_0"), val = bool(false)]; tensor concat_198 = concat(axis = concat_198_axis_0, interleave = concat_198_interleave_0, values = (expand_dims_196, concat_198_values1_0, cache_position_end, concat_198_values3_0))[name = string("concat_198")]; tensor key_states_167_perm_0 = const()[name = string("key_states_167_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_17_stride_0 = const()[name = string("key_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_167_cast_fp16 = transpose(perm = key_states_167_perm_0, x = key_states_165_cast_fp16)[name = string("transpose_551")]; tensor key_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = key_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = key_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_17_squeeze_mask_0, stride = key_cache_internal_tensor_assign_17_stride_0, update = key_states_167_cast_fp16, x = coreml_update_state_366)[name = string("key_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_17_cast_fp16, input = key_cache)[name = string("coreml_update_state_368_write_state")]; tensor coreml_update_state_368 = read_state(input = key_cache)[name = string("coreml_update_state_368")]; tensor value_states_99_perm_0 = const()[name = string("value_states_99_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_17_stride_0 = const()[name = string("value_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_99_cast_fp16 = transpose(perm = value_states_99_perm_0, x = var_6127_cast_fp16)[name = string("transpose_550")]; tensor value_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = value_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = value_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_17_squeeze_mask_0, stride = value_cache_internal_tensor_assign_17_stride_0, update = value_states_99_cast_fp16, x = coreml_update_state_367)[name = string("value_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_17_cast_fp16, input = value_cache)[name = string("coreml_update_state_369_write_state")]; tensor coreml_update_state_369 = read_state(input = value_cache)[name = string("coreml_update_state_369")]; tensor var_6221_begin_0 = const()[name = string("op_6221_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6221_end_0 = const()[name = string("op_6221_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6221_end_mask_0 = const()[name = string("op_6221_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6221_cast_fp16 = slice_by_index(begin = var_6221_begin_0, end = var_6221_end_0, end_mask = var_6221_end_mask_0, x = coreml_update_state_368)[name = string("op_6221_cast_fp16")]; tensor tile_32 = const()[name = string("tile_32"), val = tensor([1, 1])]; int32 var_6224_axis_0 = const()[name = string("op_6224_axis_0"), val = int32(1)]; tensor var_6224_cast_fp16_0, tensor var_6224_cast_fp16_1 = split(axis = var_6224_axis_0, split_sizes = tile_32, x = var_6221_cast_fp16)[name = string("op_6224_cast_fp16")]; tensor var_6231_begin_0 = const()[name = string("op_6231_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6231_end_0 = const()[name = string("op_6231_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6231_end_mask_0 = const()[name = string("op_6231_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6231_cast_fp16 = slice_by_index(begin = var_6231_begin_0, end = var_6231_end_0, end_mask = var_6231_end_mask_0, x = coreml_update_state_369)[name = string("op_6231_cast_fp16")]; tensor tile_33 = const()[name = string("tile_33"), val = tensor([1, 1])]; int32 var_6234_axis_0 = const()[name = string("op_6234_axis_0"), val = int32(1)]; tensor var_6234_cast_fp16_0, tensor var_6234_cast_fp16_1 = split(axis = var_6234_axis_0, split_sizes = tile_33, x = var_6231_cast_fp16)[name = string("op_6234_cast_fp16")]; tensor var_6237_split_sizes_0 = const()[name = string("op_6237_split_sizes_0"), val = tensor([8, 8])]; int32 var_6237_axis_0 = const()[name = string("op_6237_axis_0"), val = int32(1)]; tensor var_6237_0, tensor var_6237_1 = split(axis = var_6237_axis_0, split_sizes = var_6237_split_sizes_0, x = query_states_99_cast_fp16)[name = string("op_6237")]; bool attn_weights_257_transpose_x_0 = const()[name = string("attn_weights_257_transpose_x_0"), val = bool(false)]; bool attn_weights_257_transpose_y_0 = const()[name = string("attn_weights_257_transpose_y_0"), val = bool(false)]; tensor attn_weights_257_cast_fp16 = matmul(transpose_x = attn_weights_257_transpose_x_0, transpose_y = attn_weights_257_transpose_y_0, x = var_6224_cast_fp16_0, y = var_6237_0)[name = string("attn_weights_257_cast_fp16")]; fp16 var_6240_to_fp16 = const()[name = string("op_6240_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_259_cast_fp16 = mul(x = attn_weights_257_cast_fp16, y = var_6240_to_fp16)[name = string("attn_weights_259_cast_fp16")]; tensor attn_weights_261_cast_fp16 = add(x = attn_weights_259_cast_fp16, y = attn_mask_1)[name = string("attn_weights_261_cast_fp16")]; int32 var_6244 = const()[name = string("op_6244"), val = int32(-2)]; tensor attn_weights_263_cast_fp16 = softmax(axis = var_6244, x = attn_weights_261_cast_fp16)[name = string("attn_weights_263_cast_fp16")]; bool var_6250_transpose_x_1 = const()[name = string("op_6250_transpose_x_1"), val = bool(true)]; bool var_6250_transpose_y_1 = const()[name = string("op_6250_transpose_y_1"), val = bool(false)]; tensor var_6250_cast_fp16 = matmul(transpose_x = var_6250_transpose_x_1, transpose_y = var_6250_transpose_y_1, x = attn_weights_263_cast_fp16, y = var_6234_cast_fp16_0)[name = string("op_6250_cast_fp16")]; bool attn_weights_265_transpose_x_0 = const()[name = string("attn_weights_265_transpose_x_0"), val = bool(false)]; bool attn_weights_265_transpose_y_0 = const()[name = string("attn_weights_265_transpose_y_0"), val = bool(false)]; tensor attn_weights_265_cast_fp16 = matmul(transpose_x = attn_weights_265_transpose_x_0, transpose_y = attn_weights_265_transpose_y_0, x = var_6224_cast_fp16_1, y = var_6237_1)[name = string("attn_weights_265_cast_fp16")]; fp16 var_6252_to_fp16 = const()[name = string("op_6252_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_267_cast_fp16 = mul(x = attn_weights_265_cast_fp16, y = var_6252_to_fp16)[name = string("attn_weights_267_cast_fp16")]; tensor attn_weights_269_cast_fp16 = add(x = attn_weights_267_cast_fp16, y = attn_mask_1)[name = string("attn_weights_269_cast_fp16")]; int32 var_6256 = const()[name = string("op_6256"), val = int32(-2)]; tensor attn_weights_271_cast_fp16 = softmax(axis = var_6256, x = attn_weights_269_cast_fp16)[name = string("attn_weights_271_cast_fp16")]; bool attn_output_129_transpose_x_1 = const()[name = string("attn_output_129_transpose_x_1"), val = bool(true)]; bool attn_output_129_transpose_y_1 = const()[name = string("attn_output_129_transpose_y_1"), val = bool(false)]; tensor attn_output_129_cast_fp16 = matmul(transpose_x = attn_output_129_transpose_x_1, transpose_y = attn_output_129_transpose_y_1, x = attn_weights_271_cast_fp16, y = var_6234_cast_fp16_1)[name = string("attn_output_129_cast_fp16")]; int32 var_6264 = const()[name = string("op_6264"), val = int32(1)]; bool attn_output_131_interleave_0 = const()[name = string("attn_output_131_interleave_0"), val = bool(false)]; tensor attn_output_131_cast_fp16 = concat(axis = var_6264, interleave = attn_output_131_interleave_0, values = (var_6250_cast_fp16, attn_output_129_cast_fp16))[name = string("attn_output_131_cast_fp16")]; tensor var_6268_perm_0 = const()[name = string("op_6268_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_203x = const()[name = string("concat_203x"), val = tensor([1, 2048, 1, -1])]; tensor var_6268_cast_fp16 = transpose(perm = var_6268_perm_0, x = attn_output_131_cast_fp16)[name = string("transpose_549")]; tensor attn_output_135_cast_fp16 = reshape(shape = concat_203x, x = var_6268_cast_fp16)[name = string("attn_output_135_cast_fp16")]; tensor hidden_states_163_strides_0 = const()[name = string("hidden_states_163_strides_0"), val = tensor([1, 1])]; string hidden_states_163_pad_type_0 = const()[name = string("hidden_states_163_pad_type_0"), val = string("valid")]; tensor hidden_states_163_pad_0 = const()[name = string("hidden_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_163_dilations_0 = const()[name = string("hidden_states_163_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_163_groups_0 = const()[name = string("hidden_states_163_groups_0"), val = int32(1)]; tensor hidden_states_163_cast_fp16 = conv(dilations = hidden_states_163_dilations_0, groups = hidden_states_163_groups_0, pad = hidden_states_163_pad_0, pad_type = hidden_states_163_pad_type_0, strides = hidden_states_163_strides_0, weight = layers_16_self_attn_o_proj_weight_cast_fp16, x = attn_output_135_cast_fp16)[name = string("hidden_states_163_cast_fp16")]; tensor hidden_states_165_cast_fp16 = add(x = hidden_states_159_cast_fp16, y = hidden_states_163_cast_fp16)[name = string("hidden_states_165_cast_fp16")]; fp16 const_168_promoted_to_fp16 = const()[name = string("const_168_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6301_cast_fp16 = mul(x = hidden_states_165_cast_fp16, y = const_168_promoted_to_fp16)[name = string("op_6301_cast_fp16")]; int32 var_6299 = const()[name = string("op_6299"), val = int32(1)]; bool doubled_133_interleave_0 = const()[name = string("doubled_133_interleave_0"), val = bool(false)]; tensor doubled_133_cast_fp16 = concat(axis = var_6299, interleave = doubled_133_interleave_0, values = (hidden_states_165_cast_fp16, var_6301_cast_fp16))[name = string("doubled_133_cast_fp16")]; tensor out_67_axes_0 = const()[name = string("out_67_axes_0"), val = tensor([1])]; tensor out_67_gamma_0_to_fp16 = const()[name = string("out_67_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441663488)))]; fp16 var_6311_to_fp16 = const()[name = string("op_6311_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_67_cast_fp16 = layer_norm(axes = out_67_axes_0, epsilon = var_6311_to_fp16, gamma = out_67_gamma_0_to_fp16, x = doubled_133_cast_fp16)[name = string("out_67_cast_fp16")]; tensor var_6322_split_sizes_0 = const()[name = string("op_6322_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6322_axis_0 = const()[name = string("op_6322_axis_0"), val = int32(1)]; tensor var_6322_cast_fp16_0, tensor var_6322_cast_fp16_1 = split(axis = var_6322_axis_0, split_sizes = var_6322_split_sizes_0, x = out_67_cast_fp16)[name = string("op_6322_cast_fp16")]; tensor layers_16_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441671744)))]; tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; tensor input_33_cast_fp16 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = layers_16_mlp_gate_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("input_33_cast_fp16")]; tensor var_6339_cast_fp16 = silu(x = input_33_cast_fp16)[name = string("op_6339_cast_fp16")]; tensor layers_16_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1466837632)))]; tensor var_6345_strides_0 = const()[name = string("op_6345_strides_0"), val = tensor([1, 1])]; string var_6345_pad_type_0 = const()[name = string("op_6345_pad_type_0"), val = string("valid")]; tensor var_6345_pad_0 = const()[name = string("op_6345_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6345_dilations_0 = const()[name = string("op_6345_dilations_0"), val = tensor([1, 1])]; int32 var_6345_groups_0 = const()[name = string("op_6345_groups_0"), val = int32(1)]; tensor var_6345_cast_fp16 = conv(dilations = var_6345_dilations_0, groups = var_6345_groups_0, pad = var_6345_pad_0, pad_type = var_6345_pad_type_0, strides = var_6345_strides_0, weight = layers_16_mlp_up_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("op_6345_cast_fp16")]; tensor x_169_cast_fp16 = mul(x = var_6339_cast_fp16, y = var_6345_cast_fp16)[name = string("x_169_cast_fp16")]; tensor hidden_states_167_strides_0 = const()[name = string("hidden_states_167_strides_0"), val = tensor([1, 1])]; string hidden_states_167_pad_type_0 = const()[name = string("hidden_states_167_pad_type_0"), val = string("valid")]; tensor hidden_states_167_pad_0 = const()[name = string("hidden_states_167_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_167_dilations_0 = const()[name = string("hidden_states_167_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_167_groups_0 = const()[name = string("hidden_states_167_groups_0"), val = int32(1)]; tensor hidden_states_167_cast_fp16 = conv(dilations = hidden_states_167_dilations_0, groups = hidden_states_167_groups_0, pad = hidden_states_167_pad_0, pad_type = hidden_states_167_pad_type_0, strides = hidden_states_167_strides_0, weight = layers_16_mlp_down_proj_weight_cast_fp16, x = x_169_cast_fp16)[name = string("hidden_states_167_cast_fp16")]; tensor hidden_states_169_cast_fp16 = add(x = hidden_states_165_cast_fp16, y = hidden_states_167_cast_fp16)[name = string("hidden_states_169_cast_fp16")]; fp16 const_170_promoted_to_fp16 = const()[name = string("const_170_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6363_cast_fp16 = mul(x = hidden_states_169_cast_fp16, y = const_170_promoted_to_fp16)[name = string("op_6363_cast_fp16")]; int32 var_6361 = const()[name = string("op_6361"), val = int32(1)]; bool doubled_137_interleave_0 = const()[name = string("doubled_137_interleave_0"), val = bool(false)]; tensor doubled_137_cast_fp16 = concat(axis = var_6361, interleave = doubled_137_interleave_0, values = (hidden_states_169_cast_fp16, var_6363_cast_fp16))[name = string("doubled_137_cast_fp16")]; tensor out_69_axes_0 = const()[name = string("out_69_axes_0"), val = tensor([1])]; tensor out_69_gamma_0_to_fp16 = const()[name = string("out_69_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492003520)))]; fp16 var_6373_to_fp16 = const()[name = string("op_6373_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_69_cast_fp16 = layer_norm(axes = out_69_axes_0, epsilon = var_6373_to_fp16, gamma = out_69_gamma_0_to_fp16, x = doubled_137_cast_fp16)[name = string("out_69_cast_fp16")]; tensor var_6384_split_sizes_0 = const()[name = string("op_6384_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6384_axis_0 = const()[name = string("op_6384_axis_0"), val = int32(1)]; tensor var_6384_cast_fp16_0, tensor var_6384_cast_fp16_1 = split(axis = var_6384_axis_0, split_sizes = var_6384_split_sizes_0, x = out_69_cast_fp16)[name = string("op_6384_cast_fp16")]; tensor query_states_103_strides_0 = const()[name = string("query_states_103_strides_0"), val = tensor([1, 1])]; string query_states_103_pad_type_0 = const()[name = string("query_states_103_pad_type_0"), val = string("valid")]; tensor query_states_103_pad_0 = const()[name = string("query_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_103_dilations_0 = const()[name = string("query_states_103_dilations_0"), val = tensor([1, 1])]; int32 query_states_103_groups_0 = const()[name = string("query_states_103_groups_0"), val = int32(1)]; tensor query_states_103_cast_fp16 = conv(dilations = query_states_103_dilations_0, groups = query_states_103_groups_0, pad = query_states_103_pad_0, pad_type = query_states_103_pad_type_0, strides = query_states_103_strides_0, weight = layers_17_self_attn_q_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("query_states_103_cast_fp16")]; tensor key_states_171_strides_0 = const()[name = string("key_states_171_strides_0"), val = tensor([1, 1])]; string key_states_171_pad_type_0 = const()[name = string("key_states_171_pad_type_0"), val = string("valid")]; tensor key_states_171_pad_0 = const()[name = string("key_states_171_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_171_dilations_0 = const()[name = string("key_states_171_dilations_0"), val = tensor([1, 1])]; int32 key_states_171_groups_0 = const()[name = string("key_states_171_groups_0"), val = int32(1)]; tensor key_states_171_cast_fp16 = conv(dilations = key_states_171_dilations_0, groups = key_states_171_groups_0, pad = key_states_171_pad_0, pad_type = key_states_171_pad_type_0, strides = key_states_171_strides_0, weight = layers_17_self_attn_k_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("key_states_171_cast_fp16")]; tensor value_states_103_strides_0 = const()[name = string("value_states_103_strides_0"), val = tensor([1, 1])]; string value_states_103_pad_type_0 = const()[name = string("value_states_103_pad_type_0"), val = string("valid")]; tensor value_states_103_pad_0 = const()[name = string("value_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_103_dilations_0 = const()[name = string("value_states_103_dilations_0"), val = tensor([1, 1])]; int32 value_states_103_groups_0 = const()[name = string("value_states_103_groups_0"), val = int32(1)]; tensor value_states_103_cast_fp16 = conv(dilations = value_states_103_dilations_0, groups = value_states_103_groups_0, pad = value_states_103_pad_0, pad_type = value_states_103_pad_type_0, strides = value_states_103_strides_0, weight = layers_17_self_attn_v_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("value_states_103_cast_fp16")]; tensor concat_204x = const()[name = string("concat_204x"), val = tensor([1, 16, 128, -1])]; tensor x_171_cast_fp16 = reshape(shape = concat_204x, x = query_states_103_cast_fp16)[name = string("x_171_cast_fp16")]; tensor concat_205x = const()[name = string("concat_205x"), val = tensor([1, 2, 128, -1])]; tensor var_6441_cast_fp16 = reshape(shape = concat_205x, x = key_states_171_cast_fp16)[name = string("op_6441_cast_fp16")]; tensor concat_206x = const()[name = string("concat_206x"), val = tensor([1, 2, 128, -1])]; tensor var_6448_cast_fp16 = reshape(shape = concat_206x, x = value_states_103_cast_fp16)[name = string("op_6448_cast_fp16")]; tensor var_6452_cast_fp16 = mul(x = x_171_cast_fp16, y = var_869_cast_fp16)[name = string("op_6452_cast_fp16")]; tensor var_6453_split_sizes_0 = const()[name = string("op_6453_split_sizes_0"), val = tensor([64, 64])]; int32 var_6453_axis_0 = const()[name = string("op_6453_axis_0"), val = int32(-2)]; tensor var_6453_cast_fp16_0, tensor var_6453_cast_fp16_1 = split(axis = var_6453_axis_0, split_sizes = var_6453_split_sizes_0, x = x_171_cast_fp16)[name = string("op_6453_cast_fp16")]; fp16 const_172_promoted_to_fp16 = const()[name = string("const_172_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6455_cast_fp16 = mul(x = var_6453_cast_fp16_1, y = const_172_promoted_to_fp16)[name = string("op_6455_cast_fp16")]; int32 var_6457 = const()[name = string("op_6457"), val = int32(-2)]; bool var_6458_interleave_0 = const()[name = string("op_6458_interleave_0"), val = bool(false)]; tensor var_6458_cast_fp16 = concat(axis = var_6457, interleave = var_6458_interleave_0, values = (var_6455_cast_fp16, var_6453_cast_fp16_0))[name = string("op_6458_cast_fp16")]; tensor var_6459_cast_fp16 = mul(x = var_6458_cast_fp16, y = var_878_cast_fp16)[name = string("op_6459_cast_fp16")]; tensor query_states_105_cast_fp16 = add(x = var_6452_cast_fp16, y = var_6459_cast_fp16)[name = string("query_states_105_cast_fp16")]; tensor var_6465_cast_fp16 = mul(x = var_6441_cast_fp16, y = var_869_cast_fp16)[name = string("op_6465_cast_fp16")]; tensor var_6466_split_sizes_0 = const()[name = string("op_6466_split_sizes_0"), val = tensor([64, 64])]; int32 var_6466_axis_0 = const()[name = string("op_6466_axis_0"), val = int32(-2)]; tensor var_6466_cast_fp16_0, tensor var_6466_cast_fp16_1 = split(axis = var_6466_axis_0, split_sizes = var_6466_split_sizes_0, x = var_6441_cast_fp16)[name = string("op_6466_cast_fp16")]; fp16 const_173_promoted_to_fp16 = const()[name = string("const_173_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6468_cast_fp16 = mul(x = var_6466_cast_fp16_1, y = const_173_promoted_to_fp16)[name = string("op_6468_cast_fp16")]; int32 var_6470 = const()[name = string("op_6470"), val = int32(-2)]; bool var_6471_interleave_0 = const()[name = string("op_6471_interleave_0"), val = bool(false)]; tensor var_6471_cast_fp16 = concat(axis = var_6470, interleave = var_6471_interleave_0, values = (var_6468_cast_fp16, var_6466_cast_fp16_0))[name = string("op_6471_cast_fp16")]; tensor var_6472_cast_fp16 = mul(x = var_6471_cast_fp16, y = var_878_cast_fp16)[name = string("op_6472_cast_fp16")]; tensor key_states_175_cast_fp16 = add(x = var_6465_cast_fp16, y = var_6472_cast_fp16)[name = string("key_states_175_cast_fp16")]; tensor expand_dims_204 = const()[name = string("expand_dims_204"), val = tensor([17])]; tensor expand_dims_205 = const()[name = string("expand_dims_205"), val = tensor([0])]; tensor expand_dims_207 = const()[name = string("expand_dims_207"), val = tensor([0])]; int32 concat_209_axis_0 = const()[name = string("concat_209_axis_0"), val = int32(0)]; bool concat_209_interleave_0 = const()[name = string("concat_209_interleave_0"), val = bool(false)]; tensor concat_209 = concat(axis = concat_209_axis_0, interleave = concat_209_interleave_0, values = (expand_dims_204, expand_dims_205, position_id, expand_dims_207))[name = string("concat_209")]; tensor expand_dims_208 = const()[name = string("expand_dims_208"), val = tensor([18])]; tensor concat_210_values1_0 = const()[name = string("concat_210_values1_0"), val = tensor([0])]; tensor concat_210_values3_0 = const()[name = string("concat_210_values3_0"), val = tensor([0])]; int32 concat_210_axis_0 = const()[name = string("concat_210_axis_0"), val = int32(0)]; bool concat_210_interleave_0 = const()[name = string("concat_210_interleave_0"), val = bool(false)]; tensor concat_210 = concat(axis = concat_210_axis_0, interleave = concat_210_interleave_0, values = (expand_dims_208, concat_210_values1_0, cache_position_end, concat_210_values3_0))[name = string("concat_210")]; tensor key_states_177_perm_0 = const()[name = string("key_states_177_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_18_stride_0 = const()[name = string("key_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_177_cast_fp16 = transpose(perm = key_states_177_perm_0, x = key_states_175_cast_fp16)[name = string("transpose_548")]; tensor key_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = key_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = key_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_18_squeeze_mask_0, stride = key_cache_internal_tensor_assign_18_stride_0, update = key_states_177_cast_fp16, x = coreml_update_state_368)[name = string("key_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_18_cast_fp16, input = key_cache)[name = string("coreml_update_state_370_write_state")]; tensor coreml_update_state_370 = read_state(input = key_cache)[name = string("coreml_update_state_370")]; tensor value_states_105_perm_0 = const()[name = string("value_states_105_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_18_stride_0 = const()[name = string("value_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_105_cast_fp16 = transpose(perm = value_states_105_perm_0, x = var_6448_cast_fp16)[name = string("transpose_547")]; tensor value_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = value_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = value_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_18_squeeze_mask_0, stride = value_cache_internal_tensor_assign_18_stride_0, update = value_states_105_cast_fp16, x = coreml_update_state_369)[name = string("value_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_18_cast_fp16, input = value_cache)[name = string("coreml_update_state_371_write_state")]; tensor coreml_update_state_371 = read_state(input = value_cache)[name = string("coreml_update_state_371")]; tensor var_6542_begin_0 = const()[name = string("op_6542_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6542_end_0 = const()[name = string("op_6542_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6542_end_mask_0 = const()[name = string("op_6542_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6542_cast_fp16 = slice_by_index(begin = var_6542_begin_0, end = var_6542_end_0, end_mask = var_6542_end_mask_0, x = coreml_update_state_370)[name = string("op_6542_cast_fp16")]; tensor tile_34 = const()[name = string("tile_34"), val = tensor([1, 1])]; int32 var_6545_axis_0 = const()[name = string("op_6545_axis_0"), val = int32(1)]; tensor var_6545_cast_fp16_0, tensor var_6545_cast_fp16_1 = split(axis = var_6545_axis_0, split_sizes = tile_34, x = var_6542_cast_fp16)[name = string("op_6545_cast_fp16")]; tensor var_6552_begin_0 = const()[name = string("op_6552_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6552_end_0 = const()[name = string("op_6552_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6552_end_mask_0 = const()[name = string("op_6552_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6552_cast_fp16 = slice_by_index(begin = var_6552_begin_0, end = var_6552_end_0, end_mask = var_6552_end_mask_0, x = coreml_update_state_371)[name = string("op_6552_cast_fp16")]; tensor tile_35 = const()[name = string("tile_35"), val = tensor([1, 1])]; int32 var_6555_axis_0 = const()[name = string("op_6555_axis_0"), val = int32(1)]; tensor var_6555_cast_fp16_0, tensor var_6555_cast_fp16_1 = split(axis = var_6555_axis_0, split_sizes = tile_35, x = var_6552_cast_fp16)[name = string("op_6555_cast_fp16")]; tensor var_6558_split_sizes_0 = const()[name = string("op_6558_split_sizes_0"), val = tensor([8, 8])]; int32 var_6558_axis_0 = const()[name = string("op_6558_axis_0"), val = int32(1)]; tensor var_6558_0, tensor var_6558_1 = split(axis = var_6558_axis_0, split_sizes = var_6558_split_sizes_0, x = query_states_105_cast_fp16)[name = string("op_6558")]; bool attn_weights_273_transpose_x_0 = const()[name = string("attn_weights_273_transpose_x_0"), val = bool(false)]; bool attn_weights_273_transpose_y_0 = const()[name = string("attn_weights_273_transpose_y_0"), val = bool(false)]; tensor attn_weights_273_cast_fp16 = matmul(transpose_x = attn_weights_273_transpose_x_0, transpose_y = attn_weights_273_transpose_y_0, x = var_6545_cast_fp16_0, y = var_6558_0)[name = string("attn_weights_273_cast_fp16")]; fp16 var_6561_to_fp16 = const()[name = string("op_6561_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_275_cast_fp16 = mul(x = attn_weights_273_cast_fp16, y = var_6561_to_fp16)[name = string("attn_weights_275_cast_fp16")]; tensor attn_weights_277_cast_fp16 = add(x = attn_weights_275_cast_fp16, y = attn_mask_1)[name = string("attn_weights_277_cast_fp16")]; int32 var_6565 = const()[name = string("op_6565"), val = int32(-2)]; tensor attn_weights_279_cast_fp16 = softmax(axis = var_6565, x = attn_weights_277_cast_fp16)[name = string("attn_weights_279_cast_fp16")]; bool var_6571_transpose_x_1 = const()[name = string("op_6571_transpose_x_1"), val = bool(true)]; bool var_6571_transpose_y_1 = const()[name = string("op_6571_transpose_y_1"), val = bool(false)]; tensor var_6571_cast_fp16 = matmul(transpose_x = var_6571_transpose_x_1, transpose_y = var_6571_transpose_y_1, x = attn_weights_279_cast_fp16, y = var_6555_cast_fp16_0)[name = string("op_6571_cast_fp16")]; bool attn_weights_281_transpose_x_0 = const()[name = string("attn_weights_281_transpose_x_0"), val = bool(false)]; bool attn_weights_281_transpose_y_0 = const()[name = string("attn_weights_281_transpose_y_0"), val = bool(false)]; tensor attn_weights_281_cast_fp16 = matmul(transpose_x = attn_weights_281_transpose_x_0, transpose_y = attn_weights_281_transpose_y_0, x = var_6545_cast_fp16_1, y = var_6558_1)[name = string("attn_weights_281_cast_fp16")]; fp16 var_6573_to_fp16 = const()[name = string("op_6573_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_283_cast_fp16 = mul(x = attn_weights_281_cast_fp16, y = var_6573_to_fp16)[name = string("attn_weights_283_cast_fp16")]; tensor attn_weights_285_cast_fp16 = add(x = attn_weights_283_cast_fp16, y = attn_mask_1)[name = string("attn_weights_285_cast_fp16")]; int32 var_6577 = const()[name = string("op_6577"), val = int32(-2)]; tensor attn_weights_287_cast_fp16 = softmax(axis = var_6577, x = attn_weights_285_cast_fp16)[name = string("attn_weights_287_cast_fp16")]; bool attn_output_137_transpose_x_1 = const()[name = string("attn_output_137_transpose_x_1"), val = bool(true)]; bool attn_output_137_transpose_y_1 = const()[name = string("attn_output_137_transpose_y_1"), val = bool(false)]; tensor attn_output_137_cast_fp16 = matmul(transpose_x = attn_output_137_transpose_x_1, transpose_y = attn_output_137_transpose_y_1, x = attn_weights_287_cast_fp16, y = var_6555_cast_fp16_1)[name = string("attn_output_137_cast_fp16")]; int32 var_6585 = const()[name = string("op_6585"), val = int32(1)]; bool attn_output_139_interleave_0 = const()[name = string("attn_output_139_interleave_0"), val = bool(false)]; tensor attn_output_139_cast_fp16 = concat(axis = var_6585, interleave = attn_output_139_interleave_0, values = (var_6571_cast_fp16, attn_output_137_cast_fp16))[name = string("attn_output_139_cast_fp16")]; tensor var_6589_perm_0 = const()[name = string("op_6589_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_215x = const()[name = string("concat_215x"), val = tensor([1, 2048, 1, -1])]; tensor var_6589_cast_fp16 = transpose(perm = var_6589_perm_0, x = attn_output_139_cast_fp16)[name = string("transpose_546")]; tensor attn_output_143_cast_fp16 = reshape(shape = concat_215x, x = var_6589_cast_fp16)[name = string("attn_output_143_cast_fp16")]; tensor hidden_states_173_strides_0 = const()[name = string("hidden_states_173_strides_0"), val = tensor([1, 1])]; string hidden_states_173_pad_type_0 = const()[name = string("hidden_states_173_pad_type_0"), val = string("valid")]; tensor hidden_states_173_pad_0 = const()[name = string("hidden_states_173_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_173_dilations_0 = const()[name = string("hidden_states_173_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_173_groups_0 = const()[name = string("hidden_states_173_groups_0"), val = int32(1)]; tensor hidden_states_173_cast_fp16 = conv(dilations = hidden_states_173_dilations_0, groups = hidden_states_173_groups_0, pad = hidden_states_173_pad_0, pad_type = hidden_states_173_pad_type_0, strides = hidden_states_173_strides_0, weight = layers_17_self_attn_o_proj_weight_cast_fp16, x = attn_output_143_cast_fp16)[name = string("hidden_states_173_cast_fp16")]; tensor hidden_states_175_cast_fp16 = add(x = hidden_states_169_cast_fp16, y = hidden_states_173_cast_fp16)[name = string("hidden_states_175_cast_fp16")]; fp16 const_178_promoted_to_fp16 = const()[name = string("const_178_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6622_cast_fp16 = mul(x = hidden_states_175_cast_fp16, y = const_178_promoted_to_fp16)[name = string("op_6622_cast_fp16")]; int32 var_6620 = const()[name = string("op_6620"), val = int32(1)]; bool doubled_141_interleave_0 = const()[name = string("doubled_141_interleave_0"), val = bool(false)]; tensor doubled_141_cast_fp16 = concat(axis = var_6620, interleave = doubled_141_interleave_0, values = (hidden_states_175_cast_fp16, var_6622_cast_fp16))[name = string("doubled_141_cast_fp16")]; tensor out_71_axes_0 = const()[name = string("out_71_axes_0"), val = tensor([1])]; tensor out_71_gamma_0_to_fp16 = const()[name = string("out_71_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492011776)))]; fp16 var_6632_to_fp16 = const()[name = string("op_6632_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_71_cast_fp16 = layer_norm(axes = out_71_axes_0, epsilon = var_6632_to_fp16, gamma = out_71_gamma_0_to_fp16, x = doubled_141_cast_fp16)[name = string("out_71_cast_fp16")]; tensor var_6643_split_sizes_0 = const()[name = string("op_6643_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6643_axis_0 = const()[name = string("op_6643_axis_0"), val = int32(1)]; tensor var_6643_cast_fp16_0, tensor var_6643_cast_fp16_1 = split(axis = var_6643_axis_0, split_sizes = var_6643_split_sizes_0, x = out_71_cast_fp16)[name = string("op_6643_cast_fp16")]; tensor input_35_strides_0 = const()[name = string("input_35_strides_0"), val = tensor([1, 1])]; string input_35_pad_type_0 = const()[name = string("input_35_pad_type_0"), val = string("valid")]; tensor input_35_pad_0 = const()[name = string("input_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_35_dilations_0 = const()[name = string("input_35_dilations_0"), val = tensor([1, 1])]; int32 input_35_groups_0 = const()[name = string("input_35_groups_0"), val = int32(1)]; tensor input_35_cast_fp16 = conv(dilations = input_35_dilations_0, groups = input_35_groups_0, pad = input_35_pad_0, pad_type = input_35_pad_type_0, strides = input_35_strides_0, weight = layers_17_mlp_gate_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("input_35_cast_fp16")]; tensor var_6660_cast_fp16 = silu(x = input_35_cast_fp16)[name = string("op_6660_cast_fp16")]; tensor var_6666_strides_0 = const()[name = string("op_6666_strides_0"), val = tensor([1, 1])]; string var_6666_pad_type_0 = const()[name = string("op_6666_pad_type_0"), val = string("valid")]; tensor var_6666_pad_0 = const()[name = string("op_6666_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6666_dilations_0 = const()[name = string("op_6666_dilations_0"), val = tensor([1, 1])]; int32 var_6666_groups_0 = const()[name = string("op_6666_groups_0"), val = int32(1)]; tensor var_6666_cast_fp16 = conv(dilations = var_6666_dilations_0, groups = var_6666_groups_0, pad = var_6666_pad_0, pad_type = var_6666_pad_type_0, strides = var_6666_strides_0, weight = layers_17_mlp_up_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("op_6666_cast_fp16")]; tensor x_179_cast_fp16 = mul(x = var_6660_cast_fp16, y = var_6666_cast_fp16)[name = string("x_179_cast_fp16")]; tensor hidden_states_177_strides_0 = const()[name = string("hidden_states_177_strides_0"), val = tensor([1, 1])]; string hidden_states_177_pad_type_0 = const()[name = string("hidden_states_177_pad_type_0"), val = string("valid")]; tensor hidden_states_177_pad_0 = const()[name = string("hidden_states_177_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_177_dilations_0 = const()[name = string("hidden_states_177_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_177_groups_0 = const()[name = string("hidden_states_177_groups_0"), val = int32(1)]; tensor hidden_states_177_cast_fp16 = conv(dilations = hidden_states_177_dilations_0, groups = hidden_states_177_groups_0, pad = hidden_states_177_pad_0, pad_type = hidden_states_177_pad_type_0, strides = hidden_states_177_strides_0, weight = layers_17_mlp_down_proj_weight_cast_fp16, x = x_179_cast_fp16)[name = string("hidden_states_177_cast_fp16")]; tensor hidden_states_179_cast_fp16 = add(x = hidden_states_175_cast_fp16, y = hidden_states_177_cast_fp16)[name = string("hidden_states_179_cast_fp16")]; fp16 const_180_promoted_to_fp16 = const()[name = string("const_180_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6684_cast_fp16 = mul(x = hidden_states_179_cast_fp16, y = const_180_promoted_to_fp16)[name = string("op_6684_cast_fp16")]; int32 var_6682 = const()[name = string("op_6682"), val = int32(1)]; bool doubled_145_interleave_0 = const()[name = string("doubled_145_interleave_0"), val = bool(false)]; tensor doubled_145_cast_fp16 = concat(axis = var_6682, interleave = doubled_145_interleave_0, values = (hidden_states_179_cast_fp16, var_6684_cast_fp16))[name = string("doubled_145_cast_fp16")]; tensor out_73_axes_0 = const()[name = string("out_73_axes_0"), val = tensor([1])]; tensor out_73_gamma_0_to_fp16 = const()[name = string("out_73_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492020032)))]; fp16 var_6694_to_fp16 = const()[name = string("op_6694_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_73_cast_fp16 = layer_norm(axes = out_73_axes_0, epsilon = var_6694_to_fp16, gamma = out_73_gamma_0_to_fp16, x = doubled_145_cast_fp16)[name = string("out_73_cast_fp16")]; tensor var_6705_split_sizes_0 = const()[name = string("op_6705_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6705_axis_0 = const()[name = string("op_6705_axis_0"), val = int32(1)]; tensor var_6705_cast_fp16_0, tensor var_6705_cast_fp16_1 = split(axis = var_6705_axis_0, split_sizes = var_6705_split_sizes_0, x = out_73_cast_fp16)[name = string("op_6705_cast_fp16")]; tensor query_states_109_strides_0 = const()[name = string("query_states_109_strides_0"), val = tensor([1, 1])]; string query_states_109_pad_type_0 = const()[name = string("query_states_109_pad_type_0"), val = string("valid")]; tensor query_states_109_pad_0 = const()[name = string("query_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_109_dilations_0 = const()[name = string("query_states_109_dilations_0"), val = tensor([1, 1])]; int32 query_states_109_groups_0 = const()[name = string("query_states_109_groups_0"), val = int32(1)]; tensor query_states_109_cast_fp16 = conv(dilations = query_states_109_dilations_0, groups = query_states_109_groups_0, pad = query_states_109_pad_0, pad_type = query_states_109_pad_type_0, strides = query_states_109_strides_0, weight = layers_18_self_attn_q_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("query_states_109_cast_fp16")]; tensor key_states_181_strides_0 = const()[name = string("key_states_181_strides_0"), val = tensor([1, 1])]; string key_states_181_pad_type_0 = const()[name = string("key_states_181_pad_type_0"), val = string("valid")]; tensor key_states_181_pad_0 = const()[name = string("key_states_181_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_181_dilations_0 = const()[name = string("key_states_181_dilations_0"), val = tensor([1, 1])]; int32 key_states_181_groups_0 = const()[name = string("key_states_181_groups_0"), val = int32(1)]; tensor key_states_181_cast_fp16 = conv(dilations = key_states_181_dilations_0, groups = key_states_181_groups_0, pad = key_states_181_pad_0, pad_type = key_states_181_pad_type_0, strides = key_states_181_strides_0, weight = layers_18_self_attn_k_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("key_states_181_cast_fp16")]; tensor value_states_109_strides_0 = const()[name = string("value_states_109_strides_0"), val = tensor([1, 1])]; string value_states_109_pad_type_0 = const()[name = string("value_states_109_pad_type_0"), val = string("valid")]; tensor value_states_109_pad_0 = const()[name = string("value_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_109_dilations_0 = const()[name = string("value_states_109_dilations_0"), val = tensor([1, 1])]; int32 value_states_109_groups_0 = const()[name = string("value_states_109_groups_0"), val = int32(1)]; tensor value_states_109_cast_fp16 = conv(dilations = value_states_109_dilations_0, groups = value_states_109_groups_0, pad = value_states_109_pad_0, pad_type = value_states_109_pad_type_0, strides = value_states_109_strides_0, weight = layers_18_self_attn_v_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("value_states_109_cast_fp16")]; tensor concat_216x = const()[name = string("concat_216x"), val = tensor([1, 16, 128, -1])]; tensor x_181_cast_fp16 = reshape(shape = concat_216x, x = query_states_109_cast_fp16)[name = string("x_181_cast_fp16")]; tensor concat_217x = const()[name = string("concat_217x"), val = tensor([1, 2, 128, -1])]; tensor var_6762_cast_fp16 = reshape(shape = concat_217x, x = key_states_181_cast_fp16)[name = string("op_6762_cast_fp16")]; tensor concat_218x = const()[name = string("concat_218x"), val = tensor([1, 2, 128, -1])]; tensor var_6769_cast_fp16 = reshape(shape = concat_218x, x = value_states_109_cast_fp16)[name = string("op_6769_cast_fp16")]; tensor var_6773_cast_fp16 = mul(x = x_181_cast_fp16, y = var_869_cast_fp16)[name = string("op_6773_cast_fp16")]; tensor var_6774_split_sizes_0 = const()[name = string("op_6774_split_sizes_0"), val = tensor([64, 64])]; int32 var_6774_axis_0 = const()[name = string("op_6774_axis_0"), val = int32(-2)]; tensor var_6774_cast_fp16_0, tensor var_6774_cast_fp16_1 = split(axis = var_6774_axis_0, split_sizes = var_6774_split_sizes_0, x = x_181_cast_fp16)[name = string("op_6774_cast_fp16")]; fp16 const_182_promoted_to_fp16 = const()[name = string("const_182_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6776_cast_fp16 = mul(x = var_6774_cast_fp16_1, y = const_182_promoted_to_fp16)[name = string("op_6776_cast_fp16")]; int32 var_6778 = const()[name = string("op_6778"), val = int32(-2)]; bool var_6779_interleave_0 = const()[name = string("op_6779_interleave_0"), val = bool(false)]; tensor var_6779_cast_fp16 = concat(axis = var_6778, interleave = var_6779_interleave_0, values = (var_6776_cast_fp16, var_6774_cast_fp16_0))[name = string("op_6779_cast_fp16")]; tensor var_6780_cast_fp16 = mul(x = var_6779_cast_fp16, y = var_878_cast_fp16)[name = string("op_6780_cast_fp16")]; tensor query_states_111_cast_fp16 = add(x = var_6773_cast_fp16, y = var_6780_cast_fp16)[name = string("query_states_111_cast_fp16")]; tensor var_6786_cast_fp16 = mul(x = var_6762_cast_fp16, y = var_869_cast_fp16)[name = string("op_6786_cast_fp16")]; tensor var_6787_split_sizes_0 = const()[name = string("op_6787_split_sizes_0"), val = tensor([64, 64])]; int32 var_6787_axis_0 = const()[name = string("op_6787_axis_0"), val = int32(-2)]; tensor var_6787_cast_fp16_0, tensor var_6787_cast_fp16_1 = split(axis = var_6787_axis_0, split_sizes = var_6787_split_sizes_0, x = var_6762_cast_fp16)[name = string("op_6787_cast_fp16")]; fp16 const_183_promoted_to_fp16 = const()[name = string("const_183_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6789_cast_fp16 = mul(x = var_6787_cast_fp16_1, y = const_183_promoted_to_fp16)[name = string("op_6789_cast_fp16")]; int32 var_6791 = const()[name = string("op_6791"), val = int32(-2)]; bool var_6792_interleave_0 = const()[name = string("op_6792_interleave_0"), val = bool(false)]; tensor var_6792_cast_fp16 = concat(axis = var_6791, interleave = var_6792_interleave_0, values = (var_6789_cast_fp16, var_6787_cast_fp16_0))[name = string("op_6792_cast_fp16")]; tensor var_6793_cast_fp16 = mul(x = var_6792_cast_fp16, y = var_878_cast_fp16)[name = string("op_6793_cast_fp16")]; tensor key_states_185_cast_fp16 = add(x = var_6786_cast_fp16, y = var_6793_cast_fp16)[name = string("key_states_185_cast_fp16")]; tensor expand_dims_216 = const()[name = string("expand_dims_216"), val = tensor([18])]; tensor expand_dims_217 = const()[name = string("expand_dims_217"), val = tensor([0])]; tensor expand_dims_219 = const()[name = string("expand_dims_219"), val = tensor([0])]; int32 concat_221_axis_0 = const()[name = string("concat_221_axis_0"), val = int32(0)]; bool concat_221_interleave_0 = const()[name = string("concat_221_interleave_0"), val = bool(false)]; tensor concat_221 = concat(axis = concat_221_axis_0, interleave = concat_221_interleave_0, values = (expand_dims_216, expand_dims_217, position_id, expand_dims_219))[name = string("concat_221")]; tensor expand_dims_220 = const()[name = string("expand_dims_220"), val = tensor([19])]; tensor concat_222_values1_0 = const()[name = string("concat_222_values1_0"), val = tensor([0])]; tensor concat_222_values3_0 = const()[name = string("concat_222_values3_0"), val = tensor([0])]; int32 concat_222_axis_0 = const()[name = string("concat_222_axis_0"), val = int32(0)]; bool concat_222_interleave_0 = const()[name = string("concat_222_interleave_0"), val = bool(false)]; tensor concat_222 = concat(axis = concat_222_axis_0, interleave = concat_222_interleave_0, values = (expand_dims_220, concat_222_values1_0, cache_position_end, concat_222_values3_0))[name = string("concat_222")]; tensor key_states_187_perm_0 = const()[name = string("key_states_187_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_19_stride_0 = const()[name = string("key_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_187_cast_fp16 = transpose(perm = key_states_187_perm_0, x = key_states_185_cast_fp16)[name = string("transpose_545")]; tensor key_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = key_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = key_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_19_squeeze_mask_0, stride = key_cache_internal_tensor_assign_19_stride_0, update = key_states_187_cast_fp16, x = coreml_update_state_370)[name = string("key_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_19_cast_fp16, input = key_cache)[name = string("coreml_update_state_372_write_state")]; tensor coreml_update_state_372 = read_state(input = key_cache)[name = string("coreml_update_state_372")]; tensor value_states_111_perm_0 = const()[name = string("value_states_111_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_19_stride_0 = const()[name = string("value_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_111_cast_fp16 = transpose(perm = value_states_111_perm_0, x = var_6769_cast_fp16)[name = string("transpose_544")]; tensor value_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = value_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = value_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_19_squeeze_mask_0, stride = value_cache_internal_tensor_assign_19_stride_0, update = value_states_111_cast_fp16, x = coreml_update_state_371)[name = string("value_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_19_cast_fp16, input = value_cache)[name = string("coreml_update_state_373_write_state")]; tensor coreml_update_state_373 = read_state(input = value_cache)[name = string("coreml_update_state_373")]; tensor var_6863_begin_0 = const()[name = string("op_6863_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6863_end_0 = const()[name = string("op_6863_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6863_end_mask_0 = const()[name = string("op_6863_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6863_cast_fp16 = slice_by_index(begin = var_6863_begin_0, end = var_6863_end_0, end_mask = var_6863_end_mask_0, x = coreml_update_state_372)[name = string("op_6863_cast_fp16")]; tensor tile_36 = const()[name = string("tile_36"), val = tensor([1, 1])]; int32 var_6866_axis_0 = const()[name = string("op_6866_axis_0"), val = int32(1)]; tensor var_6866_cast_fp16_0, tensor var_6866_cast_fp16_1 = split(axis = var_6866_axis_0, split_sizes = tile_36, x = var_6863_cast_fp16)[name = string("op_6866_cast_fp16")]; tensor var_6873_begin_0 = const()[name = string("op_6873_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6873_end_0 = const()[name = string("op_6873_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6873_end_mask_0 = const()[name = string("op_6873_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6873_cast_fp16 = slice_by_index(begin = var_6873_begin_0, end = var_6873_end_0, end_mask = var_6873_end_mask_0, x = coreml_update_state_373)[name = string("op_6873_cast_fp16")]; tensor tile_37 = const()[name = string("tile_37"), val = tensor([1, 1])]; int32 var_6876_axis_0 = const()[name = string("op_6876_axis_0"), val = int32(1)]; tensor var_6876_cast_fp16_0, tensor var_6876_cast_fp16_1 = split(axis = var_6876_axis_0, split_sizes = tile_37, x = var_6873_cast_fp16)[name = string("op_6876_cast_fp16")]; tensor var_6879_split_sizes_0 = const()[name = string("op_6879_split_sizes_0"), val = tensor([8, 8])]; int32 var_6879_axis_0 = const()[name = string("op_6879_axis_0"), val = int32(1)]; tensor var_6879_0, tensor var_6879_1 = split(axis = var_6879_axis_0, split_sizes = var_6879_split_sizes_0, x = query_states_111_cast_fp16)[name = string("op_6879")]; bool attn_weights_289_transpose_x_0 = const()[name = string("attn_weights_289_transpose_x_0"), val = bool(false)]; bool attn_weights_289_transpose_y_0 = const()[name = string("attn_weights_289_transpose_y_0"), val = bool(false)]; tensor attn_weights_289_cast_fp16 = matmul(transpose_x = attn_weights_289_transpose_x_0, transpose_y = attn_weights_289_transpose_y_0, x = var_6866_cast_fp16_0, y = var_6879_0)[name = string("attn_weights_289_cast_fp16")]; fp16 var_6882_to_fp16 = const()[name = string("op_6882_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_291_cast_fp16 = mul(x = attn_weights_289_cast_fp16, y = var_6882_to_fp16)[name = string("attn_weights_291_cast_fp16")]; tensor attn_weights_293_cast_fp16 = add(x = attn_weights_291_cast_fp16, y = attn_mask_1)[name = string("attn_weights_293_cast_fp16")]; int32 var_6886 = const()[name = string("op_6886"), val = int32(-2)]; tensor attn_weights_295_cast_fp16 = softmax(axis = var_6886, x = attn_weights_293_cast_fp16)[name = string("attn_weights_295_cast_fp16")]; bool var_6892_transpose_x_1 = const()[name = string("op_6892_transpose_x_1"), val = bool(true)]; bool var_6892_transpose_y_1 = const()[name = string("op_6892_transpose_y_1"), val = bool(false)]; tensor var_6892_cast_fp16 = matmul(transpose_x = var_6892_transpose_x_1, transpose_y = var_6892_transpose_y_1, x = attn_weights_295_cast_fp16, y = var_6876_cast_fp16_0)[name = string("op_6892_cast_fp16")]; bool attn_weights_297_transpose_x_0 = const()[name = string("attn_weights_297_transpose_x_0"), val = bool(false)]; bool attn_weights_297_transpose_y_0 = const()[name = string("attn_weights_297_transpose_y_0"), val = bool(false)]; tensor attn_weights_297_cast_fp16 = matmul(transpose_x = attn_weights_297_transpose_x_0, transpose_y = attn_weights_297_transpose_y_0, x = var_6866_cast_fp16_1, y = var_6879_1)[name = string("attn_weights_297_cast_fp16")]; fp16 var_6894_to_fp16 = const()[name = string("op_6894_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_299_cast_fp16 = mul(x = attn_weights_297_cast_fp16, y = var_6894_to_fp16)[name = string("attn_weights_299_cast_fp16")]; tensor attn_weights_301_cast_fp16 = add(x = attn_weights_299_cast_fp16, y = attn_mask_1)[name = string("attn_weights_301_cast_fp16")]; int32 var_6898 = const()[name = string("op_6898"), val = int32(-2)]; tensor attn_weights_303_cast_fp16 = softmax(axis = var_6898, x = attn_weights_301_cast_fp16)[name = string("attn_weights_303_cast_fp16")]; bool attn_output_145_transpose_x_1 = const()[name = string("attn_output_145_transpose_x_1"), val = bool(true)]; bool attn_output_145_transpose_y_1 = const()[name = string("attn_output_145_transpose_y_1"), val = bool(false)]; tensor attn_output_145_cast_fp16 = matmul(transpose_x = attn_output_145_transpose_x_1, transpose_y = attn_output_145_transpose_y_1, x = attn_weights_303_cast_fp16, y = var_6876_cast_fp16_1)[name = string("attn_output_145_cast_fp16")]; int32 var_6906 = const()[name = string("op_6906"), val = int32(1)]; bool attn_output_147_interleave_0 = const()[name = string("attn_output_147_interleave_0"), val = bool(false)]; tensor attn_output_147_cast_fp16 = concat(axis = var_6906, interleave = attn_output_147_interleave_0, values = (var_6892_cast_fp16, attn_output_145_cast_fp16))[name = string("attn_output_147_cast_fp16")]; tensor var_6910_perm_0 = const()[name = string("op_6910_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_227x = const()[name = string("concat_227x"), val = tensor([1, 2048, 1, -1])]; tensor var_6910_cast_fp16 = transpose(perm = var_6910_perm_0, x = attn_output_147_cast_fp16)[name = string("transpose_543")]; tensor attn_output_151_cast_fp16 = reshape(shape = concat_227x, x = var_6910_cast_fp16)[name = string("attn_output_151_cast_fp16")]; tensor hidden_states_183_strides_0 = const()[name = string("hidden_states_183_strides_0"), val = tensor([1, 1])]; string hidden_states_183_pad_type_0 = const()[name = string("hidden_states_183_pad_type_0"), val = string("valid")]; tensor hidden_states_183_pad_0 = const()[name = string("hidden_states_183_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_183_dilations_0 = const()[name = string("hidden_states_183_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_183_groups_0 = const()[name = string("hidden_states_183_groups_0"), val = int32(1)]; tensor hidden_states_183_cast_fp16 = conv(dilations = hidden_states_183_dilations_0, groups = hidden_states_183_groups_0, pad = hidden_states_183_pad_0, pad_type = hidden_states_183_pad_type_0, strides = hidden_states_183_strides_0, weight = layers_18_self_attn_o_proj_weight_cast_fp16, x = attn_output_151_cast_fp16)[name = string("hidden_states_183_cast_fp16")]; tensor hidden_states_185_cast_fp16 = add(x = hidden_states_179_cast_fp16, y = hidden_states_183_cast_fp16)[name = string("hidden_states_185_cast_fp16")]; fp16 const_188_promoted_to_fp16 = const()[name = string("const_188_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6943_cast_fp16 = mul(x = hidden_states_185_cast_fp16, y = const_188_promoted_to_fp16)[name = string("op_6943_cast_fp16")]; int32 var_6941 = const()[name = string("op_6941"), val = int32(1)]; bool doubled_149_interleave_0 = const()[name = string("doubled_149_interleave_0"), val = bool(false)]; tensor doubled_149_cast_fp16 = concat(axis = var_6941, interleave = doubled_149_interleave_0, values = (hidden_states_185_cast_fp16, var_6943_cast_fp16))[name = string("doubled_149_cast_fp16")]; tensor out_75_axes_0 = const()[name = string("out_75_axes_0"), val = tensor([1])]; tensor out_75_gamma_0_to_fp16 = const()[name = string("out_75_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492028288)))]; fp16 var_6953_to_fp16 = const()[name = string("op_6953_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_75_cast_fp16 = layer_norm(axes = out_75_axes_0, epsilon = var_6953_to_fp16, gamma = out_75_gamma_0_to_fp16, x = doubled_149_cast_fp16)[name = string("out_75_cast_fp16")]; tensor var_6964_split_sizes_0 = const()[name = string("op_6964_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6964_axis_0 = const()[name = string("op_6964_axis_0"), val = int32(1)]; tensor var_6964_cast_fp16_0, tensor var_6964_cast_fp16_1 = split(axis = var_6964_axis_0, split_sizes = var_6964_split_sizes_0, x = out_75_cast_fp16)[name = string("op_6964_cast_fp16")]; tensor input_37_strides_0 = const()[name = string("input_37_strides_0"), val = tensor([1, 1])]; string input_37_pad_type_0 = const()[name = string("input_37_pad_type_0"), val = string("valid")]; tensor input_37_pad_0 = const()[name = string("input_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_37_dilations_0 = const()[name = string("input_37_dilations_0"), val = tensor([1, 1])]; int32 input_37_groups_0 = const()[name = string("input_37_groups_0"), val = int32(1)]; tensor input_37_cast_fp16 = conv(dilations = input_37_dilations_0, groups = input_37_groups_0, pad = input_37_pad_0, pad_type = input_37_pad_type_0, strides = input_37_strides_0, weight = layers_18_mlp_gate_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("input_37_cast_fp16")]; tensor var_6981_cast_fp16 = silu(x = input_37_cast_fp16)[name = string("op_6981_cast_fp16")]; tensor var_6987_strides_0 = const()[name = string("op_6987_strides_0"), val = tensor([1, 1])]; string var_6987_pad_type_0 = const()[name = string("op_6987_pad_type_0"), val = string("valid")]; tensor var_6987_pad_0 = const()[name = string("op_6987_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6987_dilations_0 = const()[name = string("op_6987_dilations_0"), val = tensor([1, 1])]; int32 var_6987_groups_0 = const()[name = string("op_6987_groups_0"), val = int32(1)]; tensor var_6987_cast_fp16 = conv(dilations = var_6987_dilations_0, groups = var_6987_groups_0, pad = var_6987_pad_0, pad_type = var_6987_pad_type_0, strides = var_6987_strides_0, weight = layers_18_mlp_up_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("op_6987_cast_fp16")]; tensor x_189_cast_fp16 = mul(x = var_6981_cast_fp16, y = var_6987_cast_fp16)[name = string("x_189_cast_fp16")]; tensor hidden_states_187_strides_0 = const()[name = string("hidden_states_187_strides_0"), val = tensor([1, 1])]; string hidden_states_187_pad_type_0 = const()[name = string("hidden_states_187_pad_type_0"), val = string("valid")]; tensor hidden_states_187_pad_0 = const()[name = string("hidden_states_187_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_187_dilations_0 = const()[name = string("hidden_states_187_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_187_groups_0 = const()[name = string("hidden_states_187_groups_0"), val = int32(1)]; tensor hidden_states_187_cast_fp16 = conv(dilations = hidden_states_187_dilations_0, groups = hidden_states_187_groups_0, pad = hidden_states_187_pad_0, pad_type = hidden_states_187_pad_type_0, strides = hidden_states_187_strides_0, weight = layers_18_mlp_down_proj_weight_cast_fp16, x = x_189_cast_fp16)[name = string("hidden_states_187_cast_fp16")]; tensor hidden_states_189_cast_fp16 = add(x = hidden_states_185_cast_fp16, y = hidden_states_187_cast_fp16)[name = string("hidden_states_189_cast_fp16")]; fp16 const_190_promoted_to_fp16 = const()[name = string("const_190_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7005_cast_fp16 = mul(x = hidden_states_189_cast_fp16, y = const_190_promoted_to_fp16)[name = string("op_7005_cast_fp16")]; int32 var_7003 = const()[name = string("op_7003"), val = int32(1)]; bool doubled_153_interleave_0 = const()[name = string("doubled_153_interleave_0"), val = bool(false)]; tensor doubled_153_cast_fp16 = concat(axis = var_7003, interleave = doubled_153_interleave_0, values = (hidden_states_189_cast_fp16, var_7005_cast_fp16))[name = string("doubled_153_cast_fp16")]; tensor out_77_axes_0 = const()[name = string("out_77_axes_0"), val = tensor([1])]; tensor out_77_gamma_0_to_fp16 = const()[name = string("out_77_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492036544)))]; fp16 var_7015_to_fp16 = const()[name = string("op_7015_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_77_cast_fp16 = layer_norm(axes = out_77_axes_0, epsilon = var_7015_to_fp16, gamma = out_77_gamma_0_to_fp16, x = doubled_153_cast_fp16)[name = string("out_77_cast_fp16")]; tensor var_7026_split_sizes_0 = const()[name = string("op_7026_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7026_axis_0 = const()[name = string("op_7026_axis_0"), val = int32(1)]; tensor var_7026_cast_fp16_0, tensor var_7026_cast_fp16_1 = split(axis = var_7026_axis_0, split_sizes = var_7026_split_sizes_0, x = out_77_cast_fp16)[name = string("op_7026_cast_fp16")]; tensor query_states_115_strides_0 = const()[name = string("query_states_115_strides_0"), val = tensor([1, 1])]; string query_states_115_pad_type_0 = const()[name = string("query_states_115_pad_type_0"), val = string("valid")]; tensor query_states_115_pad_0 = const()[name = string("query_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_115_dilations_0 = const()[name = string("query_states_115_dilations_0"), val = tensor([1, 1])]; int32 query_states_115_groups_0 = const()[name = string("query_states_115_groups_0"), val = int32(1)]; tensor query_states_115_cast_fp16 = conv(dilations = query_states_115_dilations_0, groups = query_states_115_groups_0, pad = query_states_115_pad_0, pad_type = query_states_115_pad_type_0, strides = query_states_115_strides_0, weight = layers_19_self_attn_q_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("query_states_115_cast_fp16")]; tensor key_states_191_strides_0 = const()[name = string("key_states_191_strides_0"), val = tensor([1, 1])]; string key_states_191_pad_type_0 = const()[name = string("key_states_191_pad_type_0"), val = string("valid")]; tensor key_states_191_pad_0 = const()[name = string("key_states_191_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_191_dilations_0 = const()[name = string("key_states_191_dilations_0"), val = tensor([1, 1])]; int32 key_states_191_groups_0 = const()[name = string("key_states_191_groups_0"), val = int32(1)]; tensor key_states_191_cast_fp16 = conv(dilations = key_states_191_dilations_0, groups = key_states_191_groups_0, pad = key_states_191_pad_0, pad_type = key_states_191_pad_type_0, strides = key_states_191_strides_0, weight = layers_19_self_attn_k_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("key_states_191_cast_fp16")]; tensor layers_19_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492044800)))]; tensor value_states_115_strides_0 = const()[name = string("value_states_115_strides_0"), val = tensor([1, 1])]; string value_states_115_pad_type_0 = const()[name = string("value_states_115_pad_type_0"), val = string("valid")]; tensor value_states_115_pad_0 = const()[name = string("value_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_115_dilations_0 = const()[name = string("value_states_115_dilations_0"), val = tensor([1, 1])]; int32 value_states_115_groups_0 = const()[name = string("value_states_115_groups_0"), val = int32(1)]; tensor value_states_115_cast_fp16 = conv(dilations = value_states_115_dilations_0, groups = value_states_115_groups_0, pad = value_states_115_pad_0, pad_type = value_states_115_pad_type_0, strides = value_states_115_strides_0, weight = layers_19_self_attn_v_proj_weight_to_fp16, x = var_7026_cast_fp16_0)[name = string("value_states_115_cast_fp16")]; tensor concat_228x = const()[name = string("concat_228x"), val = tensor([1, 16, 128, -1])]; tensor x_191_cast_fp16 = reshape(shape = concat_228x, x = query_states_115_cast_fp16)[name = string("x_191_cast_fp16")]; tensor concat_229x = const()[name = string("concat_229x"), val = tensor([1, 2, 128, -1])]; tensor var_7083_cast_fp16 = reshape(shape = concat_229x, x = key_states_191_cast_fp16)[name = string("op_7083_cast_fp16")]; tensor concat_230x = const()[name = string("concat_230x"), val = tensor([1, 2, 128, -1])]; tensor var_7090_cast_fp16 = reshape(shape = concat_230x, x = value_states_115_cast_fp16)[name = string("op_7090_cast_fp16")]; tensor var_7094_cast_fp16 = mul(x = x_191_cast_fp16, y = var_869_cast_fp16)[name = string("op_7094_cast_fp16")]; tensor var_7095_split_sizes_0 = const()[name = string("op_7095_split_sizes_0"), val = tensor([64, 64])]; int32 var_7095_axis_0 = const()[name = string("op_7095_axis_0"), val = int32(-2)]; tensor var_7095_cast_fp16_0, tensor var_7095_cast_fp16_1 = split(axis = var_7095_axis_0, split_sizes = var_7095_split_sizes_0, x = x_191_cast_fp16)[name = string("op_7095_cast_fp16")]; fp16 const_192_promoted_to_fp16 = const()[name = string("const_192_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7097_cast_fp16 = mul(x = var_7095_cast_fp16_1, y = const_192_promoted_to_fp16)[name = string("op_7097_cast_fp16")]; int32 var_7099 = const()[name = string("op_7099"), val = int32(-2)]; bool var_7100_interleave_0 = const()[name = string("op_7100_interleave_0"), val = bool(false)]; tensor var_7100_cast_fp16 = concat(axis = var_7099, interleave = var_7100_interleave_0, values = (var_7097_cast_fp16, var_7095_cast_fp16_0))[name = string("op_7100_cast_fp16")]; tensor var_7101_cast_fp16 = mul(x = var_7100_cast_fp16, y = var_878_cast_fp16)[name = string("op_7101_cast_fp16")]; tensor query_states_117_cast_fp16 = add(x = var_7094_cast_fp16, y = var_7101_cast_fp16)[name = string("query_states_117_cast_fp16")]; tensor var_7107_cast_fp16 = mul(x = var_7083_cast_fp16, y = var_869_cast_fp16)[name = string("op_7107_cast_fp16")]; tensor var_7108_split_sizes_0 = const()[name = string("op_7108_split_sizes_0"), val = tensor([64, 64])]; int32 var_7108_axis_0 = const()[name = string("op_7108_axis_0"), val = int32(-2)]; tensor var_7108_cast_fp16_0, tensor var_7108_cast_fp16_1 = split(axis = var_7108_axis_0, split_sizes = var_7108_split_sizes_0, x = var_7083_cast_fp16)[name = string("op_7108_cast_fp16")]; fp16 const_193_promoted_to_fp16 = const()[name = string("const_193_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7110_cast_fp16 = mul(x = var_7108_cast_fp16_1, y = const_193_promoted_to_fp16)[name = string("op_7110_cast_fp16")]; int32 var_7112 = const()[name = string("op_7112"), val = int32(-2)]; bool var_7113_interleave_0 = const()[name = string("op_7113_interleave_0"), val = bool(false)]; tensor var_7113_cast_fp16 = concat(axis = var_7112, interleave = var_7113_interleave_0, values = (var_7110_cast_fp16, var_7108_cast_fp16_0))[name = string("op_7113_cast_fp16")]; tensor var_7114_cast_fp16 = mul(x = var_7113_cast_fp16, y = var_878_cast_fp16)[name = string("op_7114_cast_fp16")]; tensor key_states_195_cast_fp16 = add(x = var_7107_cast_fp16, y = var_7114_cast_fp16)[name = string("key_states_195_cast_fp16")]; tensor expand_dims_228 = const()[name = string("expand_dims_228"), val = tensor([19])]; tensor expand_dims_229 = const()[name = string("expand_dims_229"), val = tensor([0])]; tensor expand_dims_231 = const()[name = string("expand_dims_231"), val = tensor([0])]; int32 concat_233_axis_0 = const()[name = string("concat_233_axis_0"), val = int32(0)]; bool concat_233_interleave_0 = const()[name = string("concat_233_interleave_0"), val = bool(false)]; tensor concat_233 = concat(axis = concat_233_axis_0, interleave = concat_233_interleave_0, values = (expand_dims_228, expand_dims_229, position_id, expand_dims_231))[name = string("concat_233")]; tensor expand_dims_232 = const()[name = string("expand_dims_232"), val = tensor([20])]; tensor concat_234_values1_0 = const()[name = string("concat_234_values1_0"), val = tensor([0])]; tensor concat_234_values3_0 = const()[name = string("concat_234_values3_0"), val = tensor([0])]; int32 concat_234_axis_0 = const()[name = string("concat_234_axis_0"), val = int32(0)]; bool concat_234_interleave_0 = const()[name = string("concat_234_interleave_0"), val = bool(false)]; tensor concat_234 = concat(axis = concat_234_axis_0, interleave = concat_234_interleave_0, values = (expand_dims_232, concat_234_values1_0, cache_position_end, concat_234_values3_0))[name = string("concat_234")]; tensor key_states_197_perm_0 = const()[name = string("key_states_197_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_20_stride_0 = const()[name = string("key_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_197_cast_fp16 = transpose(perm = key_states_197_perm_0, x = key_states_195_cast_fp16)[name = string("transpose_542")]; tensor key_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = key_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = key_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_20_squeeze_mask_0, stride = key_cache_internal_tensor_assign_20_stride_0, update = key_states_197_cast_fp16, x = coreml_update_state_372)[name = string("key_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_20_cast_fp16, input = key_cache)[name = string("coreml_update_state_374_write_state")]; tensor coreml_update_state_374 = read_state(input = key_cache)[name = string("coreml_update_state_374")]; tensor value_states_117_perm_0 = const()[name = string("value_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_20_stride_0 = const()[name = string("value_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_117_cast_fp16 = transpose(perm = value_states_117_perm_0, x = var_7090_cast_fp16)[name = string("transpose_541")]; tensor value_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = value_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = value_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_20_squeeze_mask_0, stride = value_cache_internal_tensor_assign_20_stride_0, update = value_states_117_cast_fp16, x = coreml_update_state_373)[name = string("value_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_20_cast_fp16, input = value_cache)[name = string("coreml_update_state_375_write_state")]; tensor coreml_update_state_375 = read_state(input = value_cache)[name = string("coreml_update_state_375")]; tensor var_7184_begin_0 = const()[name = string("op_7184_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7184_end_0 = const()[name = string("op_7184_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7184_end_mask_0 = const()[name = string("op_7184_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7184_cast_fp16 = slice_by_index(begin = var_7184_begin_0, end = var_7184_end_0, end_mask = var_7184_end_mask_0, x = coreml_update_state_374)[name = string("op_7184_cast_fp16")]; tensor tile_38 = const()[name = string("tile_38"), val = tensor([1, 1])]; int32 var_7187_axis_0 = const()[name = string("op_7187_axis_0"), val = int32(1)]; tensor var_7187_cast_fp16_0, tensor var_7187_cast_fp16_1 = split(axis = var_7187_axis_0, split_sizes = tile_38, x = var_7184_cast_fp16)[name = string("op_7187_cast_fp16")]; tensor var_7194_begin_0 = const()[name = string("op_7194_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7194_end_0 = const()[name = string("op_7194_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7194_end_mask_0 = const()[name = string("op_7194_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7194_cast_fp16 = slice_by_index(begin = var_7194_begin_0, end = var_7194_end_0, end_mask = var_7194_end_mask_0, x = coreml_update_state_375)[name = string("op_7194_cast_fp16")]; tensor tile_39 = const()[name = string("tile_39"), val = tensor([1, 1])]; int32 var_7197_axis_0 = const()[name = string("op_7197_axis_0"), val = int32(1)]; tensor var_7197_cast_fp16_0, tensor var_7197_cast_fp16_1 = split(axis = var_7197_axis_0, split_sizes = tile_39, x = var_7194_cast_fp16)[name = string("op_7197_cast_fp16")]; tensor var_7200_split_sizes_0 = const()[name = string("op_7200_split_sizes_0"), val = tensor([8, 8])]; int32 var_7200_axis_0 = const()[name = string("op_7200_axis_0"), val = int32(1)]; tensor var_7200_0, tensor var_7200_1 = split(axis = var_7200_axis_0, split_sizes = var_7200_split_sizes_0, x = query_states_117_cast_fp16)[name = string("op_7200")]; bool attn_weights_305_transpose_x_0 = const()[name = string("attn_weights_305_transpose_x_0"), val = bool(false)]; bool attn_weights_305_transpose_y_0 = const()[name = string("attn_weights_305_transpose_y_0"), val = bool(false)]; tensor attn_weights_305_cast_fp16 = matmul(transpose_x = attn_weights_305_transpose_x_0, transpose_y = attn_weights_305_transpose_y_0, x = var_7187_cast_fp16_0, y = var_7200_0)[name = string("attn_weights_305_cast_fp16")]; fp16 var_7203_to_fp16 = const()[name = string("op_7203_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_307_cast_fp16 = mul(x = attn_weights_305_cast_fp16, y = var_7203_to_fp16)[name = string("attn_weights_307_cast_fp16")]; tensor attn_weights_309_cast_fp16 = add(x = attn_weights_307_cast_fp16, y = attn_mask_1)[name = string("attn_weights_309_cast_fp16")]; int32 var_7207 = const()[name = string("op_7207"), val = int32(-2)]; tensor attn_weights_311_cast_fp16 = softmax(axis = var_7207, x = attn_weights_309_cast_fp16)[name = string("attn_weights_311_cast_fp16")]; bool var_7213_transpose_x_1 = const()[name = string("op_7213_transpose_x_1"), val = bool(true)]; bool var_7213_transpose_y_1 = const()[name = string("op_7213_transpose_y_1"), val = bool(false)]; tensor var_7213_cast_fp16 = matmul(transpose_x = var_7213_transpose_x_1, transpose_y = var_7213_transpose_y_1, x = attn_weights_311_cast_fp16, y = var_7197_cast_fp16_0)[name = string("op_7213_cast_fp16")]; bool attn_weights_313_transpose_x_0 = const()[name = string("attn_weights_313_transpose_x_0"), val = bool(false)]; bool attn_weights_313_transpose_y_0 = const()[name = string("attn_weights_313_transpose_y_0"), val = bool(false)]; tensor attn_weights_313_cast_fp16 = matmul(transpose_x = attn_weights_313_transpose_x_0, transpose_y = attn_weights_313_transpose_y_0, x = var_7187_cast_fp16_1, y = var_7200_1)[name = string("attn_weights_313_cast_fp16")]; fp16 var_7215_to_fp16 = const()[name = string("op_7215_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_315_cast_fp16 = mul(x = attn_weights_313_cast_fp16, y = var_7215_to_fp16)[name = string("attn_weights_315_cast_fp16")]; tensor attn_weights_317_cast_fp16 = add(x = attn_weights_315_cast_fp16, y = attn_mask_1)[name = string("attn_weights_317_cast_fp16")]; int32 var_7219 = const()[name = string("op_7219"), val = int32(-2)]; tensor attn_weights_319_cast_fp16 = softmax(axis = var_7219, x = attn_weights_317_cast_fp16)[name = string("attn_weights_319_cast_fp16")]; bool attn_output_153_transpose_x_1 = const()[name = string("attn_output_153_transpose_x_1"), val = bool(true)]; bool attn_output_153_transpose_y_1 = const()[name = string("attn_output_153_transpose_y_1"), val = bool(false)]; tensor attn_output_153_cast_fp16 = matmul(transpose_x = attn_output_153_transpose_x_1, transpose_y = attn_output_153_transpose_y_1, x = attn_weights_319_cast_fp16, y = var_7197_cast_fp16_1)[name = string("attn_output_153_cast_fp16")]; int32 var_7227 = const()[name = string("op_7227"), val = int32(1)]; bool attn_output_155_interleave_0 = const()[name = string("attn_output_155_interleave_0"), val = bool(false)]; tensor attn_output_155_cast_fp16 = concat(axis = var_7227, interleave = attn_output_155_interleave_0, values = (var_7213_cast_fp16, attn_output_153_cast_fp16))[name = string("attn_output_155_cast_fp16")]; tensor var_7231_perm_0 = const()[name = string("op_7231_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_239x = const()[name = string("concat_239x"), val = tensor([1, 2048, 1, -1])]; tensor var_7231_cast_fp16 = transpose(perm = var_7231_perm_0, x = attn_output_155_cast_fp16)[name = string("transpose_540")]; tensor attn_output_159_cast_fp16 = reshape(shape = concat_239x, x = var_7231_cast_fp16)[name = string("attn_output_159_cast_fp16")]; tensor layers_19_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1493093440)))]; tensor hidden_states_193_strides_0 = const()[name = string("hidden_states_193_strides_0"), val = tensor([1, 1])]; string hidden_states_193_pad_type_0 = const()[name = string("hidden_states_193_pad_type_0"), val = string("valid")]; tensor hidden_states_193_pad_0 = const()[name = string("hidden_states_193_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_193_dilations_0 = const()[name = string("hidden_states_193_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_193_groups_0 = const()[name = string("hidden_states_193_groups_0"), val = int32(1)]; tensor hidden_states_193_cast_fp16 = conv(dilations = hidden_states_193_dilations_0, groups = hidden_states_193_groups_0, pad = hidden_states_193_pad_0, pad_type = hidden_states_193_pad_type_0, strides = hidden_states_193_strides_0, weight = layers_19_self_attn_o_proj_weight_to_fp16, x = attn_output_159_cast_fp16)[name = string("hidden_states_193_cast_fp16")]; tensor hidden_states_195_cast_fp16 = add(x = hidden_states_189_cast_fp16, y = hidden_states_193_cast_fp16)[name = string("hidden_states_195_cast_fp16")]; fp16 const_198_promoted_to_fp16 = const()[name = string("const_198_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7264_cast_fp16 = mul(x = hidden_states_195_cast_fp16, y = const_198_promoted_to_fp16)[name = string("op_7264_cast_fp16")]; int32 var_7262 = const()[name = string("op_7262"), val = int32(1)]; bool doubled_157_interleave_0 = const()[name = string("doubled_157_interleave_0"), val = bool(false)]; tensor doubled_157_cast_fp16 = concat(axis = var_7262, interleave = doubled_157_interleave_0, values = (hidden_states_195_cast_fp16, var_7264_cast_fp16))[name = string("doubled_157_cast_fp16")]; tensor out_79_axes_0 = const()[name = string("out_79_axes_0"), val = tensor([1])]; tensor out_79_gamma_0_to_fp16 = const()[name = string("out_79_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501482112)))]; fp16 var_7274_to_fp16 = const()[name = string("op_7274_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_79_cast_fp16 = layer_norm(axes = out_79_axes_0, epsilon = var_7274_to_fp16, gamma = out_79_gamma_0_to_fp16, x = doubled_157_cast_fp16)[name = string("out_79_cast_fp16")]; tensor var_7285_split_sizes_0 = const()[name = string("op_7285_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7285_axis_0 = const()[name = string("op_7285_axis_0"), val = int32(1)]; tensor var_7285_cast_fp16_0, tensor var_7285_cast_fp16_1 = split(axis = var_7285_axis_0, split_sizes = var_7285_split_sizes_0, x = out_79_cast_fp16)[name = string("op_7285_cast_fp16")]; tensor input_39_strides_0 = const()[name = string("input_39_strides_0"), val = tensor([1, 1])]; string input_39_pad_type_0 = const()[name = string("input_39_pad_type_0"), val = string("valid")]; tensor input_39_pad_0 = const()[name = string("input_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_39_dilations_0 = const()[name = string("input_39_dilations_0"), val = tensor([1, 1])]; int32 input_39_groups_0 = const()[name = string("input_39_groups_0"), val = int32(1)]; tensor input_39_cast_fp16 = conv(dilations = input_39_dilations_0, groups = input_39_groups_0, pad = input_39_pad_0, pad_type = input_39_pad_type_0, strides = input_39_strides_0, weight = layers_19_mlp_gate_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("input_39_cast_fp16")]; tensor var_7302_cast_fp16 = silu(x = input_39_cast_fp16)[name = string("op_7302_cast_fp16")]; tensor var_7308_strides_0 = const()[name = string("op_7308_strides_0"), val = tensor([1, 1])]; string var_7308_pad_type_0 = const()[name = string("op_7308_pad_type_0"), val = string("valid")]; tensor var_7308_pad_0 = const()[name = string("op_7308_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7308_dilations_0 = const()[name = string("op_7308_dilations_0"), val = tensor([1, 1])]; int32 var_7308_groups_0 = const()[name = string("op_7308_groups_0"), val = int32(1)]; tensor var_7308_cast_fp16 = conv(dilations = var_7308_dilations_0, groups = var_7308_groups_0, pad = var_7308_pad_0, pad_type = var_7308_pad_type_0, strides = var_7308_strides_0, weight = layers_19_mlp_up_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("op_7308_cast_fp16")]; tensor x_199_cast_fp16 = mul(x = var_7302_cast_fp16, y = var_7308_cast_fp16)[name = string("x_199_cast_fp16")]; tensor hidden_states_197_strides_0 = const()[name = string("hidden_states_197_strides_0"), val = tensor([1, 1])]; string hidden_states_197_pad_type_0 = const()[name = string("hidden_states_197_pad_type_0"), val = string("valid")]; tensor hidden_states_197_pad_0 = const()[name = string("hidden_states_197_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_197_dilations_0 = const()[name = string("hidden_states_197_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_197_groups_0 = const()[name = string("hidden_states_197_groups_0"), val = int32(1)]; tensor hidden_states_197_cast_fp16 = conv(dilations = hidden_states_197_dilations_0, groups = hidden_states_197_groups_0, pad = hidden_states_197_pad_0, pad_type = hidden_states_197_pad_type_0, strides = hidden_states_197_strides_0, weight = layers_19_mlp_down_proj_weight_cast_fp16, x = x_199_cast_fp16)[name = string("hidden_states_197_cast_fp16")]; tensor hidden_states_199_cast_fp16 = add(x = hidden_states_195_cast_fp16, y = hidden_states_197_cast_fp16)[name = string("hidden_states_199_cast_fp16")]; fp16 const_200_promoted_to_fp16 = const()[name = string("const_200_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7326_cast_fp16 = mul(x = hidden_states_199_cast_fp16, y = const_200_promoted_to_fp16)[name = string("op_7326_cast_fp16")]; int32 var_7324 = const()[name = string("op_7324"), val = int32(1)]; bool doubled_161_interleave_0 = const()[name = string("doubled_161_interleave_0"), val = bool(false)]; tensor doubled_161_cast_fp16 = concat(axis = var_7324, interleave = doubled_161_interleave_0, values = (hidden_states_199_cast_fp16, var_7326_cast_fp16))[name = string("doubled_161_cast_fp16")]; tensor out_81_axes_0 = const()[name = string("out_81_axes_0"), val = tensor([1])]; tensor out_81_gamma_0_to_fp16 = const()[name = string("out_81_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501490368)))]; fp16 var_7336_to_fp16 = const()[name = string("op_7336_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_81_cast_fp16 = layer_norm(axes = out_81_axes_0, epsilon = var_7336_to_fp16, gamma = out_81_gamma_0_to_fp16, x = doubled_161_cast_fp16)[name = string("out_81_cast_fp16")]; tensor var_7347_split_sizes_0 = const()[name = string("op_7347_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7347_axis_0 = const()[name = string("op_7347_axis_0"), val = int32(1)]; tensor var_7347_cast_fp16_0, tensor var_7347_cast_fp16_1 = split(axis = var_7347_axis_0, split_sizes = var_7347_split_sizes_0, x = out_81_cast_fp16)[name = string("op_7347_cast_fp16")]; tensor query_states_121_strides_0 = const()[name = string("query_states_121_strides_0"), val = tensor([1, 1])]; string query_states_121_pad_type_0 = const()[name = string("query_states_121_pad_type_0"), val = string("valid")]; tensor query_states_121_pad_0 = const()[name = string("query_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_121_dilations_0 = const()[name = string("query_states_121_dilations_0"), val = tensor([1, 1])]; int32 query_states_121_groups_0 = const()[name = string("query_states_121_groups_0"), val = int32(1)]; tensor query_states_121_cast_fp16 = conv(dilations = query_states_121_dilations_0, groups = query_states_121_groups_0, pad = query_states_121_pad_0, pad_type = query_states_121_pad_type_0, strides = query_states_121_strides_0, weight = layers_20_self_attn_q_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("query_states_121_cast_fp16")]; tensor key_states_201_strides_0 = const()[name = string("key_states_201_strides_0"), val = tensor([1, 1])]; string key_states_201_pad_type_0 = const()[name = string("key_states_201_pad_type_0"), val = string("valid")]; tensor key_states_201_pad_0 = const()[name = string("key_states_201_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_201_dilations_0 = const()[name = string("key_states_201_dilations_0"), val = tensor([1, 1])]; int32 key_states_201_groups_0 = const()[name = string("key_states_201_groups_0"), val = int32(1)]; tensor key_states_201_cast_fp16 = conv(dilations = key_states_201_dilations_0, groups = key_states_201_groups_0, pad = key_states_201_pad_0, pad_type = key_states_201_pad_type_0, strides = key_states_201_strides_0, weight = layers_20_self_attn_k_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("key_states_201_cast_fp16")]; tensor layers_20_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_20_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501498624)))]; tensor value_states_121_strides_0 = const()[name = string("value_states_121_strides_0"), val = tensor([1, 1])]; string value_states_121_pad_type_0 = const()[name = string("value_states_121_pad_type_0"), val = string("valid")]; tensor value_states_121_pad_0 = const()[name = string("value_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_121_dilations_0 = const()[name = string("value_states_121_dilations_0"), val = tensor([1, 1])]; int32 value_states_121_groups_0 = const()[name = string("value_states_121_groups_0"), val = int32(1)]; tensor value_states_121_cast_fp16 = conv(dilations = value_states_121_dilations_0, groups = value_states_121_groups_0, pad = value_states_121_pad_0, pad_type = value_states_121_pad_type_0, strides = value_states_121_strides_0, weight = layers_20_self_attn_v_proj_weight_to_fp16, x = var_7347_cast_fp16_0)[name = string("value_states_121_cast_fp16")]; tensor concat_240x = const()[name = string("concat_240x"), val = tensor([1, 16, 128, -1])]; tensor x_201_cast_fp16 = reshape(shape = concat_240x, x = query_states_121_cast_fp16)[name = string("x_201_cast_fp16")]; tensor concat_241x = const()[name = string("concat_241x"), val = tensor([1, 2, 128, -1])]; tensor var_7404_cast_fp16 = reshape(shape = concat_241x, x = key_states_201_cast_fp16)[name = string("op_7404_cast_fp16")]; tensor concat_242x = const()[name = string("concat_242x"), val = tensor([1, 2, 128, -1])]; tensor var_7411_cast_fp16 = reshape(shape = concat_242x, x = value_states_121_cast_fp16)[name = string("op_7411_cast_fp16")]; tensor var_7415_cast_fp16 = mul(x = x_201_cast_fp16, y = var_869_cast_fp16)[name = string("op_7415_cast_fp16")]; tensor var_7416_split_sizes_0 = const()[name = string("op_7416_split_sizes_0"), val = tensor([64, 64])]; int32 var_7416_axis_0 = const()[name = string("op_7416_axis_0"), val = int32(-2)]; tensor var_7416_cast_fp16_0, tensor var_7416_cast_fp16_1 = split(axis = var_7416_axis_0, split_sizes = var_7416_split_sizes_0, x = x_201_cast_fp16)[name = string("op_7416_cast_fp16")]; fp16 const_202_promoted_to_fp16 = const()[name = string("const_202_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7418_cast_fp16 = mul(x = var_7416_cast_fp16_1, y = const_202_promoted_to_fp16)[name = string("op_7418_cast_fp16")]; int32 var_7420 = const()[name = string("op_7420"), val = int32(-2)]; bool var_7421_interleave_0 = const()[name = string("op_7421_interleave_0"), val = bool(false)]; tensor var_7421_cast_fp16 = concat(axis = var_7420, interleave = var_7421_interleave_0, values = (var_7418_cast_fp16, var_7416_cast_fp16_0))[name = string("op_7421_cast_fp16")]; tensor var_7422_cast_fp16 = mul(x = var_7421_cast_fp16, y = var_878_cast_fp16)[name = string("op_7422_cast_fp16")]; tensor query_states_123_cast_fp16 = add(x = var_7415_cast_fp16, y = var_7422_cast_fp16)[name = string("query_states_123_cast_fp16")]; tensor var_7428_cast_fp16 = mul(x = var_7404_cast_fp16, y = var_869_cast_fp16)[name = string("op_7428_cast_fp16")]; tensor var_7429_split_sizes_0 = const()[name = string("op_7429_split_sizes_0"), val = tensor([64, 64])]; int32 var_7429_axis_0 = const()[name = string("op_7429_axis_0"), val = int32(-2)]; tensor var_7429_cast_fp16_0, tensor var_7429_cast_fp16_1 = split(axis = var_7429_axis_0, split_sizes = var_7429_split_sizes_0, x = var_7404_cast_fp16)[name = string("op_7429_cast_fp16")]; fp16 const_203_promoted_to_fp16 = const()[name = string("const_203_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7431_cast_fp16 = mul(x = var_7429_cast_fp16_1, y = const_203_promoted_to_fp16)[name = string("op_7431_cast_fp16")]; int32 var_7433 = const()[name = string("op_7433"), val = int32(-2)]; bool var_7434_interleave_0 = const()[name = string("op_7434_interleave_0"), val = bool(false)]; tensor var_7434_cast_fp16 = concat(axis = var_7433, interleave = var_7434_interleave_0, values = (var_7431_cast_fp16, var_7429_cast_fp16_0))[name = string("op_7434_cast_fp16")]; tensor var_7435_cast_fp16 = mul(x = var_7434_cast_fp16, y = var_878_cast_fp16)[name = string("op_7435_cast_fp16")]; tensor key_states_205_cast_fp16 = add(x = var_7428_cast_fp16, y = var_7435_cast_fp16)[name = string("key_states_205_cast_fp16")]; tensor expand_dims_240 = const()[name = string("expand_dims_240"), val = tensor([20])]; tensor expand_dims_241 = const()[name = string("expand_dims_241"), val = tensor([0])]; tensor expand_dims_243 = const()[name = string("expand_dims_243"), val = tensor([0])]; int32 concat_245_axis_0 = const()[name = string("concat_245_axis_0"), val = int32(0)]; bool concat_245_interleave_0 = const()[name = string("concat_245_interleave_0"), val = bool(false)]; tensor concat_245 = concat(axis = concat_245_axis_0, interleave = concat_245_interleave_0, values = (expand_dims_240, expand_dims_241, position_id, expand_dims_243))[name = string("concat_245")]; tensor expand_dims_244 = const()[name = string("expand_dims_244"), val = tensor([21])]; tensor concat_246_values1_0 = const()[name = string("concat_246_values1_0"), val = tensor([0])]; tensor concat_246_values3_0 = const()[name = string("concat_246_values3_0"), val = tensor([0])]; int32 concat_246_axis_0 = const()[name = string("concat_246_axis_0"), val = int32(0)]; bool concat_246_interleave_0 = const()[name = string("concat_246_interleave_0"), val = bool(false)]; tensor concat_246 = concat(axis = concat_246_axis_0, interleave = concat_246_interleave_0, values = (expand_dims_244, concat_246_values1_0, cache_position_end, concat_246_values3_0))[name = string("concat_246")]; tensor key_states_207_perm_0 = const()[name = string("key_states_207_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_21_stride_0 = const()[name = string("key_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_207_cast_fp16 = transpose(perm = key_states_207_perm_0, x = key_states_205_cast_fp16)[name = string("transpose_539")]; tensor key_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = key_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = key_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_21_squeeze_mask_0, stride = key_cache_internal_tensor_assign_21_stride_0, update = key_states_207_cast_fp16, x = coreml_update_state_374)[name = string("key_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_21_cast_fp16, input = key_cache)[name = string("coreml_update_state_376_write_state")]; tensor coreml_update_state_376 = read_state(input = key_cache)[name = string("coreml_update_state_376")]; tensor value_states_123_perm_0 = const()[name = string("value_states_123_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_21_stride_0 = const()[name = string("value_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_123_cast_fp16 = transpose(perm = value_states_123_perm_0, x = var_7411_cast_fp16)[name = string("transpose_538")]; tensor value_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = value_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = value_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_21_squeeze_mask_0, stride = value_cache_internal_tensor_assign_21_stride_0, update = value_states_123_cast_fp16, x = coreml_update_state_375)[name = string("value_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_21_cast_fp16, input = value_cache)[name = string("coreml_update_state_377_write_state")]; tensor coreml_update_state_377 = read_state(input = value_cache)[name = string("coreml_update_state_377")]; tensor var_7505_begin_0 = const()[name = string("op_7505_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7505_end_0 = const()[name = string("op_7505_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7505_end_mask_0 = const()[name = string("op_7505_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7505_cast_fp16 = slice_by_index(begin = var_7505_begin_0, end = var_7505_end_0, end_mask = var_7505_end_mask_0, x = coreml_update_state_376)[name = string("op_7505_cast_fp16")]; tensor tile_40 = const()[name = string("tile_40"), val = tensor([1, 1])]; int32 var_7508_axis_0 = const()[name = string("op_7508_axis_0"), val = int32(1)]; tensor var_7508_cast_fp16_0, tensor var_7508_cast_fp16_1 = split(axis = var_7508_axis_0, split_sizes = tile_40, x = var_7505_cast_fp16)[name = string("op_7508_cast_fp16")]; tensor var_7515_begin_0 = const()[name = string("op_7515_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7515_end_0 = const()[name = string("op_7515_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7515_end_mask_0 = const()[name = string("op_7515_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7515_cast_fp16 = slice_by_index(begin = var_7515_begin_0, end = var_7515_end_0, end_mask = var_7515_end_mask_0, x = coreml_update_state_377)[name = string("op_7515_cast_fp16")]; tensor tile_41 = const()[name = string("tile_41"), val = tensor([1, 1])]; int32 var_7518_axis_0 = const()[name = string("op_7518_axis_0"), val = int32(1)]; tensor var_7518_cast_fp16_0, tensor var_7518_cast_fp16_1 = split(axis = var_7518_axis_0, split_sizes = tile_41, x = var_7515_cast_fp16)[name = string("op_7518_cast_fp16")]; tensor var_7521_split_sizes_0 = const()[name = string("op_7521_split_sizes_0"), val = tensor([8, 8])]; int32 var_7521_axis_0 = const()[name = string("op_7521_axis_0"), val = int32(1)]; tensor var_7521_0, tensor var_7521_1 = split(axis = var_7521_axis_0, split_sizes = var_7521_split_sizes_0, x = query_states_123_cast_fp16)[name = string("op_7521")]; bool attn_weights_321_transpose_x_0 = const()[name = string("attn_weights_321_transpose_x_0"), val = bool(false)]; bool attn_weights_321_transpose_y_0 = const()[name = string("attn_weights_321_transpose_y_0"), val = bool(false)]; tensor attn_weights_321_cast_fp16 = matmul(transpose_x = attn_weights_321_transpose_x_0, transpose_y = attn_weights_321_transpose_y_0, x = var_7508_cast_fp16_0, y = var_7521_0)[name = string("attn_weights_321_cast_fp16")]; fp16 var_7524_to_fp16 = const()[name = string("op_7524_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_323_cast_fp16 = mul(x = attn_weights_321_cast_fp16, y = var_7524_to_fp16)[name = string("attn_weights_323_cast_fp16")]; tensor attn_weights_325_cast_fp16 = add(x = attn_weights_323_cast_fp16, y = attn_mask_1)[name = string("attn_weights_325_cast_fp16")]; int32 var_7528 = const()[name = string("op_7528"), val = int32(-2)]; tensor attn_weights_327_cast_fp16 = softmax(axis = var_7528, x = attn_weights_325_cast_fp16)[name = string("attn_weights_327_cast_fp16")]; bool var_7534_transpose_x_1 = const()[name = string("op_7534_transpose_x_1"), val = bool(true)]; bool var_7534_transpose_y_1 = const()[name = string("op_7534_transpose_y_1"), val = bool(false)]; tensor var_7534_cast_fp16 = matmul(transpose_x = var_7534_transpose_x_1, transpose_y = var_7534_transpose_y_1, x = attn_weights_327_cast_fp16, y = var_7518_cast_fp16_0)[name = string("op_7534_cast_fp16")]; bool attn_weights_329_transpose_x_0 = const()[name = string("attn_weights_329_transpose_x_0"), val = bool(false)]; bool attn_weights_329_transpose_y_0 = const()[name = string("attn_weights_329_transpose_y_0"), val = bool(false)]; tensor attn_weights_329_cast_fp16 = matmul(transpose_x = attn_weights_329_transpose_x_0, transpose_y = attn_weights_329_transpose_y_0, x = var_7508_cast_fp16_1, y = var_7521_1)[name = string("attn_weights_329_cast_fp16")]; fp16 var_7536_to_fp16 = const()[name = string("op_7536_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_331_cast_fp16 = mul(x = attn_weights_329_cast_fp16, y = var_7536_to_fp16)[name = string("attn_weights_331_cast_fp16")]; tensor attn_weights_333_cast_fp16 = add(x = attn_weights_331_cast_fp16, y = attn_mask_1)[name = string("attn_weights_333_cast_fp16")]; int32 var_7540 = const()[name = string("op_7540"), val = int32(-2)]; tensor attn_weights_335_cast_fp16 = softmax(axis = var_7540, x = attn_weights_333_cast_fp16)[name = string("attn_weights_335_cast_fp16")]; bool attn_output_161_transpose_x_1 = const()[name = string("attn_output_161_transpose_x_1"), val = bool(true)]; bool attn_output_161_transpose_y_1 = const()[name = string("attn_output_161_transpose_y_1"), val = bool(false)]; tensor attn_output_161_cast_fp16 = matmul(transpose_x = attn_output_161_transpose_x_1, transpose_y = attn_output_161_transpose_y_1, x = attn_weights_335_cast_fp16, y = var_7518_cast_fp16_1)[name = string("attn_output_161_cast_fp16")]; int32 var_7548 = const()[name = string("op_7548"), val = int32(1)]; bool attn_output_163_interleave_0 = const()[name = string("attn_output_163_interleave_0"), val = bool(false)]; tensor attn_output_163_cast_fp16 = concat(axis = var_7548, interleave = attn_output_163_interleave_0, values = (var_7534_cast_fp16, attn_output_161_cast_fp16))[name = string("attn_output_163_cast_fp16")]; tensor var_7552_perm_0 = const()[name = string("op_7552_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_251x = const()[name = string("concat_251x"), val = tensor([1, 2048, 1, -1])]; tensor var_7552_cast_fp16 = transpose(perm = var_7552_perm_0, x = attn_output_163_cast_fp16)[name = string("transpose_537")]; tensor attn_output_167_cast_fp16 = reshape(shape = concat_251x, x = var_7552_cast_fp16)[name = string("attn_output_167_cast_fp16")]; tensor hidden_states_203_strides_0 = const()[name = string("hidden_states_203_strides_0"), val = tensor([1, 1])]; string hidden_states_203_pad_type_0 = const()[name = string("hidden_states_203_pad_type_0"), val = string("valid")]; tensor hidden_states_203_pad_0 = const()[name = string("hidden_states_203_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_203_dilations_0 = const()[name = string("hidden_states_203_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_203_groups_0 = const()[name = string("hidden_states_203_groups_0"), val = int32(1)]; tensor hidden_states_203_cast_fp16 = conv(dilations = hidden_states_203_dilations_0, groups = hidden_states_203_groups_0, pad = hidden_states_203_pad_0, pad_type = hidden_states_203_pad_type_0, strides = hidden_states_203_strides_0, weight = layers_20_self_attn_o_proj_weight_cast_fp16, x = attn_output_167_cast_fp16)[name = string("hidden_states_203_cast_fp16")]; tensor hidden_states_205_cast_fp16 = add(x = hidden_states_199_cast_fp16, y = hidden_states_203_cast_fp16)[name = string("hidden_states_205_cast_fp16")]; fp16 const_208_promoted_to_fp16 = const()[name = string("const_208_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7585_cast_fp16 = mul(x = hidden_states_205_cast_fp16, y = const_208_promoted_to_fp16)[name = string("op_7585_cast_fp16")]; int32 var_7583 = const()[name = string("op_7583"), val = int32(1)]; bool doubled_165_interleave_0 = const()[name = string("doubled_165_interleave_0"), val = bool(false)]; tensor doubled_165_cast_fp16 = concat(axis = var_7583, interleave = doubled_165_interleave_0, values = (hidden_states_205_cast_fp16, var_7585_cast_fp16))[name = string("doubled_165_cast_fp16")]; tensor out_83_axes_0 = const()[name = string("out_83_axes_0"), val = tensor([1])]; tensor out_83_gamma_0_to_fp16 = const()[name = string("out_83_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502547264)))]; fp16 var_7595_to_fp16 = const()[name = string("op_7595_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_83_cast_fp16 = layer_norm(axes = out_83_axes_0, epsilon = var_7595_to_fp16, gamma = out_83_gamma_0_to_fp16, x = doubled_165_cast_fp16)[name = string("out_83_cast_fp16")]; tensor var_7606_split_sizes_0 = const()[name = string("op_7606_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7606_axis_0 = const()[name = string("op_7606_axis_0"), val = int32(1)]; tensor var_7606_cast_fp16_0, tensor var_7606_cast_fp16_1 = split(axis = var_7606_axis_0, split_sizes = var_7606_split_sizes_0, x = out_83_cast_fp16)[name = string("op_7606_cast_fp16")]; tensor input_41_strides_0 = const()[name = string("input_41_strides_0"), val = tensor([1, 1])]; string input_41_pad_type_0 = const()[name = string("input_41_pad_type_0"), val = string("valid")]; tensor input_41_pad_0 = const()[name = string("input_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_41_dilations_0 = const()[name = string("input_41_dilations_0"), val = tensor([1, 1])]; int32 input_41_groups_0 = const()[name = string("input_41_groups_0"), val = int32(1)]; tensor input_41_cast_fp16 = conv(dilations = input_41_dilations_0, groups = input_41_groups_0, pad = input_41_pad_0, pad_type = input_41_pad_type_0, strides = input_41_strides_0, weight = layers_20_mlp_gate_proj_weight_cast_fp16, x = var_7606_cast_fp16_0)[name = string("input_41_cast_fp16")]; tensor var_7623_cast_fp16 = silu(x = input_41_cast_fp16)[name = string("op_7623_cast_fp16")]; tensor layers_20_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_20_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502555520)))]; tensor var_7629_strides_0 = const()[name = string("op_7629_strides_0"), val = tensor([1, 1])]; string var_7629_pad_type_0 = const()[name = string("op_7629_pad_type_0"), val = string("valid")]; tensor var_7629_pad_0 = const()[name = string("op_7629_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7629_dilations_0 = const()[name = string("op_7629_dilations_0"), val = tensor([1, 1])]; int32 var_7629_groups_0 = const()[name = string("op_7629_groups_0"), val = int32(1)]; tensor var_7629_cast_fp16 = conv(dilations = var_7629_dilations_0, groups = var_7629_groups_0, pad = var_7629_pad_0, pad_type = var_7629_pad_type_0, strides = var_7629_strides_0, weight = layers_20_mlp_up_proj_weight_to_fp16, x = var_7606_cast_fp16_0)[name = string("op_7629_cast_fp16")]; tensor x_209_cast_fp16 = mul(x = var_7623_cast_fp16, y = var_7629_cast_fp16)[name = string("x_209_cast_fp16")]; tensor hidden_states_207_strides_0 = const()[name = string("hidden_states_207_strides_0"), val = tensor([1, 1])]; string hidden_states_207_pad_type_0 = const()[name = string("hidden_states_207_pad_type_0"), val = string("valid")]; tensor hidden_states_207_pad_0 = const()[name = string("hidden_states_207_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_207_dilations_0 = const()[name = string("hidden_states_207_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_207_groups_0 = const()[name = string("hidden_states_207_groups_0"), val = int32(1)]; tensor hidden_states_207_cast_fp16 = conv(dilations = hidden_states_207_dilations_0, groups = hidden_states_207_groups_0, pad = hidden_states_207_pad_0, pad_type = hidden_states_207_pad_type_0, strides = hidden_states_207_strides_0, weight = layers_20_mlp_down_proj_weight_cast_fp16, x = x_209_cast_fp16)[name = string("hidden_states_207_cast_fp16")]; tensor hidden_states_209_cast_fp16 = add(x = hidden_states_205_cast_fp16, y = hidden_states_207_cast_fp16)[name = string("hidden_states_209_cast_fp16")]; fp16 const_210_promoted_to_fp16 = const()[name = string("const_210_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7647_cast_fp16 = mul(x = hidden_states_209_cast_fp16, y = const_210_promoted_to_fp16)[name = string("op_7647_cast_fp16")]; int32 var_7645 = const()[name = string("op_7645"), val = int32(1)]; bool doubled_169_interleave_0 = const()[name = string("doubled_169_interleave_0"), val = bool(false)]; tensor doubled_169_cast_fp16 = concat(axis = var_7645, interleave = doubled_169_interleave_0, values = (hidden_states_209_cast_fp16, var_7647_cast_fp16))[name = string("doubled_169_cast_fp16")]; tensor out_85_axes_0 = const()[name = string("out_85_axes_0"), val = tensor([1])]; tensor out_85_gamma_0_to_fp16 = const()[name = string("out_85_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527721408)))]; fp16 var_7657_to_fp16 = const()[name = string("op_7657_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_85_cast_fp16 = layer_norm(axes = out_85_axes_0, epsilon = var_7657_to_fp16, gamma = out_85_gamma_0_to_fp16, x = doubled_169_cast_fp16)[name = string("out_85_cast_fp16")]; tensor var_7668_split_sizes_0 = const()[name = string("op_7668_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7668_axis_0 = const()[name = string("op_7668_axis_0"), val = int32(1)]; tensor var_7668_cast_fp16_0, tensor var_7668_cast_fp16_1 = split(axis = var_7668_axis_0, split_sizes = var_7668_split_sizes_0, x = out_85_cast_fp16)[name = string("op_7668_cast_fp16")]; tensor query_states_127_strides_0 = const()[name = string("query_states_127_strides_0"), val = tensor([1, 1])]; string query_states_127_pad_type_0 = const()[name = string("query_states_127_pad_type_0"), val = string("valid")]; tensor query_states_127_pad_0 = const()[name = string("query_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_127_dilations_0 = const()[name = string("query_states_127_dilations_0"), val = tensor([1, 1])]; int32 query_states_127_groups_0 = const()[name = string("query_states_127_groups_0"), val = int32(1)]; tensor query_states_127_cast_fp16 = conv(dilations = query_states_127_dilations_0, groups = query_states_127_groups_0, pad = query_states_127_pad_0, pad_type = query_states_127_pad_type_0, strides = query_states_127_strides_0, weight = layers_21_self_attn_q_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("query_states_127_cast_fp16")]; tensor key_states_211_strides_0 = const()[name = string("key_states_211_strides_0"), val = tensor([1, 1])]; string key_states_211_pad_type_0 = const()[name = string("key_states_211_pad_type_0"), val = string("valid")]; tensor key_states_211_pad_0 = const()[name = string("key_states_211_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_211_dilations_0 = const()[name = string("key_states_211_dilations_0"), val = tensor([1, 1])]; int32 key_states_211_groups_0 = const()[name = string("key_states_211_groups_0"), val = int32(1)]; tensor key_states_211_cast_fp16 = conv(dilations = key_states_211_dilations_0, groups = key_states_211_groups_0, pad = key_states_211_pad_0, pad_type = key_states_211_pad_type_0, strides = key_states_211_strides_0, weight = layers_21_self_attn_k_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("key_states_211_cast_fp16")]; tensor layers_21_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_21_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527729664)))]; tensor value_states_127_strides_0 = const()[name = string("value_states_127_strides_0"), val = tensor([1, 1])]; string value_states_127_pad_type_0 = const()[name = string("value_states_127_pad_type_0"), val = string("valid")]; tensor value_states_127_pad_0 = const()[name = string("value_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_127_dilations_0 = const()[name = string("value_states_127_dilations_0"), val = tensor([1, 1])]; int32 value_states_127_groups_0 = const()[name = string("value_states_127_groups_0"), val = int32(1)]; tensor value_states_127_cast_fp16 = conv(dilations = value_states_127_dilations_0, groups = value_states_127_groups_0, pad = value_states_127_pad_0, pad_type = value_states_127_pad_type_0, strides = value_states_127_strides_0, weight = layers_21_self_attn_v_proj_weight_to_fp16, x = var_7668_cast_fp16_0)[name = string("value_states_127_cast_fp16")]; tensor concat_252x = const()[name = string("concat_252x"), val = tensor([1, 16, 128, -1])]; tensor x_211_cast_fp16 = reshape(shape = concat_252x, x = query_states_127_cast_fp16)[name = string("x_211_cast_fp16")]; tensor concat_253x = const()[name = string("concat_253x"), val = tensor([1, 2, 128, -1])]; tensor var_7725_cast_fp16 = reshape(shape = concat_253x, x = key_states_211_cast_fp16)[name = string("op_7725_cast_fp16")]; tensor concat_254x = const()[name = string("concat_254x"), val = tensor([1, 2, 128, -1])]; tensor var_7732_cast_fp16 = reshape(shape = concat_254x, x = value_states_127_cast_fp16)[name = string("op_7732_cast_fp16")]; tensor var_7736_cast_fp16 = mul(x = x_211_cast_fp16, y = var_869_cast_fp16)[name = string("op_7736_cast_fp16")]; tensor var_7737_split_sizes_0 = const()[name = string("op_7737_split_sizes_0"), val = tensor([64, 64])]; int32 var_7737_axis_0 = const()[name = string("op_7737_axis_0"), val = int32(-2)]; tensor var_7737_cast_fp16_0, tensor var_7737_cast_fp16_1 = split(axis = var_7737_axis_0, split_sizes = var_7737_split_sizes_0, x = x_211_cast_fp16)[name = string("op_7737_cast_fp16")]; fp16 const_212_promoted_to_fp16 = const()[name = string("const_212_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7739_cast_fp16 = mul(x = var_7737_cast_fp16_1, y = const_212_promoted_to_fp16)[name = string("op_7739_cast_fp16")]; int32 var_7741 = const()[name = string("op_7741"), val = int32(-2)]; bool var_7742_interleave_0 = const()[name = string("op_7742_interleave_0"), val = bool(false)]; tensor var_7742_cast_fp16 = concat(axis = var_7741, interleave = var_7742_interleave_0, values = (var_7739_cast_fp16, var_7737_cast_fp16_0))[name = string("op_7742_cast_fp16")]; tensor var_7743_cast_fp16 = mul(x = var_7742_cast_fp16, y = var_878_cast_fp16)[name = string("op_7743_cast_fp16")]; tensor query_states_129_cast_fp16 = add(x = var_7736_cast_fp16, y = var_7743_cast_fp16)[name = string("query_states_129_cast_fp16")]; tensor var_7749_cast_fp16 = mul(x = var_7725_cast_fp16, y = var_869_cast_fp16)[name = string("op_7749_cast_fp16")]; tensor var_7750_split_sizes_0 = const()[name = string("op_7750_split_sizes_0"), val = tensor([64, 64])]; int32 var_7750_axis_0 = const()[name = string("op_7750_axis_0"), val = int32(-2)]; tensor var_7750_cast_fp16_0, tensor var_7750_cast_fp16_1 = split(axis = var_7750_axis_0, split_sizes = var_7750_split_sizes_0, x = var_7725_cast_fp16)[name = string("op_7750_cast_fp16")]; fp16 const_213_promoted_to_fp16 = const()[name = string("const_213_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7752_cast_fp16 = mul(x = var_7750_cast_fp16_1, y = const_213_promoted_to_fp16)[name = string("op_7752_cast_fp16")]; int32 var_7754 = const()[name = string("op_7754"), val = int32(-2)]; bool var_7755_interleave_0 = const()[name = string("op_7755_interleave_0"), val = bool(false)]; tensor var_7755_cast_fp16 = concat(axis = var_7754, interleave = var_7755_interleave_0, values = (var_7752_cast_fp16, var_7750_cast_fp16_0))[name = string("op_7755_cast_fp16")]; tensor var_7756_cast_fp16 = mul(x = var_7755_cast_fp16, y = var_878_cast_fp16)[name = string("op_7756_cast_fp16")]; tensor key_states_215_cast_fp16 = add(x = var_7749_cast_fp16, y = var_7756_cast_fp16)[name = string("key_states_215_cast_fp16")]; tensor expand_dims_252 = const()[name = string("expand_dims_252"), val = tensor([21])]; tensor expand_dims_253 = const()[name = string("expand_dims_253"), val = tensor([0])]; tensor expand_dims_255 = const()[name = string("expand_dims_255"), val = tensor([0])]; int32 concat_257_axis_0 = const()[name = string("concat_257_axis_0"), val = int32(0)]; bool concat_257_interleave_0 = const()[name = string("concat_257_interleave_0"), val = bool(false)]; tensor concat_257 = concat(axis = concat_257_axis_0, interleave = concat_257_interleave_0, values = (expand_dims_252, expand_dims_253, position_id, expand_dims_255))[name = string("concat_257")]; tensor expand_dims_256 = const()[name = string("expand_dims_256"), val = tensor([22])]; tensor concat_258_values1_0 = const()[name = string("concat_258_values1_0"), val = tensor([0])]; tensor concat_258_values3_0 = const()[name = string("concat_258_values3_0"), val = tensor([0])]; int32 concat_258_axis_0 = const()[name = string("concat_258_axis_0"), val = int32(0)]; bool concat_258_interleave_0 = const()[name = string("concat_258_interleave_0"), val = bool(false)]; tensor concat_258 = concat(axis = concat_258_axis_0, interleave = concat_258_interleave_0, values = (expand_dims_256, concat_258_values1_0, cache_position_end, concat_258_values3_0))[name = string("concat_258")]; tensor key_states_217_perm_0 = const()[name = string("key_states_217_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_22_stride_0 = const()[name = string("key_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_217_cast_fp16 = transpose(perm = key_states_217_perm_0, x = key_states_215_cast_fp16)[name = string("transpose_536")]; tensor key_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = key_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = key_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_22_squeeze_mask_0, stride = key_cache_internal_tensor_assign_22_stride_0, update = key_states_217_cast_fp16, x = coreml_update_state_376)[name = string("key_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_22_cast_fp16, input = key_cache)[name = string("coreml_update_state_378_write_state")]; tensor coreml_update_state_378 = read_state(input = key_cache)[name = string("coreml_update_state_378")]; tensor value_states_129_perm_0 = const()[name = string("value_states_129_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_22_stride_0 = const()[name = string("value_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_129_cast_fp16 = transpose(perm = value_states_129_perm_0, x = var_7732_cast_fp16)[name = string("transpose_535")]; tensor value_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = value_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = value_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_22_squeeze_mask_0, stride = value_cache_internal_tensor_assign_22_stride_0, update = value_states_129_cast_fp16, x = coreml_update_state_377)[name = string("value_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_22_cast_fp16, input = value_cache)[name = string("coreml_update_state_379_write_state")]; tensor coreml_update_state_379 = read_state(input = value_cache)[name = string("coreml_update_state_379")]; tensor var_7826_begin_0 = const()[name = string("op_7826_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7826_end_0 = const()[name = string("op_7826_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7826_end_mask_0 = const()[name = string("op_7826_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7826_cast_fp16 = slice_by_index(begin = var_7826_begin_0, end = var_7826_end_0, end_mask = var_7826_end_mask_0, x = coreml_update_state_378)[name = string("op_7826_cast_fp16")]; tensor tile_42 = const()[name = string("tile_42"), val = tensor([1, 1])]; int32 var_7829_axis_0 = const()[name = string("op_7829_axis_0"), val = int32(1)]; tensor var_7829_cast_fp16_0, tensor var_7829_cast_fp16_1 = split(axis = var_7829_axis_0, split_sizes = tile_42, x = var_7826_cast_fp16)[name = string("op_7829_cast_fp16")]; tensor var_7836_begin_0 = const()[name = string("op_7836_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7836_end_0 = const()[name = string("op_7836_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7836_end_mask_0 = const()[name = string("op_7836_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7836_cast_fp16 = slice_by_index(begin = var_7836_begin_0, end = var_7836_end_0, end_mask = var_7836_end_mask_0, x = coreml_update_state_379)[name = string("op_7836_cast_fp16")]; tensor tile_43 = const()[name = string("tile_43"), val = tensor([1, 1])]; int32 var_7839_axis_0 = const()[name = string("op_7839_axis_0"), val = int32(1)]; tensor var_7839_cast_fp16_0, tensor var_7839_cast_fp16_1 = split(axis = var_7839_axis_0, split_sizes = tile_43, x = var_7836_cast_fp16)[name = string("op_7839_cast_fp16")]; tensor var_7842_split_sizes_0 = const()[name = string("op_7842_split_sizes_0"), val = tensor([8, 8])]; int32 var_7842_axis_0 = const()[name = string("op_7842_axis_0"), val = int32(1)]; tensor var_7842_0, tensor var_7842_1 = split(axis = var_7842_axis_0, split_sizes = var_7842_split_sizes_0, x = query_states_129_cast_fp16)[name = string("op_7842")]; bool attn_weights_337_transpose_x_0 = const()[name = string("attn_weights_337_transpose_x_0"), val = bool(false)]; bool attn_weights_337_transpose_y_0 = const()[name = string("attn_weights_337_transpose_y_0"), val = bool(false)]; tensor attn_weights_337_cast_fp16 = matmul(transpose_x = attn_weights_337_transpose_x_0, transpose_y = attn_weights_337_transpose_y_0, x = var_7829_cast_fp16_0, y = var_7842_0)[name = string("attn_weights_337_cast_fp16")]; fp16 var_7845_to_fp16 = const()[name = string("op_7845_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_339_cast_fp16 = mul(x = attn_weights_337_cast_fp16, y = var_7845_to_fp16)[name = string("attn_weights_339_cast_fp16")]; tensor attn_weights_341_cast_fp16 = add(x = attn_weights_339_cast_fp16, y = attn_mask_1)[name = string("attn_weights_341_cast_fp16")]; int32 var_7849 = const()[name = string("op_7849"), val = int32(-2)]; tensor attn_weights_343_cast_fp16 = softmax(axis = var_7849, x = attn_weights_341_cast_fp16)[name = string("attn_weights_343_cast_fp16")]; bool var_7855_transpose_x_1 = const()[name = string("op_7855_transpose_x_1"), val = bool(true)]; bool var_7855_transpose_y_1 = const()[name = string("op_7855_transpose_y_1"), val = bool(false)]; tensor var_7855_cast_fp16 = matmul(transpose_x = var_7855_transpose_x_1, transpose_y = var_7855_transpose_y_1, x = attn_weights_343_cast_fp16, y = var_7839_cast_fp16_0)[name = string("op_7855_cast_fp16")]; bool attn_weights_345_transpose_x_0 = const()[name = string("attn_weights_345_transpose_x_0"), val = bool(false)]; bool attn_weights_345_transpose_y_0 = const()[name = string("attn_weights_345_transpose_y_0"), val = bool(false)]; tensor attn_weights_345_cast_fp16 = matmul(transpose_x = attn_weights_345_transpose_x_0, transpose_y = attn_weights_345_transpose_y_0, x = var_7829_cast_fp16_1, y = var_7842_1)[name = string("attn_weights_345_cast_fp16")]; fp16 var_7857_to_fp16 = const()[name = string("op_7857_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_347_cast_fp16 = mul(x = attn_weights_345_cast_fp16, y = var_7857_to_fp16)[name = string("attn_weights_347_cast_fp16")]; tensor attn_weights_349_cast_fp16 = add(x = attn_weights_347_cast_fp16, y = attn_mask_1)[name = string("attn_weights_349_cast_fp16")]; int32 var_7861 = const()[name = string("op_7861"), val = int32(-2)]; tensor attn_weights_351_cast_fp16 = softmax(axis = var_7861, x = attn_weights_349_cast_fp16)[name = string("attn_weights_351_cast_fp16")]; bool attn_output_169_transpose_x_1 = const()[name = string("attn_output_169_transpose_x_1"), val = bool(true)]; bool attn_output_169_transpose_y_1 = const()[name = string("attn_output_169_transpose_y_1"), val = bool(false)]; tensor attn_output_169_cast_fp16 = matmul(transpose_x = attn_output_169_transpose_x_1, transpose_y = attn_output_169_transpose_y_1, x = attn_weights_351_cast_fp16, y = var_7839_cast_fp16_1)[name = string("attn_output_169_cast_fp16")]; int32 var_7869 = const()[name = string("op_7869"), val = int32(1)]; bool attn_output_171_interleave_0 = const()[name = string("attn_output_171_interleave_0"), val = bool(false)]; tensor attn_output_171_cast_fp16 = concat(axis = var_7869, interleave = attn_output_171_interleave_0, values = (var_7855_cast_fp16, attn_output_169_cast_fp16))[name = string("attn_output_171_cast_fp16")]; tensor var_7873_perm_0 = const()[name = string("op_7873_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_263x = const()[name = string("concat_263x"), val = tensor([1, 2048, 1, -1])]; tensor var_7873_cast_fp16 = transpose(perm = var_7873_perm_0, x = attn_output_171_cast_fp16)[name = string("transpose_534")]; tensor attn_output_175_cast_fp16 = reshape(shape = concat_263x, x = var_7873_cast_fp16)[name = string("attn_output_175_cast_fp16")]; tensor hidden_states_213_strides_0 = const()[name = string("hidden_states_213_strides_0"), val = tensor([1, 1])]; string hidden_states_213_pad_type_0 = const()[name = string("hidden_states_213_pad_type_0"), val = string("valid")]; tensor hidden_states_213_pad_0 = const()[name = string("hidden_states_213_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_213_dilations_0 = const()[name = string("hidden_states_213_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_213_groups_0 = const()[name = string("hidden_states_213_groups_0"), val = int32(1)]; tensor hidden_states_213_cast_fp16 = conv(dilations = hidden_states_213_dilations_0, groups = hidden_states_213_groups_0, pad = hidden_states_213_pad_0, pad_type = hidden_states_213_pad_type_0, strides = hidden_states_213_strides_0, weight = layers_21_self_attn_o_proj_weight_cast_fp16, x = attn_output_175_cast_fp16)[name = string("hidden_states_213_cast_fp16")]; tensor hidden_states_215_cast_fp16 = add(x = hidden_states_209_cast_fp16, y = hidden_states_213_cast_fp16)[name = string("hidden_states_215_cast_fp16")]; fp16 const_218_promoted_to_fp16 = const()[name = string("const_218_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7906_cast_fp16 = mul(x = hidden_states_215_cast_fp16, y = const_218_promoted_to_fp16)[name = string("op_7906_cast_fp16")]; int32 var_7904 = const()[name = string("op_7904"), val = int32(1)]; bool doubled_173_interleave_0 = const()[name = string("doubled_173_interleave_0"), val = bool(false)]; tensor doubled_173_cast_fp16 = concat(axis = var_7904, interleave = doubled_173_interleave_0, values = (hidden_states_215_cast_fp16, var_7906_cast_fp16))[name = string("doubled_173_cast_fp16")]; tensor out_87_axes_0 = const()[name = string("out_87_axes_0"), val = tensor([1])]; tensor out_87_gamma_0_to_fp16 = const()[name = string("out_87_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528778304)))]; fp16 var_7916_to_fp16 = const()[name = string("op_7916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_87_cast_fp16 = layer_norm(axes = out_87_axes_0, epsilon = var_7916_to_fp16, gamma = out_87_gamma_0_to_fp16, x = doubled_173_cast_fp16)[name = string("out_87_cast_fp16")]; tensor var_7927_split_sizes_0 = const()[name = string("op_7927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7927_axis_0 = const()[name = string("op_7927_axis_0"), val = int32(1)]; tensor var_7927_cast_fp16_0, tensor var_7927_cast_fp16_1 = split(axis = var_7927_axis_0, split_sizes = var_7927_split_sizes_0, x = out_87_cast_fp16)[name = string("op_7927_cast_fp16")]; tensor input_43_strides_0 = const()[name = string("input_43_strides_0"), val = tensor([1, 1])]; string input_43_pad_type_0 = const()[name = string("input_43_pad_type_0"), val = string("valid")]; tensor input_43_pad_0 = const()[name = string("input_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_43_dilations_0 = const()[name = string("input_43_dilations_0"), val = tensor([1, 1])]; int32 input_43_groups_0 = const()[name = string("input_43_groups_0"), val = int32(1)]; tensor input_43_cast_fp16 = conv(dilations = input_43_dilations_0, groups = input_43_groups_0, pad = input_43_pad_0, pad_type = input_43_pad_type_0, strides = input_43_strides_0, weight = layers_21_mlp_gate_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("input_43_cast_fp16")]; tensor var_7944_cast_fp16 = silu(x = input_43_cast_fp16)[name = string("op_7944_cast_fp16")]; tensor var_7950_strides_0 = const()[name = string("op_7950_strides_0"), val = tensor([1, 1])]; string var_7950_pad_type_0 = const()[name = string("op_7950_pad_type_0"), val = string("valid")]; tensor var_7950_pad_0 = const()[name = string("op_7950_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7950_dilations_0 = const()[name = string("op_7950_dilations_0"), val = tensor([1, 1])]; int32 var_7950_groups_0 = const()[name = string("op_7950_groups_0"), val = int32(1)]; tensor var_7950_cast_fp16 = conv(dilations = var_7950_dilations_0, groups = var_7950_groups_0, pad = var_7950_pad_0, pad_type = var_7950_pad_type_0, strides = var_7950_strides_0, weight = layers_21_mlp_up_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("op_7950_cast_fp16")]; tensor x_219_cast_fp16 = mul(x = var_7944_cast_fp16, y = var_7950_cast_fp16)[name = string("x_219_cast_fp16")]; tensor hidden_states_217_strides_0 = const()[name = string("hidden_states_217_strides_0"), val = tensor([1, 1])]; string hidden_states_217_pad_type_0 = const()[name = string("hidden_states_217_pad_type_0"), val = string("valid")]; tensor hidden_states_217_pad_0 = const()[name = string("hidden_states_217_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_217_dilations_0 = const()[name = string("hidden_states_217_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_217_groups_0 = const()[name = string("hidden_states_217_groups_0"), val = int32(1)]; tensor hidden_states_217_cast_fp16 = conv(dilations = hidden_states_217_dilations_0, groups = hidden_states_217_groups_0, pad = hidden_states_217_pad_0, pad_type = hidden_states_217_pad_type_0, strides = hidden_states_217_strides_0, weight = layers_21_mlp_down_proj_weight_cast_fp16, x = x_219_cast_fp16)[name = string("hidden_states_217_cast_fp16")]; tensor hidden_states_219_cast_fp16 = add(x = hidden_states_215_cast_fp16, y = hidden_states_217_cast_fp16)[name = string("hidden_states_219_cast_fp16")]; fp16 const_220_promoted_to_fp16 = const()[name = string("const_220_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7968_cast_fp16 = mul(x = hidden_states_219_cast_fp16, y = const_220_promoted_to_fp16)[name = string("op_7968_cast_fp16")]; int32 var_7966 = const()[name = string("op_7966"), val = int32(1)]; bool doubled_177_interleave_0 = const()[name = string("doubled_177_interleave_0"), val = bool(false)]; tensor doubled_177_cast_fp16 = concat(axis = var_7966, interleave = doubled_177_interleave_0, values = (hidden_states_219_cast_fp16, var_7968_cast_fp16))[name = string("doubled_177_cast_fp16")]; tensor out_89_axes_0 = const()[name = string("out_89_axes_0"), val = tensor([1])]; tensor out_89_gamma_0_to_fp16 = const()[name = string("out_89_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528786560)))]; fp16 var_7978_to_fp16 = const()[name = string("op_7978_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_89_cast_fp16 = layer_norm(axes = out_89_axes_0, epsilon = var_7978_to_fp16, gamma = out_89_gamma_0_to_fp16, x = doubled_177_cast_fp16)[name = string("out_89_cast_fp16")]; tensor var_7989_split_sizes_0 = const()[name = string("op_7989_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7989_axis_0 = const()[name = string("op_7989_axis_0"), val = int32(1)]; tensor var_7989_cast_fp16_0, tensor var_7989_cast_fp16_1 = split(axis = var_7989_axis_0, split_sizes = var_7989_split_sizes_0, x = out_89_cast_fp16)[name = string("op_7989_cast_fp16")]; tensor query_states_133_strides_0 = const()[name = string("query_states_133_strides_0"), val = tensor([1, 1])]; string query_states_133_pad_type_0 = const()[name = string("query_states_133_pad_type_0"), val = string("valid")]; tensor query_states_133_pad_0 = const()[name = string("query_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_133_dilations_0 = const()[name = string("query_states_133_dilations_0"), val = tensor([1, 1])]; int32 query_states_133_groups_0 = const()[name = string("query_states_133_groups_0"), val = int32(1)]; tensor query_states_133_cast_fp16 = conv(dilations = query_states_133_dilations_0, groups = query_states_133_groups_0, pad = query_states_133_pad_0, pad_type = query_states_133_pad_type_0, strides = query_states_133_strides_0, weight = layers_22_self_attn_q_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("query_states_133_cast_fp16")]; tensor key_states_221_strides_0 = const()[name = string("key_states_221_strides_0"), val = tensor([1, 1])]; string key_states_221_pad_type_0 = const()[name = string("key_states_221_pad_type_0"), val = string("valid")]; tensor key_states_221_pad_0 = const()[name = string("key_states_221_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_221_dilations_0 = const()[name = string("key_states_221_dilations_0"), val = tensor([1, 1])]; int32 key_states_221_groups_0 = const()[name = string("key_states_221_groups_0"), val = int32(1)]; tensor key_states_221_cast_fp16 = conv(dilations = key_states_221_dilations_0, groups = key_states_221_groups_0, pad = key_states_221_pad_0, pad_type = key_states_221_pad_type_0, strides = key_states_221_strides_0, weight = layers_22_self_attn_k_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("key_states_221_cast_fp16")]; tensor layers_22_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528794816)))]; tensor value_states_133_strides_0 = const()[name = string("value_states_133_strides_0"), val = tensor([1, 1])]; string value_states_133_pad_type_0 = const()[name = string("value_states_133_pad_type_0"), val = string("valid")]; tensor value_states_133_pad_0 = const()[name = string("value_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_133_dilations_0 = const()[name = string("value_states_133_dilations_0"), val = tensor([1, 1])]; int32 value_states_133_groups_0 = const()[name = string("value_states_133_groups_0"), val = int32(1)]; tensor value_states_133_cast_fp16 = conv(dilations = value_states_133_dilations_0, groups = value_states_133_groups_0, pad = value_states_133_pad_0, pad_type = value_states_133_pad_type_0, strides = value_states_133_strides_0, weight = layers_22_self_attn_v_proj_weight_to_fp16, x = var_7989_cast_fp16_0)[name = string("value_states_133_cast_fp16")]; tensor concat_264x = const()[name = string("concat_264x"), val = tensor([1, 16, 128, -1])]; tensor x_221_cast_fp16 = reshape(shape = concat_264x, x = query_states_133_cast_fp16)[name = string("x_221_cast_fp16")]; tensor concat_265x = const()[name = string("concat_265x"), val = tensor([1, 2, 128, -1])]; tensor var_8046_cast_fp16 = reshape(shape = concat_265x, x = key_states_221_cast_fp16)[name = string("op_8046_cast_fp16")]; tensor concat_266x = const()[name = string("concat_266x"), val = tensor([1, 2, 128, -1])]; tensor var_8053_cast_fp16 = reshape(shape = concat_266x, x = value_states_133_cast_fp16)[name = string("op_8053_cast_fp16")]; tensor var_8057_cast_fp16 = mul(x = x_221_cast_fp16, y = var_869_cast_fp16)[name = string("op_8057_cast_fp16")]; tensor var_8058_split_sizes_0 = const()[name = string("op_8058_split_sizes_0"), val = tensor([64, 64])]; int32 var_8058_axis_0 = const()[name = string("op_8058_axis_0"), val = int32(-2)]; tensor var_8058_cast_fp16_0, tensor var_8058_cast_fp16_1 = split(axis = var_8058_axis_0, split_sizes = var_8058_split_sizes_0, x = x_221_cast_fp16)[name = string("op_8058_cast_fp16")]; fp16 const_222_promoted_to_fp16 = const()[name = string("const_222_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8060_cast_fp16 = mul(x = var_8058_cast_fp16_1, y = const_222_promoted_to_fp16)[name = string("op_8060_cast_fp16")]; int32 var_8062 = const()[name = string("op_8062"), val = int32(-2)]; bool var_8063_interleave_0 = const()[name = string("op_8063_interleave_0"), val = bool(false)]; tensor var_8063_cast_fp16 = concat(axis = var_8062, interleave = var_8063_interleave_0, values = (var_8060_cast_fp16, var_8058_cast_fp16_0))[name = string("op_8063_cast_fp16")]; tensor var_8064_cast_fp16 = mul(x = var_8063_cast_fp16, y = var_878_cast_fp16)[name = string("op_8064_cast_fp16")]; tensor query_states_135_cast_fp16 = add(x = var_8057_cast_fp16, y = var_8064_cast_fp16)[name = string("query_states_135_cast_fp16")]; tensor var_8070_cast_fp16 = mul(x = var_8046_cast_fp16, y = var_869_cast_fp16)[name = string("op_8070_cast_fp16")]; tensor var_8071_split_sizes_0 = const()[name = string("op_8071_split_sizes_0"), val = tensor([64, 64])]; int32 var_8071_axis_0 = const()[name = string("op_8071_axis_0"), val = int32(-2)]; tensor var_8071_cast_fp16_0, tensor var_8071_cast_fp16_1 = split(axis = var_8071_axis_0, split_sizes = var_8071_split_sizes_0, x = var_8046_cast_fp16)[name = string("op_8071_cast_fp16")]; fp16 const_223_promoted_to_fp16 = const()[name = string("const_223_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8073_cast_fp16 = mul(x = var_8071_cast_fp16_1, y = const_223_promoted_to_fp16)[name = string("op_8073_cast_fp16")]; int32 var_8075 = const()[name = string("op_8075"), val = int32(-2)]; bool var_8076_interleave_0 = const()[name = string("op_8076_interleave_0"), val = bool(false)]; tensor var_8076_cast_fp16 = concat(axis = var_8075, interleave = var_8076_interleave_0, values = (var_8073_cast_fp16, var_8071_cast_fp16_0))[name = string("op_8076_cast_fp16")]; tensor var_8077_cast_fp16 = mul(x = var_8076_cast_fp16, y = var_878_cast_fp16)[name = string("op_8077_cast_fp16")]; tensor key_states_225_cast_fp16 = add(x = var_8070_cast_fp16, y = var_8077_cast_fp16)[name = string("key_states_225_cast_fp16")]; tensor expand_dims_264 = const()[name = string("expand_dims_264"), val = tensor([22])]; tensor expand_dims_265 = const()[name = string("expand_dims_265"), val = tensor([0])]; tensor expand_dims_267 = const()[name = string("expand_dims_267"), val = tensor([0])]; int32 concat_269_axis_0 = const()[name = string("concat_269_axis_0"), val = int32(0)]; bool concat_269_interleave_0 = const()[name = string("concat_269_interleave_0"), val = bool(false)]; tensor concat_269 = concat(axis = concat_269_axis_0, interleave = concat_269_interleave_0, values = (expand_dims_264, expand_dims_265, position_id, expand_dims_267))[name = string("concat_269")]; tensor expand_dims_268 = const()[name = string("expand_dims_268"), val = tensor([23])]; tensor concat_270_values1_0 = const()[name = string("concat_270_values1_0"), val = tensor([0])]; tensor concat_270_values3_0 = const()[name = string("concat_270_values3_0"), val = tensor([0])]; int32 concat_270_axis_0 = const()[name = string("concat_270_axis_0"), val = int32(0)]; bool concat_270_interleave_0 = const()[name = string("concat_270_interleave_0"), val = bool(false)]; tensor concat_270 = concat(axis = concat_270_axis_0, interleave = concat_270_interleave_0, values = (expand_dims_268, concat_270_values1_0, cache_position_end, concat_270_values3_0))[name = string("concat_270")]; tensor key_states_227_perm_0 = const()[name = string("key_states_227_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_23_stride_0 = const()[name = string("key_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_227_cast_fp16 = transpose(perm = key_states_227_perm_0, x = key_states_225_cast_fp16)[name = string("transpose_533")]; tensor key_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = key_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = key_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_23_squeeze_mask_0, stride = key_cache_internal_tensor_assign_23_stride_0, update = key_states_227_cast_fp16, x = coreml_update_state_378)[name = string("key_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_23_cast_fp16, input = key_cache)[name = string("coreml_update_state_380_write_state")]; tensor coreml_update_state_380 = read_state(input = key_cache)[name = string("coreml_update_state_380")]; tensor value_states_135_perm_0 = const()[name = string("value_states_135_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_23_stride_0 = const()[name = string("value_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_135_cast_fp16 = transpose(perm = value_states_135_perm_0, x = var_8053_cast_fp16)[name = string("transpose_532")]; tensor value_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = value_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = value_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_23_squeeze_mask_0, stride = value_cache_internal_tensor_assign_23_stride_0, update = value_states_135_cast_fp16, x = coreml_update_state_379)[name = string("value_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_23_cast_fp16, input = value_cache)[name = string("coreml_update_state_381_write_state")]; tensor coreml_update_state_381 = read_state(input = value_cache)[name = string("coreml_update_state_381")]; tensor var_8147_begin_0 = const()[name = string("op_8147_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8147_end_0 = const()[name = string("op_8147_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8147_end_mask_0 = const()[name = string("op_8147_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8147_cast_fp16 = slice_by_index(begin = var_8147_begin_0, end = var_8147_end_0, end_mask = var_8147_end_mask_0, x = coreml_update_state_380)[name = string("op_8147_cast_fp16")]; tensor tile_44 = const()[name = string("tile_44"), val = tensor([1, 1])]; int32 var_8150_axis_0 = const()[name = string("op_8150_axis_0"), val = int32(1)]; tensor var_8150_cast_fp16_0, tensor var_8150_cast_fp16_1 = split(axis = var_8150_axis_0, split_sizes = tile_44, x = var_8147_cast_fp16)[name = string("op_8150_cast_fp16")]; tensor var_8157_begin_0 = const()[name = string("op_8157_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8157_end_0 = const()[name = string("op_8157_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8157_end_mask_0 = const()[name = string("op_8157_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8157_cast_fp16 = slice_by_index(begin = var_8157_begin_0, end = var_8157_end_0, end_mask = var_8157_end_mask_0, x = coreml_update_state_381)[name = string("op_8157_cast_fp16")]; tensor tile_45 = const()[name = string("tile_45"), val = tensor([1, 1])]; int32 var_8160_axis_0 = const()[name = string("op_8160_axis_0"), val = int32(1)]; tensor var_8160_cast_fp16_0, tensor var_8160_cast_fp16_1 = split(axis = var_8160_axis_0, split_sizes = tile_45, x = var_8157_cast_fp16)[name = string("op_8160_cast_fp16")]; tensor var_8163_split_sizes_0 = const()[name = string("op_8163_split_sizes_0"), val = tensor([8, 8])]; int32 var_8163_axis_0 = const()[name = string("op_8163_axis_0"), val = int32(1)]; tensor var_8163_0, tensor var_8163_1 = split(axis = var_8163_axis_0, split_sizes = var_8163_split_sizes_0, x = query_states_135_cast_fp16)[name = string("op_8163")]; bool attn_weights_353_transpose_x_0 = const()[name = string("attn_weights_353_transpose_x_0"), val = bool(false)]; bool attn_weights_353_transpose_y_0 = const()[name = string("attn_weights_353_transpose_y_0"), val = bool(false)]; tensor attn_weights_353_cast_fp16 = matmul(transpose_x = attn_weights_353_transpose_x_0, transpose_y = attn_weights_353_transpose_y_0, x = var_8150_cast_fp16_0, y = var_8163_0)[name = string("attn_weights_353_cast_fp16")]; fp16 var_8166_to_fp16 = const()[name = string("op_8166_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_355_cast_fp16 = mul(x = attn_weights_353_cast_fp16, y = var_8166_to_fp16)[name = string("attn_weights_355_cast_fp16")]; tensor attn_weights_357_cast_fp16 = add(x = attn_weights_355_cast_fp16, y = attn_mask_1)[name = string("attn_weights_357_cast_fp16")]; int32 var_8170 = const()[name = string("op_8170"), val = int32(-2)]; tensor attn_weights_359_cast_fp16 = softmax(axis = var_8170, x = attn_weights_357_cast_fp16)[name = string("attn_weights_359_cast_fp16")]; bool var_8176_transpose_x_1 = const()[name = string("op_8176_transpose_x_1"), val = bool(true)]; bool var_8176_transpose_y_1 = const()[name = string("op_8176_transpose_y_1"), val = bool(false)]; tensor var_8176_cast_fp16 = matmul(transpose_x = var_8176_transpose_x_1, transpose_y = var_8176_transpose_y_1, x = attn_weights_359_cast_fp16, y = var_8160_cast_fp16_0)[name = string("op_8176_cast_fp16")]; bool attn_weights_361_transpose_x_0 = const()[name = string("attn_weights_361_transpose_x_0"), val = bool(false)]; bool attn_weights_361_transpose_y_0 = const()[name = string("attn_weights_361_transpose_y_0"), val = bool(false)]; tensor attn_weights_361_cast_fp16 = matmul(transpose_x = attn_weights_361_transpose_x_0, transpose_y = attn_weights_361_transpose_y_0, x = var_8150_cast_fp16_1, y = var_8163_1)[name = string("attn_weights_361_cast_fp16")]; fp16 var_8178_to_fp16 = const()[name = string("op_8178_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_363_cast_fp16 = mul(x = attn_weights_361_cast_fp16, y = var_8178_to_fp16)[name = string("attn_weights_363_cast_fp16")]; tensor attn_weights_365_cast_fp16 = add(x = attn_weights_363_cast_fp16, y = attn_mask_1)[name = string("attn_weights_365_cast_fp16")]; int32 var_8182 = const()[name = string("op_8182"), val = int32(-2)]; tensor attn_weights_367_cast_fp16 = softmax(axis = var_8182, x = attn_weights_365_cast_fp16)[name = string("attn_weights_367_cast_fp16")]; bool attn_output_177_transpose_x_1 = const()[name = string("attn_output_177_transpose_x_1"), val = bool(true)]; bool attn_output_177_transpose_y_1 = const()[name = string("attn_output_177_transpose_y_1"), val = bool(false)]; tensor attn_output_177_cast_fp16 = matmul(transpose_x = attn_output_177_transpose_x_1, transpose_y = attn_output_177_transpose_y_1, x = attn_weights_367_cast_fp16, y = var_8160_cast_fp16_1)[name = string("attn_output_177_cast_fp16")]; int32 var_8190 = const()[name = string("op_8190"), val = int32(1)]; bool attn_output_179_interleave_0 = const()[name = string("attn_output_179_interleave_0"), val = bool(false)]; tensor attn_output_179_cast_fp16 = concat(axis = var_8190, interleave = attn_output_179_interleave_0, values = (var_8176_cast_fp16, attn_output_177_cast_fp16))[name = string("attn_output_179_cast_fp16")]; tensor var_8194_perm_0 = const()[name = string("op_8194_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_275x = const()[name = string("concat_275x"), val = tensor([1, 2048, 1, -1])]; tensor var_8194_cast_fp16 = transpose(perm = var_8194_perm_0, x = attn_output_179_cast_fp16)[name = string("transpose_531")]; tensor attn_output_183_cast_fp16 = reshape(shape = concat_275x, x = var_8194_cast_fp16)[name = string("attn_output_183_cast_fp16")]; tensor layers_22_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1529843456)))]; tensor hidden_states_223_strides_0 = const()[name = string("hidden_states_223_strides_0"), val = tensor([1, 1])]; string hidden_states_223_pad_type_0 = const()[name = string("hidden_states_223_pad_type_0"), val = string("valid")]; tensor hidden_states_223_pad_0 = const()[name = string("hidden_states_223_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_223_dilations_0 = const()[name = string("hidden_states_223_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_223_groups_0 = const()[name = string("hidden_states_223_groups_0"), val = int32(1)]; tensor hidden_states_223_cast_fp16 = conv(dilations = hidden_states_223_dilations_0, groups = hidden_states_223_groups_0, pad = hidden_states_223_pad_0, pad_type = hidden_states_223_pad_type_0, strides = hidden_states_223_strides_0, weight = layers_22_self_attn_o_proj_weight_to_fp16, x = attn_output_183_cast_fp16)[name = string("hidden_states_223_cast_fp16")]; tensor hidden_states_225_cast_fp16 = add(x = hidden_states_219_cast_fp16, y = hidden_states_223_cast_fp16)[name = string("hidden_states_225_cast_fp16")]; fp16 const_228_promoted_to_fp16 = const()[name = string("const_228_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8227_cast_fp16 = mul(x = hidden_states_225_cast_fp16, y = const_228_promoted_to_fp16)[name = string("op_8227_cast_fp16")]; int32 var_8225 = const()[name = string("op_8225"), val = int32(1)]; bool doubled_181_interleave_0 = const()[name = string("doubled_181_interleave_0"), val = bool(false)]; tensor doubled_181_cast_fp16 = concat(axis = var_8225, interleave = doubled_181_interleave_0, values = (hidden_states_225_cast_fp16, var_8227_cast_fp16))[name = string("doubled_181_cast_fp16")]; tensor out_91_axes_0 = const()[name = string("out_91_axes_0"), val = tensor([1])]; tensor out_91_gamma_0_to_fp16 = const()[name = string("out_91_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538232128)))]; fp16 var_8237_to_fp16 = const()[name = string("op_8237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_91_cast_fp16 = layer_norm(axes = out_91_axes_0, epsilon = var_8237_to_fp16, gamma = out_91_gamma_0_to_fp16, x = doubled_181_cast_fp16)[name = string("out_91_cast_fp16")]; tensor var_8248_split_sizes_0 = const()[name = string("op_8248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8248_axis_0 = const()[name = string("op_8248_axis_0"), val = int32(1)]; tensor var_8248_cast_fp16_0, tensor var_8248_cast_fp16_1 = split(axis = var_8248_axis_0, split_sizes = var_8248_split_sizes_0, x = out_91_cast_fp16)[name = string("op_8248_cast_fp16")]; tensor input_45_strides_0 = const()[name = string("input_45_strides_0"), val = tensor([1, 1])]; string input_45_pad_type_0 = const()[name = string("input_45_pad_type_0"), val = string("valid")]; tensor input_45_pad_0 = const()[name = string("input_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_45_dilations_0 = const()[name = string("input_45_dilations_0"), val = tensor([1, 1])]; int32 input_45_groups_0 = const()[name = string("input_45_groups_0"), val = int32(1)]; tensor input_45_cast_fp16 = conv(dilations = input_45_dilations_0, groups = input_45_groups_0, pad = input_45_pad_0, pad_type = input_45_pad_type_0, strides = input_45_strides_0, weight = layers_22_mlp_gate_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("input_45_cast_fp16")]; tensor var_8265_cast_fp16 = silu(x = input_45_cast_fp16)[name = string("op_8265_cast_fp16")]; tensor var_8271_strides_0 = const()[name = string("op_8271_strides_0"), val = tensor([1, 1])]; string var_8271_pad_type_0 = const()[name = string("op_8271_pad_type_0"), val = string("valid")]; tensor var_8271_pad_0 = const()[name = string("op_8271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8271_dilations_0 = const()[name = string("op_8271_dilations_0"), val = tensor([1, 1])]; int32 var_8271_groups_0 = const()[name = string("op_8271_groups_0"), val = int32(1)]; tensor var_8271_cast_fp16 = conv(dilations = var_8271_dilations_0, groups = var_8271_groups_0, pad = var_8271_pad_0, pad_type = var_8271_pad_type_0, strides = var_8271_strides_0, weight = layers_22_mlp_up_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("op_8271_cast_fp16")]; tensor x_229_cast_fp16 = mul(x = var_8265_cast_fp16, y = var_8271_cast_fp16)[name = string("x_229_cast_fp16")]; tensor hidden_states_227_strides_0 = const()[name = string("hidden_states_227_strides_0"), val = tensor([1, 1])]; string hidden_states_227_pad_type_0 = const()[name = string("hidden_states_227_pad_type_0"), val = string("valid")]; tensor hidden_states_227_pad_0 = const()[name = string("hidden_states_227_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_227_dilations_0 = const()[name = string("hidden_states_227_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_227_groups_0 = const()[name = string("hidden_states_227_groups_0"), val = int32(1)]; tensor hidden_states_227_cast_fp16 = conv(dilations = hidden_states_227_dilations_0, groups = hidden_states_227_groups_0, pad = hidden_states_227_pad_0, pad_type = hidden_states_227_pad_type_0, strides = hidden_states_227_strides_0, weight = layers_22_mlp_down_proj_weight_cast_fp16, x = x_229_cast_fp16)[name = string("hidden_states_227_cast_fp16")]; tensor hidden_states_229_cast_fp16 = add(x = hidden_states_225_cast_fp16, y = hidden_states_227_cast_fp16)[name = string("hidden_states_229_cast_fp16")]; fp16 const_230_promoted_to_fp16 = const()[name = string("const_230_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8289_cast_fp16 = mul(x = hidden_states_229_cast_fp16, y = const_230_promoted_to_fp16)[name = string("op_8289_cast_fp16")]; int32 var_8287 = const()[name = string("op_8287"), val = int32(1)]; bool doubled_185_interleave_0 = const()[name = string("doubled_185_interleave_0"), val = bool(false)]; tensor doubled_185_cast_fp16 = concat(axis = var_8287, interleave = doubled_185_interleave_0, values = (hidden_states_229_cast_fp16, var_8289_cast_fp16))[name = string("doubled_185_cast_fp16")]; tensor out_93_axes_0 = const()[name = string("out_93_axes_0"), val = tensor([1])]; tensor out_93_gamma_0_to_fp16 = const()[name = string("out_93_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538240384)))]; fp16 var_8299_to_fp16 = const()[name = string("op_8299_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_93_cast_fp16 = layer_norm(axes = out_93_axes_0, epsilon = var_8299_to_fp16, gamma = out_93_gamma_0_to_fp16, x = doubled_185_cast_fp16)[name = string("out_93_cast_fp16")]; tensor var_8310_split_sizes_0 = const()[name = string("op_8310_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8310_axis_0 = const()[name = string("op_8310_axis_0"), val = int32(1)]; tensor var_8310_cast_fp16_0, tensor var_8310_cast_fp16_1 = split(axis = var_8310_axis_0, split_sizes = var_8310_split_sizes_0, x = out_93_cast_fp16)[name = string("op_8310_cast_fp16")]; tensor query_states_139_strides_0 = const()[name = string("query_states_139_strides_0"), val = tensor([1, 1])]; string query_states_139_pad_type_0 = const()[name = string("query_states_139_pad_type_0"), val = string("valid")]; tensor query_states_139_pad_0 = const()[name = string("query_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_139_dilations_0 = const()[name = string("query_states_139_dilations_0"), val = tensor([1, 1])]; int32 query_states_139_groups_0 = const()[name = string("query_states_139_groups_0"), val = int32(1)]; tensor query_states_139_cast_fp16 = conv(dilations = query_states_139_dilations_0, groups = query_states_139_groups_0, pad = query_states_139_pad_0, pad_type = query_states_139_pad_type_0, strides = query_states_139_strides_0, weight = layers_23_self_attn_q_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("query_states_139_cast_fp16")]; tensor key_states_231_strides_0 = const()[name = string("key_states_231_strides_0"), val = tensor([1, 1])]; string key_states_231_pad_type_0 = const()[name = string("key_states_231_pad_type_0"), val = string("valid")]; tensor key_states_231_pad_0 = const()[name = string("key_states_231_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_231_dilations_0 = const()[name = string("key_states_231_dilations_0"), val = tensor([1, 1])]; int32 key_states_231_groups_0 = const()[name = string("key_states_231_groups_0"), val = int32(1)]; tensor key_states_231_cast_fp16 = conv(dilations = key_states_231_dilations_0, groups = key_states_231_groups_0, pad = key_states_231_pad_0, pad_type = key_states_231_pad_type_0, strides = key_states_231_strides_0, weight = layers_23_self_attn_k_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("key_states_231_cast_fp16")]; tensor layers_23_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_23_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538248640)))]; tensor value_states_139_strides_0 = const()[name = string("value_states_139_strides_0"), val = tensor([1, 1])]; string value_states_139_pad_type_0 = const()[name = string("value_states_139_pad_type_0"), val = string("valid")]; tensor value_states_139_pad_0 = const()[name = string("value_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_139_dilations_0 = const()[name = string("value_states_139_dilations_0"), val = tensor([1, 1])]; int32 value_states_139_groups_0 = const()[name = string("value_states_139_groups_0"), val = int32(1)]; tensor value_states_139_cast_fp16 = conv(dilations = value_states_139_dilations_0, groups = value_states_139_groups_0, pad = value_states_139_pad_0, pad_type = value_states_139_pad_type_0, strides = value_states_139_strides_0, weight = layers_23_self_attn_v_proj_weight_to_fp16, x = var_8310_cast_fp16_0)[name = string("value_states_139_cast_fp16")]; tensor concat_276x = const()[name = string("concat_276x"), val = tensor([1, 16, 128, -1])]; tensor x_231_cast_fp16 = reshape(shape = concat_276x, x = query_states_139_cast_fp16)[name = string("x_231_cast_fp16")]; tensor concat_277x = const()[name = string("concat_277x"), val = tensor([1, 2, 128, -1])]; tensor var_8367_cast_fp16 = reshape(shape = concat_277x, x = key_states_231_cast_fp16)[name = string("op_8367_cast_fp16")]; tensor concat_278x = const()[name = string("concat_278x"), val = tensor([1, 2, 128, -1])]; tensor var_8374_cast_fp16 = reshape(shape = concat_278x, x = value_states_139_cast_fp16)[name = string("op_8374_cast_fp16")]; tensor var_8378_cast_fp16 = mul(x = x_231_cast_fp16, y = var_869_cast_fp16)[name = string("op_8378_cast_fp16")]; tensor var_8379_split_sizes_0 = const()[name = string("op_8379_split_sizes_0"), val = tensor([64, 64])]; int32 var_8379_axis_0 = const()[name = string("op_8379_axis_0"), val = int32(-2)]; tensor var_8379_cast_fp16_0, tensor var_8379_cast_fp16_1 = split(axis = var_8379_axis_0, split_sizes = var_8379_split_sizes_0, x = x_231_cast_fp16)[name = string("op_8379_cast_fp16")]; fp16 const_232_promoted_to_fp16 = const()[name = string("const_232_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8381_cast_fp16 = mul(x = var_8379_cast_fp16_1, y = const_232_promoted_to_fp16)[name = string("op_8381_cast_fp16")]; int32 var_8383 = const()[name = string("op_8383"), val = int32(-2)]; bool var_8384_interleave_0 = const()[name = string("op_8384_interleave_0"), val = bool(false)]; tensor var_8384_cast_fp16 = concat(axis = var_8383, interleave = var_8384_interleave_0, values = (var_8381_cast_fp16, var_8379_cast_fp16_0))[name = string("op_8384_cast_fp16")]; tensor var_8385_cast_fp16 = mul(x = var_8384_cast_fp16, y = var_878_cast_fp16)[name = string("op_8385_cast_fp16")]; tensor query_states_141_cast_fp16 = add(x = var_8378_cast_fp16, y = var_8385_cast_fp16)[name = string("query_states_141_cast_fp16")]; tensor var_8391_cast_fp16 = mul(x = var_8367_cast_fp16, y = var_869_cast_fp16)[name = string("op_8391_cast_fp16")]; tensor var_8392_split_sizes_0 = const()[name = string("op_8392_split_sizes_0"), val = tensor([64, 64])]; int32 var_8392_axis_0 = const()[name = string("op_8392_axis_0"), val = int32(-2)]; tensor var_8392_cast_fp16_0, tensor var_8392_cast_fp16_1 = split(axis = var_8392_axis_0, split_sizes = var_8392_split_sizes_0, x = var_8367_cast_fp16)[name = string("op_8392_cast_fp16")]; fp16 const_233_promoted_to_fp16 = const()[name = string("const_233_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8394_cast_fp16 = mul(x = var_8392_cast_fp16_1, y = const_233_promoted_to_fp16)[name = string("op_8394_cast_fp16")]; int32 var_8396 = const()[name = string("op_8396"), val = int32(-2)]; bool var_8397_interleave_0 = const()[name = string("op_8397_interleave_0"), val = bool(false)]; tensor var_8397_cast_fp16 = concat(axis = var_8396, interleave = var_8397_interleave_0, values = (var_8394_cast_fp16, var_8392_cast_fp16_0))[name = string("op_8397_cast_fp16")]; tensor var_8398_cast_fp16 = mul(x = var_8397_cast_fp16, y = var_878_cast_fp16)[name = string("op_8398_cast_fp16")]; tensor key_states_235_cast_fp16 = add(x = var_8391_cast_fp16, y = var_8398_cast_fp16)[name = string("key_states_235_cast_fp16")]; tensor expand_dims_276 = const()[name = string("expand_dims_276"), val = tensor([23])]; tensor expand_dims_277 = const()[name = string("expand_dims_277"), val = tensor([0])]; tensor expand_dims_279 = const()[name = string("expand_dims_279"), val = tensor([0])]; int32 concat_281_axis_0 = const()[name = string("concat_281_axis_0"), val = int32(0)]; bool concat_281_interleave_0 = const()[name = string("concat_281_interleave_0"), val = bool(false)]; tensor concat_281 = concat(axis = concat_281_axis_0, interleave = concat_281_interleave_0, values = (expand_dims_276, expand_dims_277, position_id, expand_dims_279))[name = string("concat_281")]; tensor expand_dims_280 = const()[name = string("expand_dims_280"), val = tensor([24])]; tensor concat_282_values1_0 = const()[name = string("concat_282_values1_0"), val = tensor([0])]; tensor concat_282_values3_0 = const()[name = string("concat_282_values3_0"), val = tensor([0])]; int32 concat_282_axis_0 = const()[name = string("concat_282_axis_0"), val = int32(0)]; bool concat_282_interleave_0 = const()[name = string("concat_282_interleave_0"), val = bool(false)]; tensor concat_282 = concat(axis = concat_282_axis_0, interleave = concat_282_interleave_0, values = (expand_dims_280, concat_282_values1_0, cache_position_end, concat_282_values3_0))[name = string("concat_282")]; tensor key_states_237_perm_0 = const()[name = string("key_states_237_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_24_stride_0 = const()[name = string("key_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_237_cast_fp16 = transpose(perm = key_states_237_perm_0, x = key_states_235_cast_fp16)[name = string("transpose_530")]; tensor key_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = key_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = key_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_24_squeeze_mask_0, stride = key_cache_internal_tensor_assign_24_stride_0, update = key_states_237_cast_fp16, x = coreml_update_state_380)[name = string("key_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_24_cast_fp16, input = key_cache)[name = string("coreml_update_state_382_write_state")]; tensor coreml_update_state_382 = read_state(input = key_cache)[name = string("coreml_update_state_382")]; tensor value_states_141_perm_0 = const()[name = string("value_states_141_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_24_stride_0 = const()[name = string("value_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_141_cast_fp16 = transpose(perm = value_states_141_perm_0, x = var_8374_cast_fp16)[name = string("transpose_529")]; tensor value_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = value_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = value_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_24_squeeze_mask_0, stride = value_cache_internal_tensor_assign_24_stride_0, update = value_states_141_cast_fp16, x = coreml_update_state_381)[name = string("value_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_24_cast_fp16, input = value_cache)[name = string("coreml_update_state_383_write_state")]; tensor coreml_update_state_383 = read_state(input = value_cache)[name = string("coreml_update_state_383")]; tensor var_8468_begin_0 = const()[name = string("op_8468_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8468_end_0 = const()[name = string("op_8468_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8468_end_mask_0 = const()[name = string("op_8468_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8468_cast_fp16 = slice_by_index(begin = var_8468_begin_0, end = var_8468_end_0, end_mask = var_8468_end_mask_0, x = coreml_update_state_382)[name = string("op_8468_cast_fp16")]; tensor tile_46 = const()[name = string("tile_46"), val = tensor([1, 1])]; int32 var_8471_axis_0 = const()[name = string("op_8471_axis_0"), val = int32(1)]; tensor var_8471_cast_fp16_0, tensor var_8471_cast_fp16_1 = split(axis = var_8471_axis_0, split_sizes = tile_46, x = var_8468_cast_fp16)[name = string("op_8471_cast_fp16")]; tensor var_8478_begin_0 = const()[name = string("op_8478_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8478_end_0 = const()[name = string("op_8478_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8478_end_mask_0 = const()[name = string("op_8478_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8478_cast_fp16 = slice_by_index(begin = var_8478_begin_0, end = var_8478_end_0, end_mask = var_8478_end_mask_0, x = coreml_update_state_383)[name = string("op_8478_cast_fp16")]; tensor tile_47 = const()[name = string("tile_47"), val = tensor([1, 1])]; int32 var_8481_axis_0 = const()[name = string("op_8481_axis_0"), val = int32(1)]; tensor var_8481_cast_fp16_0, tensor var_8481_cast_fp16_1 = split(axis = var_8481_axis_0, split_sizes = tile_47, x = var_8478_cast_fp16)[name = string("op_8481_cast_fp16")]; tensor var_8484_split_sizes_0 = const()[name = string("op_8484_split_sizes_0"), val = tensor([8, 8])]; int32 var_8484_axis_0 = const()[name = string("op_8484_axis_0"), val = int32(1)]; tensor var_8484_0, tensor var_8484_1 = split(axis = var_8484_axis_0, split_sizes = var_8484_split_sizes_0, x = query_states_141_cast_fp16)[name = string("op_8484")]; bool attn_weights_369_transpose_x_0 = const()[name = string("attn_weights_369_transpose_x_0"), val = bool(false)]; bool attn_weights_369_transpose_y_0 = const()[name = string("attn_weights_369_transpose_y_0"), val = bool(false)]; tensor attn_weights_369_cast_fp16 = matmul(transpose_x = attn_weights_369_transpose_x_0, transpose_y = attn_weights_369_transpose_y_0, x = var_8471_cast_fp16_0, y = var_8484_0)[name = string("attn_weights_369_cast_fp16")]; fp16 var_8487_to_fp16 = const()[name = string("op_8487_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_371_cast_fp16 = mul(x = attn_weights_369_cast_fp16, y = var_8487_to_fp16)[name = string("attn_weights_371_cast_fp16")]; tensor attn_weights_373_cast_fp16 = add(x = attn_weights_371_cast_fp16, y = attn_mask_1)[name = string("attn_weights_373_cast_fp16")]; int32 var_8491 = const()[name = string("op_8491"), val = int32(-2)]; tensor attn_weights_375_cast_fp16 = softmax(axis = var_8491, x = attn_weights_373_cast_fp16)[name = string("attn_weights_375_cast_fp16")]; bool var_8497_transpose_x_1 = const()[name = string("op_8497_transpose_x_1"), val = bool(true)]; bool var_8497_transpose_y_1 = const()[name = string("op_8497_transpose_y_1"), val = bool(false)]; tensor var_8497_cast_fp16 = matmul(transpose_x = var_8497_transpose_x_1, transpose_y = var_8497_transpose_y_1, x = attn_weights_375_cast_fp16, y = var_8481_cast_fp16_0)[name = string("op_8497_cast_fp16")]; bool attn_weights_377_transpose_x_0 = const()[name = string("attn_weights_377_transpose_x_0"), val = bool(false)]; bool attn_weights_377_transpose_y_0 = const()[name = string("attn_weights_377_transpose_y_0"), val = bool(false)]; tensor attn_weights_377_cast_fp16 = matmul(transpose_x = attn_weights_377_transpose_x_0, transpose_y = attn_weights_377_transpose_y_0, x = var_8471_cast_fp16_1, y = var_8484_1)[name = string("attn_weights_377_cast_fp16")]; fp16 var_8499_to_fp16 = const()[name = string("op_8499_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_379_cast_fp16 = mul(x = attn_weights_377_cast_fp16, y = var_8499_to_fp16)[name = string("attn_weights_379_cast_fp16")]; tensor attn_weights_381_cast_fp16 = add(x = attn_weights_379_cast_fp16, y = attn_mask_1)[name = string("attn_weights_381_cast_fp16")]; int32 var_8503 = const()[name = string("op_8503"), val = int32(-2)]; tensor attn_weights_383_cast_fp16 = softmax(axis = var_8503, x = attn_weights_381_cast_fp16)[name = string("attn_weights_383_cast_fp16")]; bool attn_output_185_transpose_x_1 = const()[name = string("attn_output_185_transpose_x_1"), val = bool(true)]; bool attn_output_185_transpose_y_1 = const()[name = string("attn_output_185_transpose_y_1"), val = bool(false)]; tensor attn_output_185_cast_fp16 = matmul(transpose_x = attn_output_185_transpose_x_1, transpose_y = attn_output_185_transpose_y_1, x = attn_weights_383_cast_fp16, y = var_8481_cast_fp16_1)[name = string("attn_output_185_cast_fp16")]; int32 var_8511 = const()[name = string("op_8511"), val = int32(1)]; bool attn_output_187_interleave_0 = const()[name = string("attn_output_187_interleave_0"), val = bool(false)]; tensor attn_output_187_cast_fp16 = concat(axis = var_8511, interleave = attn_output_187_interleave_0, values = (var_8497_cast_fp16, attn_output_185_cast_fp16))[name = string("attn_output_187_cast_fp16")]; tensor var_8515_perm_0 = const()[name = string("op_8515_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_287x = const()[name = string("concat_287x"), val = tensor([1, 2048, 1, -1])]; tensor var_8515_cast_fp16 = transpose(perm = var_8515_perm_0, x = attn_output_187_cast_fp16)[name = string("transpose_528")]; tensor attn_output_191_cast_fp16 = reshape(shape = concat_287x, x = var_8515_cast_fp16)[name = string("attn_output_191_cast_fp16")]; tensor hidden_states_233_strides_0 = const()[name = string("hidden_states_233_strides_0"), val = tensor([1, 1])]; string hidden_states_233_pad_type_0 = const()[name = string("hidden_states_233_pad_type_0"), val = string("valid")]; tensor hidden_states_233_pad_0 = const()[name = string("hidden_states_233_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_233_dilations_0 = const()[name = string("hidden_states_233_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_233_groups_0 = const()[name = string("hidden_states_233_groups_0"), val = int32(1)]; tensor hidden_states_233_cast_fp16 = conv(dilations = hidden_states_233_dilations_0, groups = hidden_states_233_groups_0, pad = hidden_states_233_pad_0, pad_type = hidden_states_233_pad_type_0, strides = hidden_states_233_strides_0, weight = layers_23_self_attn_o_proj_weight_cast_fp16, x = attn_output_191_cast_fp16)[name = string("hidden_states_233_cast_fp16")]; tensor hidden_states_235_cast_fp16 = add(x = hidden_states_229_cast_fp16, y = hidden_states_233_cast_fp16)[name = string("hidden_states_235_cast_fp16")]; fp16 const_238_promoted_to_fp16 = const()[name = string("const_238_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8548_cast_fp16 = mul(x = hidden_states_235_cast_fp16, y = const_238_promoted_to_fp16)[name = string("op_8548_cast_fp16")]; int32 var_8546 = const()[name = string("op_8546"), val = int32(1)]; bool doubled_189_interleave_0 = const()[name = string("doubled_189_interleave_0"), val = bool(false)]; tensor doubled_189_cast_fp16 = concat(axis = var_8546, interleave = doubled_189_interleave_0, values = (hidden_states_235_cast_fp16, var_8548_cast_fp16))[name = string("doubled_189_cast_fp16")]; tensor out_95_axes_0 = const()[name = string("out_95_axes_0"), val = tensor([1])]; tensor out_95_gamma_0_to_fp16 = const()[name = string("out_95_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539297280)))]; fp16 var_8558_to_fp16 = const()[name = string("op_8558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_95_cast_fp16 = layer_norm(axes = out_95_axes_0, epsilon = var_8558_to_fp16, gamma = out_95_gamma_0_to_fp16, x = doubled_189_cast_fp16)[name = string("out_95_cast_fp16")]; tensor var_8569_split_sizes_0 = const()[name = string("op_8569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8569_axis_0 = const()[name = string("op_8569_axis_0"), val = int32(1)]; tensor var_8569_cast_fp16_0, tensor var_8569_cast_fp16_1 = split(axis = var_8569_axis_0, split_sizes = var_8569_split_sizes_0, x = out_95_cast_fp16)[name = string("op_8569_cast_fp16")]; tensor input_47_strides_0 = const()[name = string("input_47_strides_0"), val = tensor([1, 1])]; string input_47_pad_type_0 = const()[name = string("input_47_pad_type_0"), val = string("valid")]; tensor input_47_pad_0 = const()[name = string("input_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_47_dilations_0 = const()[name = string("input_47_dilations_0"), val = tensor([1, 1])]; int32 input_47_groups_0 = const()[name = string("input_47_groups_0"), val = int32(1)]; tensor input_47_cast_fp16 = conv(dilations = input_47_dilations_0, groups = input_47_groups_0, pad = input_47_pad_0, pad_type = input_47_pad_type_0, strides = input_47_strides_0, weight = layers_23_mlp_gate_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("input_47_cast_fp16")]; tensor var_8586_cast_fp16 = silu(x = input_47_cast_fp16)[name = string("op_8586_cast_fp16")]; tensor var_8592_strides_0 = const()[name = string("op_8592_strides_0"), val = tensor([1, 1])]; string var_8592_pad_type_0 = const()[name = string("op_8592_pad_type_0"), val = string("valid")]; tensor var_8592_pad_0 = const()[name = string("op_8592_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8592_dilations_0 = const()[name = string("op_8592_dilations_0"), val = tensor([1, 1])]; int32 var_8592_groups_0 = const()[name = string("op_8592_groups_0"), val = int32(1)]; tensor var_8592_cast_fp16 = conv(dilations = var_8592_dilations_0, groups = var_8592_groups_0, pad = var_8592_pad_0, pad_type = var_8592_pad_type_0, strides = var_8592_strides_0, weight = layers_23_mlp_up_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("op_8592_cast_fp16")]; tensor x_239_cast_fp16 = mul(x = var_8586_cast_fp16, y = var_8592_cast_fp16)[name = string("x_239_cast_fp16")]; tensor hidden_states_237_strides_0 = const()[name = string("hidden_states_237_strides_0"), val = tensor([1, 1])]; string hidden_states_237_pad_type_0 = const()[name = string("hidden_states_237_pad_type_0"), val = string("valid")]; tensor hidden_states_237_pad_0 = const()[name = string("hidden_states_237_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_237_dilations_0 = const()[name = string("hidden_states_237_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_237_groups_0 = const()[name = string("hidden_states_237_groups_0"), val = int32(1)]; tensor hidden_states_237_cast_fp16 = conv(dilations = hidden_states_237_dilations_0, groups = hidden_states_237_groups_0, pad = hidden_states_237_pad_0, pad_type = hidden_states_237_pad_type_0, strides = hidden_states_237_strides_0, weight = layers_23_mlp_down_proj_weight_cast_fp16, x = x_239_cast_fp16)[name = string("hidden_states_237_cast_fp16")]; tensor hidden_states_239_cast_fp16 = add(x = hidden_states_235_cast_fp16, y = hidden_states_237_cast_fp16)[name = string("hidden_states_239_cast_fp16")]; fp16 const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8610_cast_fp16 = mul(x = hidden_states_239_cast_fp16, y = const_240_promoted_to_fp16)[name = string("op_8610_cast_fp16")]; int32 var_8608 = const()[name = string("op_8608"), val = int32(1)]; bool doubled_193_interleave_0 = const()[name = string("doubled_193_interleave_0"), val = bool(false)]; tensor doubled_193_cast_fp16 = concat(axis = var_8608, interleave = doubled_193_interleave_0, values = (hidden_states_239_cast_fp16, var_8610_cast_fp16))[name = string("doubled_193_cast_fp16")]; tensor out_97_axes_0 = const()[name = string("out_97_axes_0"), val = tensor([1])]; tensor out_97_gamma_0_to_fp16 = const()[name = string("out_97_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539305536)))]; fp16 var_8620_to_fp16 = const()[name = string("op_8620_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_97_cast_fp16 = layer_norm(axes = out_97_axes_0, epsilon = var_8620_to_fp16, gamma = out_97_gamma_0_to_fp16, x = doubled_193_cast_fp16)[name = string("out_97_cast_fp16")]; tensor var_8631_split_sizes_0 = const()[name = string("op_8631_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8631_axis_0 = const()[name = string("op_8631_axis_0"), val = int32(1)]; tensor var_8631_cast_fp16_0, tensor var_8631_cast_fp16_1 = split(axis = var_8631_axis_0, split_sizes = var_8631_split_sizes_0, x = out_97_cast_fp16)[name = string("op_8631_cast_fp16")]; tensor query_states_145_strides_0 = const()[name = string("query_states_145_strides_0"), val = tensor([1, 1])]; string query_states_145_pad_type_0 = const()[name = string("query_states_145_pad_type_0"), val = string("valid")]; tensor query_states_145_pad_0 = const()[name = string("query_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_145_dilations_0 = const()[name = string("query_states_145_dilations_0"), val = tensor([1, 1])]; int32 query_states_145_groups_0 = const()[name = string("query_states_145_groups_0"), val = int32(1)]; tensor query_states_145_cast_fp16 = conv(dilations = query_states_145_dilations_0, groups = query_states_145_groups_0, pad = query_states_145_pad_0, pad_type = query_states_145_pad_type_0, strides = query_states_145_strides_0, weight = layers_24_self_attn_q_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("query_states_145_cast_fp16")]; tensor key_states_241_strides_0 = const()[name = string("key_states_241_strides_0"), val = tensor([1, 1])]; string key_states_241_pad_type_0 = const()[name = string("key_states_241_pad_type_0"), val = string("valid")]; tensor key_states_241_pad_0 = const()[name = string("key_states_241_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_241_dilations_0 = const()[name = string("key_states_241_dilations_0"), val = tensor([1, 1])]; int32 key_states_241_groups_0 = const()[name = string("key_states_241_groups_0"), val = int32(1)]; tensor key_states_241_cast_fp16 = conv(dilations = key_states_241_dilations_0, groups = key_states_241_groups_0, pad = key_states_241_pad_0, pad_type = key_states_241_pad_type_0, strides = key_states_241_strides_0, weight = layers_24_self_attn_k_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("key_states_241_cast_fp16")]; tensor layers_24_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_24_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539313792)))]; tensor value_states_145_strides_0 = const()[name = string("value_states_145_strides_0"), val = tensor([1, 1])]; string value_states_145_pad_type_0 = const()[name = string("value_states_145_pad_type_0"), val = string("valid")]; tensor value_states_145_pad_0 = const()[name = string("value_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_145_dilations_0 = const()[name = string("value_states_145_dilations_0"), val = tensor([1, 1])]; int32 value_states_145_groups_0 = const()[name = string("value_states_145_groups_0"), val = int32(1)]; tensor value_states_145_cast_fp16 = conv(dilations = value_states_145_dilations_0, groups = value_states_145_groups_0, pad = value_states_145_pad_0, pad_type = value_states_145_pad_type_0, strides = value_states_145_strides_0, weight = layers_24_self_attn_v_proj_weight_to_fp16, x = var_8631_cast_fp16_0)[name = string("value_states_145_cast_fp16")]; tensor concat_288x = const()[name = string("concat_288x"), val = tensor([1, 16, 128, -1])]; tensor x_241_cast_fp16 = reshape(shape = concat_288x, x = query_states_145_cast_fp16)[name = string("x_241_cast_fp16")]; tensor concat_289x = const()[name = string("concat_289x"), val = tensor([1, 2, 128, -1])]; tensor var_8688_cast_fp16 = reshape(shape = concat_289x, x = key_states_241_cast_fp16)[name = string("op_8688_cast_fp16")]; tensor concat_290x = const()[name = string("concat_290x"), val = tensor([1, 2, 128, -1])]; tensor var_8695_cast_fp16 = reshape(shape = concat_290x, x = value_states_145_cast_fp16)[name = string("op_8695_cast_fp16")]; tensor var_8699_cast_fp16 = mul(x = x_241_cast_fp16, y = var_869_cast_fp16)[name = string("op_8699_cast_fp16")]; tensor var_8700_split_sizes_0 = const()[name = string("op_8700_split_sizes_0"), val = tensor([64, 64])]; int32 var_8700_axis_0 = const()[name = string("op_8700_axis_0"), val = int32(-2)]; tensor var_8700_cast_fp16_0, tensor var_8700_cast_fp16_1 = split(axis = var_8700_axis_0, split_sizes = var_8700_split_sizes_0, x = x_241_cast_fp16)[name = string("op_8700_cast_fp16")]; fp16 const_242_promoted_to_fp16 = const()[name = string("const_242_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8702_cast_fp16 = mul(x = var_8700_cast_fp16_1, y = const_242_promoted_to_fp16)[name = string("op_8702_cast_fp16")]; int32 var_8704 = const()[name = string("op_8704"), val = int32(-2)]; bool var_8705_interleave_0 = const()[name = string("op_8705_interleave_0"), val = bool(false)]; tensor var_8705_cast_fp16 = concat(axis = var_8704, interleave = var_8705_interleave_0, values = (var_8702_cast_fp16, var_8700_cast_fp16_0))[name = string("op_8705_cast_fp16")]; tensor var_8706_cast_fp16 = mul(x = var_8705_cast_fp16, y = var_878_cast_fp16)[name = string("op_8706_cast_fp16")]; tensor query_states_147_cast_fp16 = add(x = var_8699_cast_fp16, y = var_8706_cast_fp16)[name = string("query_states_147_cast_fp16")]; tensor var_8712_cast_fp16 = mul(x = var_8688_cast_fp16, y = var_869_cast_fp16)[name = string("op_8712_cast_fp16")]; tensor var_8713_split_sizes_0 = const()[name = string("op_8713_split_sizes_0"), val = tensor([64, 64])]; int32 var_8713_axis_0 = const()[name = string("op_8713_axis_0"), val = int32(-2)]; tensor var_8713_cast_fp16_0, tensor var_8713_cast_fp16_1 = split(axis = var_8713_axis_0, split_sizes = var_8713_split_sizes_0, x = var_8688_cast_fp16)[name = string("op_8713_cast_fp16")]; fp16 const_243_promoted_to_fp16 = const()[name = string("const_243_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8715_cast_fp16 = mul(x = var_8713_cast_fp16_1, y = const_243_promoted_to_fp16)[name = string("op_8715_cast_fp16")]; int32 var_8717 = const()[name = string("op_8717"), val = int32(-2)]; bool var_8718_interleave_0 = const()[name = string("op_8718_interleave_0"), val = bool(false)]; tensor var_8718_cast_fp16 = concat(axis = var_8717, interleave = var_8718_interleave_0, values = (var_8715_cast_fp16, var_8713_cast_fp16_0))[name = string("op_8718_cast_fp16")]; tensor var_8719_cast_fp16 = mul(x = var_8718_cast_fp16, y = var_878_cast_fp16)[name = string("op_8719_cast_fp16")]; tensor key_states_245_cast_fp16 = add(x = var_8712_cast_fp16, y = var_8719_cast_fp16)[name = string("key_states_245_cast_fp16")]; tensor expand_dims_288 = const()[name = string("expand_dims_288"), val = tensor([24])]; tensor expand_dims_289 = const()[name = string("expand_dims_289"), val = tensor([0])]; tensor expand_dims_291 = const()[name = string("expand_dims_291"), val = tensor([0])]; int32 concat_293_axis_0 = const()[name = string("concat_293_axis_0"), val = int32(0)]; bool concat_293_interleave_0 = const()[name = string("concat_293_interleave_0"), val = bool(false)]; tensor concat_293 = concat(axis = concat_293_axis_0, interleave = concat_293_interleave_0, values = (expand_dims_288, expand_dims_289, position_id, expand_dims_291))[name = string("concat_293")]; tensor expand_dims_292 = const()[name = string("expand_dims_292"), val = tensor([25])]; tensor concat_294_values1_0 = const()[name = string("concat_294_values1_0"), val = tensor([0])]; tensor concat_294_values3_0 = const()[name = string("concat_294_values3_0"), val = tensor([0])]; int32 concat_294_axis_0 = const()[name = string("concat_294_axis_0"), val = int32(0)]; bool concat_294_interleave_0 = const()[name = string("concat_294_interleave_0"), val = bool(false)]; tensor concat_294 = concat(axis = concat_294_axis_0, interleave = concat_294_interleave_0, values = (expand_dims_292, concat_294_values1_0, cache_position_end, concat_294_values3_0))[name = string("concat_294")]; tensor key_states_247_perm_0 = const()[name = string("key_states_247_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_25_stride_0 = const()[name = string("key_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_247_cast_fp16 = transpose(perm = key_states_247_perm_0, x = key_states_245_cast_fp16)[name = string("transpose_527")]; tensor key_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = key_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = key_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_25_squeeze_mask_0, stride = key_cache_internal_tensor_assign_25_stride_0, update = key_states_247_cast_fp16, x = coreml_update_state_382)[name = string("key_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_25_cast_fp16, input = key_cache)[name = string("coreml_update_state_384_write_state")]; tensor coreml_update_state_384 = read_state(input = key_cache)[name = string("coreml_update_state_384")]; tensor value_states_147_perm_0 = const()[name = string("value_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_25_stride_0 = const()[name = string("value_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_147_cast_fp16 = transpose(perm = value_states_147_perm_0, x = var_8695_cast_fp16)[name = string("transpose_526")]; tensor value_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = value_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = value_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_25_squeeze_mask_0, stride = value_cache_internal_tensor_assign_25_stride_0, update = value_states_147_cast_fp16, x = coreml_update_state_383)[name = string("value_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_25_cast_fp16, input = value_cache)[name = string("coreml_update_state_385_write_state")]; tensor coreml_update_state_385 = read_state(input = value_cache)[name = string("coreml_update_state_385")]; tensor var_8789_begin_0 = const()[name = string("op_8789_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8789_end_0 = const()[name = string("op_8789_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8789_end_mask_0 = const()[name = string("op_8789_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8789_cast_fp16 = slice_by_index(begin = var_8789_begin_0, end = var_8789_end_0, end_mask = var_8789_end_mask_0, x = coreml_update_state_384)[name = string("op_8789_cast_fp16")]; tensor tile_48 = const()[name = string("tile_48"), val = tensor([1, 1])]; int32 var_8792_axis_0 = const()[name = string("op_8792_axis_0"), val = int32(1)]; tensor var_8792_cast_fp16_0, tensor var_8792_cast_fp16_1 = split(axis = var_8792_axis_0, split_sizes = tile_48, x = var_8789_cast_fp16)[name = string("op_8792_cast_fp16")]; tensor var_8799_begin_0 = const()[name = string("op_8799_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8799_end_0 = const()[name = string("op_8799_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8799_end_mask_0 = const()[name = string("op_8799_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8799_cast_fp16 = slice_by_index(begin = var_8799_begin_0, end = var_8799_end_0, end_mask = var_8799_end_mask_0, x = coreml_update_state_385)[name = string("op_8799_cast_fp16")]; tensor tile_49 = const()[name = string("tile_49"), val = tensor([1, 1])]; int32 var_8802_axis_0 = const()[name = string("op_8802_axis_0"), val = int32(1)]; tensor var_8802_cast_fp16_0, tensor var_8802_cast_fp16_1 = split(axis = var_8802_axis_0, split_sizes = tile_49, x = var_8799_cast_fp16)[name = string("op_8802_cast_fp16")]; tensor var_8805_split_sizes_0 = const()[name = string("op_8805_split_sizes_0"), val = tensor([8, 8])]; int32 var_8805_axis_0 = const()[name = string("op_8805_axis_0"), val = int32(1)]; tensor var_8805_0, tensor var_8805_1 = split(axis = var_8805_axis_0, split_sizes = var_8805_split_sizes_0, x = query_states_147_cast_fp16)[name = string("op_8805")]; bool attn_weights_385_transpose_x_0 = const()[name = string("attn_weights_385_transpose_x_0"), val = bool(false)]; bool attn_weights_385_transpose_y_0 = const()[name = string("attn_weights_385_transpose_y_0"), val = bool(false)]; tensor attn_weights_385_cast_fp16 = matmul(transpose_x = attn_weights_385_transpose_x_0, transpose_y = attn_weights_385_transpose_y_0, x = var_8792_cast_fp16_0, y = var_8805_0)[name = string("attn_weights_385_cast_fp16")]; fp16 var_8808_to_fp16 = const()[name = string("op_8808_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_387_cast_fp16 = mul(x = attn_weights_385_cast_fp16, y = var_8808_to_fp16)[name = string("attn_weights_387_cast_fp16")]; tensor attn_weights_389_cast_fp16 = add(x = attn_weights_387_cast_fp16, y = attn_mask_1)[name = string("attn_weights_389_cast_fp16")]; int32 var_8812 = const()[name = string("op_8812"), val = int32(-2)]; tensor attn_weights_391_cast_fp16 = softmax(axis = var_8812, x = attn_weights_389_cast_fp16)[name = string("attn_weights_391_cast_fp16")]; bool var_8818_transpose_x_1 = const()[name = string("op_8818_transpose_x_1"), val = bool(true)]; bool var_8818_transpose_y_1 = const()[name = string("op_8818_transpose_y_1"), val = bool(false)]; tensor var_8818_cast_fp16 = matmul(transpose_x = var_8818_transpose_x_1, transpose_y = var_8818_transpose_y_1, x = attn_weights_391_cast_fp16, y = var_8802_cast_fp16_0)[name = string("op_8818_cast_fp16")]; bool attn_weights_393_transpose_x_0 = const()[name = string("attn_weights_393_transpose_x_0"), val = bool(false)]; bool attn_weights_393_transpose_y_0 = const()[name = string("attn_weights_393_transpose_y_0"), val = bool(false)]; tensor attn_weights_393_cast_fp16 = matmul(transpose_x = attn_weights_393_transpose_x_0, transpose_y = attn_weights_393_transpose_y_0, x = var_8792_cast_fp16_1, y = var_8805_1)[name = string("attn_weights_393_cast_fp16")]; fp16 var_8820_to_fp16 = const()[name = string("op_8820_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_395_cast_fp16 = mul(x = attn_weights_393_cast_fp16, y = var_8820_to_fp16)[name = string("attn_weights_395_cast_fp16")]; tensor attn_weights_397_cast_fp16 = add(x = attn_weights_395_cast_fp16, y = attn_mask_1)[name = string("attn_weights_397_cast_fp16")]; int32 var_8824 = const()[name = string("op_8824"), val = int32(-2)]; tensor attn_weights_399_cast_fp16 = softmax(axis = var_8824, x = attn_weights_397_cast_fp16)[name = string("attn_weights_399_cast_fp16")]; bool attn_output_193_transpose_x_1 = const()[name = string("attn_output_193_transpose_x_1"), val = bool(true)]; bool attn_output_193_transpose_y_1 = const()[name = string("attn_output_193_transpose_y_1"), val = bool(false)]; tensor attn_output_193_cast_fp16 = matmul(transpose_x = attn_output_193_transpose_x_1, transpose_y = attn_output_193_transpose_y_1, x = attn_weights_399_cast_fp16, y = var_8802_cast_fp16_1)[name = string("attn_output_193_cast_fp16")]; int32 var_8832 = const()[name = string("op_8832"), val = int32(1)]; bool attn_output_195_interleave_0 = const()[name = string("attn_output_195_interleave_0"), val = bool(false)]; tensor attn_output_195_cast_fp16 = concat(axis = var_8832, interleave = attn_output_195_interleave_0, values = (var_8818_cast_fp16, attn_output_193_cast_fp16))[name = string("attn_output_195_cast_fp16")]; tensor var_8836_perm_0 = const()[name = string("op_8836_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_299x = const()[name = string("concat_299x"), val = tensor([1, 2048, 1, -1])]; tensor var_8836_cast_fp16 = transpose(perm = var_8836_perm_0, x = attn_output_195_cast_fp16)[name = string("transpose_525")]; tensor attn_output_199_cast_fp16 = reshape(shape = concat_299x, x = var_8836_cast_fp16)[name = string("attn_output_199_cast_fp16")]; tensor hidden_states_243_strides_0 = const()[name = string("hidden_states_243_strides_0"), val = tensor([1, 1])]; string hidden_states_243_pad_type_0 = const()[name = string("hidden_states_243_pad_type_0"), val = string("valid")]; tensor hidden_states_243_pad_0 = const()[name = string("hidden_states_243_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_243_dilations_0 = const()[name = string("hidden_states_243_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_243_groups_0 = const()[name = string("hidden_states_243_groups_0"), val = int32(1)]; tensor hidden_states_243_cast_fp16 = conv(dilations = hidden_states_243_dilations_0, groups = hidden_states_243_groups_0, pad = hidden_states_243_pad_0, pad_type = hidden_states_243_pad_type_0, strides = hidden_states_243_strides_0, weight = layers_24_self_attn_o_proj_weight_cast_fp16, x = attn_output_199_cast_fp16)[name = string("hidden_states_243_cast_fp16")]; tensor hidden_states_245_cast_fp16 = add(x = hidden_states_239_cast_fp16, y = hidden_states_243_cast_fp16)[name = string("hidden_states_245_cast_fp16")]; fp16 const_248_promoted_to_fp16 = const()[name = string("const_248_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8869_cast_fp16 = mul(x = hidden_states_245_cast_fp16, y = const_248_promoted_to_fp16)[name = string("op_8869_cast_fp16")]; int32 var_8867 = const()[name = string("op_8867"), val = int32(1)]; bool doubled_197_interleave_0 = const()[name = string("doubled_197_interleave_0"), val = bool(false)]; tensor doubled_197_cast_fp16 = concat(axis = var_8867, interleave = doubled_197_interleave_0, values = (hidden_states_245_cast_fp16, var_8869_cast_fp16))[name = string("doubled_197_cast_fp16")]; tensor out_99_axes_0 = const()[name = string("out_99_axes_0"), val = tensor([1])]; tensor out_99_gamma_0_to_fp16 = const()[name = string("out_99_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540362432)))]; fp16 var_8879_to_fp16 = const()[name = string("op_8879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_99_cast_fp16 = layer_norm(axes = out_99_axes_0, epsilon = var_8879_to_fp16, gamma = out_99_gamma_0_to_fp16, x = doubled_197_cast_fp16)[name = string("out_99_cast_fp16")]; tensor var_8890_split_sizes_0 = const()[name = string("op_8890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8890_axis_0 = const()[name = string("op_8890_axis_0"), val = int32(1)]; tensor var_8890_cast_fp16_0, tensor var_8890_cast_fp16_1 = split(axis = var_8890_axis_0, split_sizes = var_8890_split_sizes_0, x = out_99_cast_fp16)[name = string("op_8890_cast_fp16")]; tensor input_49_strides_0 = const()[name = string("input_49_strides_0"), val = tensor([1, 1])]; string input_49_pad_type_0 = const()[name = string("input_49_pad_type_0"), val = string("valid")]; tensor input_49_pad_0 = const()[name = string("input_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_49_dilations_0 = const()[name = string("input_49_dilations_0"), val = tensor([1, 1])]; int32 input_49_groups_0 = const()[name = string("input_49_groups_0"), val = int32(1)]; tensor input_49_cast_fp16 = conv(dilations = input_49_dilations_0, groups = input_49_groups_0, pad = input_49_pad_0, pad_type = input_49_pad_type_0, strides = input_49_strides_0, weight = layers_24_mlp_gate_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("input_49_cast_fp16")]; tensor var_8907_cast_fp16 = silu(x = input_49_cast_fp16)[name = string("op_8907_cast_fp16")]; tensor var_8913_strides_0 = const()[name = string("op_8913_strides_0"), val = tensor([1, 1])]; string var_8913_pad_type_0 = const()[name = string("op_8913_pad_type_0"), val = string("valid")]; tensor var_8913_pad_0 = const()[name = string("op_8913_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8913_dilations_0 = const()[name = string("op_8913_dilations_0"), val = tensor([1, 1])]; int32 var_8913_groups_0 = const()[name = string("op_8913_groups_0"), val = int32(1)]; tensor var_8913_cast_fp16 = conv(dilations = var_8913_dilations_0, groups = var_8913_groups_0, pad = var_8913_pad_0, pad_type = var_8913_pad_type_0, strides = var_8913_strides_0, weight = layers_24_mlp_up_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("op_8913_cast_fp16")]; tensor x_249_cast_fp16 = mul(x = var_8907_cast_fp16, y = var_8913_cast_fp16)[name = string("x_249_cast_fp16")]; tensor hidden_states_247_strides_0 = const()[name = string("hidden_states_247_strides_0"), val = tensor([1, 1])]; string hidden_states_247_pad_type_0 = const()[name = string("hidden_states_247_pad_type_0"), val = string("valid")]; tensor hidden_states_247_pad_0 = const()[name = string("hidden_states_247_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_247_dilations_0 = const()[name = string("hidden_states_247_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_247_groups_0 = const()[name = string("hidden_states_247_groups_0"), val = int32(1)]; tensor hidden_states_247_cast_fp16 = conv(dilations = hidden_states_247_dilations_0, groups = hidden_states_247_groups_0, pad = hidden_states_247_pad_0, pad_type = hidden_states_247_pad_type_0, strides = hidden_states_247_strides_0, weight = layers_24_mlp_down_proj_weight_cast_fp16, x = x_249_cast_fp16)[name = string("hidden_states_247_cast_fp16")]; tensor hidden_states_249_cast_fp16 = add(x = hidden_states_245_cast_fp16, y = hidden_states_247_cast_fp16)[name = string("hidden_states_249_cast_fp16")]; fp16 const_250_promoted_to_fp16 = const()[name = string("const_250_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8931_cast_fp16 = mul(x = hidden_states_249_cast_fp16, y = const_250_promoted_to_fp16)[name = string("op_8931_cast_fp16")]; int32 var_8929 = const()[name = string("op_8929"), val = int32(1)]; bool doubled_201_interleave_0 = const()[name = string("doubled_201_interleave_0"), val = bool(false)]; tensor doubled_201_cast_fp16 = concat(axis = var_8929, interleave = doubled_201_interleave_0, values = (hidden_states_249_cast_fp16, var_8931_cast_fp16))[name = string("doubled_201_cast_fp16")]; tensor out_101_axes_0 = const()[name = string("out_101_axes_0"), val = tensor([1])]; tensor out_101_gamma_0_to_fp16 = const()[name = string("out_101_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540370688)))]; fp16 var_8941_to_fp16 = const()[name = string("op_8941_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_101_cast_fp16 = layer_norm(axes = out_101_axes_0, epsilon = var_8941_to_fp16, gamma = out_101_gamma_0_to_fp16, x = doubled_201_cast_fp16)[name = string("out_101_cast_fp16")]; tensor var_8952_split_sizes_0 = const()[name = string("op_8952_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8952_axis_0 = const()[name = string("op_8952_axis_0"), val = int32(1)]; tensor var_8952_cast_fp16_0, tensor var_8952_cast_fp16_1 = split(axis = var_8952_axis_0, split_sizes = var_8952_split_sizes_0, x = out_101_cast_fp16)[name = string("op_8952_cast_fp16")]; tensor query_states_151_strides_0 = const()[name = string("query_states_151_strides_0"), val = tensor([1, 1])]; string query_states_151_pad_type_0 = const()[name = string("query_states_151_pad_type_0"), val = string("valid")]; tensor query_states_151_pad_0 = const()[name = string("query_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_151_dilations_0 = const()[name = string("query_states_151_dilations_0"), val = tensor([1, 1])]; int32 query_states_151_groups_0 = const()[name = string("query_states_151_groups_0"), val = int32(1)]; tensor query_states_151_cast_fp16 = conv(dilations = query_states_151_dilations_0, groups = query_states_151_groups_0, pad = query_states_151_pad_0, pad_type = query_states_151_pad_type_0, strides = query_states_151_strides_0, weight = layers_25_self_attn_q_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("query_states_151_cast_fp16")]; tensor key_states_251_strides_0 = const()[name = string("key_states_251_strides_0"), val = tensor([1, 1])]; string key_states_251_pad_type_0 = const()[name = string("key_states_251_pad_type_0"), val = string("valid")]; tensor key_states_251_pad_0 = const()[name = string("key_states_251_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_251_dilations_0 = const()[name = string("key_states_251_dilations_0"), val = tensor([1, 1])]; int32 key_states_251_groups_0 = const()[name = string("key_states_251_groups_0"), val = int32(1)]; tensor key_states_251_cast_fp16 = conv(dilations = key_states_251_dilations_0, groups = key_states_251_groups_0, pad = key_states_251_pad_0, pad_type = key_states_251_pad_type_0, strides = key_states_251_strides_0, weight = layers_25_self_attn_k_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("key_states_251_cast_fp16")]; tensor layers_25_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_25_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540378944)))]; tensor value_states_151_strides_0 = const()[name = string("value_states_151_strides_0"), val = tensor([1, 1])]; string value_states_151_pad_type_0 = const()[name = string("value_states_151_pad_type_0"), val = string("valid")]; tensor value_states_151_pad_0 = const()[name = string("value_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_151_dilations_0 = const()[name = string("value_states_151_dilations_0"), val = tensor([1, 1])]; int32 value_states_151_groups_0 = const()[name = string("value_states_151_groups_0"), val = int32(1)]; tensor value_states_151_cast_fp16 = conv(dilations = value_states_151_dilations_0, groups = value_states_151_groups_0, pad = value_states_151_pad_0, pad_type = value_states_151_pad_type_0, strides = value_states_151_strides_0, weight = layers_25_self_attn_v_proj_weight_to_fp16, x = var_8952_cast_fp16_0)[name = string("value_states_151_cast_fp16")]; tensor concat_300x = const()[name = string("concat_300x"), val = tensor([1, 16, 128, -1])]; tensor x_251_cast_fp16 = reshape(shape = concat_300x, x = query_states_151_cast_fp16)[name = string("x_251_cast_fp16")]; tensor concat_301x = const()[name = string("concat_301x"), val = tensor([1, 2, 128, -1])]; tensor var_9009_cast_fp16 = reshape(shape = concat_301x, x = key_states_251_cast_fp16)[name = string("op_9009_cast_fp16")]; tensor concat_302x = const()[name = string("concat_302x"), val = tensor([1, 2, 128, -1])]; tensor var_9016_cast_fp16 = reshape(shape = concat_302x, x = value_states_151_cast_fp16)[name = string("op_9016_cast_fp16")]; tensor var_9020_cast_fp16 = mul(x = x_251_cast_fp16, y = var_869_cast_fp16)[name = string("op_9020_cast_fp16")]; tensor var_9021_split_sizes_0 = const()[name = string("op_9021_split_sizes_0"), val = tensor([64, 64])]; int32 var_9021_axis_0 = const()[name = string("op_9021_axis_0"), val = int32(-2)]; tensor var_9021_cast_fp16_0, tensor var_9021_cast_fp16_1 = split(axis = var_9021_axis_0, split_sizes = var_9021_split_sizes_0, x = x_251_cast_fp16)[name = string("op_9021_cast_fp16")]; fp16 const_252_promoted_to_fp16 = const()[name = string("const_252_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9023_cast_fp16 = mul(x = var_9021_cast_fp16_1, y = const_252_promoted_to_fp16)[name = string("op_9023_cast_fp16")]; int32 var_9025 = const()[name = string("op_9025"), val = int32(-2)]; bool var_9026_interleave_0 = const()[name = string("op_9026_interleave_0"), val = bool(false)]; tensor var_9026_cast_fp16 = concat(axis = var_9025, interleave = var_9026_interleave_0, values = (var_9023_cast_fp16, var_9021_cast_fp16_0))[name = string("op_9026_cast_fp16")]; tensor var_9027_cast_fp16 = mul(x = var_9026_cast_fp16, y = var_878_cast_fp16)[name = string("op_9027_cast_fp16")]; tensor query_states_153_cast_fp16 = add(x = var_9020_cast_fp16, y = var_9027_cast_fp16)[name = string("query_states_153_cast_fp16")]; tensor var_9033_cast_fp16 = mul(x = var_9009_cast_fp16, y = var_869_cast_fp16)[name = string("op_9033_cast_fp16")]; tensor var_9034_split_sizes_0 = const()[name = string("op_9034_split_sizes_0"), val = tensor([64, 64])]; int32 var_9034_axis_0 = const()[name = string("op_9034_axis_0"), val = int32(-2)]; tensor var_9034_cast_fp16_0, tensor var_9034_cast_fp16_1 = split(axis = var_9034_axis_0, split_sizes = var_9034_split_sizes_0, x = var_9009_cast_fp16)[name = string("op_9034_cast_fp16")]; fp16 const_253_promoted_to_fp16 = const()[name = string("const_253_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9036_cast_fp16 = mul(x = var_9034_cast_fp16_1, y = const_253_promoted_to_fp16)[name = string("op_9036_cast_fp16")]; int32 var_9038 = const()[name = string("op_9038"), val = int32(-2)]; bool var_9039_interleave_0 = const()[name = string("op_9039_interleave_0"), val = bool(false)]; tensor var_9039_cast_fp16 = concat(axis = var_9038, interleave = var_9039_interleave_0, values = (var_9036_cast_fp16, var_9034_cast_fp16_0))[name = string("op_9039_cast_fp16")]; tensor var_9040_cast_fp16 = mul(x = var_9039_cast_fp16, y = var_878_cast_fp16)[name = string("op_9040_cast_fp16")]; tensor key_states_255_cast_fp16 = add(x = var_9033_cast_fp16, y = var_9040_cast_fp16)[name = string("key_states_255_cast_fp16")]; tensor expand_dims_300 = const()[name = string("expand_dims_300"), val = tensor([25])]; tensor expand_dims_301 = const()[name = string("expand_dims_301"), val = tensor([0])]; tensor expand_dims_303 = const()[name = string("expand_dims_303"), val = tensor([0])]; int32 concat_305_axis_0 = const()[name = string("concat_305_axis_0"), val = int32(0)]; bool concat_305_interleave_0 = const()[name = string("concat_305_interleave_0"), val = bool(false)]; tensor concat_305 = concat(axis = concat_305_axis_0, interleave = concat_305_interleave_0, values = (expand_dims_300, expand_dims_301, position_id, expand_dims_303))[name = string("concat_305")]; tensor expand_dims_304 = const()[name = string("expand_dims_304"), val = tensor([26])]; tensor concat_306_values1_0 = const()[name = string("concat_306_values1_0"), val = tensor([0])]; tensor concat_306_values3_0 = const()[name = string("concat_306_values3_0"), val = tensor([0])]; int32 concat_306_axis_0 = const()[name = string("concat_306_axis_0"), val = int32(0)]; bool concat_306_interleave_0 = const()[name = string("concat_306_interleave_0"), val = bool(false)]; tensor concat_306 = concat(axis = concat_306_axis_0, interleave = concat_306_interleave_0, values = (expand_dims_304, concat_306_values1_0, cache_position_end, concat_306_values3_0))[name = string("concat_306")]; tensor key_states_257_perm_0 = const()[name = string("key_states_257_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_26_stride_0 = const()[name = string("key_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_257_cast_fp16 = transpose(perm = key_states_257_perm_0, x = key_states_255_cast_fp16)[name = string("transpose_524")]; tensor key_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = key_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = key_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_26_squeeze_mask_0, stride = key_cache_internal_tensor_assign_26_stride_0, update = key_states_257_cast_fp16, x = coreml_update_state_384)[name = string("key_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_26_cast_fp16, input = key_cache)[name = string("coreml_update_state_386_write_state")]; tensor coreml_update_state_386 = read_state(input = key_cache)[name = string("coreml_update_state_386")]; tensor value_states_153_perm_0 = const()[name = string("value_states_153_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_26_stride_0 = const()[name = string("value_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_153_cast_fp16 = transpose(perm = value_states_153_perm_0, x = var_9016_cast_fp16)[name = string("transpose_523")]; tensor value_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = value_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = value_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_26_squeeze_mask_0, stride = value_cache_internal_tensor_assign_26_stride_0, update = value_states_153_cast_fp16, x = coreml_update_state_385)[name = string("value_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_26_cast_fp16, input = value_cache)[name = string("coreml_update_state_387_write_state")]; tensor coreml_update_state_387 = read_state(input = value_cache)[name = string("coreml_update_state_387")]; tensor var_9110_begin_0 = const()[name = string("op_9110_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9110_end_0 = const()[name = string("op_9110_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9110_end_mask_0 = const()[name = string("op_9110_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9110_cast_fp16 = slice_by_index(begin = var_9110_begin_0, end = var_9110_end_0, end_mask = var_9110_end_mask_0, x = coreml_update_state_386)[name = string("op_9110_cast_fp16")]; tensor tile_50 = const()[name = string("tile_50"), val = tensor([1, 1])]; int32 var_9113_axis_0 = const()[name = string("op_9113_axis_0"), val = int32(1)]; tensor var_9113_cast_fp16_0, tensor var_9113_cast_fp16_1 = split(axis = var_9113_axis_0, split_sizes = tile_50, x = var_9110_cast_fp16)[name = string("op_9113_cast_fp16")]; tensor var_9120_begin_0 = const()[name = string("op_9120_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9120_end_0 = const()[name = string("op_9120_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9120_end_mask_0 = const()[name = string("op_9120_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9120_cast_fp16 = slice_by_index(begin = var_9120_begin_0, end = var_9120_end_0, end_mask = var_9120_end_mask_0, x = coreml_update_state_387)[name = string("op_9120_cast_fp16")]; tensor tile_51 = const()[name = string("tile_51"), val = tensor([1, 1])]; int32 var_9123_axis_0 = const()[name = string("op_9123_axis_0"), val = int32(1)]; tensor var_9123_cast_fp16_0, tensor var_9123_cast_fp16_1 = split(axis = var_9123_axis_0, split_sizes = tile_51, x = var_9120_cast_fp16)[name = string("op_9123_cast_fp16")]; tensor var_9126_split_sizes_0 = const()[name = string("op_9126_split_sizes_0"), val = tensor([8, 8])]; int32 var_9126_axis_0 = const()[name = string("op_9126_axis_0"), val = int32(1)]; tensor var_9126_0, tensor var_9126_1 = split(axis = var_9126_axis_0, split_sizes = var_9126_split_sizes_0, x = query_states_153_cast_fp16)[name = string("op_9126")]; bool attn_weights_401_transpose_x_0 = const()[name = string("attn_weights_401_transpose_x_0"), val = bool(false)]; bool attn_weights_401_transpose_y_0 = const()[name = string("attn_weights_401_transpose_y_0"), val = bool(false)]; tensor attn_weights_401_cast_fp16 = matmul(transpose_x = attn_weights_401_transpose_x_0, transpose_y = attn_weights_401_transpose_y_0, x = var_9113_cast_fp16_0, y = var_9126_0)[name = string("attn_weights_401_cast_fp16")]; fp16 var_9129_to_fp16 = const()[name = string("op_9129_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_403_cast_fp16 = mul(x = attn_weights_401_cast_fp16, y = var_9129_to_fp16)[name = string("attn_weights_403_cast_fp16")]; tensor attn_weights_405_cast_fp16 = add(x = attn_weights_403_cast_fp16, y = attn_mask_1)[name = string("attn_weights_405_cast_fp16")]; int32 var_9133 = const()[name = string("op_9133"), val = int32(-2)]; tensor attn_weights_407_cast_fp16 = softmax(axis = var_9133, x = attn_weights_405_cast_fp16)[name = string("attn_weights_407_cast_fp16")]; bool var_9139_transpose_x_1 = const()[name = string("op_9139_transpose_x_1"), val = bool(true)]; bool var_9139_transpose_y_1 = const()[name = string("op_9139_transpose_y_1"), val = bool(false)]; tensor var_9139_cast_fp16 = matmul(transpose_x = var_9139_transpose_x_1, transpose_y = var_9139_transpose_y_1, x = attn_weights_407_cast_fp16, y = var_9123_cast_fp16_0)[name = string("op_9139_cast_fp16")]; bool attn_weights_409_transpose_x_0 = const()[name = string("attn_weights_409_transpose_x_0"), val = bool(false)]; bool attn_weights_409_transpose_y_0 = const()[name = string("attn_weights_409_transpose_y_0"), val = bool(false)]; tensor attn_weights_409_cast_fp16 = matmul(transpose_x = attn_weights_409_transpose_x_0, transpose_y = attn_weights_409_transpose_y_0, x = var_9113_cast_fp16_1, y = var_9126_1)[name = string("attn_weights_409_cast_fp16")]; fp16 var_9141_to_fp16 = const()[name = string("op_9141_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_411_cast_fp16 = mul(x = attn_weights_409_cast_fp16, y = var_9141_to_fp16)[name = string("attn_weights_411_cast_fp16")]; tensor attn_weights_413_cast_fp16 = add(x = attn_weights_411_cast_fp16, y = attn_mask_1)[name = string("attn_weights_413_cast_fp16")]; int32 var_9145 = const()[name = string("op_9145"), val = int32(-2)]; tensor attn_weights_415_cast_fp16 = softmax(axis = var_9145, x = attn_weights_413_cast_fp16)[name = string("attn_weights_415_cast_fp16")]; bool attn_output_201_transpose_x_1 = const()[name = string("attn_output_201_transpose_x_1"), val = bool(true)]; bool attn_output_201_transpose_y_1 = const()[name = string("attn_output_201_transpose_y_1"), val = bool(false)]; tensor attn_output_201_cast_fp16 = matmul(transpose_x = attn_output_201_transpose_x_1, transpose_y = attn_output_201_transpose_y_1, x = attn_weights_415_cast_fp16, y = var_9123_cast_fp16_1)[name = string("attn_output_201_cast_fp16")]; int32 var_9153 = const()[name = string("op_9153"), val = int32(1)]; bool attn_output_203_interleave_0 = const()[name = string("attn_output_203_interleave_0"), val = bool(false)]; tensor attn_output_203_cast_fp16 = concat(axis = var_9153, interleave = attn_output_203_interleave_0, values = (var_9139_cast_fp16, attn_output_201_cast_fp16))[name = string("attn_output_203_cast_fp16")]; tensor var_9157_perm_0 = const()[name = string("op_9157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_311x = const()[name = string("concat_311x"), val = tensor([1, 2048, 1, -1])]; tensor var_9157_cast_fp16 = transpose(perm = var_9157_perm_0, x = attn_output_203_cast_fp16)[name = string("transpose_522")]; tensor attn_output_207_cast_fp16 = reshape(shape = concat_311x, x = var_9157_cast_fp16)[name = string("attn_output_207_cast_fp16")]; tensor hidden_states_253_strides_0 = const()[name = string("hidden_states_253_strides_0"), val = tensor([1, 1])]; string hidden_states_253_pad_type_0 = const()[name = string("hidden_states_253_pad_type_0"), val = string("valid")]; tensor hidden_states_253_pad_0 = const()[name = string("hidden_states_253_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_253_dilations_0 = const()[name = string("hidden_states_253_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_253_groups_0 = const()[name = string("hidden_states_253_groups_0"), val = int32(1)]; tensor hidden_states_253_cast_fp16 = conv(dilations = hidden_states_253_dilations_0, groups = hidden_states_253_groups_0, pad = hidden_states_253_pad_0, pad_type = hidden_states_253_pad_type_0, strides = hidden_states_253_strides_0, weight = layers_25_self_attn_o_proj_weight_cast_fp16, x = attn_output_207_cast_fp16)[name = string("hidden_states_253_cast_fp16")]; tensor hidden_states_255_cast_fp16 = add(x = hidden_states_249_cast_fp16, y = hidden_states_253_cast_fp16)[name = string("hidden_states_255_cast_fp16")]; fp16 const_258_promoted_to_fp16 = const()[name = string("const_258_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9190_cast_fp16 = mul(x = hidden_states_255_cast_fp16, y = const_258_promoted_to_fp16)[name = string("op_9190_cast_fp16")]; int32 var_9188 = const()[name = string("op_9188"), val = int32(1)]; bool doubled_205_interleave_0 = const()[name = string("doubled_205_interleave_0"), val = bool(false)]; tensor doubled_205_cast_fp16 = concat(axis = var_9188, interleave = doubled_205_interleave_0, values = (hidden_states_255_cast_fp16, var_9190_cast_fp16))[name = string("doubled_205_cast_fp16")]; tensor out_103_axes_0 = const()[name = string("out_103_axes_0"), val = tensor([1])]; tensor out_103_gamma_0_to_fp16 = const()[name = string("out_103_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541427584)))]; fp16 var_9200_to_fp16 = const()[name = string("op_9200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_103_cast_fp16 = layer_norm(axes = out_103_axes_0, epsilon = var_9200_to_fp16, gamma = out_103_gamma_0_to_fp16, x = doubled_205_cast_fp16)[name = string("out_103_cast_fp16")]; tensor var_9211_split_sizes_0 = const()[name = string("op_9211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9211_axis_0 = const()[name = string("op_9211_axis_0"), val = int32(1)]; tensor var_9211_cast_fp16_0, tensor var_9211_cast_fp16_1 = split(axis = var_9211_axis_0, split_sizes = var_9211_split_sizes_0, x = out_103_cast_fp16)[name = string("op_9211_cast_fp16")]; tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; tensor input_51_cast_fp16 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = layers_25_mlp_gate_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("input_51_cast_fp16")]; tensor var_9228_cast_fp16 = silu(x = input_51_cast_fp16)[name = string("op_9228_cast_fp16")]; tensor var_9234_strides_0 = const()[name = string("op_9234_strides_0"), val = tensor([1, 1])]; string var_9234_pad_type_0 = const()[name = string("op_9234_pad_type_0"), val = string("valid")]; tensor var_9234_pad_0 = const()[name = string("op_9234_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9234_dilations_0 = const()[name = string("op_9234_dilations_0"), val = tensor([1, 1])]; int32 var_9234_groups_0 = const()[name = string("op_9234_groups_0"), val = int32(1)]; tensor var_9234_cast_fp16 = conv(dilations = var_9234_dilations_0, groups = var_9234_groups_0, pad = var_9234_pad_0, pad_type = var_9234_pad_type_0, strides = var_9234_strides_0, weight = layers_25_mlp_up_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("op_9234_cast_fp16")]; tensor x_259_cast_fp16 = mul(x = var_9228_cast_fp16, y = var_9234_cast_fp16)[name = string("x_259_cast_fp16")]; tensor layers_25_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_25_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541435840)))]; tensor hidden_states_257_strides_0 = const()[name = string("hidden_states_257_strides_0"), val = tensor([1, 1])]; string hidden_states_257_pad_type_0 = const()[name = string("hidden_states_257_pad_type_0"), val = string("valid")]; tensor hidden_states_257_pad_0 = const()[name = string("hidden_states_257_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_257_dilations_0 = const()[name = string("hidden_states_257_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_257_groups_0 = const()[name = string("hidden_states_257_groups_0"), val = int32(1)]; tensor hidden_states_257_cast_fp16 = conv(dilations = hidden_states_257_dilations_0, groups = hidden_states_257_groups_0, pad = hidden_states_257_pad_0, pad_type = hidden_states_257_pad_type_0, strides = hidden_states_257_strides_0, weight = layers_25_mlp_down_proj_weight_to_fp16, x = x_259_cast_fp16)[name = string("hidden_states_257_cast_fp16")]; tensor hidden_states_259_cast_fp16 = add(x = hidden_states_255_cast_fp16, y = hidden_states_257_cast_fp16)[name = string("hidden_states_259_cast_fp16")]; fp16 const_260_promoted_to_fp16 = const()[name = string("const_260_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9252_cast_fp16 = mul(x = hidden_states_259_cast_fp16, y = const_260_promoted_to_fp16)[name = string("op_9252_cast_fp16")]; int32 var_9250 = const()[name = string("op_9250"), val = int32(1)]; bool doubled_209_interleave_0 = const()[name = string("doubled_209_interleave_0"), val = bool(false)]; tensor doubled_209_cast_fp16 = concat(axis = var_9250, interleave = doubled_209_interleave_0, values = (hidden_states_259_cast_fp16, var_9252_cast_fp16))[name = string("doubled_209_cast_fp16")]; tensor out_105_axes_0 = const()[name = string("out_105_axes_0"), val = tensor([1])]; tensor out_105_gamma_0_to_fp16 = const()[name = string("out_105_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566601728)))]; fp16 var_9262_to_fp16 = const()[name = string("op_9262_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_105_cast_fp16 = layer_norm(axes = out_105_axes_0, epsilon = var_9262_to_fp16, gamma = out_105_gamma_0_to_fp16, x = doubled_209_cast_fp16)[name = string("out_105_cast_fp16")]; tensor var_9273_split_sizes_0 = const()[name = string("op_9273_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9273_axis_0 = const()[name = string("op_9273_axis_0"), val = int32(1)]; tensor var_9273_cast_fp16_0, tensor var_9273_cast_fp16_1 = split(axis = var_9273_axis_0, split_sizes = var_9273_split_sizes_0, x = out_105_cast_fp16)[name = string("op_9273_cast_fp16")]; tensor query_states_157_strides_0 = const()[name = string("query_states_157_strides_0"), val = tensor([1, 1])]; string query_states_157_pad_type_0 = const()[name = string("query_states_157_pad_type_0"), val = string("valid")]; tensor query_states_157_pad_0 = const()[name = string("query_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_157_dilations_0 = const()[name = string("query_states_157_dilations_0"), val = tensor([1, 1])]; int32 query_states_157_groups_0 = const()[name = string("query_states_157_groups_0"), val = int32(1)]; tensor query_states_157_cast_fp16 = conv(dilations = query_states_157_dilations_0, groups = query_states_157_groups_0, pad = query_states_157_pad_0, pad_type = query_states_157_pad_type_0, strides = query_states_157_strides_0, weight = layers_26_self_attn_q_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("query_states_157_cast_fp16")]; tensor key_states_261_strides_0 = const()[name = string("key_states_261_strides_0"), val = tensor([1, 1])]; string key_states_261_pad_type_0 = const()[name = string("key_states_261_pad_type_0"), val = string("valid")]; tensor key_states_261_pad_0 = const()[name = string("key_states_261_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_261_dilations_0 = const()[name = string("key_states_261_dilations_0"), val = tensor([1, 1])]; int32 key_states_261_groups_0 = const()[name = string("key_states_261_groups_0"), val = int32(1)]; tensor key_states_261_cast_fp16 = conv(dilations = key_states_261_dilations_0, groups = key_states_261_groups_0, pad = key_states_261_pad_0, pad_type = key_states_261_pad_type_0, strides = key_states_261_strides_0, weight = layers_26_self_attn_k_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("key_states_261_cast_fp16")]; tensor layers_26_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_26_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566609984)))]; tensor value_states_157_strides_0 = const()[name = string("value_states_157_strides_0"), val = tensor([1, 1])]; string value_states_157_pad_type_0 = const()[name = string("value_states_157_pad_type_0"), val = string("valid")]; tensor value_states_157_pad_0 = const()[name = string("value_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_157_dilations_0 = const()[name = string("value_states_157_dilations_0"), val = tensor([1, 1])]; int32 value_states_157_groups_0 = const()[name = string("value_states_157_groups_0"), val = int32(1)]; tensor value_states_157_cast_fp16 = conv(dilations = value_states_157_dilations_0, groups = value_states_157_groups_0, pad = value_states_157_pad_0, pad_type = value_states_157_pad_type_0, strides = value_states_157_strides_0, weight = layers_26_self_attn_v_proj_weight_to_fp16, x = var_9273_cast_fp16_0)[name = string("value_states_157_cast_fp16")]; tensor concat_312x = const()[name = string("concat_312x"), val = tensor([1, 16, 128, -1])]; tensor x_261_cast_fp16 = reshape(shape = concat_312x, x = query_states_157_cast_fp16)[name = string("x_261_cast_fp16")]; tensor concat_313x = const()[name = string("concat_313x"), val = tensor([1, 2, 128, -1])]; tensor var_9330_cast_fp16 = reshape(shape = concat_313x, x = key_states_261_cast_fp16)[name = string("op_9330_cast_fp16")]; tensor concat_314x = const()[name = string("concat_314x"), val = tensor([1, 2, 128, -1])]; tensor var_9337_cast_fp16 = reshape(shape = concat_314x, x = value_states_157_cast_fp16)[name = string("op_9337_cast_fp16")]; tensor var_9341_cast_fp16 = mul(x = x_261_cast_fp16, y = var_869_cast_fp16)[name = string("op_9341_cast_fp16")]; tensor var_9342_split_sizes_0 = const()[name = string("op_9342_split_sizes_0"), val = tensor([64, 64])]; int32 var_9342_axis_0 = const()[name = string("op_9342_axis_0"), val = int32(-2)]; tensor var_9342_cast_fp16_0, tensor var_9342_cast_fp16_1 = split(axis = var_9342_axis_0, split_sizes = var_9342_split_sizes_0, x = x_261_cast_fp16)[name = string("op_9342_cast_fp16")]; fp16 const_262_promoted_to_fp16 = const()[name = string("const_262_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9344_cast_fp16 = mul(x = var_9342_cast_fp16_1, y = const_262_promoted_to_fp16)[name = string("op_9344_cast_fp16")]; int32 var_9346 = const()[name = string("op_9346"), val = int32(-2)]; bool var_9347_interleave_0 = const()[name = string("op_9347_interleave_0"), val = bool(false)]; tensor var_9347_cast_fp16 = concat(axis = var_9346, interleave = var_9347_interleave_0, values = (var_9344_cast_fp16, var_9342_cast_fp16_0))[name = string("op_9347_cast_fp16")]; tensor var_9348_cast_fp16 = mul(x = var_9347_cast_fp16, y = var_878_cast_fp16)[name = string("op_9348_cast_fp16")]; tensor query_states_159_cast_fp16 = add(x = var_9341_cast_fp16, y = var_9348_cast_fp16)[name = string("query_states_159_cast_fp16")]; tensor var_9354_cast_fp16 = mul(x = var_9330_cast_fp16, y = var_869_cast_fp16)[name = string("op_9354_cast_fp16")]; tensor var_9355_split_sizes_0 = const()[name = string("op_9355_split_sizes_0"), val = tensor([64, 64])]; int32 var_9355_axis_0 = const()[name = string("op_9355_axis_0"), val = int32(-2)]; tensor var_9355_cast_fp16_0, tensor var_9355_cast_fp16_1 = split(axis = var_9355_axis_0, split_sizes = var_9355_split_sizes_0, x = var_9330_cast_fp16)[name = string("op_9355_cast_fp16")]; fp16 const_263_promoted_to_fp16 = const()[name = string("const_263_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9357_cast_fp16 = mul(x = var_9355_cast_fp16_1, y = const_263_promoted_to_fp16)[name = string("op_9357_cast_fp16")]; int32 var_9359 = const()[name = string("op_9359"), val = int32(-2)]; bool var_9360_interleave_0 = const()[name = string("op_9360_interleave_0"), val = bool(false)]; tensor var_9360_cast_fp16 = concat(axis = var_9359, interleave = var_9360_interleave_0, values = (var_9357_cast_fp16, var_9355_cast_fp16_0))[name = string("op_9360_cast_fp16")]; tensor var_9361_cast_fp16 = mul(x = var_9360_cast_fp16, y = var_878_cast_fp16)[name = string("op_9361_cast_fp16")]; tensor key_states_265_cast_fp16 = add(x = var_9354_cast_fp16, y = var_9361_cast_fp16)[name = string("key_states_265_cast_fp16")]; tensor expand_dims_312 = const()[name = string("expand_dims_312"), val = tensor([26])]; tensor expand_dims_313 = const()[name = string("expand_dims_313"), val = tensor([0])]; tensor expand_dims_315 = const()[name = string("expand_dims_315"), val = tensor([0])]; int32 concat_317_axis_0 = const()[name = string("concat_317_axis_0"), val = int32(0)]; bool concat_317_interleave_0 = const()[name = string("concat_317_interleave_0"), val = bool(false)]; tensor concat_317 = concat(axis = concat_317_axis_0, interleave = concat_317_interleave_0, values = (expand_dims_312, expand_dims_313, position_id, expand_dims_315))[name = string("concat_317")]; tensor expand_dims_316 = const()[name = string("expand_dims_316"), val = tensor([27])]; tensor concat_318_values1_0 = const()[name = string("concat_318_values1_0"), val = tensor([0])]; tensor concat_318_values3_0 = const()[name = string("concat_318_values3_0"), val = tensor([0])]; int32 concat_318_axis_0 = const()[name = string("concat_318_axis_0"), val = int32(0)]; bool concat_318_interleave_0 = const()[name = string("concat_318_interleave_0"), val = bool(false)]; tensor concat_318 = concat(axis = concat_318_axis_0, interleave = concat_318_interleave_0, values = (expand_dims_316, concat_318_values1_0, cache_position_end, concat_318_values3_0))[name = string("concat_318")]; tensor key_states_267_perm_0 = const()[name = string("key_states_267_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_27_stride_0 = const()[name = string("key_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_267_cast_fp16 = transpose(perm = key_states_267_perm_0, x = key_states_265_cast_fp16)[name = string("transpose_521")]; tensor key_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = key_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = key_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_27_squeeze_mask_0, stride = key_cache_internal_tensor_assign_27_stride_0, update = key_states_267_cast_fp16, x = coreml_update_state_386)[name = string("key_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_27_cast_fp16, input = key_cache)[name = string("coreml_update_state_388_write_state")]; tensor coreml_update_state_388 = read_state(input = key_cache)[name = string("coreml_update_state_388")]; tensor value_states_159_perm_0 = const()[name = string("value_states_159_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_27_stride_0 = const()[name = string("value_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_159_cast_fp16 = transpose(perm = value_states_159_perm_0, x = var_9337_cast_fp16)[name = string("transpose_520")]; tensor value_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = value_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = value_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_27_squeeze_mask_0, stride = value_cache_internal_tensor_assign_27_stride_0, update = value_states_159_cast_fp16, x = coreml_update_state_387)[name = string("value_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_27_cast_fp16, input = value_cache)[name = string("coreml_update_state_389_write_state")]; tensor coreml_update_state_389 = read_state(input = value_cache)[name = string("coreml_update_state_389")]; tensor var_9431_begin_0 = const()[name = string("op_9431_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9431_end_0 = const()[name = string("op_9431_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9431_end_mask_0 = const()[name = string("op_9431_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9431_cast_fp16 = slice_by_index(begin = var_9431_begin_0, end = var_9431_end_0, end_mask = var_9431_end_mask_0, x = coreml_update_state_388)[name = string("op_9431_cast_fp16")]; tensor tile_52 = const()[name = string("tile_52"), val = tensor([1, 1])]; int32 var_9434_axis_0 = const()[name = string("op_9434_axis_0"), val = int32(1)]; tensor var_9434_cast_fp16_0, tensor var_9434_cast_fp16_1 = split(axis = var_9434_axis_0, split_sizes = tile_52, x = var_9431_cast_fp16)[name = string("op_9434_cast_fp16")]; tensor var_9441_begin_0 = const()[name = string("op_9441_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9441_end_0 = const()[name = string("op_9441_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9441_end_mask_0 = const()[name = string("op_9441_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9441_cast_fp16 = slice_by_index(begin = var_9441_begin_0, end = var_9441_end_0, end_mask = var_9441_end_mask_0, x = coreml_update_state_389)[name = string("op_9441_cast_fp16")]; tensor tile_53 = const()[name = string("tile_53"), val = tensor([1, 1])]; int32 var_9444_axis_0 = const()[name = string("op_9444_axis_0"), val = int32(1)]; tensor var_9444_cast_fp16_0, tensor var_9444_cast_fp16_1 = split(axis = var_9444_axis_0, split_sizes = tile_53, x = var_9441_cast_fp16)[name = string("op_9444_cast_fp16")]; tensor var_9447_split_sizes_0 = const()[name = string("op_9447_split_sizes_0"), val = tensor([8, 8])]; int32 var_9447_axis_0 = const()[name = string("op_9447_axis_0"), val = int32(1)]; tensor var_9447_0, tensor var_9447_1 = split(axis = var_9447_axis_0, split_sizes = var_9447_split_sizes_0, x = query_states_159_cast_fp16)[name = string("op_9447")]; bool attn_weights_417_transpose_x_0 = const()[name = string("attn_weights_417_transpose_x_0"), val = bool(false)]; bool attn_weights_417_transpose_y_0 = const()[name = string("attn_weights_417_transpose_y_0"), val = bool(false)]; tensor attn_weights_417_cast_fp16 = matmul(transpose_x = attn_weights_417_transpose_x_0, transpose_y = attn_weights_417_transpose_y_0, x = var_9434_cast_fp16_0, y = var_9447_0)[name = string("attn_weights_417_cast_fp16")]; fp16 var_9450_to_fp16 = const()[name = string("op_9450_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_419_cast_fp16 = mul(x = attn_weights_417_cast_fp16, y = var_9450_to_fp16)[name = string("attn_weights_419_cast_fp16")]; tensor attn_weights_421_cast_fp16 = add(x = attn_weights_419_cast_fp16, y = attn_mask_1)[name = string("attn_weights_421_cast_fp16")]; int32 var_9454 = const()[name = string("op_9454"), val = int32(-2)]; tensor attn_weights_423_cast_fp16 = softmax(axis = var_9454, x = attn_weights_421_cast_fp16)[name = string("attn_weights_423_cast_fp16")]; bool var_9460_transpose_x_1 = const()[name = string("op_9460_transpose_x_1"), val = bool(true)]; bool var_9460_transpose_y_1 = const()[name = string("op_9460_transpose_y_1"), val = bool(false)]; tensor var_9460_cast_fp16 = matmul(transpose_x = var_9460_transpose_x_1, transpose_y = var_9460_transpose_y_1, x = attn_weights_423_cast_fp16, y = var_9444_cast_fp16_0)[name = string("op_9460_cast_fp16")]; bool attn_weights_425_transpose_x_0 = const()[name = string("attn_weights_425_transpose_x_0"), val = bool(false)]; bool attn_weights_425_transpose_y_0 = const()[name = string("attn_weights_425_transpose_y_0"), val = bool(false)]; tensor attn_weights_425_cast_fp16 = matmul(transpose_x = attn_weights_425_transpose_x_0, transpose_y = attn_weights_425_transpose_y_0, x = var_9434_cast_fp16_1, y = var_9447_1)[name = string("attn_weights_425_cast_fp16")]; fp16 var_9462_to_fp16 = const()[name = string("op_9462_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_427_cast_fp16 = mul(x = attn_weights_425_cast_fp16, y = var_9462_to_fp16)[name = string("attn_weights_427_cast_fp16")]; tensor attn_weights_429_cast_fp16 = add(x = attn_weights_427_cast_fp16, y = attn_mask_1)[name = string("attn_weights_429_cast_fp16")]; int32 var_9466 = const()[name = string("op_9466"), val = int32(-2)]; tensor attn_weights_431_cast_fp16 = softmax(axis = var_9466, x = attn_weights_429_cast_fp16)[name = string("attn_weights_431_cast_fp16")]; bool attn_output_209_transpose_x_1 = const()[name = string("attn_output_209_transpose_x_1"), val = bool(true)]; bool attn_output_209_transpose_y_1 = const()[name = string("attn_output_209_transpose_y_1"), val = bool(false)]; tensor attn_output_209_cast_fp16 = matmul(transpose_x = attn_output_209_transpose_x_1, transpose_y = attn_output_209_transpose_y_1, x = attn_weights_431_cast_fp16, y = var_9444_cast_fp16_1)[name = string("attn_output_209_cast_fp16")]; int32 var_9474 = const()[name = string("op_9474"), val = int32(1)]; bool attn_output_211_interleave_0 = const()[name = string("attn_output_211_interleave_0"), val = bool(false)]; tensor attn_output_211_cast_fp16 = concat(axis = var_9474, interleave = attn_output_211_interleave_0, values = (var_9460_cast_fp16, attn_output_209_cast_fp16))[name = string("attn_output_211_cast_fp16")]; tensor var_9478_perm_0 = const()[name = string("op_9478_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_323x = const()[name = string("concat_323x"), val = tensor([1, 2048, 1, -1])]; tensor var_9478_cast_fp16 = transpose(perm = var_9478_perm_0, x = attn_output_211_cast_fp16)[name = string("transpose_519")]; tensor attn_output_215_cast_fp16 = reshape(shape = concat_323x, x = var_9478_cast_fp16)[name = string("attn_output_215_cast_fp16")]; tensor hidden_states_263_strides_0 = const()[name = string("hidden_states_263_strides_0"), val = tensor([1, 1])]; string hidden_states_263_pad_type_0 = const()[name = string("hidden_states_263_pad_type_0"), val = string("valid")]; tensor hidden_states_263_pad_0 = const()[name = string("hidden_states_263_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_263_dilations_0 = const()[name = string("hidden_states_263_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_263_groups_0 = const()[name = string("hidden_states_263_groups_0"), val = int32(1)]; tensor hidden_states_263_cast_fp16 = conv(dilations = hidden_states_263_dilations_0, groups = hidden_states_263_groups_0, pad = hidden_states_263_pad_0, pad_type = hidden_states_263_pad_type_0, strides = hidden_states_263_strides_0, weight = layers_26_self_attn_o_proj_weight_cast_fp16, x = attn_output_215_cast_fp16)[name = string("hidden_states_263_cast_fp16")]; tensor hidden_states_265_cast_fp16 = add(x = hidden_states_259_cast_fp16, y = hidden_states_263_cast_fp16)[name = string("hidden_states_265_cast_fp16")]; fp16 const_268_promoted_to_fp16 = const()[name = string("const_268_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9511_cast_fp16 = mul(x = hidden_states_265_cast_fp16, y = const_268_promoted_to_fp16)[name = string("op_9511_cast_fp16")]; int32 var_9509 = const()[name = string("op_9509"), val = int32(1)]; bool doubled_213_interleave_0 = const()[name = string("doubled_213_interleave_0"), val = bool(false)]; tensor doubled_213_cast_fp16 = concat(axis = var_9509, interleave = doubled_213_interleave_0, values = (hidden_states_265_cast_fp16, var_9511_cast_fp16))[name = string("doubled_213_cast_fp16")]; tensor out_107_axes_0 = const()[name = string("out_107_axes_0"), val = tensor([1])]; tensor out_107_gamma_0_to_fp16 = const()[name = string("out_107_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567658624)))]; fp16 var_9521_to_fp16 = const()[name = string("op_9521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_107_cast_fp16 = layer_norm(axes = out_107_axes_0, epsilon = var_9521_to_fp16, gamma = out_107_gamma_0_to_fp16, x = doubled_213_cast_fp16)[name = string("out_107_cast_fp16")]; tensor var_9532_split_sizes_0 = const()[name = string("op_9532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9532_axis_0 = const()[name = string("op_9532_axis_0"), val = int32(1)]; tensor var_9532_cast_fp16_0, tensor var_9532_cast_fp16_1 = split(axis = var_9532_axis_0, split_sizes = var_9532_split_sizes_0, x = out_107_cast_fp16)[name = string("op_9532_cast_fp16")]; tensor input_53_strides_0 = const()[name = string("input_53_strides_0"), val = tensor([1, 1])]; string input_53_pad_type_0 = const()[name = string("input_53_pad_type_0"), val = string("valid")]; tensor input_53_pad_0 = const()[name = string("input_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_53_dilations_0 = const()[name = string("input_53_dilations_0"), val = tensor([1, 1])]; int32 input_53_groups_0 = const()[name = string("input_53_groups_0"), val = int32(1)]; tensor input_53_cast_fp16 = conv(dilations = input_53_dilations_0, groups = input_53_groups_0, pad = input_53_pad_0, pad_type = input_53_pad_type_0, strides = input_53_strides_0, weight = layers_26_mlp_gate_proj_weight_cast_fp16, x = var_9532_cast_fp16_0)[name = string("input_53_cast_fp16")]; tensor var_9549_cast_fp16 = silu(x = input_53_cast_fp16)[name = string("op_9549_cast_fp16")]; tensor layers_26_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567666880)))]; tensor var_9555_strides_0 = const()[name = string("op_9555_strides_0"), val = tensor([1, 1])]; string var_9555_pad_type_0 = const()[name = string("op_9555_pad_type_0"), val = string("valid")]; tensor var_9555_pad_0 = const()[name = string("op_9555_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9555_dilations_0 = const()[name = string("op_9555_dilations_0"), val = tensor([1, 1])]; int32 var_9555_groups_0 = const()[name = string("op_9555_groups_0"), val = int32(1)]; tensor var_9555_cast_fp16 = conv(dilations = var_9555_dilations_0, groups = var_9555_groups_0, pad = var_9555_pad_0, pad_type = var_9555_pad_type_0, strides = var_9555_strides_0, weight = layers_26_mlp_up_proj_weight_to_fp16, x = var_9532_cast_fp16_0)[name = string("op_9555_cast_fp16")]; tensor x_269_cast_fp16 = mul(x = var_9549_cast_fp16, y = var_9555_cast_fp16)[name = string("x_269_cast_fp16")]; tensor layers_26_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1592832768)))]; tensor hidden_states_267_strides_0 = const()[name = string("hidden_states_267_strides_0"), val = tensor([1, 1])]; string hidden_states_267_pad_type_0 = const()[name = string("hidden_states_267_pad_type_0"), val = string("valid")]; tensor hidden_states_267_pad_0 = const()[name = string("hidden_states_267_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_267_dilations_0 = const()[name = string("hidden_states_267_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_267_groups_0 = const()[name = string("hidden_states_267_groups_0"), val = int32(1)]; tensor hidden_states_267_cast_fp16 = conv(dilations = hidden_states_267_dilations_0, groups = hidden_states_267_groups_0, pad = hidden_states_267_pad_0, pad_type = hidden_states_267_pad_type_0, strides = hidden_states_267_strides_0, weight = layers_26_mlp_down_proj_weight_to_fp16, x = x_269_cast_fp16)[name = string("hidden_states_267_cast_fp16")]; tensor hidden_states_269_cast_fp16 = add(x = hidden_states_265_cast_fp16, y = hidden_states_267_cast_fp16)[name = string("hidden_states_269_cast_fp16")]; fp16 const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9573_cast_fp16 = mul(x = hidden_states_269_cast_fp16, y = const_270_promoted_to_fp16)[name = string("op_9573_cast_fp16")]; int32 var_9571 = const()[name = string("op_9571"), val = int32(1)]; bool doubled_217_interleave_0 = const()[name = string("doubled_217_interleave_0"), val = bool(false)]; tensor doubled_217_cast_fp16 = concat(axis = var_9571, interleave = doubled_217_interleave_0, values = (hidden_states_269_cast_fp16, var_9573_cast_fp16))[name = string("doubled_217_cast_fp16")]; tensor out_109_axes_0 = const()[name = string("out_109_axes_0"), val = tensor([1])]; tensor out_109_gamma_0_to_fp16 = const()[name = string("out_109_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1617998656)))]; fp16 var_9583_to_fp16 = const()[name = string("op_9583_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_109_cast_fp16 = layer_norm(axes = out_109_axes_0, epsilon = var_9583_to_fp16, gamma = out_109_gamma_0_to_fp16, x = doubled_217_cast_fp16)[name = string("out_109_cast_fp16")]; tensor var_9594_split_sizes_0 = const()[name = string("op_9594_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9594_axis_0 = const()[name = string("op_9594_axis_0"), val = int32(1)]; tensor var_9594_cast_fp16_0, tensor var_9594_cast_fp16_1 = split(axis = var_9594_axis_0, split_sizes = var_9594_split_sizes_0, x = out_109_cast_fp16)[name = string("op_9594_cast_fp16")]; tensor layers_27_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1618006912)))]; tensor query_states_163_strides_0 = const()[name = string("query_states_163_strides_0"), val = tensor([1, 1])]; string query_states_163_pad_type_0 = const()[name = string("query_states_163_pad_type_0"), val = string("valid")]; tensor query_states_163_pad_0 = const()[name = string("query_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_163_dilations_0 = const()[name = string("query_states_163_dilations_0"), val = tensor([1, 1])]; int32 query_states_163_groups_0 = const()[name = string("query_states_163_groups_0"), val = int32(1)]; tensor query_states_163_cast_fp16 = conv(dilations = query_states_163_dilations_0, groups = query_states_163_groups_0, pad = query_states_163_pad_0, pad_type = query_states_163_pad_type_0, strides = query_states_163_strides_0, weight = layers_27_self_attn_q_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("query_states_163_cast_fp16")]; tensor layers_27_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1626395584)))]; tensor key_states_271_strides_0 = const()[name = string("key_states_271_strides_0"), val = tensor([1, 1])]; string key_states_271_pad_type_0 = const()[name = string("key_states_271_pad_type_0"), val = string("valid")]; tensor key_states_271_pad_0 = const()[name = string("key_states_271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_271_dilations_0 = const()[name = string("key_states_271_dilations_0"), val = tensor([1, 1])]; int32 key_states_271_groups_0 = const()[name = string("key_states_271_groups_0"), val = int32(1)]; tensor key_states_271_cast_fp16 = conv(dilations = key_states_271_dilations_0, groups = key_states_271_groups_0, pad = key_states_271_pad_0, pad_type = key_states_271_pad_type_0, strides = key_states_271_strides_0, weight = layers_27_self_attn_k_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("key_states_271_cast_fp16")]; tensor layers_27_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1627444224)))]; tensor value_states_163_strides_0 = const()[name = string("value_states_163_strides_0"), val = tensor([1, 1])]; string value_states_163_pad_type_0 = const()[name = string("value_states_163_pad_type_0"), val = string("valid")]; tensor value_states_163_pad_0 = const()[name = string("value_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_163_dilations_0 = const()[name = string("value_states_163_dilations_0"), val = tensor([1, 1])]; int32 value_states_163_groups_0 = const()[name = string("value_states_163_groups_0"), val = int32(1)]; tensor value_states_163_cast_fp16 = conv(dilations = value_states_163_dilations_0, groups = value_states_163_groups_0, pad = value_states_163_pad_0, pad_type = value_states_163_pad_type_0, strides = value_states_163_strides_0, weight = layers_27_self_attn_v_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("value_states_163_cast_fp16")]; tensor concat_324x = const()[name = string("concat_324x"), val = tensor([1, 16, 128, -1])]; tensor x_271_cast_fp16 = reshape(shape = concat_324x, x = query_states_163_cast_fp16)[name = string("x_271_cast_fp16")]; tensor concat_325x = const()[name = string("concat_325x"), val = tensor([1, 2, 128, -1])]; tensor var_9651_cast_fp16 = reshape(shape = concat_325x, x = key_states_271_cast_fp16)[name = string("op_9651_cast_fp16")]; tensor concat_326x = const()[name = string("concat_326x"), val = tensor([1, 2, 128, -1])]; tensor var_9658_cast_fp16 = reshape(shape = concat_326x, x = value_states_163_cast_fp16)[name = string("op_9658_cast_fp16")]; tensor var_9662_cast_fp16 = mul(x = x_271_cast_fp16, y = var_869_cast_fp16)[name = string("op_9662_cast_fp16")]; tensor var_9663_split_sizes_0 = const()[name = string("op_9663_split_sizes_0"), val = tensor([64, 64])]; int32 var_9663_axis_0 = const()[name = string("op_9663_axis_0"), val = int32(-2)]; tensor var_9663_cast_fp16_0, tensor var_9663_cast_fp16_1 = split(axis = var_9663_axis_0, split_sizes = var_9663_split_sizes_0, x = x_271_cast_fp16)[name = string("op_9663_cast_fp16")]; fp16 const_272_promoted_to_fp16 = const()[name = string("const_272_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9665_cast_fp16 = mul(x = var_9663_cast_fp16_1, y = const_272_promoted_to_fp16)[name = string("op_9665_cast_fp16")]; int32 var_9667 = const()[name = string("op_9667"), val = int32(-2)]; bool var_9668_interleave_0 = const()[name = string("op_9668_interleave_0"), val = bool(false)]; tensor var_9668_cast_fp16 = concat(axis = var_9667, interleave = var_9668_interleave_0, values = (var_9665_cast_fp16, var_9663_cast_fp16_0))[name = string("op_9668_cast_fp16")]; tensor var_9669_cast_fp16 = mul(x = var_9668_cast_fp16, y = var_878_cast_fp16)[name = string("op_9669_cast_fp16")]; tensor query_states_165_cast_fp16 = add(x = var_9662_cast_fp16, y = var_9669_cast_fp16)[name = string("query_states_165_cast_fp16")]; tensor var_9675_cast_fp16 = mul(x = var_9651_cast_fp16, y = var_869_cast_fp16)[name = string("op_9675_cast_fp16")]; tensor var_9676_split_sizes_0 = const()[name = string("op_9676_split_sizes_0"), val = tensor([64, 64])]; int32 var_9676_axis_0 = const()[name = string("op_9676_axis_0"), val = int32(-2)]; tensor var_9676_cast_fp16_0, tensor var_9676_cast_fp16_1 = split(axis = var_9676_axis_0, split_sizes = var_9676_split_sizes_0, x = var_9651_cast_fp16)[name = string("op_9676_cast_fp16")]; fp16 const_273_promoted_to_fp16 = const()[name = string("const_273_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9678_cast_fp16 = mul(x = var_9676_cast_fp16_1, y = const_273_promoted_to_fp16)[name = string("op_9678_cast_fp16")]; int32 var_9680 = const()[name = string("op_9680"), val = int32(-2)]; bool var_9681_interleave_0 = const()[name = string("op_9681_interleave_0"), val = bool(false)]; tensor var_9681_cast_fp16 = concat(axis = var_9680, interleave = var_9681_interleave_0, values = (var_9678_cast_fp16, var_9676_cast_fp16_0))[name = string("op_9681_cast_fp16")]; tensor var_9682_cast_fp16 = mul(x = var_9681_cast_fp16, y = var_878_cast_fp16)[name = string("op_9682_cast_fp16")]; tensor key_states_275_cast_fp16 = add(x = var_9675_cast_fp16, y = var_9682_cast_fp16)[name = string("key_states_275_cast_fp16")]; tensor expand_dims_324 = const()[name = string("expand_dims_324"), val = tensor([27])]; tensor expand_dims_325 = const()[name = string("expand_dims_325"), val = tensor([0])]; tensor expand_dims_327 = const()[name = string("expand_dims_327"), val = tensor([0])]; int32 concat_329_axis_0 = const()[name = string("concat_329_axis_0"), val = int32(0)]; bool concat_329_interleave_0 = const()[name = string("concat_329_interleave_0"), val = bool(false)]; tensor concat_329 = concat(axis = concat_329_axis_0, interleave = concat_329_interleave_0, values = (expand_dims_324, expand_dims_325, position_id, expand_dims_327))[name = string("concat_329")]; tensor expand_dims_328 = const()[name = string("expand_dims_328"), val = tensor([28])]; tensor concat_330_values1_0 = const()[name = string("concat_330_values1_0"), val = tensor([0])]; tensor concat_330_values3_0 = const()[name = string("concat_330_values3_0"), val = tensor([0])]; int32 concat_330_axis_0 = const()[name = string("concat_330_axis_0"), val = int32(0)]; bool concat_330_interleave_0 = const()[name = string("concat_330_interleave_0"), val = bool(false)]; tensor concat_330 = concat(axis = concat_330_axis_0, interleave = concat_330_interleave_0, values = (expand_dims_328, concat_330_values1_0, cache_position_end, concat_330_values3_0))[name = string("concat_330")]; tensor key_states_277_perm_0 = const()[name = string("key_states_277_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_28_stride_0 = const()[name = string("key_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_277_cast_fp16 = transpose(perm = key_states_277_perm_0, x = key_states_275_cast_fp16)[name = string("transpose_518")]; tensor key_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = key_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = key_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_28_squeeze_mask_0, stride = key_cache_internal_tensor_assign_28_stride_0, update = key_states_277_cast_fp16, x = coreml_update_state_388)[name = string("key_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_28_cast_fp16, input = key_cache)[name = string("coreml_update_state_390_write_state")]; tensor coreml_update_state_390 = read_state(input = key_cache)[name = string("coreml_update_state_390")]; tensor value_states_165_perm_0 = const()[name = string("value_states_165_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_28_stride_0 = const()[name = string("value_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_165_cast_fp16 = transpose(perm = value_states_165_perm_0, x = var_9658_cast_fp16)[name = string("transpose_517")]; tensor value_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = value_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = value_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_28_squeeze_mask_0, stride = value_cache_internal_tensor_assign_28_stride_0, update = value_states_165_cast_fp16, x = coreml_update_state_389)[name = string("value_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_28_cast_fp16, input = value_cache)[name = string("coreml_update_state_391_write_state")]; tensor coreml_update_state_391 = read_state(input = value_cache)[name = string("coreml_update_state_391")]; tensor var_9752_begin_0 = const()[name = string("op_9752_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9752_end_0 = const()[name = string("op_9752_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9752_end_mask_0 = const()[name = string("op_9752_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9752_cast_fp16 = slice_by_index(begin = var_9752_begin_0, end = var_9752_end_0, end_mask = var_9752_end_mask_0, x = coreml_update_state_390)[name = string("op_9752_cast_fp16")]; tensor tile_54 = const()[name = string("tile_54"), val = tensor([1, 1])]; int32 var_9755_axis_0 = const()[name = string("op_9755_axis_0"), val = int32(1)]; tensor var_9755_cast_fp16_0, tensor var_9755_cast_fp16_1 = split(axis = var_9755_axis_0, split_sizes = tile_54, x = var_9752_cast_fp16)[name = string("op_9755_cast_fp16")]; tensor var_9762_begin_0 = const()[name = string("op_9762_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9762_end_0 = const()[name = string("op_9762_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9762_end_mask_0 = const()[name = string("op_9762_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9762_cast_fp16 = slice_by_index(begin = var_9762_begin_0, end = var_9762_end_0, end_mask = var_9762_end_mask_0, x = coreml_update_state_391)[name = string("op_9762_cast_fp16")]; tensor tile_55 = const()[name = string("tile_55"), val = tensor([1, 1])]; int32 var_9765_axis_0 = const()[name = string("op_9765_axis_0"), val = int32(1)]; tensor var_9765_cast_fp16_0, tensor var_9765_cast_fp16_1 = split(axis = var_9765_axis_0, split_sizes = tile_55, x = var_9762_cast_fp16)[name = string("op_9765_cast_fp16")]; tensor var_9768_split_sizes_0 = const()[name = string("op_9768_split_sizes_0"), val = tensor([8, 8])]; int32 var_9768_axis_0 = const()[name = string("op_9768_axis_0"), val = int32(1)]; tensor var_9768_0, tensor var_9768_1 = split(axis = var_9768_axis_0, split_sizes = var_9768_split_sizes_0, x = query_states_165_cast_fp16)[name = string("op_9768")]; bool attn_weights_433_transpose_x_0 = const()[name = string("attn_weights_433_transpose_x_0"), val = bool(false)]; bool attn_weights_433_transpose_y_0 = const()[name = string("attn_weights_433_transpose_y_0"), val = bool(false)]; tensor attn_weights_433_cast_fp16 = matmul(transpose_x = attn_weights_433_transpose_x_0, transpose_y = attn_weights_433_transpose_y_0, x = var_9755_cast_fp16_0, y = var_9768_0)[name = string("attn_weights_433_cast_fp16")]; fp16 var_9771_to_fp16 = const()[name = string("op_9771_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_435_cast_fp16 = mul(x = attn_weights_433_cast_fp16, y = var_9771_to_fp16)[name = string("attn_weights_435_cast_fp16")]; tensor attn_weights_437_cast_fp16 = add(x = attn_weights_435_cast_fp16, y = attn_mask_1)[name = string("attn_weights_437_cast_fp16")]; int32 var_9775 = const()[name = string("op_9775"), val = int32(-2)]; tensor attn_weights_439_cast_fp16 = softmax(axis = var_9775, x = attn_weights_437_cast_fp16)[name = string("attn_weights_439_cast_fp16")]; bool var_9781_transpose_x_1 = const()[name = string("op_9781_transpose_x_1"), val = bool(true)]; bool var_9781_transpose_y_1 = const()[name = string("op_9781_transpose_y_1"), val = bool(false)]; tensor var_9781_cast_fp16 = matmul(transpose_x = var_9781_transpose_x_1, transpose_y = var_9781_transpose_y_1, x = attn_weights_439_cast_fp16, y = var_9765_cast_fp16_0)[name = string("op_9781_cast_fp16")]; bool attn_weights_441_transpose_x_0 = const()[name = string("attn_weights_441_transpose_x_0"), val = bool(false)]; bool attn_weights_441_transpose_y_0 = const()[name = string("attn_weights_441_transpose_y_0"), val = bool(false)]; tensor attn_weights_441_cast_fp16 = matmul(transpose_x = attn_weights_441_transpose_x_0, transpose_y = attn_weights_441_transpose_y_0, x = var_9755_cast_fp16_1, y = var_9768_1)[name = string("attn_weights_441_cast_fp16")]; fp16 var_9783_to_fp16 = const()[name = string("op_9783_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_443_cast_fp16 = mul(x = attn_weights_441_cast_fp16, y = var_9783_to_fp16)[name = string("attn_weights_443_cast_fp16")]; tensor attn_weights_445_cast_fp16 = add(x = attn_weights_443_cast_fp16, y = attn_mask_1)[name = string("attn_weights_445_cast_fp16")]; int32 var_9787 = const()[name = string("op_9787"), val = int32(-2)]; tensor attn_weights_cast_fp16 = softmax(axis = var_9787, x = attn_weights_445_cast_fp16)[name = string("attn_weights_cast_fp16")]; bool attn_output_217_transpose_x_1 = const()[name = string("attn_output_217_transpose_x_1"), val = bool(true)]; bool attn_output_217_transpose_y_1 = const()[name = string("attn_output_217_transpose_y_1"), val = bool(false)]; tensor attn_output_217_cast_fp16 = matmul(transpose_x = attn_output_217_transpose_x_1, transpose_y = attn_output_217_transpose_y_1, x = attn_weights_cast_fp16, y = var_9765_cast_fp16_1)[name = string("attn_output_217_cast_fp16")]; int32 var_9795 = const()[name = string("op_9795"), val = int32(1)]; bool attn_output_219_interleave_0 = const()[name = string("attn_output_219_interleave_0"), val = bool(false)]; tensor attn_output_219_cast_fp16 = concat(axis = var_9795, interleave = attn_output_219_interleave_0, values = (var_9781_cast_fp16, attn_output_217_cast_fp16))[name = string("attn_output_219_cast_fp16")]; tensor var_9799_perm_0 = const()[name = string("op_9799_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_335x = const()[name = string("concat_335x"), val = tensor([1, 2048, 1, -1])]; tensor var_9799_cast_fp16 = transpose(perm = var_9799_perm_0, x = attn_output_219_cast_fp16)[name = string("transpose_516")]; tensor attn_output_cast_fp16 = reshape(shape = concat_335x, x = var_9799_cast_fp16)[name = string("attn_output_cast_fp16")]; tensor layers_27_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1628492864)))]; tensor hidden_states_273_strides_0 = const()[name = string("hidden_states_273_strides_0"), val = tensor([1, 1])]; string hidden_states_273_pad_type_0 = const()[name = string("hidden_states_273_pad_type_0"), val = string("valid")]; tensor hidden_states_273_pad_0 = const()[name = string("hidden_states_273_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_273_dilations_0 = const()[name = string("hidden_states_273_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_273_groups_0 = const()[name = string("hidden_states_273_groups_0"), val = int32(1)]; tensor hidden_states_273_cast_fp16 = conv(dilations = hidden_states_273_dilations_0, groups = hidden_states_273_groups_0, pad = hidden_states_273_pad_0, pad_type = hidden_states_273_pad_type_0, strides = hidden_states_273_strides_0, weight = layers_27_self_attn_o_proj_weight_to_fp16, x = attn_output_cast_fp16)[name = string("hidden_states_273_cast_fp16")]; tensor hidden_states_275_cast_fp16 = add(x = hidden_states_269_cast_fp16, y = hidden_states_273_cast_fp16)[name = string("hidden_states_275_cast_fp16")]; fp16 const_278_promoted_to_fp16 = const()[name = string("const_278_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9832_cast_fp16 = mul(x = hidden_states_275_cast_fp16, y = const_278_promoted_to_fp16)[name = string("op_9832_cast_fp16")]; int32 var_9830 = const()[name = string("op_9830"), val = int32(1)]; bool doubled_221_interleave_0 = const()[name = string("doubled_221_interleave_0"), val = bool(false)]; tensor doubled_221_cast_fp16 = concat(axis = var_9830, interleave = doubled_221_interleave_0, values = (hidden_states_275_cast_fp16, var_9832_cast_fp16))[name = string("doubled_221_cast_fp16")]; tensor out_111_axes_0 = const()[name = string("out_111_axes_0"), val = tensor([1])]; tensor out_111_gamma_0_to_fp16 = const()[name = string("out_111_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636881536)))]; fp16 var_9842_to_fp16 = const()[name = string("op_9842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_111_cast_fp16 = layer_norm(axes = out_111_axes_0, epsilon = var_9842_to_fp16, gamma = out_111_gamma_0_to_fp16, x = doubled_221_cast_fp16)[name = string("out_111_cast_fp16")]; tensor var_9853_split_sizes_0 = const()[name = string("op_9853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9853_axis_0 = const()[name = string("op_9853_axis_0"), val = int32(1)]; tensor var_9853_cast_fp16_0, tensor var_9853_cast_fp16_1 = split(axis = var_9853_axis_0, split_sizes = var_9853_split_sizes_0, x = out_111_cast_fp16)[name = string("op_9853_cast_fp16")]; tensor layers_27_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636889792)))]; tensor input_strides_0 = const()[name = string("input_strides_0"), val = tensor([1, 1])]; string input_pad_type_0 = const()[name = string("input_pad_type_0"), val = string("valid")]; tensor input_pad_0 = const()[name = string("input_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_dilations_0 = const()[name = string("input_dilations_0"), val = tensor([1, 1])]; int32 input_groups_0 = const()[name = string("input_groups_0"), val = int32(1)]; tensor input_cast_fp16 = conv(dilations = input_dilations_0, groups = input_groups_0, pad = input_pad_0, pad_type = input_pad_type_0, strides = input_strides_0, weight = layers_27_mlp_gate_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("input_cast_fp16")]; tensor var_9870_cast_fp16 = silu(x = input_cast_fp16)[name = string("op_9870_cast_fp16")]; tensor layers_27_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1662055680)))]; tensor var_9876_strides_0 = const()[name = string("op_9876_strides_0"), val = tensor([1, 1])]; string var_9876_pad_type_0 = const()[name = string("op_9876_pad_type_0"), val = string("valid")]; tensor var_9876_pad_0 = const()[name = string("op_9876_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9876_dilations_0 = const()[name = string("op_9876_dilations_0"), val = tensor([1, 1])]; int32 var_9876_groups_0 = const()[name = string("op_9876_groups_0"), val = int32(1)]; tensor var_9876_cast_fp16 = conv(dilations = var_9876_dilations_0, groups = var_9876_groups_0, pad = var_9876_pad_0, pad_type = var_9876_pad_type_0, strides = var_9876_strides_0, weight = layers_27_mlp_up_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("op_9876_cast_fp16")]; tensor x_cast_fp16 = mul(x = var_9870_cast_fp16, y = var_9876_cast_fp16)[name = string("x_cast_fp16")]; tensor layers_27_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1687221568)))]; tensor hidden_states_277_strides_0 = const()[name = string("hidden_states_277_strides_0"), val = tensor([1, 1])]; string hidden_states_277_pad_type_0 = const()[name = string("hidden_states_277_pad_type_0"), val = string("valid")]; tensor hidden_states_277_pad_0 = const()[name = string("hidden_states_277_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_277_dilations_0 = const()[name = string("hidden_states_277_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_277_groups_0 = const()[name = string("hidden_states_277_groups_0"), val = int32(1)]; tensor hidden_states_277_cast_fp16 = conv(dilations = hidden_states_277_dilations_0, groups = hidden_states_277_groups_0, pad = hidden_states_277_pad_0, pad_type = hidden_states_277_pad_type_0, strides = hidden_states_277_strides_0, weight = layers_27_mlp_down_proj_weight_to_fp16, x = x_cast_fp16)[name = string("hidden_states_277_cast_fp16")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_275_cast_fp16, y = hidden_states_277_cast_fp16)[name = string("hidden_states_cast_fp16")]; fp16 const_280_promoted_to_fp16 = const()[name = string("const_280_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9894_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_280_promoted_to_fp16)[name = string("op_9894_cast_fp16")]; int32 var_9892 = const()[name = string("op_9892"), val = int32(1)]; bool doubled_225_interleave_0 = const()[name = string("doubled_225_interleave_0"), val = bool(false)]; tensor doubled_225_cast_fp16 = concat(axis = var_9892, interleave = doubled_225_interleave_0, values = (hidden_states_cast_fp16, var_9894_cast_fp16))[name = string("doubled_225_cast_fp16")]; tensor out_axes_0 = const()[name = string("out_axes_0"), val = tensor([1])]; tensor out_gamma_0_to_fp16 = const()[name = string("out_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1712387456)))]; fp16 var_9904_to_fp16 = const()[name = string("op_9904_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_cast_fp16 = layer_norm(axes = out_axes_0, epsilon = var_9904_to_fp16, gamma = out_gamma_0_to_fp16, x = doubled_225_cast_fp16)[name = string("out_cast_fp16")]; tensor var_9915_split_sizes_0 = const()[name = string("op_9915_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9915_axis_0 = const()[name = string("op_9915_axis_0"), val = int32(1)]; tensor hidden_states, tensor var_9915_cast_fp16_1 = split(axis = var_9915_axis_0, split_sizes = var_9915_split_sizes_0, x = out_cast_fp16)[name = string("op_9915_cast_fp16")]; } -> (hidden_states); func length_16(tensor inputs_embeds, state> key_cache, tensor position_id, tensor position_index_seed, state> value_cache) { tensor layers_1_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524992))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524416))))[name = string("layers_1_self_attn_v_proj_weight_cast_fp16")]; tensor layers_1_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(525312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13120640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13108288))))[name = string("layers_1_mlp_up_proj_weight_cast_fp16")]; tensor layers_2_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13126848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651776))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651200))))[name = string("layers_2_self_attn_v_proj_weight_cast_fp16")]; tensor layers_2_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13652096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26247424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26235072))))[name = string("layers_2_mlp_up_proj_weight_cast_fp16")]; tensor layers_3_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26253632))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778560))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26777984))))[name = string("layers_3_self_attn_v_proj_weight_cast_fp16")]; tensor layers_3_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30977408))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30973248))))[name = string("layers_3_self_attn_o_proj_weight_cast_fp16")]; tensor layers_3_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30979520))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43566656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43562496))))[name = string("layers_3_mlp_down_proj_weight_cast_fp16")]; tensor layers_4_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43568768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093696))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093120))))[name = string("layers_4_self_attn_v_proj_weight_cast_fp16")]; tensor layers_4_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44094016))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48292544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48288384))))[name = string("layers_4_self_attn_o_proj_weight_cast_fp16")]; tensor layers_4_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48294656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60877632))))[name = string("layers_4_mlp_gate_proj_weight_cast_fp16")]; tensor layers_4_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60896192))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73491520))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73479168))))[name = string("layers_4_mlp_up_proj_weight_cast_fp16")]; tensor layers_4_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73497728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86084864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86080704))))[name = string("layers_4_mlp_down_proj_weight_cast_fp16")]; tensor layers_5_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86086976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611328))))[name = string("layers_5_self_attn_v_proj_weight_cast_fp16")]; tensor layers_5_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86612224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90810752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90806592))))[name = string("layers_5_self_attn_o_proj_weight_cast_fp16")]; tensor layers_5_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90812864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103408192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103395840))))[name = string("layers_5_mlp_up_proj_weight_cast_fp16")]; tensor layers_5_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103414400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116001536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115997376))))[name = string("layers_5_mlp_down_proj_weight_cast_fp16")]; tensor layers_6_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116003648))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528576))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528000))))[name = string("layers_6_self_attn_v_proj_weight_cast_fp16")]; tensor layers_6_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528896))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120727424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120723264))))[name = string("layers_6_self_attn_o_proj_weight_cast_fp16")]; tensor layers_6_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120729536))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133324864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133312512))))[name = string("layers_6_mlp_gate_proj_weight_cast_fp16")]; tensor layers_6_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133331072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145926400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145914048))))[name = string("layers_6_mlp_up_proj_weight_cast_fp16")]; tensor layers_6_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145932608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158519744))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158515584))))[name = string("layers_6_mlp_down_proj_weight_cast_fp16")]; tensor layers_7_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158521856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046208))))[name = string("layers_7_self_attn_v_proj_weight_cast_fp16")]; tensor layers_7_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159047104))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163245632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163241472))))[name = string("layers_7_self_attn_o_proj_weight_cast_fp16")]; tensor layers_7_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163247744))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175843072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175830720))))[name = string("layers_7_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175849280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374208))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176373632))))[name = string("layers_8_self_attn_v_proj_weight_cast_fp16")]; tensor layers_8_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374528))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180573056))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180568896))))[name = string("layers_8_self_attn_o_proj_weight_cast_fp16")]; tensor layers_8_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180575168))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193170496))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193158144))))[name = string("layers_8_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193176704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205772032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205759680))))[name = string("layers_8_mlp_up_proj_weight_cast_fp16")]; tensor layers_8_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205778240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218365376))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218361216))))[name = string("layers_8_mlp_down_proj_weight_cast_fp16")]; tensor layers_9_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218367488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218891840))))[name = string("layers_9_self_attn_v_proj_weight_cast_fp16")]; tensor layers_9_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223091264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223087104))))[name = string("layers_9_self_attn_o_proj_weight_cast_fp16")]; tensor layers_9_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223093376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235688704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235676352))))[name = string("layers_9_mlp_gate_proj_weight_cast_fp16")]; tensor layers_9_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235694912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248290240))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248277888))))[name = string("layers_9_mlp_up_proj_weight_cast_fp16")]; tensor layers_9_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248296448))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260883584))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260879424))))[name = string("layers_9_mlp_down_proj_weight_cast_fp16")]; tensor layers_10_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260885696))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410048))))[name = string("layers_10_self_attn_v_proj_weight_cast_fp16")]; tensor layers_10_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410944))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265609472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265605312))))[name = string("layers_10_self_attn_o_proj_weight_cast_fp16")]; tensor layers_10_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265611584))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278206912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278194560))))[name = string("layers_10_mlp_gate_proj_weight_cast_fp16")]; tensor layers_10_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278213120))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290808448))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290796096))))[name = string("layers_10_mlp_up_proj_weight_cast_fp16")]; tensor layers_10_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290814656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303401792))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303397632))))[name = string("layers_10_mlp_down_proj_weight_cast_fp16")]; tensor layers_11_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303403904))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307602432))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307598272))))[name = string("layers_11_self_attn_q_proj_weight_cast_fp16")]; tensor layers_11_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307604544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308128896))))[name = string("layers_11_self_attn_k_proj_weight_cast_fp16")]; tensor layers_11_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129792))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654720))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654144))))[name = string("layers_11_self_attn_v_proj_weight_cast_fp16")]; tensor layers_11_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308655040))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312853568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312849408))))[name = string("layers_11_self_attn_o_proj_weight_cast_fp16")]; tensor layers_11_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312855680))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325451008))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325438656))))[name = string("layers_11_mlp_gate_proj_weight_cast_fp16")]; tensor layers_11_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325457216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338052544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338040192))))[name = string("layers_11_mlp_up_proj_weight_cast_fp16")]; tensor layers_11_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338058752))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350645888))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350641728))))[name = string("layers_11_mlp_down_proj_weight_cast_fp16")]; tensor layers_12_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350648000))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354846528))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354842368))))[name = string("layers_12_self_attn_q_proj_weight_cast_fp16")]; tensor layers_12_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354848640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355372992))))[name = string("layers_12_self_attn_k_proj_weight_cast_fp16")]; tensor layers_12_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373888))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898816))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898240))))[name = string("layers_12_self_attn_v_proj_weight_cast_fp16")]; tensor layers_12_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355899136))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360097664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360093504))))[name = string("layers_12_self_attn_o_proj_weight_cast_fp16")]; tensor layers_12_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360099776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372695104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372682752))))[name = string("layers_12_mlp_gate_proj_weight_cast_fp16")]; tensor layers_12_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372701312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385296640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385284288))))[name = string("layers_12_mlp_up_proj_weight_cast_fp16")]; tensor layers_12_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385302848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397885824))))[name = string("layers_12_mlp_down_proj_weight_cast_fp16")]; tensor layers_13_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397892096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402090624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402086464))))[name = string("layers_13_self_attn_q_proj_weight_cast_fp16")]; tensor layers_13_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402092736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617088))))[name = string("layers_13_self_attn_k_proj_weight_cast_fp16")]; tensor layers_13_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617984))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142336))))[name = string("layers_13_self_attn_v_proj_weight_cast_fp16")]; tensor layers_13_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403143232))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407341760))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407337600))))[name = string("layers_13_self_attn_o_proj_weight_cast_fp16")]; tensor layers_13_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407343872))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419939200))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419926848))))[name = string("layers_13_mlp_gate_proj_weight_cast_fp16")]; tensor layers_13_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419945408))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432532544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432528384))))[name = string("layers_13_mlp_down_proj_weight_cast_fp16")]; tensor layers_14_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432534656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436733184))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436729024))))[name = string("layers_14_self_attn_q_proj_weight_cast_fp16")]; tensor layers_14_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436735296))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437259648))))[name = string("layers_14_self_attn_v_proj_weight_cast_fp16")]; tensor layers_14_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441459072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441454912))))[name = string("layers_14_self_attn_o_proj_weight_cast_fp16")]; tensor layers_14_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441461184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454056512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454044160))))[name = string("layers_14_mlp_gate_proj_weight_cast_fp16")]; tensor layers_14_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454062720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466658048))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466645696))))[name = string("layers_14_mlp_up_proj_weight_cast_fp16")]; tensor layers_14_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466664256))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479251392))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479247232))))[name = string("layers_14_mlp_down_proj_weight_cast_fp16")]; tensor layers_15_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479253504))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483452032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483447872))))[name = string("layers_15_self_attn_q_proj_weight_cast_fp16")]; tensor layers_15_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483454144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483978496))))[name = string("layers_15_self_attn_k_proj_weight_cast_fp16")]; tensor layers_15_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504320))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484503744))))[name = string("layers_15_self_attn_v_proj_weight_cast_fp16")]; tensor layers_15_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488703168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488699008))))[name = string("layers_15_self_attn_o_proj_weight_cast_fp16")]; tensor layers_15_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488705280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501300608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501288256))))[name = string("layers_15_mlp_gate_proj_weight_cast_fp16")]; tensor layers_15_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501306816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513902144))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513889792))))[name = string("layers_15_mlp_up_proj_weight_cast_fp16")]; tensor layers_15_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513908352))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526495488))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526491328))))[name = string("layers_15_mlp_down_proj_weight_cast_fp16")]; tensor layers_16_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526497600))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530696128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530691968))))[name = string("layers_16_self_attn_q_proj_weight_cast_fp16")]; tensor layers_16_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530698240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531222592))))[name = string("layers_16_self_attn_k_proj_weight_cast_fp16")]; tensor layers_16_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531747840))))[name = string("layers_16_self_attn_v_proj_weight_cast_fp16")]; tensor layers_16_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535947264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535943104))))[name = string("layers_16_self_attn_o_proj_weight_cast_fp16")]; tensor layers_16_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535949376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548536512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548532352))))[name = string("layers_16_mlp_down_proj_weight_cast_fp16")]; tensor layers_17_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548538624))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552737152))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552732992))))[name = string("layers_17_self_attn_q_proj_weight_cast_fp16")]; tensor layers_17_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552739264))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553263616))))[name = string("layers_17_self_attn_k_proj_weight_cast_fp16")]; tensor layers_17_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264512))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553788864))))[name = string("layers_17_self_attn_v_proj_weight_cast_fp16")]; tensor layers_17_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789760))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557988288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557984128))))[name = string("layers_17_self_attn_o_proj_weight_cast_fp16")]; tensor layers_17_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557990400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570585728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570573376))))[name = string("layers_17_mlp_gate_proj_weight_cast_fp16")]; tensor layers_17_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570591936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583187264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583174912))))[name = string("layers_17_mlp_up_proj_weight_cast_fp16")]; tensor layers_17_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583193472))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595780608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595776448))))[name = string("layers_17_mlp_down_proj_weight_cast_fp16")]; tensor layers_18_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595782720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599981248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599977088))))[name = string("layers_18_self_attn_q_proj_weight_cast_fp16")]; tensor layers_18_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599983360))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600507712))))[name = string("layers_18_self_attn_k_proj_weight_cast_fp16")]; tensor layers_18_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601032960))))[name = string("layers_18_self_attn_v_proj_weight_cast_fp16")]; tensor layers_18_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605232384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605228224))))[name = string("layers_18_self_attn_o_proj_weight_cast_fp16")]; tensor layers_18_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605234496))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617829824))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617817472))))[name = string("layers_18_mlp_gate_proj_weight_cast_fp16")]; tensor layers_18_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617836032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630431360))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630419008))))[name = string("layers_18_mlp_up_proj_weight_cast_fp16")]; tensor layers_18_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630437568))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643024704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643020544))))[name = string("layers_18_mlp_down_proj_weight_cast_fp16")]; tensor layers_19_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643026816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647225344))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647221184))))[name = string("layers_19_self_attn_q_proj_weight_cast_fp16")]; tensor layers_19_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647227456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647751808))))[name = string("layers_19_self_attn_k_proj_weight_cast_fp16")]; tensor layers_19_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660348032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660335680))))[name = string("layers_19_mlp_gate_proj_weight_cast_fp16")]; tensor layers_19_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660354240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672949568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672937216))))[name = string("layers_19_mlp_up_proj_weight_cast_fp16")]; tensor layers_19_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672955776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685542912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685538752))))[name = string("layers_19_mlp_down_proj_weight_cast_fp16")]; tensor layers_20_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685545024))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689743552))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689739392))))[name = string("layers_20_self_attn_q_proj_weight_cast_fp16")]; tensor layers_20_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689745664))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270592))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270016))))[name = string("layers_20_self_attn_k_proj_weight_cast_fp16")]; tensor layers_20_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694469440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694465280))))[name = string("layers_20_self_attn_o_proj_weight_cast_fp16")]; tensor layers_20_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694471552))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707066880))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707054528))))[name = string("layers_20_mlp_gate_proj_weight_cast_fp16")]; tensor layers_20_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707073088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719660224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719656064))))[name = string("layers_20_mlp_down_proj_weight_cast_fp16")]; tensor layers_21_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719662336))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723860864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723856704))))[name = string("layers_21_self_attn_q_proj_weight_cast_fp16")]; tensor layers_21_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723862976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387328))))[name = string("layers_21_self_attn_k_proj_weight_cast_fp16")]; tensor layers_21_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724388224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728586752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728582592))))[name = string("layers_21_self_attn_o_proj_weight_cast_fp16")]; tensor layers_21_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728588864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741184192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741171840))))[name = string("layers_21_mlp_gate_proj_weight_cast_fp16")]; tensor layers_21_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741190400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753785728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753773376))))[name = string("layers_21_mlp_up_proj_weight_cast_fp16")]; tensor layers_21_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753791936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766379072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766374912))))[name = string("layers_21_mlp_down_proj_weight_cast_fp16")]; tensor layers_22_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766381184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770579712))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770575552))))[name = string("layers_22_self_attn_q_proj_weight_cast_fp16")]; tensor layers_22_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770581824))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106176))))[name = string("layers_22_self_attn_k_proj_weight_cast_fp16")]; tensor layers_22_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771107072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783702400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783690048))))[name = string("layers_22_mlp_gate_proj_weight_cast_fp16")]; tensor layers_22_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783708608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796303936))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796291584))))[name = string("layers_22_mlp_up_proj_weight_cast_fp16")]; tensor layers_22_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796310144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808897280))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808893120))))[name = string("layers_22_mlp_down_proj_weight_cast_fp16")]; tensor layers_23_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808899392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813097920))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813093760))))[name = string("layers_23_self_attn_q_proj_weight_cast_fp16")]; tensor layers_23_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813100032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624960))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624384))))[name = string("layers_23_self_attn_k_proj_weight_cast_fp16")]; tensor layers_23_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813625280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817823808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817819648))))[name = string("layers_23_self_attn_o_proj_weight_cast_fp16")]; tensor layers_23_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817825920))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830421248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830408896))))[name = string("layers_23_mlp_gate_proj_weight_cast_fp16")]; tensor layers_23_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830427456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843022784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843010432))))[name = string("layers_23_mlp_up_proj_weight_cast_fp16")]; tensor layers_23_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843028992))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855616128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855611968))))[name = string("layers_23_mlp_down_proj_weight_cast_fp16")]; tensor layers_24_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855618240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859816768))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859812608))))[name = string("layers_24_self_attn_q_proj_weight_cast_fp16")]; tensor layers_24_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859818880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343232))))[name = string("layers_24_self_attn_k_proj_weight_cast_fp16")]; tensor layers_24_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860344128))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864542656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864538496))))[name = string("layers_24_self_attn_o_proj_weight_cast_fp16")]; tensor layers_24_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864544768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877140096))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877127744))))[name = string("layers_24_mlp_gate_proj_weight_cast_fp16")]; tensor layers_24_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877146304))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889741632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889729280))))[name = string("layers_24_mlp_up_proj_weight_cast_fp16")]; tensor layers_24_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889747840))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902334976))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902330816))))[name = string("layers_24_mlp_down_proj_weight_cast_fp16")]; tensor layers_25_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902337088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906535616))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906531456))))[name = string("layers_25_self_attn_q_proj_weight_cast_fp16")]; tensor layers_25_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906537728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062080))))[name = string("layers_25_self_attn_k_proj_weight_cast_fp16")]; tensor layers_25_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911261504))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911257344))))[name = string("layers_25_self_attn_o_proj_weight_cast_fp16")]; tensor layers_25_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911263616))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923858944))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923846592))))[name = string("layers_25_mlp_gate_proj_weight_cast_fp16")]; tensor layers_25_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923865152))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936460480))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936448128))))[name = string("layers_25_mlp_up_proj_weight_cast_fp16")]; tensor layers_26_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936466688))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940665216))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940661056))))[name = string("layers_26_self_attn_q_proj_weight_cast_fp16")]; tensor layers_26_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940667328))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192256))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941191680))))[name = string("layers_26_self_attn_k_proj_weight_cast_fp16")]; tensor layers_26_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192576))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945391104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945386944))))[name = string("layers_26_self_attn_o_proj_weight_cast_fp16")]; tensor layers_26_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945393216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957988544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957976192))))[name = string("layers_26_mlp_gate_proj_weight_cast_fp16")]; int32 var_765 = const()[name = string("op_765"), val = int32(0)]; tensor var_766 = mul(x = position_index_seed, y = var_765)[name = string("op_766")]; int32 var_768 = const()[name = string("op_768"), val = int32(1)]; tensor ones = add(x = var_766, y = var_768)[name = string("ones")]; int32 var_770 = const()[name = string("op_770"), val = int32(0)]; bool var_772_exclusive_0 = const()[name = string("op_772_exclusive_0"), val = bool(false)]; bool var_772_reverse_0 = const()[name = string("op_772_reverse_0"), val = bool(false)]; tensor var_772 = cumsum(axis = var_770, exclusive = var_772_exclusive_0, reverse = var_772_reverse_0, x = ones)[name = string("op_772")]; int32 var_774 = const()[name = string("op_774"), val = int32(1)]; tensor position_offsets = sub(x = var_772, y = var_774)[name = string("position_offsets")]; tensor position_ids_1 = add(x = position_offsets, y = position_id)[name = string("position_ids_1")]; bool var_784_keep_dims_0 = const()[name = string("op_784_keep_dims_0"), val = bool(false)]; int32 var_784 = reduce_sum(keep_dims = var_784_keep_dims_0, x = ones)[name = string("op_784")]; int32 var_786 = const()[name = string("op_786"), val = int32(1)]; int32 offset = sub(x = var_784, y = var_786)[name = string("offset")]; tensor var_789 = add(x = position_id, y = offset)[name = string("op_789")]; int32 var_791 = const()[name = string("op_791"), val = int32(1)]; tensor cache_position_end = add(x = var_789, y = var_791)[name = string("cache_position_end")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor greater_equal_0 = greater_equal(x = position_ids_1, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(32768)]; tensor add_0 = add(x = position_ids_1, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = position_ids_1, b = add_0, cond = greater_equal_0)[name = string("select_0")]; tensor rope_emb_cos_cached_to_fp16 = const()[name = string("rope_emb_cos_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957994752)))]; int32 cos_1_batch_dims_0 = const()[name = string("cos_1_batch_dims_0"), val = int32(0)]; bool cos_1_validate_indices_0 = const()[name = string("cos_1_validate_indices_0"), val = bool(false)]; int32 greater_equal_6_y_0 = const()[name = string("greater_equal_6_y_0"), val = int32(0)]; tensor greater_equal_6 = greater_equal(x = select_0, y = greater_equal_6_y_0)[name = string("greater_equal_6")]; int32 slice_by_index_6 = const()[name = string("slice_by_index_6"), val = int32(32768)]; tensor add_6 = add(x = select_0, y = slice_by_index_6)[name = string("add_6")]; tensor select_6 = select(a = select_0, b = add_6, cond = greater_equal_6)[name = string("select_6")]; int32 cos_1_cast_fp16_axis_3 = const()[name = string("cos_1_cast_fp16_axis_3"), val = int32(0)]; tensor cos_1_cast_fp16 = gather(axis = cos_1_cast_fp16_axis_3, batch_dims = cos_1_batch_dims_0, indices = select_6, validate_indices = cos_1_validate_indices_0, x = rope_emb_cos_cached_to_fp16)[name = string("cos_1_cast_fp16")]; tensor rope_emb_sin_cached_to_fp16 = const()[name = string("rope_emb_sin_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(966383424)))]; int32 sin_1_batch_dims_0 = const()[name = string("sin_1_batch_dims_0"), val = int32(0)]; bool sin_1_validate_indices_0 = const()[name = string("sin_1_validate_indices_0"), val = bool(false)]; int32 sin_1_cast_fp16_axis_3 = const()[name = string("sin_1_cast_fp16_axis_3"), val = int32(0)]; tensor sin_1_cast_fp16 = gather(axis = sin_1_cast_fp16_axis_3, batch_dims = sin_1_batch_dims_0, indices = select_6, validate_indices = sin_1_validate_indices_0, x = rope_emb_sin_cached_to_fp16)[name = string("sin_1_cast_fp16")]; tensor var_865_perm_0 = const()[name = string("op_865_perm_0"), val = tensor([-1, -2])]; tensor var_867_axes_0 = const()[name = string("op_867_axes_0"), val = tensor([0])]; tensor var_865_cast_fp16 = transpose(perm = var_865_perm_0, x = cos_1_cast_fp16)[name = string("transpose_343")]; tensor var_867_cast_fp16 = expand_dims(axes = var_867_axes_0, x = var_865_cast_fp16)[name = string("op_867_cast_fp16")]; tensor var_869_axes_0 = const()[name = string("op_869_axes_0"), val = tensor([0])]; tensor var_869_cast_fp16 = expand_dims(axes = var_869_axes_0, x = var_867_cast_fp16)[name = string("op_869_cast_fp16")]; tensor var_874_perm_0 = const()[name = string("op_874_perm_0"), val = tensor([-1, -2])]; tensor var_876_axes_0 = const()[name = string("op_876_axes_0"), val = tensor([0])]; tensor var_874_cast_fp16 = transpose(perm = var_874_perm_0, x = sin_1_cast_fp16)[name = string("transpose_342")]; tensor var_876_cast_fp16 = expand_dims(axes = var_876_axes_0, x = var_874_cast_fp16)[name = string("op_876_cast_fp16")]; tensor var_878_axes_0 = const()[name = string("op_878_axes_0"), val = tensor([0])]; tensor var_878_cast_fp16 = expand_dims(axes = var_878_axes_0, x = var_876_cast_fp16)[name = string("op_878_cast_fp16")]; string position_ids_1_to_uint16_dtype_0 = const()[name = string("position_ids_1_to_uint16_dtype_0"), val = string("uint16")]; tensor causal_mask = const()[name = string("causal_mask"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(974772096)))]; int32 mask_axis_0 = const()[name = string("mask_axis_0"), val = int32(1)]; int32 mask_batch_dims_0 = const()[name = string("mask_batch_dims_0"), val = int32(0)]; bool mask_validate_indices_0 = const()[name = string("mask_validate_indices_0"), val = bool(false)]; tensor position_ids_1_to_uint16 = cast(dtype = position_ids_1_to_uint16_dtype_0, x = position_ids_1)[name = string("cast_7")]; tensor mask_cast_uint16 = gather(axis = mask_axis_0, batch_dims = mask_batch_dims_0, indices = position_ids_1_to_uint16, validate_indices = mask_validate_indices_0, x = causal_mask)[name = string("mask_cast_uint16")]; tensor var_895_axes_0 = const()[name = string("op_895_axes_0"), val = tensor([0])]; tensor var_895 = expand_dims(axes = var_895_axes_0, x = mask_cast_uint16)[name = string("op_895")]; tensor attn_mask_1_axes_0 = const()[name = string("attn_mask_1_axes_0"), val = tensor([0])]; tensor attn_mask_1 = expand_dims(axes = attn_mask_1_axes_0, x = var_895)[name = string("attn_mask_1")]; string inputs_embeds_to_fp16_dtype_0 = const()[name = string("inputs_embeds_to_fp16_dtype_0"), val = string("fp16")]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor inputs_embeds_to_fp16 = cast(dtype = inputs_embeds_to_fp16_dtype_0, x = inputs_embeds)[name = string("cast_6")]; tensor var_906_cast_fp16 = mul(x = inputs_embeds_to_fp16, y = const_0_promoted_to_fp16)[name = string("op_906_cast_fp16")]; int32 var_904 = const()[name = string("op_904"), val = int32(1)]; bool doubled_1_interleave_0 = const()[name = string("doubled_1_interleave_0"), val = bool(false)]; tensor doubled_1_cast_fp16 = concat(axis = var_904, interleave = doubled_1_interleave_0, values = (inputs_embeds_to_fp16, var_906_cast_fp16))[name = string("doubled_1_cast_fp16")]; tensor out_1_axes_0 = const()[name = string("out_1_axes_0"), val = tensor([1])]; tensor out_1_gamma_0_to_fp16 = const()[name = string("out_1_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983160768)))]; fp16 var_916_to_fp16 = const()[name = string("op_916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_1_cast_fp16 = layer_norm(axes = out_1_axes_0, epsilon = var_916_to_fp16, gamma = out_1_gamma_0_to_fp16, x = doubled_1_cast_fp16)[name = string("out_1_cast_fp16")]; tensor var_927_split_sizes_0 = const()[name = string("op_927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_927_axis_0 = const()[name = string("op_927_axis_0"), val = int32(1)]; tensor var_927_cast_fp16_0, tensor var_927_cast_fp16_1 = split(axis = var_927_axis_0, split_sizes = var_927_split_sizes_0, x = out_1_cast_fp16)[name = string("op_927_cast_fp16")]; tensor layers_0_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983169024)))]; tensor query_states_1_strides_0 = const()[name = string("query_states_1_strides_0"), val = tensor([1, 1])]; string query_states_1_pad_type_0 = const()[name = string("query_states_1_pad_type_0"), val = string("valid")]; tensor query_states_1_pad_0 = const()[name = string("query_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_1_dilations_0 = const()[name = string("query_states_1_dilations_0"), val = tensor([1, 1])]; int32 query_states_1_groups_0 = const()[name = string("query_states_1_groups_0"), val = int32(1)]; tensor query_states_1_cast_fp16 = conv(dilations = query_states_1_dilations_0, groups = query_states_1_groups_0, pad = query_states_1_pad_0, pad_type = query_states_1_pad_type_0, strides = query_states_1_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("query_states_1_cast_fp16")]; tensor layers_0_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(991557696)))]; tensor key_states_1_strides_0 = const()[name = string("key_states_1_strides_0"), val = tensor([1, 1])]; string key_states_1_pad_type_0 = const()[name = string("key_states_1_pad_type_0"), val = string("valid")]; tensor key_states_1_pad_0 = const()[name = string("key_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_1_dilations_0 = const()[name = string("key_states_1_dilations_0"), val = tensor([1, 1])]; int32 key_states_1_groups_0 = const()[name = string("key_states_1_groups_0"), val = int32(1)]; tensor key_states_1_cast_fp16 = conv(dilations = key_states_1_dilations_0, groups = key_states_1_groups_0, pad = key_states_1_pad_0, pad_type = key_states_1_pad_type_0, strides = key_states_1_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("key_states_1_cast_fp16")]; tensor layers_0_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(992606336)))]; tensor value_states_1_strides_0 = const()[name = string("value_states_1_strides_0"), val = tensor([1, 1])]; string value_states_1_pad_type_0 = const()[name = string("value_states_1_pad_type_0"), val = string("valid")]; tensor value_states_1_pad_0 = const()[name = string("value_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_1_dilations_0 = const()[name = string("value_states_1_dilations_0"), val = tensor([1, 1])]; int32 value_states_1_groups_0 = const()[name = string("value_states_1_groups_0"), val = int32(1)]; tensor value_states_1_cast_fp16 = conv(dilations = value_states_1_dilations_0, groups = value_states_1_groups_0, pad = value_states_1_pad_0, pad_type = value_states_1_pad_type_0, strides = value_states_1_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("value_states_1_cast_fp16")]; tensor concat_0x = const()[name = string("concat_0x"), val = tensor([1, 16, 128, -1])]; tensor x_1_cast_fp16 = reshape(shape = concat_0x, x = query_states_1_cast_fp16)[name = string("x_1_cast_fp16")]; tensor concat_1x = const()[name = string("concat_1x"), val = tensor([1, 2, 128, -1])]; tensor var_984_cast_fp16 = reshape(shape = concat_1x, x = key_states_1_cast_fp16)[name = string("op_984_cast_fp16")]; tensor concat_2x = const()[name = string("concat_2x"), val = tensor([1, 2, 128, -1])]; tensor var_991_cast_fp16 = reshape(shape = concat_2x, x = value_states_1_cast_fp16)[name = string("op_991_cast_fp16")]; tensor var_995_cast_fp16 = mul(x = x_1_cast_fp16, y = var_869_cast_fp16)[name = string("op_995_cast_fp16")]; tensor var_996_split_sizes_0 = const()[name = string("op_996_split_sizes_0"), val = tensor([64, 64])]; int32 var_996_axis_0 = const()[name = string("op_996_axis_0"), val = int32(-2)]; tensor var_996_cast_fp16_0, tensor var_996_cast_fp16_1 = split(axis = var_996_axis_0, split_sizes = var_996_split_sizes_0, x = x_1_cast_fp16)[name = string("op_996_cast_fp16")]; fp16 const_2_promoted_to_fp16 = const()[name = string("const_2_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_998_cast_fp16 = mul(x = var_996_cast_fp16_1, y = const_2_promoted_to_fp16)[name = string("op_998_cast_fp16")]; int32 var_1000 = const()[name = string("op_1000"), val = int32(-2)]; bool var_1001_interleave_0 = const()[name = string("op_1001_interleave_0"), val = bool(false)]; tensor var_1001_cast_fp16 = concat(axis = var_1000, interleave = var_1001_interleave_0, values = (var_998_cast_fp16, var_996_cast_fp16_0))[name = string("op_1001_cast_fp16")]; tensor var_1002_cast_fp16 = mul(x = var_1001_cast_fp16, y = var_878_cast_fp16)[name = string("op_1002_cast_fp16")]; tensor query_states_3_cast_fp16 = add(x = var_995_cast_fp16, y = var_1002_cast_fp16)[name = string("query_states_3_cast_fp16")]; tensor var_1008_cast_fp16 = mul(x = var_984_cast_fp16, y = var_869_cast_fp16)[name = string("op_1008_cast_fp16")]; tensor var_1009_split_sizes_0 = const()[name = string("op_1009_split_sizes_0"), val = tensor([64, 64])]; int32 var_1009_axis_0 = const()[name = string("op_1009_axis_0"), val = int32(-2)]; tensor var_1009_cast_fp16_0, tensor var_1009_cast_fp16_1 = split(axis = var_1009_axis_0, split_sizes = var_1009_split_sizes_0, x = var_984_cast_fp16)[name = string("op_1009_cast_fp16")]; fp16 const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1011_cast_fp16 = mul(x = var_1009_cast_fp16_1, y = const_3_promoted_to_fp16)[name = string("op_1011_cast_fp16")]; int32 var_1013 = const()[name = string("op_1013"), val = int32(-2)]; bool var_1014_interleave_0 = const()[name = string("op_1014_interleave_0"), val = bool(false)]; tensor var_1014_cast_fp16 = concat(axis = var_1013, interleave = var_1014_interleave_0, values = (var_1011_cast_fp16, var_1009_cast_fp16_0))[name = string("op_1014_cast_fp16")]; tensor var_1015_cast_fp16 = mul(x = var_1014_cast_fp16, y = var_878_cast_fp16)[name = string("op_1015_cast_fp16")]; tensor key_states_5_cast_fp16 = add(x = var_1008_cast_fp16, y = var_1015_cast_fp16)[name = string("key_states_5_cast_fp16")]; tensor read_state_0 = read_state(input = key_cache)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; int32 concat_5_axis_0 = const()[name = string("concat_5_axis_0"), val = int32(0)]; bool concat_5_interleave_0 = const()[name = string("concat_5_interleave_0"), val = bool(false)]; tensor concat_5 = concat(axis = concat_5_axis_0, interleave = concat_5_interleave_0, values = (expand_dims_0, expand_dims_1, position_id, expand_dims_3))[name = string("concat_5")]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; tensor concat_6_values1_0 = const()[name = string("concat_6_values1_0"), val = tensor([0])]; tensor concat_6_values3_0 = const()[name = string("concat_6_values3_0"), val = tensor([0])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_4, concat_6_values1_0, cache_position_end, concat_6_values3_0))[name = string("concat_6")]; tensor key_states_7_perm_0 = const()[name = string("key_states_7_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_1_stride_0 = const()[name = string("key_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_7_cast_fp16 = transpose(perm = key_states_7_perm_0, x = key_states_5_cast_fp16)[name = string("transpose_341")]; tensor key_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = key_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = key_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_1_squeeze_mask_0, stride = key_cache_internal_tensor_assign_1_stride_0, update = key_states_7_cast_fp16, x = read_state_0)[name = string("key_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_1_cast_fp16, input = key_cache)[name = string("coreml_update_state_168_write_state")]; tensor coreml_update_state_168 = read_state(input = key_cache)[name = string("coreml_update_state_168")]; tensor read_state_1 = read_state(input = value_cache)[name = string("read_state_1")]; tensor value_states_3_perm_0 = const()[name = string("value_states_3_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_1_stride_0 = const()[name = string("value_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_3_cast_fp16 = transpose(perm = value_states_3_perm_0, x = var_991_cast_fp16)[name = string("transpose_340")]; tensor value_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = value_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = value_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_1_squeeze_mask_0, stride = value_cache_internal_tensor_assign_1_stride_0, update = value_states_3_cast_fp16, x = read_state_1)[name = string("value_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_1_cast_fp16, input = value_cache)[name = string("coreml_update_state_169_write_state")]; tensor coreml_update_state_169 = read_state(input = value_cache)[name = string("coreml_update_state_169")]; tensor var_1085_begin_0 = const()[name = string("op_1085_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1085_end_0 = const()[name = string("op_1085_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1085_end_mask_0 = const()[name = string("op_1085_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1085_cast_fp16 = slice_by_index(begin = var_1085_begin_0, end = var_1085_end_0, end_mask = var_1085_end_mask_0, x = coreml_update_state_168)[name = string("op_1085_cast_fp16")]; tensor tile_0 = const()[name = string("tile_0"), val = tensor([1, 1])]; int32 var_1088_axis_0 = const()[name = string("op_1088_axis_0"), val = int32(1)]; tensor var_1088_cast_fp16_0, tensor var_1088_cast_fp16_1 = split(axis = var_1088_axis_0, split_sizes = tile_0, x = var_1085_cast_fp16)[name = string("op_1088_cast_fp16")]; tensor var_1095_begin_0 = const()[name = string("op_1095_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1095_end_0 = const()[name = string("op_1095_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1095_end_mask_0 = const()[name = string("op_1095_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1095_cast_fp16 = slice_by_index(begin = var_1095_begin_0, end = var_1095_end_0, end_mask = var_1095_end_mask_0, x = coreml_update_state_169)[name = string("op_1095_cast_fp16")]; tensor tile_1 = const()[name = string("tile_1"), val = tensor([1, 1])]; int32 var_1098_axis_0 = const()[name = string("op_1098_axis_0"), val = int32(1)]; tensor var_1098_cast_fp16_0, tensor var_1098_cast_fp16_1 = split(axis = var_1098_axis_0, split_sizes = tile_1, x = var_1095_cast_fp16)[name = string("op_1098_cast_fp16")]; tensor var_1101_split_sizes_0 = const()[name = string("op_1101_split_sizes_0"), val = tensor([8, 8])]; int32 var_1101_axis_0 = const()[name = string("op_1101_axis_0"), val = int32(1)]; tensor var_1101_0, tensor var_1101_1 = split(axis = var_1101_axis_0, split_sizes = var_1101_split_sizes_0, x = query_states_3_cast_fp16)[name = string("op_1101")]; bool attn_weights_1_transpose_x_0 = const()[name = string("attn_weights_1_transpose_x_0"), val = bool(false)]; bool attn_weights_1_transpose_y_0 = const()[name = string("attn_weights_1_transpose_y_0"), val = bool(false)]; tensor attn_weights_1_cast_fp16 = matmul(transpose_x = attn_weights_1_transpose_x_0, transpose_y = attn_weights_1_transpose_y_0, x = var_1088_cast_fp16_0, y = var_1101_0)[name = string("attn_weights_1_cast_fp16")]; fp16 var_1104_to_fp16 = const()[name = string("op_1104_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_3_cast_fp16 = mul(x = attn_weights_1_cast_fp16, y = var_1104_to_fp16)[name = string("attn_weights_3_cast_fp16")]; tensor attn_weights_5_cast_fp16 = add(x = attn_weights_3_cast_fp16, y = attn_mask_1)[name = string("attn_weights_5_cast_fp16")]; int32 var_1108 = const()[name = string("op_1108"), val = int32(-2)]; tensor attn_weights_7_cast_fp16 = softmax(axis = var_1108, x = attn_weights_5_cast_fp16)[name = string("attn_weights_7_cast_fp16")]; bool var_1114_transpose_x_1 = const()[name = string("op_1114_transpose_x_1"), val = bool(true)]; bool var_1114_transpose_y_1 = const()[name = string("op_1114_transpose_y_1"), val = bool(false)]; tensor var_1114_cast_fp16 = matmul(transpose_x = var_1114_transpose_x_1, transpose_y = var_1114_transpose_y_1, x = attn_weights_7_cast_fp16, y = var_1098_cast_fp16_0)[name = string("op_1114_cast_fp16")]; bool attn_weights_9_transpose_x_0 = const()[name = string("attn_weights_9_transpose_x_0"), val = bool(false)]; bool attn_weights_9_transpose_y_0 = const()[name = string("attn_weights_9_transpose_y_0"), val = bool(false)]; tensor attn_weights_9_cast_fp16 = matmul(transpose_x = attn_weights_9_transpose_x_0, transpose_y = attn_weights_9_transpose_y_0, x = var_1088_cast_fp16_1, y = var_1101_1)[name = string("attn_weights_9_cast_fp16")]; fp16 var_1116_to_fp16 = const()[name = string("op_1116_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_11_cast_fp16 = mul(x = attn_weights_9_cast_fp16, y = var_1116_to_fp16)[name = string("attn_weights_11_cast_fp16")]; tensor attn_weights_13_cast_fp16 = add(x = attn_weights_11_cast_fp16, y = attn_mask_1)[name = string("attn_weights_13_cast_fp16")]; int32 var_1120 = const()[name = string("op_1120"), val = int32(-2)]; tensor attn_weights_15_cast_fp16 = softmax(axis = var_1120, x = attn_weights_13_cast_fp16)[name = string("attn_weights_15_cast_fp16")]; bool attn_output_1_transpose_x_1 = const()[name = string("attn_output_1_transpose_x_1"), val = bool(true)]; bool attn_output_1_transpose_y_1 = const()[name = string("attn_output_1_transpose_y_1"), val = bool(false)]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_1, transpose_y = attn_output_1_transpose_y_1, x = attn_weights_15_cast_fp16, y = var_1098_cast_fp16_1)[name = string("attn_output_1_cast_fp16")]; int32 var_1128 = const()[name = string("op_1128"), val = int32(1)]; bool attn_output_3_interleave_0 = const()[name = string("attn_output_3_interleave_0"), val = bool(false)]; tensor attn_output_3_cast_fp16 = concat(axis = var_1128, interleave = attn_output_3_interleave_0, values = (var_1114_cast_fp16, attn_output_1_cast_fp16))[name = string("attn_output_3_cast_fp16")]; tensor var_1132_perm_0 = const()[name = string("op_1132_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_11x = const()[name = string("concat_11x"), val = tensor([1, 2048, 1, -1])]; tensor var_1132_cast_fp16 = transpose(perm = var_1132_perm_0, x = attn_output_3_cast_fp16)[name = string("transpose_339")]; tensor attn_output_7_cast_fp16 = reshape(shape = concat_11x, x = var_1132_cast_fp16)[name = string("attn_output_7_cast_fp16")]; tensor layers_0_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(993654976)))]; tensor hidden_states_3_strides_0 = const()[name = string("hidden_states_3_strides_0"), val = tensor([1, 1])]; string hidden_states_3_pad_type_0 = const()[name = string("hidden_states_3_pad_type_0"), val = string("valid")]; tensor hidden_states_3_pad_0 = const()[name = string("hidden_states_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_3_dilations_0 = const()[name = string("hidden_states_3_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_3_groups_0 = const()[name = string("hidden_states_3_groups_0"), val = int32(1)]; tensor hidden_states_3_cast_fp16 = conv(dilations = hidden_states_3_dilations_0, groups = hidden_states_3_groups_0, pad = hidden_states_3_pad_0, pad_type = hidden_states_3_pad_type_0, strides = hidden_states_3_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16, x = attn_output_7_cast_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor hidden_states_5_cast_fp16 = add(x = inputs_embeds_to_fp16, y = hidden_states_3_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1165_cast_fp16 = mul(x = hidden_states_5_cast_fp16, y = const_8_promoted_to_fp16)[name = string("op_1165_cast_fp16")]; int32 var_1163 = const()[name = string("op_1163"), val = int32(1)]; bool doubled_5_interleave_0 = const()[name = string("doubled_5_interleave_0"), val = bool(false)]; tensor doubled_5_cast_fp16 = concat(axis = var_1163, interleave = doubled_5_interleave_0, values = (hidden_states_5_cast_fp16, var_1165_cast_fp16))[name = string("doubled_5_cast_fp16")]; tensor out_3_axes_0 = const()[name = string("out_3_axes_0"), val = tensor([1])]; tensor out_3_gamma_0_to_fp16 = const()[name = string("out_3_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002043648)))]; fp16 var_1175_to_fp16 = const()[name = string("op_1175_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_3_cast_fp16 = layer_norm(axes = out_3_axes_0, epsilon = var_1175_to_fp16, gamma = out_3_gamma_0_to_fp16, x = doubled_5_cast_fp16)[name = string("out_3_cast_fp16")]; tensor var_1186_split_sizes_0 = const()[name = string("op_1186_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1186_axis_0 = const()[name = string("op_1186_axis_0"), val = int32(1)]; tensor var_1186_cast_fp16_0, tensor var_1186_cast_fp16_1 = split(axis = var_1186_axis_0, split_sizes = var_1186_split_sizes_0, x = out_3_cast_fp16)[name = string("op_1186_cast_fp16")]; tensor layers_0_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002051904)))]; tensor input_1_strides_0 = const()[name = string("input_1_strides_0"), val = tensor([1, 1])]; string input_1_pad_type_0 = const()[name = string("input_1_pad_type_0"), val = string("valid")]; tensor input_1_pad_0 = const()[name = string("input_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_1_dilations_0 = const()[name = string("input_1_dilations_0"), val = tensor([1, 1])]; int32 input_1_groups_0 = const()[name = string("input_1_groups_0"), val = int32(1)]; tensor input_1_cast_fp16 = conv(dilations = input_1_dilations_0, groups = input_1_groups_0, pad = input_1_pad_0, pad_type = input_1_pad_type_0, strides = input_1_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("input_1_cast_fp16")]; tensor var_1203_cast_fp16 = silu(x = input_1_cast_fp16)[name = string("op_1203_cast_fp16")]; tensor layers_0_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1027217792)))]; tensor var_1209_strides_0 = const()[name = string("op_1209_strides_0"), val = tensor([1, 1])]; string var_1209_pad_type_0 = const()[name = string("op_1209_pad_type_0"), val = string("valid")]; tensor var_1209_pad_0 = const()[name = string("op_1209_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1209_dilations_0 = const()[name = string("op_1209_dilations_0"), val = tensor([1, 1])]; int32 var_1209_groups_0 = const()[name = string("op_1209_groups_0"), val = int32(1)]; tensor var_1209_cast_fp16 = conv(dilations = var_1209_dilations_0, groups = var_1209_groups_0, pad = var_1209_pad_0, pad_type = var_1209_pad_type_0, strides = var_1209_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("op_1209_cast_fp16")]; tensor x_9_cast_fp16 = mul(x = var_1203_cast_fp16, y = var_1209_cast_fp16)[name = string("x_9_cast_fp16")]; tensor layers_0_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1052383680)))]; tensor hidden_states_7_strides_0 = const()[name = string("hidden_states_7_strides_0"), val = tensor([1, 1])]; string hidden_states_7_pad_type_0 = const()[name = string("hidden_states_7_pad_type_0"), val = string("valid")]; tensor hidden_states_7_pad_0 = const()[name = string("hidden_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_7_dilations_0 = const()[name = string("hidden_states_7_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_7_groups_0 = const()[name = string("hidden_states_7_groups_0"), val = int32(1)]; tensor hidden_states_7_cast_fp16 = conv(dilations = hidden_states_7_dilations_0, groups = hidden_states_7_groups_0, pad = hidden_states_7_pad_0, pad_type = hidden_states_7_pad_type_0, strides = hidden_states_7_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16, x = x_9_cast_fp16)[name = string("hidden_states_7_cast_fp16")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = hidden_states_7_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; fp16 const_10_promoted_to_fp16 = const()[name = string("const_10_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1227_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_10_promoted_to_fp16)[name = string("op_1227_cast_fp16")]; int32 var_1225 = const()[name = string("op_1225"), val = int32(1)]; bool doubled_9_interleave_0 = const()[name = string("doubled_9_interleave_0"), val = bool(false)]; tensor doubled_9_cast_fp16 = concat(axis = var_1225, interleave = doubled_9_interleave_0, values = (hidden_states_9_cast_fp16, var_1227_cast_fp16))[name = string("doubled_9_cast_fp16")]; tensor out_5_axes_0 = const()[name = string("out_5_axes_0"), val = tensor([1])]; tensor out_5_gamma_0_to_fp16 = const()[name = string("out_5_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077549568)))]; fp16 var_1237_to_fp16 = const()[name = string("op_1237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_5_cast_fp16 = layer_norm(axes = out_5_axes_0, epsilon = var_1237_to_fp16, gamma = out_5_gamma_0_to_fp16, x = doubled_9_cast_fp16)[name = string("out_5_cast_fp16")]; tensor var_1248_split_sizes_0 = const()[name = string("op_1248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1248_axis_0 = const()[name = string("op_1248_axis_0"), val = int32(1)]; tensor var_1248_cast_fp16_0, tensor var_1248_cast_fp16_1 = split(axis = var_1248_axis_0, split_sizes = var_1248_split_sizes_0, x = out_5_cast_fp16)[name = string("op_1248_cast_fp16")]; tensor layers_1_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077557824)))]; tensor query_states_7_strides_0 = const()[name = string("query_states_7_strides_0"), val = tensor([1, 1])]; string query_states_7_pad_type_0 = const()[name = string("query_states_7_pad_type_0"), val = string("valid")]; tensor query_states_7_pad_0 = const()[name = string("query_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_7_dilations_0 = const()[name = string("query_states_7_dilations_0"), val = tensor([1, 1])]; int32 query_states_7_groups_0 = const()[name = string("query_states_7_groups_0"), val = int32(1)]; tensor query_states_7_cast_fp16 = conv(dilations = query_states_7_dilations_0, groups = query_states_7_groups_0, pad = query_states_7_pad_0, pad_type = query_states_7_pad_type_0, strides = query_states_7_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("query_states_7_cast_fp16")]; tensor layers_1_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1085946496)))]; tensor key_states_11_strides_0 = const()[name = string("key_states_11_strides_0"), val = tensor([1, 1])]; string key_states_11_pad_type_0 = const()[name = string("key_states_11_pad_type_0"), val = string("valid")]; tensor key_states_11_pad_0 = const()[name = string("key_states_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_11_dilations_0 = const()[name = string("key_states_11_dilations_0"), val = tensor([1, 1])]; int32 key_states_11_groups_0 = const()[name = string("key_states_11_groups_0"), val = int32(1)]; tensor key_states_11_cast_fp16 = conv(dilations = key_states_11_dilations_0, groups = key_states_11_groups_0, pad = key_states_11_pad_0, pad_type = key_states_11_pad_type_0, strides = key_states_11_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("key_states_11_cast_fp16")]; tensor value_states_7_strides_0 = const()[name = string("value_states_7_strides_0"), val = tensor([1, 1])]; string value_states_7_pad_type_0 = const()[name = string("value_states_7_pad_type_0"), val = string("valid")]; tensor value_states_7_pad_0 = const()[name = string("value_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_7_dilations_0 = const()[name = string("value_states_7_dilations_0"), val = tensor([1, 1])]; int32 value_states_7_groups_0 = const()[name = string("value_states_7_groups_0"), val = int32(1)]; tensor value_states_7_cast_fp16 = conv(dilations = value_states_7_dilations_0, groups = value_states_7_groups_0, pad = value_states_7_pad_0, pad_type = value_states_7_pad_type_0, strides = value_states_7_strides_0, weight = layers_1_self_attn_v_proj_weight_cast_fp16, x = var_1248_cast_fp16_0)[name = string("value_states_7_cast_fp16")]; tensor concat_12x = const()[name = string("concat_12x"), val = tensor([1, 16, 128, -1])]; tensor x_11_cast_fp16 = reshape(shape = concat_12x, x = query_states_7_cast_fp16)[name = string("x_11_cast_fp16")]; tensor concat_13x = const()[name = string("concat_13x"), val = tensor([1, 2, 128, -1])]; tensor var_1305_cast_fp16 = reshape(shape = concat_13x, x = key_states_11_cast_fp16)[name = string("op_1305_cast_fp16")]; tensor concat_14x = const()[name = string("concat_14x"), val = tensor([1, 2, 128, -1])]; tensor var_1312_cast_fp16 = reshape(shape = concat_14x, x = value_states_7_cast_fp16)[name = string("op_1312_cast_fp16")]; tensor var_1316_cast_fp16 = mul(x = x_11_cast_fp16, y = var_869_cast_fp16)[name = string("op_1316_cast_fp16")]; tensor var_1317_split_sizes_0 = const()[name = string("op_1317_split_sizes_0"), val = tensor([64, 64])]; int32 var_1317_axis_0 = const()[name = string("op_1317_axis_0"), val = int32(-2)]; tensor var_1317_cast_fp16_0, tensor var_1317_cast_fp16_1 = split(axis = var_1317_axis_0, split_sizes = var_1317_split_sizes_0, x = x_11_cast_fp16)[name = string("op_1317_cast_fp16")]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1319_cast_fp16 = mul(x = var_1317_cast_fp16_1, y = const_12_promoted_to_fp16)[name = string("op_1319_cast_fp16")]; int32 var_1321 = const()[name = string("op_1321"), val = int32(-2)]; bool var_1322_interleave_0 = const()[name = string("op_1322_interleave_0"), val = bool(false)]; tensor var_1322_cast_fp16 = concat(axis = var_1321, interleave = var_1322_interleave_0, values = (var_1319_cast_fp16, var_1317_cast_fp16_0))[name = string("op_1322_cast_fp16")]; tensor var_1323_cast_fp16 = mul(x = var_1322_cast_fp16, y = var_878_cast_fp16)[name = string("op_1323_cast_fp16")]; tensor query_states_9_cast_fp16 = add(x = var_1316_cast_fp16, y = var_1323_cast_fp16)[name = string("query_states_9_cast_fp16")]; tensor var_1329_cast_fp16 = mul(x = var_1305_cast_fp16, y = var_869_cast_fp16)[name = string("op_1329_cast_fp16")]; tensor var_1330_split_sizes_0 = const()[name = string("op_1330_split_sizes_0"), val = tensor([64, 64])]; int32 var_1330_axis_0 = const()[name = string("op_1330_axis_0"), val = int32(-2)]; tensor var_1330_cast_fp16_0, tensor var_1330_cast_fp16_1 = split(axis = var_1330_axis_0, split_sizes = var_1330_split_sizes_0, x = var_1305_cast_fp16)[name = string("op_1330_cast_fp16")]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1332_cast_fp16 = mul(x = var_1330_cast_fp16_1, y = const_13_promoted_to_fp16)[name = string("op_1332_cast_fp16")]; int32 var_1334 = const()[name = string("op_1334"), val = int32(-2)]; bool var_1335_interleave_0 = const()[name = string("op_1335_interleave_0"), val = bool(false)]; tensor var_1335_cast_fp16 = concat(axis = var_1334, interleave = var_1335_interleave_0, values = (var_1332_cast_fp16, var_1330_cast_fp16_0))[name = string("op_1335_cast_fp16")]; tensor var_1336_cast_fp16 = mul(x = var_1335_cast_fp16, y = var_878_cast_fp16)[name = string("op_1336_cast_fp16")]; tensor key_states_15_cast_fp16 = add(x = var_1329_cast_fp16, y = var_1336_cast_fp16)[name = string("key_states_15_cast_fp16")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; int32 concat_17_axis_0 = const()[name = string("concat_17_axis_0"), val = int32(0)]; bool concat_17_interleave_0 = const()[name = string("concat_17_interleave_0"), val = bool(false)]; tensor concat_17 = concat(axis = concat_17_axis_0, interleave = concat_17_interleave_0, values = (expand_dims_12, expand_dims_13, position_id, expand_dims_15))[name = string("concat_17")]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; tensor concat_18_values1_0 = const()[name = string("concat_18_values1_0"), val = tensor([0])]; tensor concat_18_values3_0 = const()[name = string("concat_18_values3_0"), val = tensor([0])]; int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_16, concat_18_values1_0, cache_position_end, concat_18_values3_0))[name = string("concat_18")]; tensor key_states_17_perm_0 = const()[name = string("key_states_17_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_2_stride_0 = const()[name = string("key_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_17_cast_fp16 = transpose(perm = key_states_17_perm_0, x = key_states_15_cast_fp16)[name = string("transpose_338")]; tensor key_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = key_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = key_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_2_squeeze_mask_0, stride = key_cache_internal_tensor_assign_2_stride_0, update = key_states_17_cast_fp16, x = coreml_update_state_168)[name = string("key_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_2_cast_fp16, input = key_cache)[name = string("coreml_update_state_170_write_state")]; tensor coreml_update_state_170 = read_state(input = key_cache)[name = string("coreml_update_state_170")]; tensor value_states_9_perm_0 = const()[name = string("value_states_9_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_2_stride_0 = const()[name = string("value_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_9_cast_fp16 = transpose(perm = value_states_9_perm_0, x = var_1312_cast_fp16)[name = string("transpose_337")]; tensor value_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = value_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = value_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_2_squeeze_mask_0, stride = value_cache_internal_tensor_assign_2_stride_0, update = value_states_9_cast_fp16, x = coreml_update_state_169)[name = string("value_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_2_cast_fp16, input = value_cache)[name = string("coreml_update_state_171_write_state")]; tensor coreml_update_state_171 = read_state(input = value_cache)[name = string("coreml_update_state_171")]; tensor var_1406_begin_0 = const()[name = string("op_1406_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1406_end_0 = const()[name = string("op_1406_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1406_end_mask_0 = const()[name = string("op_1406_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1406_cast_fp16 = slice_by_index(begin = var_1406_begin_0, end = var_1406_end_0, end_mask = var_1406_end_mask_0, x = coreml_update_state_170)[name = string("op_1406_cast_fp16")]; tensor tile_2 = const()[name = string("tile_2"), val = tensor([1, 1])]; int32 var_1409_axis_0 = const()[name = string("op_1409_axis_0"), val = int32(1)]; tensor var_1409_cast_fp16_0, tensor var_1409_cast_fp16_1 = split(axis = var_1409_axis_0, split_sizes = tile_2, x = var_1406_cast_fp16)[name = string("op_1409_cast_fp16")]; tensor var_1416_begin_0 = const()[name = string("op_1416_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1416_end_0 = const()[name = string("op_1416_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1416_end_mask_0 = const()[name = string("op_1416_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1416_cast_fp16 = slice_by_index(begin = var_1416_begin_0, end = var_1416_end_0, end_mask = var_1416_end_mask_0, x = coreml_update_state_171)[name = string("op_1416_cast_fp16")]; tensor tile_3 = const()[name = string("tile_3"), val = tensor([1, 1])]; int32 var_1419_axis_0 = const()[name = string("op_1419_axis_0"), val = int32(1)]; tensor var_1419_cast_fp16_0, tensor var_1419_cast_fp16_1 = split(axis = var_1419_axis_0, split_sizes = tile_3, x = var_1416_cast_fp16)[name = string("op_1419_cast_fp16")]; tensor var_1422_split_sizes_0 = const()[name = string("op_1422_split_sizes_0"), val = tensor([8, 8])]; int32 var_1422_axis_0 = const()[name = string("op_1422_axis_0"), val = int32(1)]; tensor var_1422_0, tensor var_1422_1 = split(axis = var_1422_axis_0, split_sizes = var_1422_split_sizes_0, x = query_states_9_cast_fp16)[name = string("op_1422")]; bool attn_weights_17_transpose_x_0 = const()[name = string("attn_weights_17_transpose_x_0"), val = bool(false)]; bool attn_weights_17_transpose_y_0 = const()[name = string("attn_weights_17_transpose_y_0"), val = bool(false)]; tensor attn_weights_17_cast_fp16 = matmul(transpose_x = attn_weights_17_transpose_x_0, transpose_y = attn_weights_17_transpose_y_0, x = var_1409_cast_fp16_0, y = var_1422_0)[name = string("attn_weights_17_cast_fp16")]; fp16 var_1425_to_fp16 = const()[name = string("op_1425_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_19_cast_fp16 = mul(x = attn_weights_17_cast_fp16, y = var_1425_to_fp16)[name = string("attn_weights_19_cast_fp16")]; tensor attn_weights_21_cast_fp16 = add(x = attn_weights_19_cast_fp16, y = attn_mask_1)[name = string("attn_weights_21_cast_fp16")]; int32 var_1429 = const()[name = string("op_1429"), val = int32(-2)]; tensor attn_weights_23_cast_fp16 = softmax(axis = var_1429, x = attn_weights_21_cast_fp16)[name = string("attn_weights_23_cast_fp16")]; bool var_1435_transpose_x_1 = const()[name = string("op_1435_transpose_x_1"), val = bool(true)]; bool var_1435_transpose_y_1 = const()[name = string("op_1435_transpose_y_1"), val = bool(false)]; tensor var_1435_cast_fp16 = matmul(transpose_x = var_1435_transpose_x_1, transpose_y = var_1435_transpose_y_1, x = attn_weights_23_cast_fp16, y = var_1419_cast_fp16_0)[name = string("op_1435_cast_fp16")]; bool attn_weights_25_transpose_x_0 = const()[name = string("attn_weights_25_transpose_x_0"), val = bool(false)]; bool attn_weights_25_transpose_y_0 = const()[name = string("attn_weights_25_transpose_y_0"), val = bool(false)]; tensor attn_weights_25_cast_fp16 = matmul(transpose_x = attn_weights_25_transpose_x_0, transpose_y = attn_weights_25_transpose_y_0, x = var_1409_cast_fp16_1, y = var_1422_1)[name = string("attn_weights_25_cast_fp16")]; fp16 var_1437_to_fp16 = const()[name = string("op_1437_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_27_cast_fp16 = mul(x = attn_weights_25_cast_fp16, y = var_1437_to_fp16)[name = string("attn_weights_27_cast_fp16")]; tensor attn_weights_29_cast_fp16 = add(x = attn_weights_27_cast_fp16, y = attn_mask_1)[name = string("attn_weights_29_cast_fp16")]; int32 var_1441 = const()[name = string("op_1441"), val = int32(-2)]; tensor attn_weights_31_cast_fp16 = softmax(axis = var_1441, x = attn_weights_29_cast_fp16)[name = string("attn_weights_31_cast_fp16")]; bool attn_output_9_transpose_x_1 = const()[name = string("attn_output_9_transpose_x_1"), val = bool(true)]; bool attn_output_9_transpose_y_1 = const()[name = string("attn_output_9_transpose_y_1"), val = bool(false)]; tensor attn_output_9_cast_fp16 = matmul(transpose_x = attn_output_9_transpose_x_1, transpose_y = attn_output_9_transpose_y_1, x = attn_weights_31_cast_fp16, y = var_1419_cast_fp16_1)[name = string("attn_output_9_cast_fp16")]; int32 var_1449 = const()[name = string("op_1449"), val = int32(1)]; bool attn_output_11_interleave_0 = const()[name = string("attn_output_11_interleave_0"), val = bool(false)]; tensor attn_output_11_cast_fp16 = concat(axis = var_1449, interleave = attn_output_11_interleave_0, values = (var_1435_cast_fp16, attn_output_9_cast_fp16))[name = string("attn_output_11_cast_fp16")]; tensor var_1453_perm_0 = const()[name = string("op_1453_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_23x = const()[name = string("concat_23x"), val = tensor([1, 2048, 1, -1])]; tensor var_1453_cast_fp16 = transpose(perm = var_1453_perm_0, x = attn_output_11_cast_fp16)[name = string("transpose_336")]; tensor attn_output_15_cast_fp16 = reshape(shape = concat_23x, x = var_1453_cast_fp16)[name = string("attn_output_15_cast_fp16")]; tensor layers_1_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086995136)))]; tensor hidden_states_13_strides_0 = const()[name = string("hidden_states_13_strides_0"), val = tensor([1, 1])]; string hidden_states_13_pad_type_0 = const()[name = string("hidden_states_13_pad_type_0"), val = string("valid")]; tensor hidden_states_13_pad_0 = const()[name = string("hidden_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_13_dilations_0 = const()[name = string("hidden_states_13_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_13_groups_0 = const()[name = string("hidden_states_13_groups_0"), val = int32(1)]; tensor hidden_states_13_cast_fp16 = conv(dilations = hidden_states_13_dilations_0, groups = hidden_states_13_groups_0, pad = hidden_states_13_pad_0, pad_type = hidden_states_13_pad_type_0, strides = hidden_states_13_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16, x = attn_output_15_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; tensor hidden_states_15_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = hidden_states_13_cast_fp16)[name = string("hidden_states_15_cast_fp16")]; fp16 const_18_promoted_to_fp16 = const()[name = string("const_18_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1486_cast_fp16 = mul(x = hidden_states_15_cast_fp16, y = const_18_promoted_to_fp16)[name = string("op_1486_cast_fp16")]; int32 var_1484 = const()[name = string("op_1484"), val = int32(1)]; bool doubled_13_interleave_0 = const()[name = string("doubled_13_interleave_0"), val = bool(false)]; tensor doubled_13_cast_fp16 = concat(axis = var_1484, interleave = doubled_13_interleave_0, values = (hidden_states_15_cast_fp16, var_1486_cast_fp16))[name = string("doubled_13_cast_fp16")]; tensor out_7_axes_0 = const()[name = string("out_7_axes_0"), val = tensor([1])]; tensor out_7_gamma_0_to_fp16 = const()[name = string("out_7_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095383808)))]; fp16 var_1496_to_fp16 = const()[name = string("op_1496_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_7_cast_fp16 = layer_norm(axes = out_7_axes_0, epsilon = var_1496_to_fp16, gamma = out_7_gamma_0_to_fp16, x = doubled_13_cast_fp16)[name = string("out_7_cast_fp16")]; tensor var_1507_split_sizes_0 = const()[name = string("op_1507_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1507_axis_0 = const()[name = string("op_1507_axis_0"), val = int32(1)]; tensor var_1507_cast_fp16_0, tensor var_1507_cast_fp16_1 = split(axis = var_1507_axis_0, split_sizes = var_1507_split_sizes_0, x = out_7_cast_fp16)[name = string("op_1507_cast_fp16")]; tensor layers_1_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095392064)))]; tensor input_3_strides_0 = const()[name = string("input_3_strides_0"), val = tensor([1, 1])]; string input_3_pad_type_0 = const()[name = string("input_3_pad_type_0"), val = string("valid")]; tensor input_3_pad_0 = const()[name = string("input_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_3_dilations_0 = const()[name = string("input_3_dilations_0"), val = tensor([1, 1])]; int32 input_3_groups_0 = const()[name = string("input_3_groups_0"), val = int32(1)]; tensor input_3_cast_fp16 = conv(dilations = input_3_dilations_0, groups = input_3_groups_0, pad = input_3_pad_0, pad_type = input_3_pad_type_0, strides = input_3_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16, x = var_1507_cast_fp16_0)[name = string("input_3_cast_fp16")]; tensor var_1524_cast_fp16 = silu(x = input_3_cast_fp16)[name = string("op_1524_cast_fp16")]; tensor var_1530_strides_0 = const()[name = string("op_1530_strides_0"), val = tensor([1, 1])]; string var_1530_pad_type_0 = const()[name = string("op_1530_pad_type_0"), val = string("valid")]; tensor var_1530_pad_0 = const()[name = string("op_1530_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1530_dilations_0 = const()[name = string("op_1530_dilations_0"), val = tensor([1, 1])]; int32 var_1530_groups_0 = const()[name = string("op_1530_groups_0"), val = int32(1)]; tensor var_1530_cast_fp16 = conv(dilations = var_1530_dilations_0, groups = var_1530_groups_0, pad = var_1530_pad_0, pad_type = var_1530_pad_type_0, strides = var_1530_strides_0, weight = layers_1_mlp_up_proj_weight_cast_fp16, x = var_1507_cast_fp16_0)[name = string("op_1530_cast_fp16")]; tensor x_19_cast_fp16 = mul(x = var_1524_cast_fp16, y = var_1530_cast_fp16)[name = string("x_19_cast_fp16")]; tensor layers_1_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1120557952)))]; tensor hidden_states_17_strides_0 = const()[name = string("hidden_states_17_strides_0"), val = tensor([1, 1])]; string hidden_states_17_pad_type_0 = const()[name = string("hidden_states_17_pad_type_0"), val = string("valid")]; tensor hidden_states_17_pad_0 = const()[name = string("hidden_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_17_dilations_0 = const()[name = string("hidden_states_17_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_17_groups_0 = const()[name = string("hidden_states_17_groups_0"), val = int32(1)]; tensor hidden_states_17_cast_fp16 = conv(dilations = hidden_states_17_dilations_0, groups = hidden_states_17_groups_0, pad = hidden_states_17_pad_0, pad_type = hidden_states_17_pad_type_0, strides = hidden_states_17_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16, x = x_19_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; tensor hidden_states_19_cast_fp16 = add(x = hidden_states_15_cast_fp16, y = hidden_states_17_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1548_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_1548_cast_fp16")]; int32 var_1546 = const()[name = string("op_1546"), val = int32(1)]; bool doubled_17_interleave_0 = const()[name = string("doubled_17_interleave_0"), val = bool(false)]; tensor doubled_17_cast_fp16 = concat(axis = var_1546, interleave = doubled_17_interleave_0, values = (hidden_states_19_cast_fp16, var_1548_cast_fp16))[name = string("doubled_17_cast_fp16")]; tensor out_9_axes_0 = const()[name = string("out_9_axes_0"), val = tensor([1])]; tensor out_9_gamma_0_to_fp16 = const()[name = string("out_9_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145723840)))]; fp16 var_1558_to_fp16 = const()[name = string("op_1558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_9_cast_fp16 = layer_norm(axes = out_9_axes_0, epsilon = var_1558_to_fp16, gamma = out_9_gamma_0_to_fp16, x = doubled_17_cast_fp16)[name = string("out_9_cast_fp16")]; tensor var_1569_split_sizes_0 = const()[name = string("op_1569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1569_axis_0 = const()[name = string("op_1569_axis_0"), val = int32(1)]; tensor var_1569_cast_fp16_0, tensor var_1569_cast_fp16_1 = split(axis = var_1569_axis_0, split_sizes = var_1569_split_sizes_0, x = out_9_cast_fp16)[name = string("op_1569_cast_fp16")]; tensor layers_2_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145732096)))]; tensor query_states_13_strides_0 = const()[name = string("query_states_13_strides_0"), val = tensor([1, 1])]; string query_states_13_pad_type_0 = const()[name = string("query_states_13_pad_type_0"), val = string("valid")]; tensor query_states_13_pad_0 = const()[name = string("query_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_13_dilations_0 = const()[name = string("query_states_13_dilations_0"), val = tensor([1, 1])]; int32 query_states_13_groups_0 = const()[name = string("query_states_13_groups_0"), val = int32(1)]; tensor query_states_13_cast_fp16 = conv(dilations = query_states_13_dilations_0, groups = query_states_13_groups_0, pad = query_states_13_pad_0, pad_type = query_states_13_pad_type_0, strides = query_states_13_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("query_states_13_cast_fp16")]; tensor layers_2_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1154120768)))]; tensor key_states_21_strides_0 = const()[name = string("key_states_21_strides_0"), val = tensor([1, 1])]; string key_states_21_pad_type_0 = const()[name = string("key_states_21_pad_type_0"), val = string("valid")]; tensor key_states_21_pad_0 = const()[name = string("key_states_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_21_dilations_0 = const()[name = string("key_states_21_dilations_0"), val = tensor([1, 1])]; int32 key_states_21_groups_0 = const()[name = string("key_states_21_groups_0"), val = int32(1)]; tensor key_states_21_cast_fp16 = conv(dilations = key_states_21_dilations_0, groups = key_states_21_groups_0, pad = key_states_21_pad_0, pad_type = key_states_21_pad_type_0, strides = key_states_21_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("key_states_21_cast_fp16")]; tensor value_states_13_strides_0 = const()[name = string("value_states_13_strides_0"), val = tensor([1, 1])]; string value_states_13_pad_type_0 = const()[name = string("value_states_13_pad_type_0"), val = string("valid")]; tensor value_states_13_pad_0 = const()[name = string("value_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_13_dilations_0 = const()[name = string("value_states_13_dilations_0"), val = tensor([1, 1])]; int32 value_states_13_groups_0 = const()[name = string("value_states_13_groups_0"), val = int32(1)]; tensor value_states_13_cast_fp16 = conv(dilations = value_states_13_dilations_0, groups = value_states_13_groups_0, pad = value_states_13_pad_0, pad_type = value_states_13_pad_type_0, strides = value_states_13_strides_0, weight = layers_2_self_attn_v_proj_weight_cast_fp16, x = var_1569_cast_fp16_0)[name = string("value_states_13_cast_fp16")]; tensor concat_24x = const()[name = string("concat_24x"), val = tensor([1, 16, 128, -1])]; tensor x_21_cast_fp16 = reshape(shape = concat_24x, x = query_states_13_cast_fp16)[name = string("x_21_cast_fp16")]; tensor concat_25x = const()[name = string("concat_25x"), val = tensor([1, 2, 128, -1])]; tensor var_1626_cast_fp16 = reshape(shape = concat_25x, x = key_states_21_cast_fp16)[name = string("op_1626_cast_fp16")]; tensor concat_26x = const()[name = string("concat_26x"), val = tensor([1, 2, 128, -1])]; tensor var_1633_cast_fp16 = reshape(shape = concat_26x, x = value_states_13_cast_fp16)[name = string("op_1633_cast_fp16")]; tensor var_1637_cast_fp16 = mul(x = x_21_cast_fp16, y = var_869_cast_fp16)[name = string("op_1637_cast_fp16")]; tensor var_1638_split_sizes_0 = const()[name = string("op_1638_split_sizes_0"), val = tensor([64, 64])]; int32 var_1638_axis_0 = const()[name = string("op_1638_axis_0"), val = int32(-2)]; tensor var_1638_cast_fp16_0, tensor var_1638_cast_fp16_1 = split(axis = var_1638_axis_0, split_sizes = var_1638_split_sizes_0, x = x_21_cast_fp16)[name = string("op_1638_cast_fp16")]; fp16 const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1640_cast_fp16 = mul(x = var_1638_cast_fp16_1, y = const_22_promoted_to_fp16)[name = string("op_1640_cast_fp16")]; int32 var_1642 = const()[name = string("op_1642"), val = int32(-2)]; bool var_1643_interleave_0 = const()[name = string("op_1643_interleave_0"), val = bool(false)]; tensor var_1643_cast_fp16 = concat(axis = var_1642, interleave = var_1643_interleave_0, values = (var_1640_cast_fp16, var_1638_cast_fp16_0))[name = string("op_1643_cast_fp16")]; tensor var_1644_cast_fp16 = mul(x = var_1643_cast_fp16, y = var_878_cast_fp16)[name = string("op_1644_cast_fp16")]; tensor query_states_15_cast_fp16 = add(x = var_1637_cast_fp16, y = var_1644_cast_fp16)[name = string("query_states_15_cast_fp16")]; tensor var_1650_cast_fp16 = mul(x = var_1626_cast_fp16, y = var_869_cast_fp16)[name = string("op_1650_cast_fp16")]; tensor var_1651_split_sizes_0 = const()[name = string("op_1651_split_sizes_0"), val = tensor([64, 64])]; int32 var_1651_axis_0 = const()[name = string("op_1651_axis_0"), val = int32(-2)]; tensor var_1651_cast_fp16_0, tensor var_1651_cast_fp16_1 = split(axis = var_1651_axis_0, split_sizes = var_1651_split_sizes_0, x = var_1626_cast_fp16)[name = string("op_1651_cast_fp16")]; fp16 const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1653_cast_fp16 = mul(x = var_1651_cast_fp16_1, y = const_23_promoted_to_fp16)[name = string("op_1653_cast_fp16")]; int32 var_1655 = const()[name = string("op_1655"), val = int32(-2)]; bool var_1656_interleave_0 = const()[name = string("op_1656_interleave_0"), val = bool(false)]; tensor var_1656_cast_fp16 = concat(axis = var_1655, interleave = var_1656_interleave_0, values = (var_1653_cast_fp16, var_1651_cast_fp16_0))[name = string("op_1656_cast_fp16")]; tensor var_1657_cast_fp16 = mul(x = var_1656_cast_fp16, y = var_878_cast_fp16)[name = string("op_1657_cast_fp16")]; tensor key_states_25_cast_fp16 = add(x = var_1650_cast_fp16, y = var_1657_cast_fp16)[name = string("key_states_25_cast_fp16")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; int32 concat_29_axis_0 = const()[name = string("concat_29_axis_0"), val = int32(0)]; bool concat_29_interleave_0 = const()[name = string("concat_29_interleave_0"), val = bool(false)]; tensor concat_29 = concat(axis = concat_29_axis_0, interleave = concat_29_interleave_0, values = (expand_dims_24, expand_dims_25, position_id, expand_dims_27))[name = string("concat_29")]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; tensor concat_30_values1_0 = const()[name = string("concat_30_values1_0"), val = tensor([0])]; tensor concat_30_values3_0 = const()[name = string("concat_30_values3_0"), val = tensor([0])]; int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_28, concat_30_values1_0, cache_position_end, concat_30_values3_0))[name = string("concat_30")]; tensor key_states_27_perm_0 = const()[name = string("key_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_3_stride_0 = const()[name = string("key_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_27_cast_fp16 = transpose(perm = key_states_27_perm_0, x = key_states_25_cast_fp16)[name = string("transpose_335")]; tensor key_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = key_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = key_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_3_squeeze_mask_0, stride = key_cache_internal_tensor_assign_3_stride_0, update = key_states_27_cast_fp16, x = coreml_update_state_170)[name = string("key_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_3_cast_fp16, input = key_cache)[name = string("coreml_update_state_172_write_state")]; tensor coreml_update_state_172 = read_state(input = key_cache)[name = string("coreml_update_state_172")]; tensor value_states_15_perm_0 = const()[name = string("value_states_15_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_3_stride_0 = const()[name = string("value_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_15_cast_fp16 = transpose(perm = value_states_15_perm_0, x = var_1633_cast_fp16)[name = string("transpose_334")]; tensor value_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = value_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = value_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_3_squeeze_mask_0, stride = value_cache_internal_tensor_assign_3_stride_0, update = value_states_15_cast_fp16, x = coreml_update_state_171)[name = string("value_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_3_cast_fp16, input = value_cache)[name = string("coreml_update_state_173_write_state")]; tensor coreml_update_state_173 = read_state(input = value_cache)[name = string("coreml_update_state_173")]; tensor var_1727_begin_0 = const()[name = string("op_1727_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1727_end_0 = const()[name = string("op_1727_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1727_end_mask_0 = const()[name = string("op_1727_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1727_cast_fp16 = slice_by_index(begin = var_1727_begin_0, end = var_1727_end_0, end_mask = var_1727_end_mask_0, x = coreml_update_state_172)[name = string("op_1727_cast_fp16")]; tensor tile_4 = const()[name = string("tile_4"), val = tensor([1, 1])]; int32 var_1730_axis_0 = const()[name = string("op_1730_axis_0"), val = int32(1)]; tensor var_1730_cast_fp16_0, tensor var_1730_cast_fp16_1 = split(axis = var_1730_axis_0, split_sizes = tile_4, x = var_1727_cast_fp16)[name = string("op_1730_cast_fp16")]; tensor var_1737_begin_0 = const()[name = string("op_1737_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1737_end_0 = const()[name = string("op_1737_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1737_end_mask_0 = const()[name = string("op_1737_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1737_cast_fp16 = slice_by_index(begin = var_1737_begin_0, end = var_1737_end_0, end_mask = var_1737_end_mask_0, x = coreml_update_state_173)[name = string("op_1737_cast_fp16")]; tensor tile_5 = const()[name = string("tile_5"), val = tensor([1, 1])]; int32 var_1740_axis_0 = const()[name = string("op_1740_axis_0"), val = int32(1)]; tensor var_1740_cast_fp16_0, tensor var_1740_cast_fp16_1 = split(axis = var_1740_axis_0, split_sizes = tile_5, x = var_1737_cast_fp16)[name = string("op_1740_cast_fp16")]; tensor var_1743_split_sizes_0 = const()[name = string("op_1743_split_sizes_0"), val = tensor([8, 8])]; int32 var_1743_axis_0 = const()[name = string("op_1743_axis_0"), val = int32(1)]; tensor var_1743_0, tensor var_1743_1 = split(axis = var_1743_axis_0, split_sizes = var_1743_split_sizes_0, x = query_states_15_cast_fp16)[name = string("op_1743")]; bool attn_weights_33_transpose_x_0 = const()[name = string("attn_weights_33_transpose_x_0"), val = bool(false)]; bool attn_weights_33_transpose_y_0 = const()[name = string("attn_weights_33_transpose_y_0"), val = bool(false)]; tensor attn_weights_33_cast_fp16 = matmul(transpose_x = attn_weights_33_transpose_x_0, transpose_y = attn_weights_33_transpose_y_0, x = var_1730_cast_fp16_0, y = var_1743_0)[name = string("attn_weights_33_cast_fp16")]; fp16 var_1746_to_fp16 = const()[name = string("op_1746_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_35_cast_fp16 = mul(x = attn_weights_33_cast_fp16, y = var_1746_to_fp16)[name = string("attn_weights_35_cast_fp16")]; tensor attn_weights_37_cast_fp16 = add(x = attn_weights_35_cast_fp16, y = attn_mask_1)[name = string("attn_weights_37_cast_fp16")]; int32 var_1750 = const()[name = string("op_1750"), val = int32(-2)]; tensor attn_weights_39_cast_fp16 = softmax(axis = var_1750, x = attn_weights_37_cast_fp16)[name = string("attn_weights_39_cast_fp16")]; bool var_1756_transpose_x_1 = const()[name = string("op_1756_transpose_x_1"), val = bool(true)]; bool var_1756_transpose_y_1 = const()[name = string("op_1756_transpose_y_1"), val = bool(false)]; tensor var_1756_cast_fp16 = matmul(transpose_x = var_1756_transpose_x_1, transpose_y = var_1756_transpose_y_1, x = attn_weights_39_cast_fp16, y = var_1740_cast_fp16_0)[name = string("op_1756_cast_fp16")]; bool attn_weights_41_transpose_x_0 = const()[name = string("attn_weights_41_transpose_x_0"), val = bool(false)]; bool attn_weights_41_transpose_y_0 = const()[name = string("attn_weights_41_transpose_y_0"), val = bool(false)]; tensor attn_weights_41_cast_fp16 = matmul(transpose_x = attn_weights_41_transpose_x_0, transpose_y = attn_weights_41_transpose_y_0, x = var_1730_cast_fp16_1, y = var_1743_1)[name = string("attn_weights_41_cast_fp16")]; fp16 var_1758_to_fp16 = const()[name = string("op_1758_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_43_cast_fp16 = mul(x = attn_weights_41_cast_fp16, y = var_1758_to_fp16)[name = string("attn_weights_43_cast_fp16")]; tensor attn_weights_45_cast_fp16 = add(x = attn_weights_43_cast_fp16, y = attn_mask_1)[name = string("attn_weights_45_cast_fp16")]; int32 var_1762 = const()[name = string("op_1762"), val = int32(-2)]; tensor attn_weights_47_cast_fp16 = softmax(axis = var_1762, x = attn_weights_45_cast_fp16)[name = string("attn_weights_47_cast_fp16")]; bool attn_output_17_transpose_x_1 = const()[name = string("attn_output_17_transpose_x_1"), val = bool(true)]; bool attn_output_17_transpose_y_1 = const()[name = string("attn_output_17_transpose_y_1"), val = bool(false)]; tensor attn_output_17_cast_fp16 = matmul(transpose_x = attn_output_17_transpose_x_1, transpose_y = attn_output_17_transpose_y_1, x = attn_weights_47_cast_fp16, y = var_1740_cast_fp16_1)[name = string("attn_output_17_cast_fp16")]; int32 var_1770 = const()[name = string("op_1770"), val = int32(1)]; bool attn_output_19_interleave_0 = const()[name = string("attn_output_19_interleave_0"), val = bool(false)]; tensor attn_output_19_cast_fp16 = concat(axis = var_1770, interleave = attn_output_19_interleave_0, values = (var_1756_cast_fp16, attn_output_17_cast_fp16))[name = string("attn_output_19_cast_fp16")]; tensor var_1774_perm_0 = const()[name = string("op_1774_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_35x = const()[name = string("concat_35x"), val = tensor([1, 2048, 1, -1])]; tensor var_1774_cast_fp16 = transpose(perm = var_1774_perm_0, x = attn_output_19_cast_fp16)[name = string("transpose_333")]; tensor attn_output_23_cast_fp16 = reshape(shape = concat_35x, x = var_1774_cast_fp16)[name = string("attn_output_23_cast_fp16")]; tensor layers_2_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1155169408)))]; tensor hidden_states_23_strides_0 = const()[name = string("hidden_states_23_strides_0"), val = tensor([1, 1])]; string hidden_states_23_pad_type_0 = const()[name = string("hidden_states_23_pad_type_0"), val = string("valid")]; tensor hidden_states_23_pad_0 = const()[name = string("hidden_states_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_23_dilations_0 = const()[name = string("hidden_states_23_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_23_groups_0 = const()[name = string("hidden_states_23_groups_0"), val = int32(1)]; tensor hidden_states_23_cast_fp16 = conv(dilations = hidden_states_23_dilations_0, groups = hidden_states_23_groups_0, pad = hidden_states_23_pad_0, pad_type = hidden_states_23_pad_type_0, strides = hidden_states_23_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16, x = attn_output_23_cast_fp16)[name = string("hidden_states_23_cast_fp16")]; tensor hidden_states_25_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = hidden_states_23_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1807_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_1807_cast_fp16")]; int32 var_1805 = const()[name = string("op_1805"), val = int32(1)]; bool doubled_21_interleave_0 = const()[name = string("doubled_21_interleave_0"), val = bool(false)]; tensor doubled_21_cast_fp16 = concat(axis = var_1805, interleave = doubled_21_interleave_0, values = (hidden_states_25_cast_fp16, var_1807_cast_fp16))[name = string("doubled_21_cast_fp16")]; tensor out_11_axes_0 = const()[name = string("out_11_axes_0"), val = tensor([1])]; tensor out_11_gamma_0_to_fp16 = const()[name = string("out_11_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163558080)))]; fp16 var_1817_to_fp16 = const()[name = string("op_1817_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_11_cast_fp16 = layer_norm(axes = out_11_axes_0, epsilon = var_1817_to_fp16, gamma = out_11_gamma_0_to_fp16, x = doubled_21_cast_fp16)[name = string("out_11_cast_fp16")]; tensor var_1828_split_sizes_0 = const()[name = string("op_1828_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1828_axis_0 = const()[name = string("op_1828_axis_0"), val = int32(1)]; tensor var_1828_cast_fp16_0, tensor var_1828_cast_fp16_1 = split(axis = var_1828_axis_0, split_sizes = var_1828_split_sizes_0, x = out_11_cast_fp16)[name = string("op_1828_cast_fp16")]; tensor layers_2_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163566336)))]; tensor input_5_strides_0 = const()[name = string("input_5_strides_0"), val = tensor([1, 1])]; string input_5_pad_type_0 = const()[name = string("input_5_pad_type_0"), val = string("valid")]; tensor input_5_pad_0 = const()[name = string("input_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_5_dilations_0 = const()[name = string("input_5_dilations_0"), val = tensor([1, 1])]; int32 input_5_groups_0 = const()[name = string("input_5_groups_0"), val = int32(1)]; tensor input_5_cast_fp16 = conv(dilations = input_5_dilations_0, groups = input_5_groups_0, pad = input_5_pad_0, pad_type = input_5_pad_type_0, strides = input_5_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16, x = var_1828_cast_fp16_0)[name = string("input_5_cast_fp16")]; tensor var_1845_cast_fp16 = silu(x = input_5_cast_fp16)[name = string("op_1845_cast_fp16")]; tensor var_1851_strides_0 = const()[name = string("op_1851_strides_0"), val = tensor([1, 1])]; string var_1851_pad_type_0 = const()[name = string("op_1851_pad_type_0"), val = string("valid")]; tensor var_1851_pad_0 = const()[name = string("op_1851_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1851_dilations_0 = const()[name = string("op_1851_dilations_0"), val = tensor([1, 1])]; int32 var_1851_groups_0 = const()[name = string("op_1851_groups_0"), val = int32(1)]; tensor var_1851_cast_fp16 = conv(dilations = var_1851_dilations_0, groups = var_1851_groups_0, pad = var_1851_pad_0, pad_type = var_1851_pad_type_0, strides = var_1851_strides_0, weight = layers_2_mlp_up_proj_weight_cast_fp16, x = var_1828_cast_fp16_0)[name = string("op_1851_cast_fp16")]; tensor x_29_cast_fp16 = mul(x = var_1845_cast_fp16, y = var_1851_cast_fp16)[name = string("x_29_cast_fp16")]; tensor layers_2_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1188732224)))]; tensor hidden_states_27_strides_0 = const()[name = string("hidden_states_27_strides_0"), val = tensor([1, 1])]; string hidden_states_27_pad_type_0 = const()[name = string("hidden_states_27_pad_type_0"), val = string("valid")]; tensor hidden_states_27_pad_0 = const()[name = string("hidden_states_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_27_dilations_0 = const()[name = string("hidden_states_27_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_27_groups_0 = const()[name = string("hidden_states_27_groups_0"), val = int32(1)]; tensor hidden_states_27_cast_fp16 = conv(dilations = hidden_states_27_dilations_0, groups = hidden_states_27_groups_0, pad = hidden_states_27_pad_0, pad_type = hidden_states_27_pad_type_0, strides = hidden_states_27_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16, x = x_29_cast_fp16)[name = string("hidden_states_27_cast_fp16")]; tensor hidden_states_29_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = hidden_states_27_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; fp16 const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1869_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_30_promoted_to_fp16)[name = string("op_1869_cast_fp16")]; int32 var_1867 = const()[name = string("op_1867"), val = int32(1)]; bool doubled_25_interleave_0 = const()[name = string("doubled_25_interleave_0"), val = bool(false)]; tensor doubled_25_cast_fp16 = concat(axis = var_1867, interleave = doubled_25_interleave_0, values = (hidden_states_29_cast_fp16, var_1869_cast_fp16))[name = string("doubled_25_cast_fp16")]; tensor out_13_axes_0 = const()[name = string("out_13_axes_0"), val = tensor([1])]; tensor out_13_gamma_0_to_fp16 = const()[name = string("out_13_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213898112)))]; fp16 var_1879_to_fp16 = const()[name = string("op_1879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_13_cast_fp16 = layer_norm(axes = out_13_axes_0, epsilon = var_1879_to_fp16, gamma = out_13_gamma_0_to_fp16, x = doubled_25_cast_fp16)[name = string("out_13_cast_fp16")]; tensor var_1890_split_sizes_0 = const()[name = string("op_1890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1890_axis_0 = const()[name = string("op_1890_axis_0"), val = int32(1)]; tensor var_1890_cast_fp16_0, tensor var_1890_cast_fp16_1 = split(axis = var_1890_axis_0, split_sizes = var_1890_split_sizes_0, x = out_13_cast_fp16)[name = string("op_1890_cast_fp16")]; tensor layers_3_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213906368)))]; tensor query_states_19_strides_0 = const()[name = string("query_states_19_strides_0"), val = tensor([1, 1])]; string query_states_19_pad_type_0 = const()[name = string("query_states_19_pad_type_0"), val = string("valid")]; tensor query_states_19_pad_0 = const()[name = string("query_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_19_dilations_0 = const()[name = string("query_states_19_dilations_0"), val = tensor([1, 1])]; int32 query_states_19_groups_0 = const()[name = string("query_states_19_groups_0"), val = int32(1)]; tensor query_states_19_cast_fp16 = conv(dilations = query_states_19_dilations_0, groups = query_states_19_groups_0, pad = query_states_19_pad_0, pad_type = query_states_19_pad_type_0, strides = query_states_19_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("query_states_19_cast_fp16")]; tensor layers_3_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1222295040)))]; tensor key_states_31_strides_0 = const()[name = string("key_states_31_strides_0"), val = tensor([1, 1])]; string key_states_31_pad_type_0 = const()[name = string("key_states_31_pad_type_0"), val = string("valid")]; tensor key_states_31_pad_0 = const()[name = string("key_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_31_dilations_0 = const()[name = string("key_states_31_dilations_0"), val = tensor([1, 1])]; int32 key_states_31_groups_0 = const()[name = string("key_states_31_groups_0"), val = int32(1)]; tensor key_states_31_cast_fp16 = conv(dilations = key_states_31_dilations_0, groups = key_states_31_groups_0, pad = key_states_31_pad_0, pad_type = key_states_31_pad_type_0, strides = key_states_31_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("key_states_31_cast_fp16")]; tensor value_states_19_strides_0 = const()[name = string("value_states_19_strides_0"), val = tensor([1, 1])]; string value_states_19_pad_type_0 = const()[name = string("value_states_19_pad_type_0"), val = string("valid")]; tensor value_states_19_pad_0 = const()[name = string("value_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_19_dilations_0 = const()[name = string("value_states_19_dilations_0"), val = tensor([1, 1])]; int32 value_states_19_groups_0 = const()[name = string("value_states_19_groups_0"), val = int32(1)]; tensor value_states_19_cast_fp16 = conv(dilations = value_states_19_dilations_0, groups = value_states_19_groups_0, pad = value_states_19_pad_0, pad_type = value_states_19_pad_type_0, strides = value_states_19_strides_0, weight = layers_3_self_attn_v_proj_weight_cast_fp16, x = var_1890_cast_fp16_0)[name = string("value_states_19_cast_fp16")]; tensor concat_36x = const()[name = string("concat_36x"), val = tensor([1, 16, 128, -1])]; tensor x_31_cast_fp16 = reshape(shape = concat_36x, x = query_states_19_cast_fp16)[name = string("x_31_cast_fp16")]; tensor concat_37x = const()[name = string("concat_37x"), val = tensor([1, 2, 128, -1])]; tensor var_1947_cast_fp16 = reshape(shape = concat_37x, x = key_states_31_cast_fp16)[name = string("op_1947_cast_fp16")]; tensor concat_38x = const()[name = string("concat_38x"), val = tensor([1, 2, 128, -1])]; tensor var_1954_cast_fp16 = reshape(shape = concat_38x, x = value_states_19_cast_fp16)[name = string("op_1954_cast_fp16")]; tensor var_1958_cast_fp16 = mul(x = x_31_cast_fp16, y = var_869_cast_fp16)[name = string("op_1958_cast_fp16")]; tensor var_1959_split_sizes_0 = const()[name = string("op_1959_split_sizes_0"), val = tensor([64, 64])]; int32 var_1959_axis_0 = const()[name = string("op_1959_axis_0"), val = int32(-2)]; tensor var_1959_cast_fp16_0, tensor var_1959_cast_fp16_1 = split(axis = var_1959_axis_0, split_sizes = var_1959_split_sizes_0, x = x_31_cast_fp16)[name = string("op_1959_cast_fp16")]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1961_cast_fp16 = mul(x = var_1959_cast_fp16_1, y = const_32_promoted_to_fp16)[name = string("op_1961_cast_fp16")]; int32 var_1963 = const()[name = string("op_1963"), val = int32(-2)]; bool var_1964_interleave_0 = const()[name = string("op_1964_interleave_0"), val = bool(false)]; tensor var_1964_cast_fp16 = concat(axis = var_1963, interleave = var_1964_interleave_0, values = (var_1961_cast_fp16, var_1959_cast_fp16_0))[name = string("op_1964_cast_fp16")]; tensor var_1965_cast_fp16 = mul(x = var_1964_cast_fp16, y = var_878_cast_fp16)[name = string("op_1965_cast_fp16")]; tensor query_states_21_cast_fp16 = add(x = var_1958_cast_fp16, y = var_1965_cast_fp16)[name = string("query_states_21_cast_fp16")]; tensor var_1971_cast_fp16 = mul(x = var_1947_cast_fp16, y = var_869_cast_fp16)[name = string("op_1971_cast_fp16")]; tensor var_1972_split_sizes_0 = const()[name = string("op_1972_split_sizes_0"), val = tensor([64, 64])]; int32 var_1972_axis_0 = const()[name = string("op_1972_axis_0"), val = int32(-2)]; tensor var_1972_cast_fp16_0, tensor var_1972_cast_fp16_1 = split(axis = var_1972_axis_0, split_sizes = var_1972_split_sizes_0, x = var_1947_cast_fp16)[name = string("op_1972_cast_fp16")]; fp16 const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1974_cast_fp16 = mul(x = var_1972_cast_fp16_1, y = const_33_promoted_to_fp16)[name = string("op_1974_cast_fp16")]; int32 var_1976 = const()[name = string("op_1976"), val = int32(-2)]; bool var_1977_interleave_0 = const()[name = string("op_1977_interleave_0"), val = bool(false)]; tensor var_1977_cast_fp16 = concat(axis = var_1976, interleave = var_1977_interleave_0, values = (var_1974_cast_fp16, var_1972_cast_fp16_0))[name = string("op_1977_cast_fp16")]; tensor var_1978_cast_fp16 = mul(x = var_1977_cast_fp16, y = var_878_cast_fp16)[name = string("op_1978_cast_fp16")]; tensor key_states_35_cast_fp16 = add(x = var_1971_cast_fp16, y = var_1978_cast_fp16)[name = string("key_states_35_cast_fp16")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; int32 concat_41_axis_0 = const()[name = string("concat_41_axis_0"), val = int32(0)]; bool concat_41_interleave_0 = const()[name = string("concat_41_interleave_0"), val = bool(false)]; tensor concat_41 = concat(axis = concat_41_axis_0, interleave = concat_41_interleave_0, values = (expand_dims_36, expand_dims_37, position_id, expand_dims_39))[name = string("concat_41")]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; tensor concat_42_values1_0 = const()[name = string("concat_42_values1_0"), val = tensor([0])]; tensor concat_42_values3_0 = const()[name = string("concat_42_values3_0"), val = tensor([0])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_40, concat_42_values1_0, cache_position_end, concat_42_values3_0))[name = string("concat_42")]; tensor key_states_37_perm_0 = const()[name = string("key_states_37_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_4_stride_0 = const()[name = string("key_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_37_cast_fp16 = transpose(perm = key_states_37_perm_0, x = key_states_35_cast_fp16)[name = string("transpose_332")]; tensor key_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = key_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = key_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_4_squeeze_mask_0, stride = key_cache_internal_tensor_assign_4_stride_0, update = key_states_37_cast_fp16, x = coreml_update_state_172)[name = string("key_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_4_cast_fp16, input = key_cache)[name = string("coreml_update_state_174_write_state")]; tensor coreml_update_state_174 = read_state(input = key_cache)[name = string("coreml_update_state_174")]; tensor value_states_21_perm_0 = const()[name = string("value_states_21_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_4_stride_0 = const()[name = string("value_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_21_cast_fp16 = transpose(perm = value_states_21_perm_0, x = var_1954_cast_fp16)[name = string("transpose_331")]; tensor value_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = value_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = value_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_4_squeeze_mask_0, stride = value_cache_internal_tensor_assign_4_stride_0, update = value_states_21_cast_fp16, x = coreml_update_state_173)[name = string("value_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_4_cast_fp16, input = value_cache)[name = string("coreml_update_state_175_write_state")]; tensor coreml_update_state_175 = read_state(input = value_cache)[name = string("coreml_update_state_175")]; tensor var_2048_begin_0 = const()[name = string("op_2048_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2048_end_0 = const()[name = string("op_2048_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2048_end_mask_0 = const()[name = string("op_2048_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2048_cast_fp16 = slice_by_index(begin = var_2048_begin_0, end = var_2048_end_0, end_mask = var_2048_end_mask_0, x = coreml_update_state_174)[name = string("op_2048_cast_fp16")]; tensor tile_6 = const()[name = string("tile_6"), val = tensor([1, 1])]; int32 var_2051_axis_0 = const()[name = string("op_2051_axis_0"), val = int32(1)]; tensor var_2051_cast_fp16_0, tensor var_2051_cast_fp16_1 = split(axis = var_2051_axis_0, split_sizes = tile_6, x = var_2048_cast_fp16)[name = string("op_2051_cast_fp16")]; tensor var_2058_begin_0 = const()[name = string("op_2058_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2058_end_0 = const()[name = string("op_2058_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2058_end_mask_0 = const()[name = string("op_2058_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2058_cast_fp16 = slice_by_index(begin = var_2058_begin_0, end = var_2058_end_0, end_mask = var_2058_end_mask_0, x = coreml_update_state_175)[name = string("op_2058_cast_fp16")]; tensor tile_7 = const()[name = string("tile_7"), val = tensor([1, 1])]; int32 var_2061_axis_0 = const()[name = string("op_2061_axis_0"), val = int32(1)]; tensor var_2061_cast_fp16_0, tensor var_2061_cast_fp16_1 = split(axis = var_2061_axis_0, split_sizes = tile_7, x = var_2058_cast_fp16)[name = string("op_2061_cast_fp16")]; tensor var_2064_split_sizes_0 = const()[name = string("op_2064_split_sizes_0"), val = tensor([8, 8])]; int32 var_2064_axis_0 = const()[name = string("op_2064_axis_0"), val = int32(1)]; tensor var_2064_0, tensor var_2064_1 = split(axis = var_2064_axis_0, split_sizes = var_2064_split_sizes_0, x = query_states_21_cast_fp16)[name = string("op_2064")]; bool attn_weights_49_transpose_x_0 = const()[name = string("attn_weights_49_transpose_x_0"), val = bool(false)]; bool attn_weights_49_transpose_y_0 = const()[name = string("attn_weights_49_transpose_y_0"), val = bool(false)]; tensor attn_weights_49_cast_fp16 = matmul(transpose_x = attn_weights_49_transpose_x_0, transpose_y = attn_weights_49_transpose_y_0, x = var_2051_cast_fp16_0, y = var_2064_0)[name = string("attn_weights_49_cast_fp16")]; fp16 var_2067_to_fp16 = const()[name = string("op_2067_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_51_cast_fp16 = mul(x = attn_weights_49_cast_fp16, y = var_2067_to_fp16)[name = string("attn_weights_51_cast_fp16")]; tensor attn_weights_53_cast_fp16 = add(x = attn_weights_51_cast_fp16, y = attn_mask_1)[name = string("attn_weights_53_cast_fp16")]; int32 var_2071 = const()[name = string("op_2071"), val = int32(-2)]; tensor attn_weights_55_cast_fp16 = softmax(axis = var_2071, x = attn_weights_53_cast_fp16)[name = string("attn_weights_55_cast_fp16")]; bool var_2077_transpose_x_1 = const()[name = string("op_2077_transpose_x_1"), val = bool(true)]; bool var_2077_transpose_y_1 = const()[name = string("op_2077_transpose_y_1"), val = bool(false)]; tensor var_2077_cast_fp16 = matmul(transpose_x = var_2077_transpose_x_1, transpose_y = var_2077_transpose_y_1, x = attn_weights_55_cast_fp16, y = var_2061_cast_fp16_0)[name = string("op_2077_cast_fp16")]; bool attn_weights_57_transpose_x_0 = const()[name = string("attn_weights_57_transpose_x_0"), val = bool(false)]; bool attn_weights_57_transpose_y_0 = const()[name = string("attn_weights_57_transpose_y_0"), val = bool(false)]; tensor attn_weights_57_cast_fp16 = matmul(transpose_x = attn_weights_57_transpose_x_0, transpose_y = attn_weights_57_transpose_y_0, x = var_2051_cast_fp16_1, y = var_2064_1)[name = string("attn_weights_57_cast_fp16")]; fp16 var_2079_to_fp16 = const()[name = string("op_2079_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_59_cast_fp16 = mul(x = attn_weights_57_cast_fp16, y = var_2079_to_fp16)[name = string("attn_weights_59_cast_fp16")]; tensor attn_weights_61_cast_fp16 = add(x = attn_weights_59_cast_fp16, y = attn_mask_1)[name = string("attn_weights_61_cast_fp16")]; int32 var_2083 = const()[name = string("op_2083"), val = int32(-2)]; tensor attn_weights_63_cast_fp16 = softmax(axis = var_2083, x = attn_weights_61_cast_fp16)[name = string("attn_weights_63_cast_fp16")]; bool attn_output_25_transpose_x_1 = const()[name = string("attn_output_25_transpose_x_1"), val = bool(true)]; bool attn_output_25_transpose_y_1 = const()[name = string("attn_output_25_transpose_y_1"), val = bool(false)]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_1, transpose_y = attn_output_25_transpose_y_1, x = attn_weights_63_cast_fp16, y = var_2061_cast_fp16_1)[name = string("attn_output_25_cast_fp16")]; int32 var_2091 = const()[name = string("op_2091"), val = int32(1)]; bool attn_output_27_interleave_0 = const()[name = string("attn_output_27_interleave_0"), val = bool(false)]; tensor attn_output_27_cast_fp16 = concat(axis = var_2091, interleave = attn_output_27_interleave_0, values = (var_2077_cast_fp16, attn_output_25_cast_fp16))[name = string("attn_output_27_cast_fp16")]; tensor var_2095_perm_0 = const()[name = string("op_2095_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_47x = const()[name = string("concat_47x"), val = tensor([1, 2048, 1, -1])]; tensor var_2095_cast_fp16 = transpose(perm = var_2095_perm_0, x = attn_output_27_cast_fp16)[name = string("transpose_330")]; tensor attn_output_31_cast_fp16 = reshape(shape = concat_47x, x = var_2095_cast_fp16)[name = string("attn_output_31_cast_fp16")]; tensor hidden_states_33_strides_0 = const()[name = string("hidden_states_33_strides_0"), val = tensor([1, 1])]; string hidden_states_33_pad_type_0 = const()[name = string("hidden_states_33_pad_type_0"), val = string("valid")]; tensor hidden_states_33_pad_0 = const()[name = string("hidden_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_33_dilations_0 = const()[name = string("hidden_states_33_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_33_groups_0 = const()[name = string("hidden_states_33_groups_0"), val = int32(1)]; tensor hidden_states_33_cast_fp16 = conv(dilations = hidden_states_33_dilations_0, groups = hidden_states_33_groups_0, pad = hidden_states_33_pad_0, pad_type = hidden_states_33_pad_type_0, strides = hidden_states_33_strides_0, weight = layers_3_self_attn_o_proj_weight_cast_fp16, x = attn_output_31_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; tensor hidden_states_35_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = hidden_states_33_cast_fp16)[name = string("hidden_states_35_cast_fp16")]; fp16 const_38_promoted_to_fp16 = const()[name = string("const_38_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2128_cast_fp16 = mul(x = hidden_states_35_cast_fp16, y = const_38_promoted_to_fp16)[name = string("op_2128_cast_fp16")]; int32 var_2126 = const()[name = string("op_2126"), val = int32(1)]; bool doubled_29_interleave_0 = const()[name = string("doubled_29_interleave_0"), val = bool(false)]; tensor doubled_29_cast_fp16 = concat(axis = var_2126, interleave = doubled_29_interleave_0, values = (hidden_states_35_cast_fp16, var_2128_cast_fp16))[name = string("doubled_29_cast_fp16")]; tensor out_15_axes_0 = const()[name = string("out_15_axes_0"), val = tensor([1])]; tensor out_15_gamma_0_to_fp16 = const()[name = string("out_15_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223343680)))]; fp16 var_2138_to_fp16 = const()[name = string("op_2138_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_15_cast_fp16 = layer_norm(axes = out_15_axes_0, epsilon = var_2138_to_fp16, gamma = out_15_gamma_0_to_fp16, x = doubled_29_cast_fp16)[name = string("out_15_cast_fp16")]; tensor var_2149_split_sizes_0 = const()[name = string("op_2149_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2149_axis_0 = const()[name = string("op_2149_axis_0"), val = int32(1)]; tensor var_2149_cast_fp16_0, tensor var_2149_cast_fp16_1 = split(axis = var_2149_axis_0, split_sizes = var_2149_split_sizes_0, x = out_15_cast_fp16)[name = string("op_2149_cast_fp16")]; tensor layers_3_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223351936)))]; tensor input_7_strides_0 = const()[name = string("input_7_strides_0"), val = tensor([1, 1])]; string input_7_pad_type_0 = const()[name = string("input_7_pad_type_0"), val = string("valid")]; tensor input_7_pad_0 = const()[name = string("input_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_7_dilations_0 = const()[name = string("input_7_dilations_0"), val = tensor([1, 1])]; int32 input_7_groups_0 = const()[name = string("input_7_groups_0"), val = int32(1)]; tensor input_7_cast_fp16 = conv(dilations = input_7_dilations_0, groups = input_7_groups_0, pad = input_7_pad_0, pad_type = input_7_pad_type_0, strides = input_7_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("input_7_cast_fp16")]; tensor var_2166_cast_fp16 = silu(x = input_7_cast_fp16)[name = string("op_2166_cast_fp16")]; tensor layers_3_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1248517824)))]; tensor var_2172_strides_0 = const()[name = string("op_2172_strides_0"), val = tensor([1, 1])]; string var_2172_pad_type_0 = const()[name = string("op_2172_pad_type_0"), val = string("valid")]; tensor var_2172_pad_0 = const()[name = string("op_2172_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2172_dilations_0 = const()[name = string("op_2172_dilations_0"), val = tensor([1, 1])]; int32 var_2172_groups_0 = const()[name = string("op_2172_groups_0"), val = int32(1)]; tensor var_2172_cast_fp16 = conv(dilations = var_2172_dilations_0, groups = var_2172_groups_0, pad = var_2172_pad_0, pad_type = var_2172_pad_type_0, strides = var_2172_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("op_2172_cast_fp16")]; tensor x_39_cast_fp16 = mul(x = var_2166_cast_fp16, y = var_2172_cast_fp16)[name = string("x_39_cast_fp16")]; tensor hidden_states_37_strides_0 = const()[name = string("hidden_states_37_strides_0"), val = tensor([1, 1])]; string hidden_states_37_pad_type_0 = const()[name = string("hidden_states_37_pad_type_0"), val = string("valid")]; tensor hidden_states_37_pad_0 = const()[name = string("hidden_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_37_dilations_0 = const()[name = string("hidden_states_37_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_37_groups_0 = const()[name = string("hidden_states_37_groups_0"), val = int32(1)]; tensor hidden_states_37_cast_fp16 = conv(dilations = hidden_states_37_dilations_0, groups = hidden_states_37_groups_0, pad = hidden_states_37_pad_0, pad_type = hidden_states_37_pad_type_0, strides = hidden_states_37_strides_0, weight = layers_3_mlp_down_proj_weight_cast_fp16, x = x_39_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; tensor hidden_states_39_cast_fp16 = add(x = hidden_states_35_cast_fp16, y = hidden_states_37_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; fp16 const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2190_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_40_promoted_to_fp16)[name = string("op_2190_cast_fp16")]; int32 var_2188 = const()[name = string("op_2188"), val = int32(1)]; bool doubled_33_interleave_0 = const()[name = string("doubled_33_interleave_0"), val = bool(false)]; tensor doubled_33_cast_fp16 = concat(axis = var_2188, interleave = doubled_33_interleave_0, values = (hidden_states_39_cast_fp16, var_2190_cast_fp16))[name = string("doubled_33_cast_fp16")]; tensor out_17_axes_0 = const()[name = string("out_17_axes_0"), val = tensor([1])]; tensor out_17_gamma_0_to_fp16 = const()[name = string("out_17_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273683712)))]; fp16 var_2200_to_fp16 = const()[name = string("op_2200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_17_cast_fp16 = layer_norm(axes = out_17_axes_0, epsilon = var_2200_to_fp16, gamma = out_17_gamma_0_to_fp16, x = doubled_33_cast_fp16)[name = string("out_17_cast_fp16")]; tensor var_2211_split_sizes_0 = const()[name = string("op_2211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2211_axis_0 = const()[name = string("op_2211_axis_0"), val = int32(1)]; tensor var_2211_cast_fp16_0, tensor var_2211_cast_fp16_1 = split(axis = var_2211_axis_0, split_sizes = var_2211_split_sizes_0, x = out_17_cast_fp16)[name = string("op_2211_cast_fp16")]; tensor layers_4_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273691968)))]; tensor query_states_25_strides_0 = const()[name = string("query_states_25_strides_0"), val = tensor([1, 1])]; string query_states_25_pad_type_0 = const()[name = string("query_states_25_pad_type_0"), val = string("valid")]; tensor query_states_25_pad_0 = const()[name = string("query_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_25_dilations_0 = const()[name = string("query_states_25_dilations_0"), val = tensor([1, 1])]; int32 query_states_25_groups_0 = const()[name = string("query_states_25_groups_0"), val = int32(1)]; tensor query_states_25_cast_fp16 = conv(dilations = query_states_25_dilations_0, groups = query_states_25_groups_0, pad = query_states_25_pad_0, pad_type = query_states_25_pad_type_0, strides = query_states_25_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("query_states_25_cast_fp16")]; tensor layers_4_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1282080640)))]; tensor key_states_41_strides_0 = const()[name = string("key_states_41_strides_0"), val = tensor([1, 1])]; string key_states_41_pad_type_0 = const()[name = string("key_states_41_pad_type_0"), val = string("valid")]; tensor key_states_41_pad_0 = const()[name = string("key_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_41_dilations_0 = const()[name = string("key_states_41_dilations_0"), val = tensor([1, 1])]; int32 key_states_41_groups_0 = const()[name = string("key_states_41_groups_0"), val = int32(1)]; tensor key_states_41_cast_fp16 = conv(dilations = key_states_41_dilations_0, groups = key_states_41_groups_0, pad = key_states_41_pad_0, pad_type = key_states_41_pad_type_0, strides = key_states_41_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("key_states_41_cast_fp16")]; tensor value_states_25_strides_0 = const()[name = string("value_states_25_strides_0"), val = tensor([1, 1])]; string value_states_25_pad_type_0 = const()[name = string("value_states_25_pad_type_0"), val = string("valid")]; tensor value_states_25_pad_0 = const()[name = string("value_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_25_dilations_0 = const()[name = string("value_states_25_dilations_0"), val = tensor([1, 1])]; int32 value_states_25_groups_0 = const()[name = string("value_states_25_groups_0"), val = int32(1)]; tensor value_states_25_cast_fp16 = conv(dilations = value_states_25_dilations_0, groups = value_states_25_groups_0, pad = value_states_25_pad_0, pad_type = value_states_25_pad_type_0, strides = value_states_25_strides_0, weight = layers_4_self_attn_v_proj_weight_cast_fp16, x = var_2211_cast_fp16_0)[name = string("value_states_25_cast_fp16")]; tensor concat_48x = const()[name = string("concat_48x"), val = tensor([1, 16, 128, -1])]; tensor x_41_cast_fp16 = reshape(shape = concat_48x, x = query_states_25_cast_fp16)[name = string("x_41_cast_fp16")]; tensor concat_49x = const()[name = string("concat_49x"), val = tensor([1, 2, 128, -1])]; tensor var_2268_cast_fp16 = reshape(shape = concat_49x, x = key_states_41_cast_fp16)[name = string("op_2268_cast_fp16")]; tensor concat_50x = const()[name = string("concat_50x"), val = tensor([1, 2, 128, -1])]; tensor var_2275_cast_fp16 = reshape(shape = concat_50x, x = value_states_25_cast_fp16)[name = string("op_2275_cast_fp16")]; tensor var_2279_cast_fp16 = mul(x = x_41_cast_fp16, y = var_869_cast_fp16)[name = string("op_2279_cast_fp16")]; tensor var_2280_split_sizes_0 = const()[name = string("op_2280_split_sizes_0"), val = tensor([64, 64])]; int32 var_2280_axis_0 = const()[name = string("op_2280_axis_0"), val = int32(-2)]; tensor var_2280_cast_fp16_0, tensor var_2280_cast_fp16_1 = split(axis = var_2280_axis_0, split_sizes = var_2280_split_sizes_0, x = x_41_cast_fp16)[name = string("op_2280_cast_fp16")]; fp16 const_42_promoted_to_fp16 = const()[name = string("const_42_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2282_cast_fp16 = mul(x = var_2280_cast_fp16_1, y = const_42_promoted_to_fp16)[name = string("op_2282_cast_fp16")]; int32 var_2284 = const()[name = string("op_2284"), val = int32(-2)]; bool var_2285_interleave_0 = const()[name = string("op_2285_interleave_0"), val = bool(false)]; tensor var_2285_cast_fp16 = concat(axis = var_2284, interleave = var_2285_interleave_0, values = (var_2282_cast_fp16, var_2280_cast_fp16_0))[name = string("op_2285_cast_fp16")]; tensor var_2286_cast_fp16 = mul(x = var_2285_cast_fp16, y = var_878_cast_fp16)[name = string("op_2286_cast_fp16")]; tensor query_states_27_cast_fp16 = add(x = var_2279_cast_fp16, y = var_2286_cast_fp16)[name = string("query_states_27_cast_fp16")]; tensor var_2292_cast_fp16 = mul(x = var_2268_cast_fp16, y = var_869_cast_fp16)[name = string("op_2292_cast_fp16")]; tensor var_2293_split_sizes_0 = const()[name = string("op_2293_split_sizes_0"), val = tensor([64, 64])]; int32 var_2293_axis_0 = const()[name = string("op_2293_axis_0"), val = int32(-2)]; tensor var_2293_cast_fp16_0, tensor var_2293_cast_fp16_1 = split(axis = var_2293_axis_0, split_sizes = var_2293_split_sizes_0, x = var_2268_cast_fp16)[name = string("op_2293_cast_fp16")]; fp16 const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2295_cast_fp16 = mul(x = var_2293_cast_fp16_1, y = const_43_promoted_to_fp16)[name = string("op_2295_cast_fp16")]; int32 var_2297 = const()[name = string("op_2297"), val = int32(-2)]; bool var_2298_interleave_0 = const()[name = string("op_2298_interleave_0"), val = bool(false)]; tensor var_2298_cast_fp16 = concat(axis = var_2297, interleave = var_2298_interleave_0, values = (var_2295_cast_fp16, var_2293_cast_fp16_0))[name = string("op_2298_cast_fp16")]; tensor var_2299_cast_fp16 = mul(x = var_2298_cast_fp16, y = var_878_cast_fp16)[name = string("op_2299_cast_fp16")]; tensor key_states_45_cast_fp16 = add(x = var_2292_cast_fp16, y = var_2299_cast_fp16)[name = string("key_states_45_cast_fp16")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; int32 concat_53_axis_0 = const()[name = string("concat_53_axis_0"), val = int32(0)]; bool concat_53_interleave_0 = const()[name = string("concat_53_interleave_0"), val = bool(false)]; tensor concat_53 = concat(axis = concat_53_axis_0, interleave = concat_53_interleave_0, values = (expand_dims_48, expand_dims_49, position_id, expand_dims_51))[name = string("concat_53")]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; tensor concat_54_values1_0 = const()[name = string("concat_54_values1_0"), val = tensor([0])]; tensor concat_54_values3_0 = const()[name = string("concat_54_values3_0"), val = tensor([0])]; int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_52, concat_54_values1_0, cache_position_end, concat_54_values3_0))[name = string("concat_54")]; tensor key_states_47_perm_0 = const()[name = string("key_states_47_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_5_stride_0 = const()[name = string("key_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_47_cast_fp16 = transpose(perm = key_states_47_perm_0, x = key_states_45_cast_fp16)[name = string("transpose_329")]; tensor key_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = key_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = key_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_5_squeeze_mask_0, stride = key_cache_internal_tensor_assign_5_stride_0, update = key_states_47_cast_fp16, x = coreml_update_state_174)[name = string("key_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_5_cast_fp16, input = key_cache)[name = string("coreml_update_state_176_write_state")]; tensor coreml_update_state_176 = read_state(input = key_cache)[name = string("coreml_update_state_176")]; tensor value_states_27_perm_0 = const()[name = string("value_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_5_stride_0 = const()[name = string("value_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_27_cast_fp16 = transpose(perm = value_states_27_perm_0, x = var_2275_cast_fp16)[name = string("transpose_328")]; tensor value_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = value_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = value_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_5_squeeze_mask_0, stride = value_cache_internal_tensor_assign_5_stride_0, update = value_states_27_cast_fp16, x = coreml_update_state_175)[name = string("value_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_5_cast_fp16, input = value_cache)[name = string("coreml_update_state_177_write_state")]; tensor coreml_update_state_177 = read_state(input = value_cache)[name = string("coreml_update_state_177")]; tensor var_2369_begin_0 = const()[name = string("op_2369_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2369_end_0 = const()[name = string("op_2369_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2369_end_mask_0 = const()[name = string("op_2369_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2369_cast_fp16 = slice_by_index(begin = var_2369_begin_0, end = var_2369_end_0, end_mask = var_2369_end_mask_0, x = coreml_update_state_176)[name = string("op_2369_cast_fp16")]; tensor tile_8 = const()[name = string("tile_8"), val = tensor([1, 1])]; int32 var_2372_axis_0 = const()[name = string("op_2372_axis_0"), val = int32(1)]; tensor var_2372_cast_fp16_0, tensor var_2372_cast_fp16_1 = split(axis = var_2372_axis_0, split_sizes = tile_8, x = var_2369_cast_fp16)[name = string("op_2372_cast_fp16")]; tensor var_2379_begin_0 = const()[name = string("op_2379_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2379_end_0 = const()[name = string("op_2379_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2379_end_mask_0 = const()[name = string("op_2379_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2379_cast_fp16 = slice_by_index(begin = var_2379_begin_0, end = var_2379_end_0, end_mask = var_2379_end_mask_0, x = coreml_update_state_177)[name = string("op_2379_cast_fp16")]; tensor tile_9 = const()[name = string("tile_9"), val = tensor([1, 1])]; int32 var_2382_axis_0 = const()[name = string("op_2382_axis_0"), val = int32(1)]; tensor var_2382_cast_fp16_0, tensor var_2382_cast_fp16_1 = split(axis = var_2382_axis_0, split_sizes = tile_9, x = var_2379_cast_fp16)[name = string("op_2382_cast_fp16")]; tensor var_2385_split_sizes_0 = const()[name = string("op_2385_split_sizes_0"), val = tensor([8, 8])]; int32 var_2385_axis_0 = const()[name = string("op_2385_axis_0"), val = int32(1)]; tensor var_2385_0, tensor var_2385_1 = split(axis = var_2385_axis_0, split_sizes = var_2385_split_sizes_0, x = query_states_27_cast_fp16)[name = string("op_2385")]; bool attn_weights_65_transpose_x_0 = const()[name = string("attn_weights_65_transpose_x_0"), val = bool(false)]; bool attn_weights_65_transpose_y_0 = const()[name = string("attn_weights_65_transpose_y_0"), val = bool(false)]; tensor attn_weights_65_cast_fp16 = matmul(transpose_x = attn_weights_65_transpose_x_0, transpose_y = attn_weights_65_transpose_y_0, x = var_2372_cast_fp16_0, y = var_2385_0)[name = string("attn_weights_65_cast_fp16")]; fp16 var_2388_to_fp16 = const()[name = string("op_2388_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_67_cast_fp16 = mul(x = attn_weights_65_cast_fp16, y = var_2388_to_fp16)[name = string("attn_weights_67_cast_fp16")]; tensor attn_weights_69_cast_fp16 = add(x = attn_weights_67_cast_fp16, y = attn_mask_1)[name = string("attn_weights_69_cast_fp16")]; int32 var_2392 = const()[name = string("op_2392"), val = int32(-2)]; tensor attn_weights_71_cast_fp16 = softmax(axis = var_2392, x = attn_weights_69_cast_fp16)[name = string("attn_weights_71_cast_fp16")]; bool var_2398_transpose_x_1 = const()[name = string("op_2398_transpose_x_1"), val = bool(true)]; bool var_2398_transpose_y_1 = const()[name = string("op_2398_transpose_y_1"), val = bool(false)]; tensor var_2398_cast_fp16 = matmul(transpose_x = var_2398_transpose_x_1, transpose_y = var_2398_transpose_y_1, x = attn_weights_71_cast_fp16, y = var_2382_cast_fp16_0)[name = string("op_2398_cast_fp16")]; bool attn_weights_73_transpose_x_0 = const()[name = string("attn_weights_73_transpose_x_0"), val = bool(false)]; bool attn_weights_73_transpose_y_0 = const()[name = string("attn_weights_73_transpose_y_0"), val = bool(false)]; tensor attn_weights_73_cast_fp16 = matmul(transpose_x = attn_weights_73_transpose_x_0, transpose_y = attn_weights_73_transpose_y_0, x = var_2372_cast_fp16_1, y = var_2385_1)[name = string("attn_weights_73_cast_fp16")]; fp16 var_2400_to_fp16 = const()[name = string("op_2400_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_75_cast_fp16 = mul(x = attn_weights_73_cast_fp16, y = var_2400_to_fp16)[name = string("attn_weights_75_cast_fp16")]; tensor attn_weights_77_cast_fp16 = add(x = attn_weights_75_cast_fp16, y = attn_mask_1)[name = string("attn_weights_77_cast_fp16")]; int32 var_2404 = const()[name = string("op_2404"), val = int32(-2)]; tensor attn_weights_79_cast_fp16 = softmax(axis = var_2404, x = attn_weights_77_cast_fp16)[name = string("attn_weights_79_cast_fp16")]; bool attn_output_33_transpose_x_1 = const()[name = string("attn_output_33_transpose_x_1"), val = bool(true)]; bool attn_output_33_transpose_y_1 = const()[name = string("attn_output_33_transpose_y_1"), val = bool(false)]; tensor attn_output_33_cast_fp16 = matmul(transpose_x = attn_output_33_transpose_x_1, transpose_y = attn_output_33_transpose_y_1, x = attn_weights_79_cast_fp16, y = var_2382_cast_fp16_1)[name = string("attn_output_33_cast_fp16")]; int32 var_2412 = const()[name = string("op_2412"), val = int32(1)]; bool attn_output_35_interleave_0 = const()[name = string("attn_output_35_interleave_0"), val = bool(false)]; tensor attn_output_35_cast_fp16 = concat(axis = var_2412, interleave = attn_output_35_interleave_0, values = (var_2398_cast_fp16, attn_output_33_cast_fp16))[name = string("attn_output_35_cast_fp16")]; tensor var_2416_perm_0 = const()[name = string("op_2416_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_59x = const()[name = string("concat_59x"), val = tensor([1, 2048, 1, -1])]; tensor var_2416_cast_fp16 = transpose(perm = var_2416_perm_0, x = attn_output_35_cast_fp16)[name = string("transpose_327")]; tensor attn_output_39_cast_fp16 = reshape(shape = concat_59x, x = var_2416_cast_fp16)[name = string("attn_output_39_cast_fp16")]; tensor hidden_states_43_strides_0 = const()[name = string("hidden_states_43_strides_0"), val = tensor([1, 1])]; string hidden_states_43_pad_type_0 = const()[name = string("hidden_states_43_pad_type_0"), val = string("valid")]; tensor hidden_states_43_pad_0 = const()[name = string("hidden_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_43_dilations_0 = const()[name = string("hidden_states_43_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_43_groups_0 = const()[name = string("hidden_states_43_groups_0"), val = int32(1)]; tensor hidden_states_43_cast_fp16 = conv(dilations = hidden_states_43_dilations_0, groups = hidden_states_43_groups_0, pad = hidden_states_43_pad_0, pad_type = hidden_states_43_pad_type_0, strides = hidden_states_43_strides_0, weight = layers_4_self_attn_o_proj_weight_cast_fp16, x = attn_output_39_cast_fp16)[name = string("hidden_states_43_cast_fp16")]; tensor hidden_states_45_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = hidden_states_43_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; fp16 const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2449_cast_fp16 = mul(x = hidden_states_45_cast_fp16, y = const_48_promoted_to_fp16)[name = string("op_2449_cast_fp16")]; int32 var_2447 = const()[name = string("op_2447"), val = int32(1)]; bool doubled_37_interleave_0 = const()[name = string("doubled_37_interleave_0"), val = bool(false)]; tensor doubled_37_cast_fp16 = concat(axis = var_2447, interleave = doubled_37_interleave_0, values = (hidden_states_45_cast_fp16, var_2449_cast_fp16))[name = string("doubled_37_cast_fp16")]; tensor out_19_axes_0 = const()[name = string("out_19_axes_0"), val = tensor([1])]; tensor out_19_gamma_0_to_fp16 = const()[name = string("out_19_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283129280)))]; fp16 var_2459_to_fp16 = const()[name = string("op_2459_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_19_cast_fp16 = layer_norm(axes = out_19_axes_0, epsilon = var_2459_to_fp16, gamma = out_19_gamma_0_to_fp16, x = doubled_37_cast_fp16)[name = string("out_19_cast_fp16")]; tensor var_2470_split_sizes_0 = const()[name = string("op_2470_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2470_axis_0 = const()[name = string("op_2470_axis_0"), val = int32(1)]; tensor var_2470_cast_fp16_0, tensor var_2470_cast_fp16_1 = split(axis = var_2470_axis_0, split_sizes = var_2470_split_sizes_0, x = out_19_cast_fp16)[name = string("op_2470_cast_fp16")]; tensor input_9_strides_0 = const()[name = string("input_9_strides_0"), val = tensor([1, 1])]; string input_9_pad_type_0 = const()[name = string("input_9_pad_type_0"), val = string("valid")]; tensor input_9_pad_0 = const()[name = string("input_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_9_dilations_0 = const()[name = string("input_9_dilations_0"), val = tensor([1, 1])]; int32 input_9_groups_0 = const()[name = string("input_9_groups_0"), val = int32(1)]; tensor input_9_cast_fp16 = conv(dilations = input_9_dilations_0, groups = input_9_groups_0, pad = input_9_pad_0, pad_type = input_9_pad_type_0, strides = input_9_strides_0, weight = layers_4_mlp_gate_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("input_9_cast_fp16")]; tensor var_2487_cast_fp16 = silu(x = input_9_cast_fp16)[name = string("op_2487_cast_fp16")]; tensor var_2493_strides_0 = const()[name = string("op_2493_strides_0"), val = tensor([1, 1])]; string var_2493_pad_type_0 = const()[name = string("op_2493_pad_type_0"), val = string("valid")]; tensor var_2493_pad_0 = const()[name = string("op_2493_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2493_dilations_0 = const()[name = string("op_2493_dilations_0"), val = tensor([1, 1])]; int32 var_2493_groups_0 = const()[name = string("op_2493_groups_0"), val = int32(1)]; tensor var_2493_cast_fp16 = conv(dilations = var_2493_dilations_0, groups = var_2493_groups_0, pad = var_2493_pad_0, pad_type = var_2493_pad_type_0, strides = var_2493_strides_0, weight = layers_4_mlp_up_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("op_2493_cast_fp16")]; tensor x_49_cast_fp16 = mul(x = var_2487_cast_fp16, y = var_2493_cast_fp16)[name = string("x_49_cast_fp16")]; tensor hidden_states_47_strides_0 = const()[name = string("hidden_states_47_strides_0"), val = tensor([1, 1])]; string hidden_states_47_pad_type_0 = const()[name = string("hidden_states_47_pad_type_0"), val = string("valid")]; tensor hidden_states_47_pad_0 = const()[name = string("hidden_states_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_47_dilations_0 = const()[name = string("hidden_states_47_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_47_groups_0 = const()[name = string("hidden_states_47_groups_0"), val = int32(1)]; tensor hidden_states_47_cast_fp16 = conv(dilations = hidden_states_47_dilations_0, groups = hidden_states_47_groups_0, pad = hidden_states_47_pad_0, pad_type = hidden_states_47_pad_type_0, strides = hidden_states_47_strides_0, weight = layers_4_mlp_down_proj_weight_cast_fp16, x = x_49_cast_fp16)[name = string("hidden_states_47_cast_fp16")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = hidden_states_47_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; fp16 const_50_promoted_to_fp16 = const()[name = string("const_50_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2511_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_50_promoted_to_fp16)[name = string("op_2511_cast_fp16")]; int32 var_2509 = const()[name = string("op_2509"), val = int32(1)]; bool doubled_41_interleave_0 = const()[name = string("doubled_41_interleave_0"), val = bool(false)]; tensor doubled_41_cast_fp16 = concat(axis = var_2509, interleave = doubled_41_interleave_0, values = (hidden_states_49_cast_fp16, var_2511_cast_fp16))[name = string("doubled_41_cast_fp16")]; tensor out_21_axes_0 = const()[name = string("out_21_axes_0"), val = tensor([1])]; tensor out_21_gamma_0_to_fp16 = const()[name = string("out_21_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283137536)))]; fp16 var_2521_to_fp16 = const()[name = string("op_2521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_21_cast_fp16 = layer_norm(axes = out_21_axes_0, epsilon = var_2521_to_fp16, gamma = out_21_gamma_0_to_fp16, x = doubled_41_cast_fp16)[name = string("out_21_cast_fp16")]; tensor var_2532_split_sizes_0 = const()[name = string("op_2532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2532_axis_0 = const()[name = string("op_2532_axis_0"), val = int32(1)]; tensor var_2532_cast_fp16_0, tensor var_2532_cast_fp16_1 = split(axis = var_2532_axis_0, split_sizes = var_2532_split_sizes_0, x = out_21_cast_fp16)[name = string("op_2532_cast_fp16")]; tensor layers_5_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283145792)))]; tensor query_states_31_strides_0 = const()[name = string("query_states_31_strides_0"), val = tensor([1, 1])]; string query_states_31_pad_type_0 = const()[name = string("query_states_31_pad_type_0"), val = string("valid")]; tensor query_states_31_pad_0 = const()[name = string("query_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_31_dilations_0 = const()[name = string("query_states_31_dilations_0"), val = tensor([1, 1])]; int32 query_states_31_groups_0 = const()[name = string("query_states_31_groups_0"), val = int32(1)]; tensor query_states_31_cast_fp16 = conv(dilations = query_states_31_dilations_0, groups = query_states_31_groups_0, pad = query_states_31_pad_0, pad_type = query_states_31_pad_type_0, strides = query_states_31_strides_0, weight = layers_5_self_attn_q_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("query_states_31_cast_fp16")]; tensor layers_5_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1291534464)))]; tensor key_states_51_strides_0 = const()[name = string("key_states_51_strides_0"), val = tensor([1, 1])]; string key_states_51_pad_type_0 = const()[name = string("key_states_51_pad_type_0"), val = string("valid")]; tensor key_states_51_pad_0 = const()[name = string("key_states_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_51_dilations_0 = const()[name = string("key_states_51_dilations_0"), val = tensor([1, 1])]; int32 key_states_51_groups_0 = const()[name = string("key_states_51_groups_0"), val = int32(1)]; tensor key_states_51_cast_fp16 = conv(dilations = key_states_51_dilations_0, groups = key_states_51_groups_0, pad = key_states_51_pad_0, pad_type = key_states_51_pad_type_0, strides = key_states_51_strides_0, weight = layers_5_self_attn_k_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("key_states_51_cast_fp16")]; tensor value_states_31_strides_0 = const()[name = string("value_states_31_strides_0"), val = tensor([1, 1])]; string value_states_31_pad_type_0 = const()[name = string("value_states_31_pad_type_0"), val = string("valid")]; tensor value_states_31_pad_0 = const()[name = string("value_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_31_dilations_0 = const()[name = string("value_states_31_dilations_0"), val = tensor([1, 1])]; int32 value_states_31_groups_0 = const()[name = string("value_states_31_groups_0"), val = int32(1)]; tensor value_states_31_cast_fp16 = conv(dilations = value_states_31_dilations_0, groups = value_states_31_groups_0, pad = value_states_31_pad_0, pad_type = value_states_31_pad_type_0, strides = value_states_31_strides_0, weight = layers_5_self_attn_v_proj_weight_cast_fp16, x = var_2532_cast_fp16_0)[name = string("value_states_31_cast_fp16")]; tensor concat_60x = const()[name = string("concat_60x"), val = tensor([1, 16, 128, -1])]; tensor x_51_cast_fp16 = reshape(shape = concat_60x, x = query_states_31_cast_fp16)[name = string("x_51_cast_fp16")]; tensor concat_61x = const()[name = string("concat_61x"), val = tensor([1, 2, 128, -1])]; tensor var_2589_cast_fp16 = reshape(shape = concat_61x, x = key_states_51_cast_fp16)[name = string("op_2589_cast_fp16")]; tensor concat_62x = const()[name = string("concat_62x"), val = tensor([1, 2, 128, -1])]; tensor var_2596_cast_fp16 = reshape(shape = concat_62x, x = value_states_31_cast_fp16)[name = string("op_2596_cast_fp16")]; tensor var_2600_cast_fp16 = mul(x = x_51_cast_fp16, y = var_869_cast_fp16)[name = string("op_2600_cast_fp16")]; tensor var_2601_split_sizes_0 = const()[name = string("op_2601_split_sizes_0"), val = tensor([64, 64])]; int32 var_2601_axis_0 = const()[name = string("op_2601_axis_0"), val = int32(-2)]; tensor var_2601_cast_fp16_0, tensor var_2601_cast_fp16_1 = split(axis = var_2601_axis_0, split_sizes = var_2601_split_sizes_0, x = x_51_cast_fp16)[name = string("op_2601_cast_fp16")]; fp16 const_52_promoted_to_fp16 = const()[name = string("const_52_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2603_cast_fp16 = mul(x = var_2601_cast_fp16_1, y = const_52_promoted_to_fp16)[name = string("op_2603_cast_fp16")]; int32 var_2605 = const()[name = string("op_2605"), val = int32(-2)]; bool var_2606_interleave_0 = const()[name = string("op_2606_interleave_0"), val = bool(false)]; tensor var_2606_cast_fp16 = concat(axis = var_2605, interleave = var_2606_interleave_0, values = (var_2603_cast_fp16, var_2601_cast_fp16_0))[name = string("op_2606_cast_fp16")]; tensor var_2607_cast_fp16 = mul(x = var_2606_cast_fp16, y = var_878_cast_fp16)[name = string("op_2607_cast_fp16")]; tensor query_states_33_cast_fp16 = add(x = var_2600_cast_fp16, y = var_2607_cast_fp16)[name = string("query_states_33_cast_fp16")]; tensor var_2613_cast_fp16 = mul(x = var_2589_cast_fp16, y = var_869_cast_fp16)[name = string("op_2613_cast_fp16")]; tensor var_2614_split_sizes_0 = const()[name = string("op_2614_split_sizes_0"), val = tensor([64, 64])]; int32 var_2614_axis_0 = const()[name = string("op_2614_axis_0"), val = int32(-2)]; tensor var_2614_cast_fp16_0, tensor var_2614_cast_fp16_1 = split(axis = var_2614_axis_0, split_sizes = var_2614_split_sizes_0, x = var_2589_cast_fp16)[name = string("op_2614_cast_fp16")]; fp16 const_53_promoted_to_fp16 = const()[name = string("const_53_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2616_cast_fp16 = mul(x = var_2614_cast_fp16_1, y = const_53_promoted_to_fp16)[name = string("op_2616_cast_fp16")]; int32 var_2618 = const()[name = string("op_2618"), val = int32(-2)]; bool var_2619_interleave_0 = const()[name = string("op_2619_interleave_0"), val = bool(false)]; tensor var_2619_cast_fp16 = concat(axis = var_2618, interleave = var_2619_interleave_0, values = (var_2616_cast_fp16, var_2614_cast_fp16_0))[name = string("op_2619_cast_fp16")]; tensor var_2620_cast_fp16 = mul(x = var_2619_cast_fp16, y = var_878_cast_fp16)[name = string("op_2620_cast_fp16")]; tensor key_states_55_cast_fp16 = add(x = var_2613_cast_fp16, y = var_2620_cast_fp16)[name = string("key_states_55_cast_fp16")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; int32 concat_65_axis_0 = const()[name = string("concat_65_axis_0"), val = int32(0)]; bool concat_65_interleave_0 = const()[name = string("concat_65_interleave_0"), val = bool(false)]; tensor concat_65 = concat(axis = concat_65_axis_0, interleave = concat_65_interleave_0, values = (expand_dims_60, expand_dims_61, position_id, expand_dims_63))[name = string("concat_65")]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; tensor concat_66_values1_0 = const()[name = string("concat_66_values1_0"), val = tensor([0])]; tensor concat_66_values3_0 = const()[name = string("concat_66_values3_0"), val = tensor([0])]; int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_64, concat_66_values1_0, cache_position_end, concat_66_values3_0))[name = string("concat_66")]; tensor key_states_57_perm_0 = const()[name = string("key_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_6_stride_0 = const()[name = string("key_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_57_cast_fp16 = transpose(perm = key_states_57_perm_0, x = key_states_55_cast_fp16)[name = string("transpose_326")]; tensor key_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = key_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = key_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_6_squeeze_mask_0, stride = key_cache_internal_tensor_assign_6_stride_0, update = key_states_57_cast_fp16, x = coreml_update_state_176)[name = string("key_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_6_cast_fp16, input = key_cache)[name = string("coreml_update_state_178_write_state")]; tensor coreml_update_state_178 = read_state(input = key_cache)[name = string("coreml_update_state_178")]; tensor value_states_33_perm_0 = const()[name = string("value_states_33_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_6_stride_0 = const()[name = string("value_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_33_cast_fp16 = transpose(perm = value_states_33_perm_0, x = var_2596_cast_fp16)[name = string("transpose_325")]; tensor value_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = value_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = value_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_6_squeeze_mask_0, stride = value_cache_internal_tensor_assign_6_stride_0, update = value_states_33_cast_fp16, x = coreml_update_state_177)[name = string("value_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_6_cast_fp16, input = value_cache)[name = string("coreml_update_state_179_write_state")]; tensor coreml_update_state_179 = read_state(input = value_cache)[name = string("coreml_update_state_179")]; tensor var_2690_begin_0 = const()[name = string("op_2690_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2690_end_0 = const()[name = string("op_2690_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2690_end_mask_0 = const()[name = string("op_2690_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2690_cast_fp16 = slice_by_index(begin = var_2690_begin_0, end = var_2690_end_0, end_mask = var_2690_end_mask_0, x = coreml_update_state_178)[name = string("op_2690_cast_fp16")]; tensor tile_10 = const()[name = string("tile_10"), val = tensor([1, 1])]; int32 var_2693_axis_0 = const()[name = string("op_2693_axis_0"), val = int32(1)]; tensor var_2693_cast_fp16_0, tensor var_2693_cast_fp16_1 = split(axis = var_2693_axis_0, split_sizes = tile_10, x = var_2690_cast_fp16)[name = string("op_2693_cast_fp16")]; tensor var_2700_begin_0 = const()[name = string("op_2700_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2700_end_0 = const()[name = string("op_2700_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2700_end_mask_0 = const()[name = string("op_2700_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2700_cast_fp16 = slice_by_index(begin = var_2700_begin_0, end = var_2700_end_0, end_mask = var_2700_end_mask_0, x = coreml_update_state_179)[name = string("op_2700_cast_fp16")]; tensor tile_11 = const()[name = string("tile_11"), val = tensor([1, 1])]; int32 var_2703_axis_0 = const()[name = string("op_2703_axis_0"), val = int32(1)]; tensor var_2703_cast_fp16_0, tensor var_2703_cast_fp16_1 = split(axis = var_2703_axis_0, split_sizes = tile_11, x = var_2700_cast_fp16)[name = string("op_2703_cast_fp16")]; tensor var_2706_split_sizes_0 = const()[name = string("op_2706_split_sizes_0"), val = tensor([8, 8])]; int32 var_2706_axis_0 = const()[name = string("op_2706_axis_0"), val = int32(1)]; tensor var_2706_0, tensor var_2706_1 = split(axis = var_2706_axis_0, split_sizes = var_2706_split_sizes_0, x = query_states_33_cast_fp16)[name = string("op_2706")]; bool attn_weights_81_transpose_x_0 = const()[name = string("attn_weights_81_transpose_x_0"), val = bool(false)]; bool attn_weights_81_transpose_y_0 = const()[name = string("attn_weights_81_transpose_y_0"), val = bool(false)]; tensor attn_weights_81_cast_fp16 = matmul(transpose_x = attn_weights_81_transpose_x_0, transpose_y = attn_weights_81_transpose_y_0, x = var_2693_cast_fp16_0, y = var_2706_0)[name = string("attn_weights_81_cast_fp16")]; fp16 var_2709_to_fp16 = const()[name = string("op_2709_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_83_cast_fp16 = mul(x = attn_weights_81_cast_fp16, y = var_2709_to_fp16)[name = string("attn_weights_83_cast_fp16")]; tensor attn_weights_85_cast_fp16 = add(x = attn_weights_83_cast_fp16, y = attn_mask_1)[name = string("attn_weights_85_cast_fp16")]; int32 var_2713 = const()[name = string("op_2713"), val = int32(-2)]; tensor attn_weights_87_cast_fp16 = softmax(axis = var_2713, x = attn_weights_85_cast_fp16)[name = string("attn_weights_87_cast_fp16")]; bool var_2719_transpose_x_1 = const()[name = string("op_2719_transpose_x_1"), val = bool(true)]; bool var_2719_transpose_y_1 = const()[name = string("op_2719_transpose_y_1"), val = bool(false)]; tensor var_2719_cast_fp16 = matmul(transpose_x = var_2719_transpose_x_1, transpose_y = var_2719_transpose_y_1, x = attn_weights_87_cast_fp16, y = var_2703_cast_fp16_0)[name = string("op_2719_cast_fp16")]; bool attn_weights_89_transpose_x_0 = const()[name = string("attn_weights_89_transpose_x_0"), val = bool(false)]; bool attn_weights_89_transpose_y_0 = const()[name = string("attn_weights_89_transpose_y_0"), val = bool(false)]; tensor attn_weights_89_cast_fp16 = matmul(transpose_x = attn_weights_89_transpose_x_0, transpose_y = attn_weights_89_transpose_y_0, x = var_2693_cast_fp16_1, y = var_2706_1)[name = string("attn_weights_89_cast_fp16")]; fp16 var_2721_to_fp16 = const()[name = string("op_2721_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_91_cast_fp16 = mul(x = attn_weights_89_cast_fp16, y = var_2721_to_fp16)[name = string("attn_weights_91_cast_fp16")]; tensor attn_weights_93_cast_fp16 = add(x = attn_weights_91_cast_fp16, y = attn_mask_1)[name = string("attn_weights_93_cast_fp16")]; int32 var_2725 = const()[name = string("op_2725"), val = int32(-2)]; tensor attn_weights_95_cast_fp16 = softmax(axis = var_2725, x = attn_weights_93_cast_fp16)[name = string("attn_weights_95_cast_fp16")]; bool attn_output_41_transpose_x_1 = const()[name = string("attn_output_41_transpose_x_1"), val = bool(true)]; bool attn_output_41_transpose_y_1 = const()[name = string("attn_output_41_transpose_y_1"), val = bool(false)]; tensor attn_output_41_cast_fp16 = matmul(transpose_x = attn_output_41_transpose_x_1, transpose_y = attn_output_41_transpose_y_1, x = attn_weights_95_cast_fp16, y = var_2703_cast_fp16_1)[name = string("attn_output_41_cast_fp16")]; int32 var_2733 = const()[name = string("op_2733"), val = int32(1)]; bool attn_output_43_interleave_0 = const()[name = string("attn_output_43_interleave_0"), val = bool(false)]; tensor attn_output_43_cast_fp16 = concat(axis = var_2733, interleave = attn_output_43_interleave_0, values = (var_2719_cast_fp16, attn_output_41_cast_fp16))[name = string("attn_output_43_cast_fp16")]; tensor var_2737_perm_0 = const()[name = string("op_2737_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_71x = const()[name = string("concat_71x"), val = tensor([1, 2048, 1, -1])]; tensor var_2737_cast_fp16 = transpose(perm = var_2737_perm_0, x = attn_output_43_cast_fp16)[name = string("transpose_324")]; tensor attn_output_47_cast_fp16 = reshape(shape = concat_71x, x = var_2737_cast_fp16)[name = string("attn_output_47_cast_fp16")]; tensor hidden_states_53_strides_0 = const()[name = string("hidden_states_53_strides_0"), val = tensor([1, 1])]; string hidden_states_53_pad_type_0 = const()[name = string("hidden_states_53_pad_type_0"), val = string("valid")]; tensor hidden_states_53_pad_0 = const()[name = string("hidden_states_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_53_dilations_0 = const()[name = string("hidden_states_53_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_53_groups_0 = const()[name = string("hidden_states_53_groups_0"), val = int32(1)]; tensor hidden_states_53_cast_fp16 = conv(dilations = hidden_states_53_dilations_0, groups = hidden_states_53_groups_0, pad = hidden_states_53_pad_0, pad_type = hidden_states_53_pad_type_0, strides = hidden_states_53_strides_0, weight = layers_5_self_attn_o_proj_weight_cast_fp16, x = attn_output_47_cast_fp16)[name = string("hidden_states_53_cast_fp16")]; tensor hidden_states_55_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = hidden_states_53_cast_fp16)[name = string("hidden_states_55_cast_fp16")]; fp16 const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2770_cast_fp16 = mul(x = hidden_states_55_cast_fp16, y = const_58_promoted_to_fp16)[name = string("op_2770_cast_fp16")]; int32 var_2768 = const()[name = string("op_2768"), val = int32(1)]; bool doubled_45_interleave_0 = const()[name = string("doubled_45_interleave_0"), val = bool(false)]; tensor doubled_45_cast_fp16 = concat(axis = var_2768, interleave = doubled_45_interleave_0, values = (hidden_states_55_cast_fp16, var_2770_cast_fp16))[name = string("doubled_45_cast_fp16")]; tensor out_23_axes_0 = const()[name = string("out_23_axes_0"), val = tensor([1])]; tensor out_23_gamma_0_to_fp16 = const()[name = string("out_23_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292583104)))]; fp16 var_2780_to_fp16 = const()[name = string("op_2780_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_23_cast_fp16 = layer_norm(axes = out_23_axes_0, epsilon = var_2780_to_fp16, gamma = out_23_gamma_0_to_fp16, x = doubled_45_cast_fp16)[name = string("out_23_cast_fp16")]; tensor var_2791_split_sizes_0 = const()[name = string("op_2791_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2791_axis_0 = const()[name = string("op_2791_axis_0"), val = int32(1)]; tensor var_2791_cast_fp16_0, tensor var_2791_cast_fp16_1 = split(axis = var_2791_axis_0, split_sizes = var_2791_split_sizes_0, x = out_23_cast_fp16)[name = string("op_2791_cast_fp16")]; tensor layers_5_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_5_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292591360)))]; tensor input_11_strides_0 = const()[name = string("input_11_strides_0"), val = tensor([1, 1])]; string input_11_pad_type_0 = const()[name = string("input_11_pad_type_0"), val = string("valid")]; tensor input_11_pad_0 = const()[name = string("input_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_11_dilations_0 = const()[name = string("input_11_dilations_0"), val = tensor([1, 1])]; int32 input_11_groups_0 = const()[name = string("input_11_groups_0"), val = int32(1)]; tensor input_11_cast_fp16 = conv(dilations = input_11_dilations_0, groups = input_11_groups_0, pad = input_11_pad_0, pad_type = input_11_pad_type_0, strides = input_11_strides_0, weight = layers_5_mlp_gate_proj_weight_to_fp16, x = var_2791_cast_fp16_0)[name = string("input_11_cast_fp16")]; tensor var_2808_cast_fp16 = silu(x = input_11_cast_fp16)[name = string("op_2808_cast_fp16")]; tensor var_2814_strides_0 = const()[name = string("op_2814_strides_0"), val = tensor([1, 1])]; string var_2814_pad_type_0 = const()[name = string("op_2814_pad_type_0"), val = string("valid")]; tensor var_2814_pad_0 = const()[name = string("op_2814_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2814_dilations_0 = const()[name = string("op_2814_dilations_0"), val = tensor([1, 1])]; int32 var_2814_groups_0 = const()[name = string("op_2814_groups_0"), val = int32(1)]; tensor var_2814_cast_fp16 = conv(dilations = var_2814_dilations_0, groups = var_2814_groups_0, pad = var_2814_pad_0, pad_type = var_2814_pad_type_0, strides = var_2814_strides_0, weight = layers_5_mlp_up_proj_weight_cast_fp16, x = var_2791_cast_fp16_0)[name = string("op_2814_cast_fp16")]; tensor x_59_cast_fp16 = mul(x = var_2808_cast_fp16, y = var_2814_cast_fp16)[name = string("x_59_cast_fp16")]; tensor hidden_states_57_strides_0 = const()[name = string("hidden_states_57_strides_0"), val = tensor([1, 1])]; string hidden_states_57_pad_type_0 = const()[name = string("hidden_states_57_pad_type_0"), val = string("valid")]; tensor hidden_states_57_pad_0 = const()[name = string("hidden_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_57_dilations_0 = const()[name = string("hidden_states_57_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_57_groups_0 = const()[name = string("hidden_states_57_groups_0"), val = int32(1)]; tensor hidden_states_57_cast_fp16 = conv(dilations = hidden_states_57_dilations_0, groups = hidden_states_57_groups_0, pad = hidden_states_57_pad_0, pad_type = hidden_states_57_pad_type_0, strides = hidden_states_57_strides_0, weight = layers_5_mlp_down_proj_weight_cast_fp16, x = x_59_cast_fp16)[name = string("hidden_states_57_cast_fp16")]; tensor hidden_states_59_cast_fp16 = add(x = hidden_states_55_cast_fp16, y = hidden_states_57_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2832_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_2832_cast_fp16")]; int32 var_2830 = const()[name = string("op_2830"), val = int32(1)]; bool doubled_49_interleave_0 = const()[name = string("doubled_49_interleave_0"), val = bool(false)]; tensor doubled_49_cast_fp16 = concat(axis = var_2830, interleave = doubled_49_interleave_0, values = (hidden_states_59_cast_fp16, var_2832_cast_fp16))[name = string("doubled_49_cast_fp16")]; tensor out_25_axes_0 = const()[name = string("out_25_axes_0"), val = tensor([1])]; tensor out_25_gamma_0_to_fp16 = const()[name = string("out_25_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317757248)))]; fp16 var_2842_to_fp16 = const()[name = string("op_2842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_25_cast_fp16 = layer_norm(axes = out_25_axes_0, epsilon = var_2842_to_fp16, gamma = out_25_gamma_0_to_fp16, x = doubled_49_cast_fp16)[name = string("out_25_cast_fp16")]; tensor var_2853_split_sizes_0 = const()[name = string("op_2853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2853_axis_0 = const()[name = string("op_2853_axis_0"), val = int32(1)]; tensor var_2853_cast_fp16_0, tensor var_2853_cast_fp16_1 = split(axis = var_2853_axis_0, split_sizes = var_2853_split_sizes_0, x = out_25_cast_fp16)[name = string("op_2853_cast_fp16")]; tensor layers_6_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317765504)))]; tensor query_states_37_strides_0 = const()[name = string("query_states_37_strides_0"), val = tensor([1, 1])]; string query_states_37_pad_type_0 = const()[name = string("query_states_37_pad_type_0"), val = string("valid")]; tensor query_states_37_pad_0 = const()[name = string("query_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_37_dilations_0 = const()[name = string("query_states_37_dilations_0"), val = tensor([1, 1])]; int32 query_states_37_groups_0 = const()[name = string("query_states_37_groups_0"), val = int32(1)]; tensor query_states_37_cast_fp16 = conv(dilations = query_states_37_dilations_0, groups = query_states_37_groups_0, pad = query_states_37_pad_0, pad_type = query_states_37_pad_type_0, strides = query_states_37_strides_0, weight = layers_6_self_attn_q_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("query_states_37_cast_fp16")]; tensor layers_6_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1326154176)))]; tensor key_states_61_strides_0 = const()[name = string("key_states_61_strides_0"), val = tensor([1, 1])]; string key_states_61_pad_type_0 = const()[name = string("key_states_61_pad_type_0"), val = string("valid")]; tensor key_states_61_pad_0 = const()[name = string("key_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_61_dilations_0 = const()[name = string("key_states_61_dilations_0"), val = tensor([1, 1])]; int32 key_states_61_groups_0 = const()[name = string("key_states_61_groups_0"), val = int32(1)]; tensor key_states_61_cast_fp16 = conv(dilations = key_states_61_dilations_0, groups = key_states_61_groups_0, pad = key_states_61_pad_0, pad_type = key_states_61_pad_type_0, strides = key_states_61_strides_0, weight = layers_6_self_attn_k_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("key_states_61_cast_fp16")]; tensor value_states_37_strides_0 = const()[name = string("value_states_37_strides_0"), val = tensor([1, 1])]; string value_states_37_pad_type_0 = const()[name = string("value_states_37_pad_type_0"), val = string("valid")]; tensor value_states_37_pad_0 = const()[name = string("value_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_37_dilations_0 = const()[name = string("value_states_37_dilations_0"), val = tensor([1, 1])]; int32 value_states_37_groups_0 = const()[name = string("value_states_37_groups_0"), val = int32(1)]; tensor value_states_37_cast_fp16 = conv(dilations = value_states_37_dilations_0, groups = value_states_37_groups_0, pad = value_states_37_pad_0, pad_type = value_states_37_pad_type_0, strides = value_states_37_strides_0, weight = layers_6_self_attn_v_proj_weight_cast_fp16, x = var_2853_cast_fp16_0)[name = string("value_states_37_cast_fp16")]; tensor concat_72x = const()[name = string("concat_72x"), val = tensor([1, 16, 128, -1])]; tensor x_61_cast_fp16 = reshape(shape = concat_72x, x = query_states_37_cast_fp16)[name = string("x_61_cast_fp16")]; tensor concat_73x = const()[name = string("concat_73x"), val = tensor([1, 2, 128, -1])]; tensor var_2910_cast_fp16 = reshape(shape = concat_73x, x = key_states_61_cast_fp16)[name = string("op_2910_cast_fp16")]; tensor concat_74x = const()[name = string("concat_74x"), val = tensor([1, 2, 128, -1])]; tensor var_2917_cast_fp16 = reshape(shape = concat_74x, x = value_states_37_cast_fp16)[name = string("op_2917_cast_fp16")]; tensor var_2921_cast_fp16 = mul(x = x_61_cast_fp16, y = var_869_cast_fp16)[name = string("op_2921_cast_fp16")]; tensor var_2922_split_sizes_0 = const()[name = string("op_2922_split_sizes_0"), val = tensor([64, 64])]; int32 var_2922_axis_0 = const()[name = string("op_2922_axis_0"), val = int32(-2)]; tensor var_2922_cast_fp16_0, tensor var_2922_cast_fp16_1 = split(axis = var_2922_axis_0, split_sizes = var_2922_split_sizes_0, x = x_61_cast_fp16)[name = string("op_2922_cast_fp16")]; fp16 const_62_promoted_to_fp16 = const()[name = string("const_62_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2924_cast_fp16 = mul(x = var_2922_cast_fp16_1, y = const_62_promoted_to_fp16)[name = string("op_2924_cast_fp16")]; int32 var_2926 = const()[name = string("op_2926"), val = int32(-2)]; bool var_2927_interleave_0 = const()[name = string("op_2927_interleave_0"), val = bool(false)]; tensor var_2927_cast_fp16 = concat(axis = var_2926, interleave = var_2927_interleave_0, values = (var_2924_cast_fp16, var_2922_cast_fp16_0))[name = string("op_2927_cast_fp16")]; tensor var_2928_cast_fp16 = mul(x = var_2927_cast_fp16, y = var_878_cast_fp16)[name = string("op_2928_cast_fp16")]; tensor query_states_39_cast_fp16 = add(x = var_2921_cast_fp16, y = var_2928_cast_fp16)[name = string("query_states_39_cast_fp16")]; tensor var_2934_cast_fp16 = mul(x = var_2910_cast_fp16, y = var_869_cast_fp16)[name = string("op_2934_cast_fp16")]; tensor var_2935_split_sizes_0 = const()[name = string("op_2935_split_sizes_0"), val = tensor([64, 64])]; int32 var_2935_axis_0 = const()[name = string("op_2935_axis_0"), val = int32(-2)]; tensor var_2935_cast_fp16_0, tensor var_2935_cast_fp16_1 = split(axis = var_2935_axis_0, split_sizes = var_2935_split_sizes_0, x = var_2910_cast_fp16)[name = string("op_2935_cast_fp16")]; fp16 const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2937_cast_fp16 = mul(x = var_2935_cast_fp16_1, y = const_63_promoted_to_fp16)[name = string("op_2937_cast_fp16")]; int32 var_2939 = const()[name = string("op_2939"), val = int32(-2)]; bool var_2940_interleave_0 = const()[name = string("op_2940_interleave_0"), val = bool(false)]; tensor var_2940_cast_fp16 = concat(axis = var_2939, interleave = var_2940_interleave_0, values = (var_2937_cast_fp16, var_2935_cast_fp16_0))[name = string("op_2940_cast_fp16")]; tensor var_2941_cast_fp16 = mul(x = var_2940_cast_fp16, y = var_878_cast_fp16)[name = string("op_2941_cast_fp16")]; tensor key_states_65_cast_fp16 = add(x = var_2934_cast_fp16, y = var_2941_cast_fp16)[name = string("key_states_65_cast_fp16")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; int32 concat_77_axis_0 = const()[name = string("concat_77_axis_0"), val = int32(0)]; bool concat_77_interleave_0 = const()[name = string("concat_77_interleave_0"), val = bool(false)]; tensor concat_77 = concat(axis = concat_77_axis_0, interleave = concat_77_interleave_0, values = (expand_dims_72, expand_dims_73, position_id, expand_dims_75))[name = string("concat_77")]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; tensor concat_78_values1_0 = const()[name = string("concat_78_values1_0"), val = tensor([0])]; tensor concat_78_values3_0 = const()[name = string("concat_78_values3_0"), val = tensor([0])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_76, concat_78_values1_0, cache_position_end, concat_78_values3_0))[name = string("concat_78")]; tensor key_states_67_perm_0 = const()[name = string("key_states_67_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_7_stride_0 = const()[name = string("key_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_67_cast_fp16 = transpose(perm = key_states_67_perm_0, x = key_states_65_cast_fp16)[name = string("transpose_323")]; tensor key_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = key_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = key_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_7_squeeze_mask_0, stride = key_cache_internal_tensor_assign_7_stride_0, update = key_states_67_cast_fp16, x = coreml_update_state_178)[name = string("key_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_7_cast_fp16, input = key_cache)[name = string("coreml_update_state_180_write_state")]; tensor coreml_update_state_180 = read_state(input = key_cache)[name = string("coreml_update_state_180")]; tensor value_states_39_perm_0 = const()[name = string("value_states_39_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_7_stride_0 = const()[name = string("value_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_39_cast_fp16 = transpose(perm = value_states_39_perm_0, x = var_2917_cast_fp16)[name = string("transpose_322")]; tensor value_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = value_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = value_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_7_squeeze_mask_0, stride = value_cache_internal_tensor_assign_7_stride_0, update = value_states_39_cast_fp16, x = coreml_update_state_179)[name = string("value_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_7_cast_fp16, input = value_cache)[name = string("coreml_update_state_181_write_state")]; tensor coreml_update_state_181 = read_state(input = value_cache)[name = string("coreml_update_state_181")]; tensor var_3011_begin_0 = const()[name = string("op_3011_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3011_end_0 = const()[name = string("op_3011_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3011_end_mask_0 = const()[name = string("op_3011_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3011_cast_fp16 = slice_by_index(begin = var_3011_begin_0, end = var_3011_end_0, end_mask = var_3011_end_mask_0, x = coreml_update_state_180)[name = string("op_3011_cast_fp16")]; tensor tile_12 = const()[name = string("tile_12"), val = tensor([1, 1])]; int32 var_3014_axis_0 = const()[name = string("op_3014_axis_0"), val = int32(1)]; tensor var_3014_cast_fp16_0, tensor var_3014_cast_fp16_1 = split(axis = var_3014_axis_0, split_sizes = tile_12, x = var_3011_cast_fp16)[name = string("op_3014_cast_fp16")]; tensor var_3021_begin_0 = const()[name = string("op_3021_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3021_end_0 = const()[name = string("op_3021_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3021_end_mask_0 = const()[name = string("op_3021_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3021_cast_fp16 = slice_by_index(begin = var_3021_begin_0, end = var_3021_end_0, end_mask = var_3021_end_mask_0, x = coreml_update_state_181)[name = string("op_3021_cast_fp16")]; tensor tile_13 = const()[name = string("tile_13"), val = tensor([1, 1])]; int32 var_3024_axis_0 = const()[name = string("op_3024_axis_0"), val = int32(1)]; tensor var_3024_cast_fp16_0, tensor var_3024_cast_fp16_1 = split(axis = var_3024_axis_0, split_sizes = tile_13, x = var_3021_cast_fp16)[name = string("op_3024_cast_fp16")]; tensor var_3027_split_sizes_0 = const()[name = string("op_3027_split_sizes_0"), val = tensor([8, 8])]; int32 var_3027_axis_0 = const()[name = string("op_3027_axis_0"), val = int32(1)]; tensor var_3027_0, tensor var_3027_1 = split(axis = var_3027_axis_0, split_sizes = var_3027_split_sizes_0, x = query_states_39_cast_fp16)[name = string("op_3027")]; bool attn_weights_97_transpose_x_0 = const()[name = string("attn_weights_97_transpose_x_0"), val = bool(false)]; bool attn_weights_97_transpose_y_0 = const()[name = string("attn_weights_97_transpose_y_0"), val = bool(false)]; tensor attn_weights_97_cast_fp16 = matmul(transpose_x = attn_weights_97_transpose_x_0, transpose_y = attn_weights_97_transpose_y_0, x = var_3014_cast_fp16_0, y = var_3027_0)[name = string("attn_weights_97_cast_fp16")]; fp16 var_3030_to_fp16 = const()[name = string("op_3030_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_99_cast_fp16 = mul(x = attn_weights_97_cast_fp16, y = var_3030_to_fp16)[name = string("attn_weights_99_cast_fp16")]; tensor attn_weights_101_cast_fp16 = add(x = attn_weights_99_cast_fp16, y = attn_mask_1)[name = string("attn_weights_101_cast_fp16")]; int32 var_3034 = const()[name = string("op_3034"), val = int32(-2)]; tensor attn_weights_103_cast_fp16 = softmax(axis = var_3034, x = attn_weights_101_cast_fp16)[name = string("attn_weights_103_cast_fp16")]; bool var_3040_transpose_x_1 = const()[name = string("op_3040_transpose_x_1"), val = bool(true)]; bool var_3040_transpose_y_1 = const()[name = string("op_3040_transpose_y_1"), val = bool(false)]; tensor var_3040_cast_fp16 = matmul(transpose_x = var_3040_transpose_x_1, transpose_y = var_3040_transpose_y_1, x = attn_weights_103_cast_fp16, y = var_3024_cast_fp16_0)[name = string("op_3040_cast_fp16")]; bool attn_weights_105_transpose_x_0 = const()[name = string("attn_weights_105_transpose_x_0"), val = bool(false)]; bool attn_weights_105_transpose_y_0 = const()[name = string("attn_weights_105_transpose_y_0"), val = bool(false)]; tensor attn_weights_105_cast_fp16 = matmul(transpose_x = attn_weights_105_transpose_x_0, transpose_y = attn_weights_105_transpose_y_0, x = var_3014_cast_fp16_1, y = var_3027_1)[name = string("attn_weights_105_cast_fp16")]; fp16 var_3042_to_fp16 = const()[name = string("op_3042_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_107_cast_fp16 = mul(x = attn_weights_105_cast_fp16, y = var_3042_to_fp16)[name = string("attn_weights_107_cast_fp16")]; tensor attn_weights_109_cast_fp16 = add(x = attn_weights_107_cast_fp16, y = attn_mask_1)[name = string("attn_weights_109_cast_fp16")]; int32 var_3046 = const()[name = string("op_3046"), val = int32(-2)]; tensor attn_weights_111_cast_fp16 = softmax(axis = var_3046, x = attn_weights_109_cast_fp16)[name = string("attn_weights_111_cast_fp16")]; bool attn_output_49_transpose_x_1 = const()[name = string("attn_output_49_transpose_x_1"), val = bool(true)]; bool attn_output_49_transpose_y_1 = const()[name = string("attn_output_49_transpose_y_1"), val = bool(false)]; tensor attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_1, transpose_y = attn_output_49_transpose_y_1, x = attn_weights_111_cast_fp16, y = var_3024_cast_fp16_1)[name = string("attn_output_49_cast_fp16")]; int32 var_3054 = const()[name = string("op_3054"), val = int32(1)]; bool attn_output_51_interleave_0 = const()[name = string("attn_output_51_interleave_0"), val = bool(false)]; tensor attn_output_51_cast_fp16 = concat(axis = var_3054, interleave = attn_output_51_interleave_0, values = (var_3040_cast_fp16, attn_output_49_cast_fp16))[name = string("attn_output_51_cast_fp16")]; tensor var_3058_perm_0 = const()[name = string("op_3058_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_83x = const()[name = string("concat_83x"), val = tensor([1, 2048, 1, -1])]; tensor var_3058_cast_fp16 = transpose(perm = var_3058_perm_0, x = attn_output_51_cast_fp16)[name = string("transpose_321")]; tensor attn_output_55_cast_fp16 = reshape(shape = concat_83x, x = var_3058_cast_fp16)[name = string("attn_output_55_cast_fp16")]; tensor hidden_states_63_strides_0 = const()[name = string("hidden_states_63_strides_0"), val = tensor([1, 1])]; string hidden_states_63_pad_type_0 = const()[name = string("hidden_states_63_pad_type_0"), val = string("valid")]; tensor hidden_states_63_pad_0 = const()[name = string("hidden_states_63_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_63_dilations_0 = const()[name = string("hidden_states_63_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_63_groups_0 = const()[name = string("hidden_states_63_groups_0"), val = int32(1)]; tensor hidden_states_63_cast_fp16 = conv(dilations = hidden_states_63_dilations_0, groups = hidden_states_63_groups_0, pad = hidden_states_63_pad_0, pad_type = hidden_states_63_pad_type_0, strides = hidden_states_63_strides_0, weight = layers_6_self_attn_o_proj_weight_cast_fp16, x = attn_output_55_cast_fp16)[name = string("hidden_states_63_cast_fp16")]; tensor hidden_states_65_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = hidden_states_63_cast_fp16)[name = string("hidden_states_65_cast_fp16")]; fp16 const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3091_cast_fp16 = mul(x = hidden_states_65_cast_fp16, y = const_68_promoted_to_fp16)[name = string("op_3091_cast_fp16")]; int32 var_3089 = const()[name = string("op_3089"), val = int32(1)]; bool doubled_53_interleave_0 = const()[name = string("doubled_53_interleave_0"), val = bool(false)]; tensor doubled_53_cast_fp16 = concat(axis = var_3089, interleave = doubled_53_interleave_0, values = (hidden_states_65_cast_fp16, var_3091_cast_fp16))[name = string("doubled_53_cast_fp16")]; tensor out_27_axes_0 = const()[name = string("out_27_axes_0"), val = tensor([1])]; tensor out_27_gamma_0_to_fp16 = const()[name = string("out_27_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327202816)))]; fp16 var_3101_to_fp16 = const()[name = string("op_3101_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_27_cast_fp16 = layer_norm(axes = out_27_axes_0, epsilon = var_3101_to_fp16, gamma = out_27_gamma_0_to_fp16, x = doubled_53_cast_fp16)[name = string("out_27_cast_fp16")]; tensor var_3112_split_sizes_0 = const()[name = string("op_3112_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3112_axis_0 = const()[name = string("op_3112_axis_0"), val = int32(1)]; tensor var_3112_cast_fp16_0, tensor var_3112_cast_fp16_1 = split(axis = var_3112_axis_0, split_sizes = var_3112_split_sizes_0, x = out_27_cast_fp16)[name = string("op_3112_cast_fp16")]; tensor input_13_strides_0 = const()[name = string("input_13_strides_0"), val = tensor([1, 1])]; string input_13_pad_type_0 = const()[name = string("input_13_pad_type_0"), val = string("valid")]; tensor input_13_pad_0 = const()[name = string("input_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_13_dilations_0 = const()[name = string("input_13_dilations_0"), val = tensor([1, 1])]; int32 input_13_groups_0 = const()[name = string("input_13_groups_0"), val = int32(1)]; tensor input_13_cast_fp16 = conv(dilations = input_13_dilations_0, groups = input_13_groups_0, pad = input_13_pad_0, pad_type = input_13_pad_type_0, strides = input_13_strides_0, weight = layers_6_mlp_gate_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("input_13_cast_fp16")]; tensor var_3129_cast_fp16 = silu(x = input_13_cast_fp16)[name = string("op_3129_cast_fp16")]; tensor var_3135_strides_0 = const()[name = string("op_3135_strides_0"), val = tensor([1, 1])]; string var_3135_pad_type_0 = const()[name = string("op_3135_pad_type_0"), val = string("valid")]; tensor var_3135_pad_0 = const()[name = string("op_3135_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3135_dilations_0 = const()[name = string("op_3135_dilations_0"), val = tensor([1, 1])]; int32 var_3135_groups_0 = const()[name = string("op_3135_groups_0"), val = int32(1)]; tensor var_3135_cast_fp16 = conv(dilations = var_3135_dilations_0, groups = var_3135_groups_0, pad = var_3135_pad_0, pad_type = var_3135_pad_type_0, strides = var_3135_strides_0, weight = layers_6_mlp_up_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("op_3135_cast_fp16")]; tensor x_69_cast_fp16 = mul(x = var_3129_cast_fp16, y = var_3135_cast_fp16)[name = string("x_69_cast_fp16")]; tensor hidden_states_67_strides_0 = const()[name = string("hidden_states_67_strides_0"), val = tensor([1, 1])]; string hidden_states_67_pad_type_0 = const()[name = string("hidden_states_67_pad_type_0"), val = string("valid")]; tensor hidden_states_67_pad_0 = const()[name = string("hidden_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_67_dilations_0 = const()[name = string("hidden_states_67_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_67_groups_0 = const()[name = string("hidden_states_67_groups_0"), val = int32(1)]; tensor hidden_states_67_cast_fp16 = conv(dilations = hidden_states_67_dilations_0, groups = hidden_states_67_groups_0, pad = hidden_states_67_pad_0, pad_type = hidden_states_67_pad_type_0, strides = hidden_states_67_strides_0, weight = layers_6_mlp_down_proj_weight_cast_fp16, x = x_69_cast_fp16)[name = string("hidden_states_67_cast_fp16")]; tensor hidden_states_69_cast_fp16 = add(x = hidden_states_65_cast_fp16, y = hidden_states_67_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3153_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_70_promoted_to_fp16)[name = string("op_3153_cast_fp16")]; int32 var_3151 = const()[name = string("op_3151"), val = int32(1)]; bool doubled_57_interleave_0 = const()[name = string("doubled_57_interleave_0"), val = bool(false)]; tensor doubled_57_cast_fp16 = concat(axis = var_3151, interleave = doubled_57_interleave_0, values = (hidden_states_69_cast_fp16, var_3153_cast_fp16))[name = string("doubled_57_cast_fp16")]; tensor out_29_axes_0 = const()[name = string("out_29_axes_0"), val = tensor([1])]; tensor out_29_gamma_0_to_fp16 = const()[name = string("out_29_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327211072)))]; fp16 var_3163_to_fp16 = const()[name = string("op_3163_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_29_cast_fp16 = layer_norm(axes = out_29_axes_0, epsilon = var_3163_to_fp16, gamma = out_29_gamma_0_to_fp16, x = doubled_57_cast_fp16)[name = string("out_29_cast_fp16")]; tensor var_3174_split_sizes_0 = const()[name = string("op_3174_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3174_axis_0 = const()[name = string("op_3174_axis_0"), val = int32(1)]; tensor var_3174_cast_fp16_0, tensor var_3174_cast_fp16_1 = split(axis = var_3174_axis_0, split_sizes = var_3174_split_sizes_0, x = out_29_cast_fp16)[name = string("op_3174_cast_fp16")]; tensor layers_7_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327219328)))]; tensor query_states_43_strides_0 = const()[name = string("query_states_43_strides_0"), val = tensor([1, 1])]; string query_states_43_pad_type_0 = const()[name = string("query_states_43_pad_type_0"), val = string("valid")]; tensor query_states_43_pad_0 = const()[name = string("query_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_43_dilations_0 = const()[name = string("query_states_43_dilations_0"), val = tensor([1, 1])]; int32 query_states_43_groups_0 = const()[name = string("query_states_43_groups_0"), val = int32(1)]; tensor query_states_43_cast_fp16 = conv(dilations = query_states_43_dilations_0, groups = query_states_43_groups_0, pad = query_states_43_pad_0, pad_type = query_states_43_pad_type_0, strides = query_states_43_strides_0, weight = layers_7_self_attn_q_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("query_states_43_cast_fp16")]; tensor layers_7_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1335608000)))]; tensor key_states_71_strides_0 = const()[name = string("key_states_71_strides_0"), val = tensor([1, 1])]; string key_states_71_pad_type_0 = const()[name = string("key_states_71_pad_type_0"), val = string("valid")]; tensor key_states_71_pad_0 = const()[name = string("key_states_71_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_71_dilations_0 = const()[name = string("key_states_71_dilations_0"), val = tensor([1, 1])]; int32 key_states_71_groups_0 = const()[name = string("key_states_71_groups_0"), val = int32(1)]; tensor key_states_71_cast_fp16 = conv(dilations = key_states_71_dilations_0, groups = key_states_71_groups_0, pad = key_states_71_pad_0, pad_type = key_states_71_pad_type_0, strides = key_states_71_strides_0, weight = layers_7_self_attn_k_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("key_states_71_cast_fp16")]; tensor value_states_43_strides_0 = const()[name = string("value_states_43_strides_0"), val = tensor([1, 1])]; string value_states_43_pad_type_0 = const()[name = string("value_states_43_pad_type_0"), val = string("valid")]; tensor value_states_43_pad_0 = const()[name = string("value_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_43_dilations_0 = const()[name = string("value_states_43_dilations_0"), val = tensor([1, 1])]; int32 value_states_43_groups_0 = const()[name = string("value_states_43_groups_0"), val = int32(1)]; tensor value_states_43_cast_fp16 = conv(dilations = value_states_43_dilations_0, groups = value_states_43_groups_0, pad = value_states_43_pad_0, pad_type = value_states_43_pad_type_0, strides = value_states_43_strides_0, weight = layers_7_self_attn_v_proj_weight_cast_fp16, x = var_3174_cast_fp16_0)[name = string("value_states_43_cast_fp16")]; tensor concat_84x = const()[name = string("concat_84x"), val = tensor([1, 16, 128, -1])]; tensor x_71_cast_fp16 = reshape(shape = concat_84x, x = query_states_43_cast_fp16)[name = string("x_71_cast_fp16")]; tensor concat_85x = const()[name = string("concat_85x"), val = tensor([1, 2, 128, -1])]; tensor var_3231_cast_fp16 = reshape(shape = concat_85x, x = key_states_71_cast_fp16)[name = string("op_3231_cast_fp16")]; tensor concat_86x = const()[name = string("concat_86x"), val = tensor([1, 2, 128, -1])]; tensor var_3238_cast_fp16 = reshape(shape = concat_86x, x = value_states_43_cast_fp16)[name = string("op_3238_cast_fp16")]; tensor var_3242_cast_fp16 = mul(x = x_71_cast_fp16, y = var_869_cast_fp16)[name = string("op_3242_cast_fp16")]; tensor var_3243_split_sizes_0 = const()[name = string("op_3243_split_sizes_0"), val = tensor([64, 64])]; int32 var_3243_axis_0 = const()[name = string("op_3243_axis_0"), val = int32(-2)]; tensor var_3243_cast_fp16_0, tensor var_3243_cast_fp16_1 = split(axis = var_3243_axis_0, split_sizes = var_3243_split_sizes_0, x = x_71_cast_fp16)[name = string("op_3243_cast_fp16")]; fp16 const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3245_cast_fp16 = mul(x = var_3243_cast_fp16_1, y = const_72_promoted_to_fp16)[name = string("op_3245_cast_fp16")]; int32 var_3247 = const()[name = string("op_3247"), val = int32(-2)]; bool var_3248_interleave_0 = const()[name = string("op_3248_interleave_0"), val = bool(false)]; tensor var_3248_cast_fp16 = concat(axis = var_3247, interleave = var_3248_interleave_0, values = (var_3245_cast_fp16, var_3243_cast_fp16_0))[name = string("op_3248_cast_fp16")]; tensor var_3249_cast_fp16 = mul(x = var_3248_cast_fp16, y = var_878_cast_fp16)[name = string("op_3249_cast_fp16")]; tensor query_states_45_cast_fp16 = add(x = var_3242_cast_fp16, y = var_3249_cast_fp16)[name = string("query_states_45_cast_fp16")]; tensor var_3255_cast_fp16 = mul(x = var_3231_cast_fp16, y = var_869_cast_fp16)[name = string("op_3255_cast_fp16")]; tensor var_3256_split_sizes_0 = const()[name = string("op_3256_split_sizes_0"), val = tensor([64, 64])]; int32 var_3256_axis_0 = const()[name = string("op_3256_axis_0"), val = int32(-2)]; tensor var_3256_cast_fp16_0, tensor var_3256_cast_fp16_1 = split(axis = var_3256_axis_0, split_sizes = var_3256_split_sizes_0, x = var_3231_cast_fp16)[name = string("op_3256_cast_fp16")]; fp16 const_73_promoted_to_fp16 = const()[name = string("const_73_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3258_cast_fp16 = mul(x = var_3256_cast_fp16_1, y = const_73_promoted_to_fp16)[name = string("op_3258_cast_fp16")]; int32 var_3260 = const()[name = string("op_3260"), val = int32(-2)]; bool var_3261_interleave_0 = const()[name = string("op_3261_interleave_0"), val = bool(false)]; tensor var_3261_cast_fp16 = concat(axis = var_3260, interleave = var_3261_interleave_0, values = (var_3258_cast_fp16, var_3256_cast_fp16_0))[name = string("op_3261_cast_fp16")]; tensor var_3262_cast_fp16 = mul(x = var_3261_cast_fp16, y = var_878_cast_fp16)[name = string("op_3262_cast_fp16")]; tensor key_states_75_cast_fp16 = add(x = var_3255_cast_fp16, y = var_3262_cast_fp16)[name = string("key_states_75_cast_fp16")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; int32 concat_89_axis_0 = const()[name = string("concat_89_axis_0"), val = int32(0)]; bool concat_89_interleave_0 = const()[name = string("concat_89_interleave_0"), val = bool(false)]; tensor concat_89 = concat(axis = concat_89_axis_0, interleave = concat_89_interleave_0, values = (expand_dims_84, expand_dims_85, position_id, expand_dims_87))[name = string("concat_89")]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; tensor concat_90_values1_0 = const()[name = string("concat_90_values1_0"), val = tensor([0])]; tensor concat_90_values3_0 = const()[name = string("concat_90_values3_0"), val = tensor([0])]; int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_88, concat_90_values1_0, cache_position_end, concat_90_values3_0))[name = string("concat_90")]; tensor key_states_77_perm_0 = const()[name = string("key_states_77_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_8_stride_0 = const()[name = string("key_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_77_cast_fp16 = transpose(perm = key_states_77_perm_0, x = key_states_75_cast_fp16)[name = string("transpose_320")]; tensor key_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = key_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = key_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_8_squeeze_mask_0, stride = key_cache_internal_tensor_assign_8_stride_0, update = key_states_77_cast_fp16, x = coreml_update_state_180)[name = string("key_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_8_cast_fp16, input = key_cache)[name = string("coreml_update_state_182_write_state")]; tensor coreml_update_state_182 = read_state(input = key_cache)[name = string("coreml_update_state_182")]; tensor value_states_45_perm_0 = const()[name = string("value_states_45_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_8_stride_0 = const()[name = string("value_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_45_cast_fp16 = transpose(perm = value_states_45_perm_0, x = var_3238_cast_fp16)[name = string("transpose_319")]; tensor value_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = value_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = value_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_8_squeeze_mask_0, stride = value_cache_internal_tensor_assign_8_stride_0, update = value_states_45_cast_fp16, x = coreml_update_state_181)[name = string("value_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_8_cast_fp16, input = value_cache)[name = string("coreml_update_state_183_write_state")]; tensor coreml_update_state_183 = read_state(input = value_cache)[name = string("coreml_update_state_183")]; tensor var_3332_begin_0 = const()[name = string("op_3332_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3332_end_0 = const()[name = string("op_3332_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3332_end_mask_0 = const()[name = string("op_3332_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3332_cast_fp16 = slice_by_index(begin = var_3332_begin_0, end = var_3332_end_0, end_mask = var_3332_end_mask_0, x = coreml_update_state_182)[name = string("op_3332_cast_fp16")]; tensor tile_14 = const()[name = string("tile_14"), val = tensor([1, 1])]; int32 var_3335_axis_0 = const()[name = string("op_3335_axis_0"), val = int32(1)]; tensor var_3335_cast_fp16_0, tensor var_3335_cast_fp16_1 = split(axis = var_3335_axis_0, split_sizes = tile_14, x = var_3332_cast_fp16)[name = string("op_3335_cast_fp16")]; tensor var_3342_begin_0 = const()[name = string("op_3342_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3342_end_0 = const()[name = string("op_3342_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3342_end_mask_0 = const()[name = string("op_3342_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3342_cast_fp16 = slice_by_index(begin = var_3342_begin_0, end = var_3342_end_0, end_mask = var_3342_end_mask_0, x = coreml_update_state_183)[name = string("op_3342_cast_fp16")]; tensor tile_15 = const()[name = string("tile_15"), val = tensor([1, 1])]; int32 var_3345_axis_0 = const()[name = string("op_3345_axis_0"), val = int32(1)]; tensor var_3345_cast_fp16_0, tensor var_3345_cast_fp16_1 = split(axis = var_3345_axis_0, split_sizes = tile_15, x = var_3342_cast_fp16)[name = string("op_3345_cast_fp16")]; tensor var_3348_split_sizes_0 = const()[name = string("op_3348_split_sizes_0"), val = tensor([8, 8])]; int32 var_3348_axis_0 = const()[name = string("op_3348_axis_0"), val = int32(1)]; tensor var_3348_0, tensor var_3348_1 = split(axis = var_3348_axis_0, split_sizes = var_3348_split_sizes_0, x = query_states_45_cast_fp16)[name = string("op_3348")]; bool attn_weights_113_transpose_x_0 = const()[name = string("attn_weights_113_transpose_x_0"), val = bool(false)]; bool attn_weights_113_transpose_y_0 = const()[name = string("attn_weights_113_transpose_y_0"), val = bool(false)]; tensor attn_weights_113_cast_fp16 = matmul(transpose_x = attn_weights_113_transpose_x_0, transpose_y = attn_weights_113_transpose_y_0, x = var_3335_cast_fp16_0, y = var_3348_0)[name = string("attn_weights_113_cast_fp16")]; fp16 var_3351_to_fp16 = const()[name = string("op_3351_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_115_cast_fp16 = mul(x = attn_weights_113_cast_fp16, y = var_3351_to_fp16)[name = string("attn_weights_115_cast_fp16")]; tensor attn_weights_117_cast_fp16 = add(x = attn_weights_115_cast_fp16, y = attn_mask_1)[name = string("attn_weights_117_cast_fp16")]; int32 var_3355 = const()[name = string("op_3355"), val = int32(-2)]; tensor attn_weights_119_cast_fp16 = softmax(axis = var_3355, x = attn_weights_117_cast_fp16)[name = string("attn_weights_119_cast_fp16")]; bool var_3361_transpose_x_1 = const()[name = string("op_3361_transpose_x_1"), val = bool(true)]; bool var_3361_transpose_y_1 = const()[name = string("op_3361_transpose_y_1"), val = bool(false)]; tensor var_3361_cast_fp16 = matmul(transpose_x = var_3361_transpose_x_1, transpose_y = var_3361_transpose_y_1, x = attn_weights_119_cast_fp16, y = var_3345_cast_fp16_0)[name = string("op_3361_cast_fp16")]; bool attn_weights_121_transpose_x_0 = const()[name = string("attn_weights_121_transpose_x_0"), val = bool(false)]; bool attn_weights_121_transpose_y_0 = const()[name = string("attn_weights_121_transpose_y_0"), val = bool(false)]; tensor attn_weights_121_cast_fp16 = matmul(transpose_x = attn_weights_121_transpose_x_0, transpose_y = attn_weights_121_transpose_y_0, x = var_3335_cast_fp16_1, y = var_3348_1)[name = string("attn_weights_121_cast_fp16")]; fp16 var_3363_to_fp16 = const()[name = string("op_3363_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_123_cast_fp16 = mul(x = attn_weights_121_cast_fp16, y = var_3363_to_fp16)[name = string("attn_weights_123_cast_fp16")]; tensor attn_weights_125_cast_fp16 = add(x = attn_weights_123_cast_fp16, y = attn_mask_1)[name = string("attn_weights_125_cast_fp16")]; int32 var_3367 = const()[name = string("op_3367"), val = int32(-2)]; tensor attn_weights_127_cast_fp16 = softmax(axis = var_3367, x = attn_weights_125_cast_fp16)[name = string("attn_weights_127_cast_fp16")]; bool attn_output_57_transpose_x_1 = const()[name = string("attn_output_57_transpose_x_1"), val = bool(true)]; bool attn_output_57_transpose_y_1 = const()[name = string("attn_output_57_transpose_y_1"), val = bool(false)]; tensor attn_output_57_cast_fp16 = matmul(transpose_x = attn_output_57_transpose_x_1, transpose_y = attn_output_57_transpose_y_1, x = attn_weights_127_cast_fp16, y = var_3345_cast_fp16_1)[name = string("attn_output_57_cast_fp16")]; int32 var_3375 = const()[name = string("op_3375"), val = int32(1)]; bool attn_output_59_interleave_0 = const()[name = string("attn_output_59_interleave_0"), val = bool(false)]; tensor attn_output_59_cast_fp16 = concat(axis = var_3375, interleave = attn_output_59_interleave_0, values = (var_3361_cast_fp16, attn_output_57_cast_fp16))[name = string("attn_output_59_cast_fp16")]; tensor var_3379_perm_0 = const()[name = string("op_3379_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_95x = const()[name = string("concat_95x"), val = tensor([1, 2048, 1, -1])]; tensor var_3379_cast_fp16 = transpose(perm = var_3379_perm_0, x = attn_output_59_cast_fp16)[name = string("transpose_318")]; tensor attn_output_63_cast_fp16 = reshape(shape = concat_95x, x = var_3379_cast_fp16)[name = string("attn_output_63_cast_fp16")]; tensor hidden_states_73_strides_0 = const()[name = string("hidden_states_73_strides_0"), val = tensor([1, 1])]; string hidden_states_73_pad_type_0 = const()[name = string("hidden_states_73_pad_type_0"), val = string("valid")]; tensor hidden_states_73_pad_0 = const()[name = string("hidden_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_73_dilations_0 = const()[name = string("hidden_states_73_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_73_groups_0 = const()[name = string("hidden_states_73_groups_0"), val = int32(1)]; tensor hidden_states_73_cast_fp16 = conv(dilations = hidden_states_73_dilations_0, groups = hidden_states_73_groups_0, pad = hidden_states_73_pad_0, pad_type = hidden_states_73_pad_type_0, strides = hidden_states_73_strides_0, weight = layers_7_self_attn_o_proj_weight_cast_fp16, x = attn_output_63_cast_fp16)[name = string("hidden_states_73_cast_fp16")]; tensor hidden_states_75_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = hidden_states_73_cast_fp16)[name = string("hidden_states_75_cast_fp16")]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3412_cast_fp16 = mul(x = hidden_states_75_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_3412_cast_fp16")]; int32 var_3410 = const()[name = string("op_3410"), val = int32(1)]; bool doubled_61_interleave_0 = const()[name = string("doubled_61_interleave_0"), val = bool(false)]; tensor doubled_61_cast_fp16 = concat(axis = var_3410, interleave = doubled_61_interleave_0, values = (hidden_states_75_cast_fp16, var_3412_cast_fp16))[name = string("doubled_61_cast_fp16")]; tensor out_31_axes_0 = const()[name = string("out_31_axes_0"), val = tensor([1])]; tensor out_31_gamma_0_to_fp16 = const()[name = string("out_31_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336656640)))]; fp16 var_3422_to_fp16 = const()[name = string("op_3422_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_31_cast_fp16 = layer_norm(axes = out_31_axes_0, epsilon = var_3422_to_fp16, gamma = out_31_gamma_0_to_fp16, x = doubled_61_cast_fp16)[name = string("out_31_cast_fp16")]; tensor var_3433_split_sizes_0 = const()[name = string("op_3433_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3433_axis_0 = const()[name = string("op_3433_axis_0"), val = int32(1)]; tensor var_3433_cast_fp16_0, tensor var_3433_cast_fp16_1 = split(axis = var_3433_axis_0, split_sizes = var_3433_split_sizes_0, x = out_31_cast_fp16)[name = string("op_3433_cast_fp16")]; tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; tensor input_15_cast_fp16 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = layers_7_mlp_gate_proj_weight_cast_fp16, x = var_3433_cast_fp16_0)[name = string("input_15_cast_fp16")]; tensor var_3450_cast_fp16 = silu(x = input_15_cast_fp16)[name = string("op_3450_cast_fp16")]; tensor layers_7_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336664896)))]; tensor var_3456_strides_0 = const()[name = string("op_3456_strides_0"), val = tensor([1, 1])]; string var_3456_pad_type_0 = const()[name = string("op_3456_pad_type_0"), val = string("valid")]; tensor var_3456_pad_0 = const()[name = string("op_3456_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3456_dilations_0 = const()[name = string("op_3456_dilations_0"), val = tensor([1, 1])]; int32 var_3456_groups_0 = const()[name = string("op_3456_groups_0"), val = int32(1)]; tensor var_3456_cast_fp16 = conv(dilations = var_3456_dilations_0, groups = var_3456_groups_0, pad = var_3456_pad_0, pad_type = var_3456_pad_type_0, strides = var_3456_strides_0, weight = layers_7_mlp_up_proj_weight_to_fp16, x = var_3433_cast_fp16_0)[name = string("op_3456_cast_fp16")]; tensor x_79_cast_fp16 = mul(x = var_3450_cast_fp16, y = var_3456_cast_fp16)[name = string("x_79_cast_fp16")]; tensor layers_7_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1361830784)))]; tensor hidden_states_77_strides_0 = const()[name = string("hidden_states_77_strides_0"), val = tensor([1, 1])]; string hidden_states_77_pad_type_0 = const()[name = string("hidden_states_77_pad_type_0"), val = string("valid")]; tensor hidden_states_77_pad_0 = const()[name = string("hidden_states_77_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_77_dilations_0 = const()[name = string("hidden_states_77_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_77_groups_0 = const()[name = string("hidden_states_77_groups_0"), val = int32(1)]; tensor hidden_states_77_cast_fp16 = conv(dilations = hidden_states_77_dilations_0, groups = hidden_states_77_groups_0, pad = hidden_states_77_pad_0, pad_type = hidden_states_77_pad_type_0, strides = hidden_states_77_strides_0, weight = layers_7_mlp_down_proj_weight_to_fp16, x = x_79_cast_fp16)[name = string("hidden_states_77_cast_fp16")]; tensor hidden_states_79_cast_fp16 = add(x = hidden_states_75_cast_fp16, y = hidden_states_77_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3474_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_80_promoted_to_fp16)[name = string("op_3474_cast_fp16")]; int32 var_3472 = const()[name = string("op_3472"), val = int32(1)]; bool doubled_65_interleave_0 = const()[name = string("doubled_65_interleave_0"), val = bool(false)]; tensor doubled_65_cast_fp16 = concat(axis = var_3472, interleave = doubled_65_interleave_0, values = (hidden_states_79_cast_fp16, var_3474_cast_fp16))[name = string("doubled_65_cast_fp16")]; tensor out_33_axes_0 = const()[name = string("out_33_axes_0"), val = tensor([1])]; tensor out_33_gamma_0_to_fp16 = const()[name = string("out_33_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1386996672)))]; fp16 var_3484_to_fp16 = const()[name = string("op_3484_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_33_cast_fp16 = layer_norm(axes = out_33_axes_0, epsilon = var_3484_to_fp16, gamma = out_33_gamma_0_to_fp16, x = doubled_65_cast_fp16)[name = string("out_33_cast_fp16")]; tensor var_3495_split_sizes_0 = const()[name = string("op_3495_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3495_axis_0 = const()[name = string("op_3495_axis_0"), val = int32(1)]; tensor var_3495_cast_fp16_0, tensor var_3495_cast_fp16_1 = split(axis = var_3495_axis_0, split_sizes = var_3495_split_sizes_0, x = out_33_cast_fp16)[name = string("op_3495_cast_fp16")]; tensor layers_8_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1387004928)))]; tensor query_states_49_strides_0 = const()[name = string("query_states_49_strides_0"), val = tensor([1, 1])]; string query_states_49_pad_type_0 = const()[name = string("query_states_49_pad_type_0"), val = string("valid")]; tensor query_states_49_pad_0 = const()[name = string("query_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_49_dilations_0 = const()[name = string("query_states_49_dilations_0"), val = tensor([1, 1])]; int32 query_states_49_groups_0 = const()[name = string("query_states_49_groups_0"), val = int32(1)]; tensor query_states_49_cast_fp16 = conv(dilations = query_states_49_dilations_0, groups = query_states_49_groups_0, pad = query_states_49_pad_0, pad_type = query_states_49_pad_type_0, strides = query_states_49_strides_0, weight = layers_8_self_attn_q_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("query_states_49_cast_fp16")]; tensor layers_8_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1395393600)))]; tensor key_states_81_strides_0 = const()[name = string("key_states_81_strides_0"), val = tensor([1, 1])]; string key_states_81_pad_type_0 = const()[name = string("key_states_81_pad_type_0"), val = string("valid")]; tensor key_states_81_pad_0 = const()[name = string("key_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_81_dilations_0 = const()[name = string("key_states_81_dilations_0"), val = tensor([1, 1])]; int32 key_states_81_groups_0 = const()[name = string("key_states_81_groups_0"), val = int32(1)]; tensor key_states_81_cast_fp16 = conv(dilations = key_states_81_dilations_0, groups = key_states_81_groups_0, pad = key_states_81_pad_0, pad_type = key_states_81_pad_type_0, strides = key_states_81_strides_0, weight = layers_8_self_attn_k_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("key_states_81_cast_fp16")]; tensor value_states_49_strides_0 = const()[name = string("value_states_49_strides_0"), val = tensor([1, 1])]; string value_states_49_pad_type_0 = const()[name = string("value_states_49_pad_type_0"), val = string("valid")]; tensor value_states_49_pad_0 = const()[name = string("value_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_49_dilations_0 = const()[name = string("value_states_49_dilations_0"), val = tensor([1, 1])]; int32 value_states_49_groups_0 = const()[name = string("value_states_49_groups_0"), val = int32(1)]; tensor value_states_49_cast_fp16 = conv(dilations = value_states_49_dilations_0, groups = value_states_49_groups_0, pad = value_states_49_pad_0, pad_type = value_states_49_pad_type_0, strides = value_states_49_strides_0, weight = layers_8_self_attn_v_proj_weight_cast_fp16, x = var_3495_cast_fp16_0)[name = string("value_states_49_cast_fp16")]; tensor concat_96x = const()[name = string("concat_96x"), val = tensor([1, 16, 128, -1])]; tensor x_81_cast_fp16 = reshape(shape = concat_96x, x = query_states_49_cast_fp16)[name = string("x_81_cast_fp16")]; tensor concat_97x = const()[name = string("concat_97x"), val = tensor([1, 2, 128, -1])]; tensor var_3552_cast_fp16 = reshape(shape = concat_97x, x = key_states_81_cast_fp16)[name = string("op_3552_cast_fp16")]; tensor concat_98x = const()[name = string("concat_98x"), val = tensor([1, 2, 128, -1])]; tensor var_3559_cast_fp16 = reshape(shape = concat_98x, x = value_states_49_cast_fp16)[name = string("op_3559_cast_fp16")]; tensor var_3563_cast_fp16 = mul(x = x_81_cast_fp16, y = var_869_cast_fp16)[name = string("op_3563_cast_fp16")]; tensor var_3564_split_sizes_0 = const()[name = string("op_3564_split_sizes_0"), val = tensor([64, 64])]; int32 var_3564_axis_0 = const()[name = string("op_3564_axis_0"), val = int32(-2)]; tensor var_3564_cast_fp16_0, tensor var_3564_cast_fp16_1 = split(axis = var_3564_axis_0, split_sizes = var_3564_split_sizes_0, x = x_81_cast_fp16)[name = string("op_3564_cast_fp16")]; fp16 const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3566_cast_fp16 = mul(x = var_3564_cast_fp16_1, y = const_82_promoted_to_fp16)[name = string("op_3566_cast_fp16")]; int32 var_3568 = const()[name = string("op_3568"), val = int32(-2)]; bool var_3569_interleave_0 = const()[name = string("op_3569_interleave_0"), val = bool(false)]; tensor var_3569_cast_fp16 = concat(axis = var_3568, interleave = var_3569_interleave_0, values = (var_3566_cast_fp16, var_3564_cast_fp16_0))[name = string("op_3569_cast_fp16")]; tensor var_3570_cast_fp16 = mul(x = var_3569_cast_fp16, y = var_878_cast_fp16)[name = string("op_3570_cast_fp16")]; tensor query_states_51_cast_fp16 = add(x = var_3563_cast_fp16, y = var_3570_cast_fp16)[name = string("query_states_51_cast_fp16")]; tensor var_3576_cast_fp16 = mul(x = var_3552_cast_fp16, y = var_869_cast_fp16)[name = string("op_3576_cast_fp16")]; tensor var_3577_split_sizes_0 = const()[name = string("op_3577_split_sizes_0"), val = tensor([64, 64])]; int32 var_3577_axis_0 = const()[name = string("op_3577_axis_0"), val = int32(-2)]; tensor var_3577_cast_fp16_0, tensor var_3577_cast_fp16_1 = split(axis = var_3577_axis_0, split_sizes = var_3577_split_sizes_0, x = var_3552_cast_fp16)[name = string("op_3577_cast_fp16")]; fp16 const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3579_cast_fp16 = mul(x = var_3577_cast_fp16_1, y = const_83_promoted_to_fp16)[name = string("op_3579_cast_fp16")]; int32 var_3581 = const()[name = string("op_3581"), val = int32(-2)]; bool var_3582_interleave_0 = const()[name = string("op_3582_interleave_0"), val = bool(false)]; tensor var_3582_cast_fp16 = concat(axis = var_3581, interleave = var_3582_interleave_0, values = (var_3579_cast_fp16, var_3577_cast_fp16_0))[name = string("op_3582_cast_fp16")]; tensor var_3583_cast_fp16 = mul(x = var_3582_cast_fp16, y = var_878_cast_fp16)[name = string("op_3583_cast_fp16")]; tensor key_states_85_cast_fp16 = add(x = var_3576_cast_fp16, y = var_3583_cast_fp16)[name = string("key_states_85_cast_fp16")]; tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; int32 concat_101_axis_0 = const()[name = string("concat_101_axis_0"), val = int32(0)]; bool concat_101_interleave_0 = const()[name = string("concat_101_interleave_0"), val = bool(false)]; tensor concat_101 = concat(axis = concat_101_axis_0, interleave = concat_101_interleave_0, values = (expand_dims_96, expand_dims_97, position_id, expand_dims_99))[name = string("concat_101")]; tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; tensor concat_102_values1_0 = const()[name = string("concat_102_values1_0"), val = tensor([0])]; tensor concat_102_values3_0 = const()[name = string("concat_102_values3_0"), val = tensor([0])]; int32 concat_102_axis_0 = const()[name = string("concat_102_axis_0"), val = int32(0)]; bool concat_102_interleave_0 = const()[name = string("concat_102_interleave_0"), val = bool(false)]; tensor concat_102 = concat(axis = concat_102_axis_0, interleave = concat_102_interleave_0, values = (expand_dims_100, concat_102_values1_0, cache_position_end, concat_102_values3_0))[name = string("concat_102")]; tensor key_states_87_perm_0 = const()[name = string("key_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_9_stride_0 = const()[name = string("key_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_87_cast_fp16 = transpose(perm = key_states_87_perm_0, x = key_states_85_cast_fp16)[name = string("transpose_317")]; tensor key_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = key_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = key_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_9_squeeze_mask_0, stride = key_cache_internal_tensor_assign_9_stride_0, update = key_states_87_cast_fp16, x = coreml_update_state_182)[name = string("key_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_9_cast_fp16, input = key_cache)[name = string("coreml_update_state_184_write_state")]; tensor coreml_update_state_184 = read_state(input = key_cache)[name = string("coreml_update_state_184")]; tensor value_states_51_perm_0 = const()[name = string("value_states_51_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_9_stride_0 = const()[name = string("value_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_51_cast_fp16 = transpose(perm = value_states_51_perm_0, x = var_3559_cast_fp16)[name = string("transpose_316")]; tensor value_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = value_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = value_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_9_squeeze_mask_0, stride = value_cache_internal_tensor_assign_9_stride_0, update = value_states_51_cast_fp16, x = coreml_update_state_183)[name = string("value_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_9_cast_fp16, input = value_cache)[name = string("coreml_update_state_185_write_state")]; tensor coreml_update_state_185 = read_state(input = value_cache)[name = string("coreml_update_state_185")]; tensor var_3653_begin_0 = const()[name = string("op_3653_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3653_end_0 = const()[name = string("op_3653_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3653_end_mask_0 = const()[name = string("op_3653_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3653_cast_fp16 = slice_by_index(begin = var_3653_begin_0, end = var_3653_end_0, end_mask = var_3653_end_mask_0, x = coreml_update_state_184)[name = string("op_3653_cast_fp16")]; tensor tile_16 = const()[name = string("tile_16"), val = tensor([1, 1])]; int32 var_3656_axis_0 = const()[name = string("op_3656_axis_0"), val = int32(1)]; tensor var_3656_cast_fp16_0, tensor var_3656_cast_fp16_1 = split(axis = var_3656_axis_0, split_sizes = tile_16, x = var_3653_cast_fp16)[name = string("op_3656_cast_fp16")]; tensor var_3663_begin_0 = const()[name = string("op_3663_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3663_end_0 = const()[name = string("op_3663_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3663_end_mask_0 = const()[name = string("op_3663_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3663_cast_fp16 = slice_by_index(begin = var_3663_begin_0, end = var_3663_end_0, end_mask = var_3663_end_mask_0, x = coreml_update_state_185)[name = string("op_3663_cast_fp16")]; tensor tile_17 = const()[name = string("tile_17"), val = tensor([1, 1])]; int32 var_3666_axis_0 = const()[name = string("op_3666_axis_0"), val = int32(1)]; tensor var_3666_cast_fp16_0, tensor var_3666_cast_fp16_1 = split(axis = var_3666_axis_0, split_sizes = tile_17, x = var_3663_cast_fp16)[name = string("op_3666_cast_fp16")]; tensor var_3669_split_sizes_0 = const()[name = string("op_3669_split_sizes_0"), val = tensor([8, 8])]; int32 var_3669_axis_0 = const()[name = string("op_3669_axis_0"), val = int32(1)]; tensor var_3669_0, tensor var_3669_1 = split(axis = var_3669_axis_0, split_sizes = var_3669_split_sizes_0, x = query_states_51_cast_fp16)[name = string("op_3669")]; bool attn_weights_129_transpose_x_0 = const()[name = string("attn_weights_129_transpose_x_0"), val = bool(false)]; bool attn_weights_129_transpose_y_0 = const()[name = string("attn_weights_129_transpose_y_0"), val = bool(false)]; tensor attn_weights_129_cast_fp16 = matmul(transpose_x = attn_weights_129_transpose_x_0, transpose_y = attn_weights_129_transpose_y_0, x = var_3656_cast_fp16_0, y = var_3669_0)[name = string("attn_weights_129_cast_fp16")]; fp16 var_3672_to_fp16 = const()[name = string("op_3672_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_131_cast_fp16 = mul(x = attn_weights_129_cast_fp16, y = var_3672_to_fp16)[name = string("attn_weights_131_cast_fp16")]; tensor attn_weights_133_cast_fp16 = add(x = attn_weights_131_cast_fp16, y = attn_mask_1)[name = string("attn_weights_133_cast_fp16")]; int32 var_3676 = const()[name = string("op_3676"), val = int32(-2)]; tensor attn_weights_135_cast_fp16 = softmax(axis = var_3676, x = attn_weights_133_cast_fp16)[name = string("attn_weights_135_cast_fp16")]; bool var_3682_transpose_x_1 = const()[name = string("op_3682_transpose_x_1"), val = bool(true)]; bool var_3682_transpose_y_1 = const()[name = string("op_3682_transpose_y_1"), val = bool(false)]; tensor var_3682_cast_fp16 = matmul(transpose_x = var_3682_transpose_x_1, transpose_y = var_3682_transpose_y_1, x = attn_weights_135_cast_fp16, y = var_3666_cast_fp16_0)[name = string("op_3682_cast_fp16")]; bool attn_weights_137_transpose_x_0 = const()[name = string("attn_weights_137_transpose_x_0"), val = bool(false)]; bool attn_weights_137_transpose_y_0 = const()[name = string("attn_weights_137_transpose_y_0"), val = bool(false)]; tensor attn_weights_137_cast_fp16 = matmul(transpose_x = attn_weights_137_transpose_x_0, transpose_y = attn_weights_137_transpose_y_0, x = var_3656_cast_fp16_1, y = var_3669_1)[name = string("attn_weights_137_cast_fp16")]; fp16 var_3684_to_fp16 = const()[name = string("op_3684_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_139_cast_fp16 = mul(x = attn_weights_137_cast_fp16, y = var_3684_to_fp16)[name = string("attn_weights_139_cast_fp16")]; tensor attn_weights_141_cast_fp16 = add(x = attn_weights_139_cast_fp16, y = attn_mask_1)[name = string("attn_weights_141_cast_fp16")]; int32 var_3688 = const()[name = string("op_3688"), val = int32(-2)]; tensor attn_weights_143_cast_fp16 = softmax(axis = var_3688, x = attn_weights_141_cast_fp16)[name = string("attn_weights_143_cast_fp16")]; bool attn_output_65_transpose_x_1 = const()[name = string("attn_output_65_transpose_x_1"), val = bool(true)]; bool attn_output_65_transpose_y_1 = const()[name = string("attn_output_65_transpose_y_1"), val = bool(false)]; tensor attn_output_65_cast_fp16 = matmul(transpose_x = attn_output_65_transpose_x_1, transpose_y = attn_output_65_transpose_y_1, x = attn_weights_143_cast_fp16, y = var_3666_cast_fp16_1)[name = string("attn_output_65_cast_fp16")]; int32 var_3696 = const()[name = string("op_3696"), val = int32(1)]; bool attn_output_67_interleave_0 = const()[name = string("attn_output_67_interleave_0"), val = bool(false)]; tensor attn_output_67_cast_fp16 = concat(axis = var_3696, interleave = attn_output_67_interleave_0, values = (var_3682_cast_fp16, attn_output_65_cast_fp16))[name = string("attn_output_67_cast_fp16")]; tensor var_3700_perm_0 = const()[name = string("op_3700_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_107x = const()[name = string("concat_107x"), val = tensor([1, 2048, 1, -1])]; tensor var_3700_cast_fp16 = transpose(perm = var_3700_perm_0, x = attn_output_67_cast_fp16)[name = string("transpose_315")]; tensor attn_output_71_cast_fp16 = reshape(shape = concat_107x, x = var_3700_cast_fp16)[name = string("attn_output_71_cast_fp16")]; tensor hidden_states_83_strides_0 = const()[name = string("hidden_states_83_strides_0"), val = tensor([1, 1])]; string hidden_states_83_pad_type_0 = const()[name = string("hidden_states_83_pad_type_0"), val = string("valid")]; tensor hidden_states_83_pad_0 = const()[name = string("hidden_states_83_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_83_dilations_0 = const()[name = string("hidden_states_83_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_83_groups_0 = const()[name = string("hidden_states_83_groups_0"), val = int32(1)]; tensor hidden_states_83_cast_fp16 = conv(dilations = hidden_states_83_dilations_0, groups = hidden_states_83_groups_0, pad = hidden_states_83_pad_0, pad_type = hidden_states_83_pad_type_0, strides = hidden_states_83_strides_0, weight = layers_8_self_attn_o_proj_weight_cast_fp16, x = attn_output_71_cast_fp16)[name = string("hidden_states_83_cast_fp16")]; tensor hidden_states_85_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = hidden_states_83_cast_fp16)[name = string("hidden_states_85_cast_fp16")]; fp16 const_88_promoted_to_fp16 = const()[name = string("const_88_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3733_cast_fp16 = mul(x = hidden_states_85_cast_fp16, y = const_88_promoted_to_fp16)[name = string("op_3733_cast_fp16")]; int32 var_3731 = const()[name = string("op_3731"), val = int32(1)]; bool doubled_69_interleave_0 = const()[name = string("doubled_69_interleave_0"), val = bool(false)]; tensor doubled_69_cast_fp16 = concat(axis = var_3731, interleave = doubled_69_interleave_0, values = (hidden_states_85_cast_fp16, var_3733_cast_fp16))[name = string("doubled_69_cast_fp16")]; tensor out_35_axes_0 = const()[name = string("out_35_axes_0"), val = tensor([1])]; tensor out_35_gamma_0_to_fp16 = const()[name = string("out_35_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396442240)))]; fp16 var_3743_to_fp16 = const()[name = string("op_3743_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_35_cast_fp16 = layer_norm(axes = out_35_axes_0, epsilon = var_3743_to_fp16, gamma = out_35_gamma_0_to_fp16, x = doubled_69_cast_fp16)[name = string("out_35_cast_fp16")]; tensor var_3754_split_sizes_0 = const()[name = string("op_3754_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3754_axis_0 = const()[name = string("op_3754_axis_0"), val = int32(1)]; tensor var_3754_cast_fp16_0, tensor var_3754_cast_fp16_1 = split(axis = var_3754_axis_0, split_sizes = var_3754_split_sizes_0, x = out_35_cast_fp16)[name = string("op_3754_cast_fp16")]; tensor input_17_strides_0 = const()[name = string("input_17_strides_0"), val = tensor([1, 1])]; string input_17_pad_type_0 = const()[name = string("input_17_pad_type_0"), val = string("valid")]; tensor input_17_pad_0 = const()[name = string("input_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_17_dilations_0 = const()[name = string("input_17_dilations_0"), val = tensor([1, 1])]; int32 input_17_groups_0 = const()[name = string("input_17_groups_0"), val = int32(1)]; tensor input_17_cast_fp16 = conv(dilations = input_17_dilations_0, groups = input_17_groups_0, pad = input_17_pad_0, pad_type = input_17_pad_type_0, strides = input_17_strides_0, weight = layers_8_mlp_gate_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("input_17_cast_fp16")]; tensor var_3771_cast_fp16 = silu(x = input_17_cast_fp16)[name = string("op_3771_cast_fp16")]; tensor var_3777_strides_0 = const()[name = string("op_3777_strides_0"), val = tensor([1, 1])]; string var_3777_pad_type_0 = const()[name = string("op_3777_pad_type_0"), val = string("valid")]; tensor var_3777_pad_0 = const()[name = string("op_3777_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3777_dilations_0 = const()[name = string("op_3777_dilations_0"), val = tensor([1, 1])]; int32 var_3777_groups_0 = const()[name = string("op_3777_groups_0"), val = int32(1)]; tensor var_3777_cast_fp16 = conv(dilations = var_3777_dilations_0, groups = var_3777_groups_0, pad = var_3777_pad_0, pad_type = var_3777_pad_type_0, strides = var_3777_strides_0, weight = layers_8_mlp_up_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("op_3777_cast_fp16")]; tensor x_89_cast_fp16 = mul(x = var_3771_cast_fp16, y = var_3777_cast_fp16)[name = string("x_89_cast_fp16")]; tensor hidden_states_87_strides_0 = const()[name = string("hidden_states_87_strides_0"), val = tensor([1, 1])]; string hidden_states_87_pad_type_0 = const()[name = string("hidden_states_87_pad_type_0"), val = string("valid")]; tensor hidden_states_87_pad_0 = const()[name = string("hidden_states_87_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_87_dilations_0 = const()[name = string("hidden_states_87_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_87_groups_0 = const()[name = string("hidden_states_87_groups_0"), val = int32(1)]; tensor hidden_states_87_cast_fp16 = conv(dilations = hidden_states_87_dilations_0, groups = hidden_states_87_groups_0, pad = hidden_states_87_pad_0, pad_type = hidden_states_87_pad_type_0, strides = hidden_states_87_strides_0, weight = layers_8_mlp_down_proj_weight_cast_fp16, x = x_89_cast_fp16)[name = string("hidden_states_87_cast_fp16")]; tensor hidden_states_89_cast_fp16 = add(x = hidden_states_85_cast_fp16, y = hidden_states_87_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3795_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_3795_cast_fp16")]; int32 var_3793 = const()[name = string("op_3793"), val = int32(1)]; bool doubled_73_interleave_0 = const()[name = string("doubled_73_interleave_0"), val = bool(false)]; tensor doubled_73_cast_fp16 = concat(axis = var_3793, interleave = doubled_73_interleave_0, values = (hidden_states_89_cast_fp16, var_3795_cast_fp16))[name = string("doubled_73_cast_fp16")]; tensor out_37_axes_0 = const()[name = string("out_37_axes_0"), val = tensor([1])]; tensor out_37_gamma_0_to_fp16 = const()[name = string("out_37_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396450496)))]; fp16 var_3805_to_fp16 = const()[name = string("op_3805_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_37_cast_fp16 = layer_norm(axes = out_37_axes_0, epsilon = var_3805_to_fp16, gamma = out_37_gamma_0_to_fp16, x = doubled_73_cast_fp16)[name = string("out_37_cast_fp16")]; tensor var_3816_split_sizes_0 = const()[name = string("op_3816_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3816_axis_0 = const()[name = string("op_3816_axis_0"), val = int32(1)]; tensor var_3816_cast_fp16_0, tensor var_3816_cast_fp16_1 = split(axis = var_3816_axis_0, split_sizes = var_3816_split_sizes_0, x = out_37_cast_fp16)[name = string("op_3816_cast_fp16")]; tensor layers_9_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396458752)))]; tensor query_states_55_strides_0 = const()[name = string("query_states_55_strides_0"), val = tensor([1, 1])]; string query_states_55_pad_type_0 = const()[name = string("query_states_55_pad_type_0"), val = string("valid")]; tensor query_states_55_pad_0 = const()[name = string("query_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_55_dilations_0 = const()[name = string("query_states_55_dilations_0"), val = tensor([1, 1])]; int32 query_states_55_groups_0 = const()[name = string("query_states_55_groups_0"), val = int32(1)]; tensor query_states_55_cast_fp16 = conv(dilations = query_states_55_dilations_0, groups = query_states_55_groups_0, pad = query_states_55_pad_0, pad_type = query_states_55_pad_type_0, strides = query_states_55_strides_0, weight = layers_9_self_attn_q_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("query_states_55_cast_fp16")]; tensor layers_9_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1404847424)))]; tensor key_states_91_strides_0 = const()[name = string("key_states_91_strides_0"), val = tensor([1, 1])]; string key_states_91_pad_type_0 = const()[name = string("key_states_91_pad_type_0"), val = string("valid")]; tensor key_states_91_pad_0 = const()[name = string("key_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_91_dilations_0 = const()[name = string("key_states_91_dilations_0"), val = tensor([1, 1])]; int32 key_states_91_groups_0 = const()[name = string("key_states_91_groups_0"), val = int32(1)]; tensor key_states_91_cast_fp16 = conv(dilations = key_states_91_dilations_0, groups = key_states_91_groups_0, pad = key_states_91_pad_0, pad_type = key_states_91_pad_type_0, strides = key_states_91_strides_0, weight = layers_9_self_attn_k_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("key_states_91_cast_fp16")]; tensor value_states_55_strides_0 = const()[name = string("value_states_55_strides_0"), val = tensor([1, 1])]; string value_states_55_pad_type_0 = const()[name = string("value_states_55_pad_type_0"), val = string("valid")]; tensor value_states_55_pad_0 = const()[name = string("value_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_55_dilations_0 = const()[name = string("value_states_55_dilations_0"), val = tensor([1, 1])]; int32 value_states_55_groups_0 = const()[name = string("value_states_55_groups_0"), val = int32(1)]; tensor value_states_55_cast_fp16 = conv(dilations = value_states_55_dilations_0, groups = value_states_55_groups_0, pad = value_states_55_pad_0, pad_type = value_states_55_pad_type_0, strides = value_states_55_strides_0, weight = layers_9_self_attn_v_proj_weight_cast_fp16, x = var_3816_cast_fp16_0)[name = string("value_states_55_cast_fp16")]; tensor concat_108x = const()[name = string("concat_108x"), val = tensor([1, 16, 128, -1])]; tensor x_91_cast_fp16 = reshape(shape = concat_108x, x = query_states_55_cast_fp16)[name = string("x_91_cast_fp16")]; tensor concat_109x = const()[name = string("concat_109x"), val = tensor([1, 2, 128, -1])]; tensor var_3873_cast_fp16 = reshape(shape = concat_109x, x = key_states_91_cast_fp16)[name = string("op_3873_cast_fp16")]; tensor concat_110x = const()[name = string("concat_110x"), val = tensor([1, 2, 128, -1])]; tensor var_3880_cast_fp16 = reshape(shape = concat_110x, x = value_states_55_cast_fp16)[name = string("op_3880_cast_fp16")]; tensor var_3884_cast_fp16 = mul(x = x_91_cast_fp16, y = var_869_cast_fp16)[name = string("op_3884_cast_fp16")]; tensor var_3885_split_sizes_0 = const()[name = string("op_3885_split_sizes_0"), val = tensor([64, 64])]; int32 var_3885_axis_0 = const()[name = string("op_3885_axis_0"), val = int32(-2)]; tensor var_3885_cast_fp16_0, tensor var_3885_cast_fp16_1 = split(axis = var_3885_axis_0, split_sizes = var_3885_split_sizes_0, x = x_91_cast_fp16)[name = string("op_3885_cast_fp16")]; fp16 const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3887_cast_fp16 = mul(x = var_3885_cast_fp16_1, y = const_92_promoted_to_fp16)[name = string("op_3887_cast_fp16")]; int32 var_3889 = const()[name = string("op_3889"), val = int32(-2)]; bool var_3890_interleave_0 = const()[name = string("op_3890_interleave_0"), val = bool(false)]; tensor var_3890_cast_fp16 = concat(axis = var_3889, interleave = var_3890_interleave_0, values = (var_3887_cast_fp16, var_3885_cast_fp16_0))[name = string("op_3890_cast_fp16")]; tensor var_3891_cast_fp16 = mul(x = var_3890_cast_fp16, y = var_878_cast_fp16)[name = string("op_3891_cast_fp16")]; tensor query_states_57_cast_fp16 = add(x = var_3884_cast_fp16, y = var_3891_cast_fp16)[name = string("query_states_57_cast_fp16")]; tensor var_3897_cast_fp16 = mul(x = var_3873_cast_fp16, y = var_869_cast_fp16)[name = string("op_3897_cast_fp16")]; tensor var_3898_split_sizes_0 = const()[name = string("op_3898_split_sizes_0"), val = tensor([64, 64])]; int32 var_3898_axis_0 = const()[name = string("op_3898_axis_0"), val = int32(-2)]; tensor var_3898_cast_fp16_0, tensor var_3898_cast_fp16_1 = split(axis = var_3898_axis_0, split_sizes = var_3898_split_sizes_0, x = var_3873_cast_fp16)[name = string("op_3898_cast_fp16")]; fp16 const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3900_cast_fp16 = mul(x = var_3898_cast_fp16_1, y = const_93_promoted_to_fp16)[name = string("op_3900_cast_fp16")]; int32 var_3902 = const()[name = string("op_3902"), val = int32(-2)]; bool var_3903_interleave_0 = const()[name = string("op_3903_interleave_0"), val = bool(false)]; tensor var_3903_cast_fp16 = concat(axis = var_3902, interleave = var_3903_interleave_0, values = (var_3900_cast_fp16, var_3898_cast_fp16_0))[name = string("op_3903_cast_fp16")]; tensor var_3904_cast_fp16 = mul(x = var_3903_cast_fp16, y = var_878_cast_fp16)[name = string("op_3904_cast_fp16")]; tensor key_states_95_cast_fp16 = add(x = var_3897_cast_fp16, y = var_3904_cast_fp16)[name = string("key_states_95_cast_fp16")]; tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; int32 concat_113_axis_0 = const()[name = string("concat_113_axis_0"), val = int32(0)]; bool concat_113_interleave_0 = const()[name = string("concat_113_interleave_0"), val = bool(false)]; tensor concat_113 = concat(axis = concat_113_axis_0, interleave = concat_113_interleave_0, values = (expand_dims_108, expand_dims_109, position_id, expand_dims_111))[name = string("concat_113")]; tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; tensor concat_114_values1_0 = const()[name = string("concat_114_values1_0"), val = tensor([0])]; tensor concat_114_values3_0 = const()[name = string("concat_114_values3_0"), val = tensor([0])]; int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_112, concat_114_values1_0, cache_position_end, concat_114_values3_0))[name = string("concat_114")]; tensor key_states_97_perm_0 = const()[name = string("key_states_97_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_10_stride_0 = const()[name = string("key_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_97_cast_fp16 = transpose(perm = key_states_97_perm_0, x = key_states_95_cast_fp16)[name = string("transpose_314")]; tensor key_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = key_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = key_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_10_squeeze_mask_0, stride = key_cache_internal_tensor_assign_10_stride_0, update = key_states_97_cast_fp16, x = coreml_update_state_184)[name = string("key_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_10_cast_fp16, input = key_cache)[name = string("coreml_update_state_186_write_state")]; tensor coreml_update_state_186 = read_state(input = key_cache)[name = string("coreml_update_state_186")]; tensor value_states_57_perm_0 = const()[name = string("value_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_10_stride_0 = const()[name = string("value_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_57_cast_fp16 = transpose(perm = value_states_57_perm_0, x = var_3880_cast_fp16)[name = string("transpose_313")]; tensor value_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = value_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = value_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_10_squeeze_mask_0, stride = value_cache_internal_tensor_assign_10_stride_0, update = value_states_57_cast_fp16, x = coreml_update_state_185)[name = string("value_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_10_cast_fp16, input = value_cache)[name = string("coreml_update_state_187_write_state")]; tensor coreml_update_state_187 = read_state(input = value_cache)[name = string("coreml_update_state_187")]; tensor var_3974_begin_0 = const()[name = string("op_3974_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3974_end_0 = const()[name = string("op_3974_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3974_end_mask_0 = const()[name = string("op_3974_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3974_cast_fp16 = slice_by_index(begin = var_3974_begin_0, end = var_3974_end_0, end_mask = var_3974_end_mask_0, x = coreml_update_state_186)[name = string("op_3974_cast_fp16")]; tensor tile_18 = const()[name = string("tile_18"), val = tensor([1, 1])]; int32 var_3977_axis_0 = const()[name = string("op_3977_axis_0"), val = int32(1)]; tensor var_3977_cast_fp16_0, tensor var_3977_cast_fp16_1 = split(axis = var_3977_axis_0, split_sizes = tile_18, x = var_3974_cast_fp16)[name = string("op_3977_cast_fp16")]; tensor var_3984_begin_0 = const()[name = string("op_3984_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3984_end_0 = const()[name = string("op_3984_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3984_end_mask_0 = const()[name = string("op_3984_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3984_cast_fp16 = slice_by_index(begin = var_3984_begin_0, end = var_3984_end_0, end_mask = var_3984_end_mask_0, x = coreml_update_state_187)[name = string("op_3984_cast_fp16")]; tensor tile_19 = const()[name = string("tile_19"), val = tensor([1, 1])]; int32 var_3987_axis_0 = const()[name = string("op_3987_axis_0"), val = int32(1)]; tensor var_3987_cast_fp16_0, tensor var_3987_cast_fp16_1 = split(axis = var_3987_axis_0, split_sizes = tile_19, x = var_3984_cast_fp16)[name = string("op_3987_cast_fp16")]; tensor var_3990_split_sizes_0 = const()[name = string("op_3990_split_sizes_0"), val = tensor([8, 8])]; int32 var_3990_axis_0 = const()[name = string("op_3990_axis_0"), val = int32(1)]; tensor var_3990_0, tensor var_3990_1 = split(axis = var_3990_axis_0, split_sizes = var_3990_split_sizes_0, x = query_states_57_cast_fp16)[name = string("op_3990")]; bool attn_weights_145_transpose_x_0 = const()[name = string("attn_weights_145_transpose_x_0"), val = bool(false)]; bool attn_weights_145_transpose_y_0 = const()[name = string("attn_weights_145_transpose_y_0"), val = bool(false)]; tensor attn_weights_145_cast_fp16 = matmul(transpose_x = attn_weights_145_transpose_x_0, transpose_y = attn_weights_145_transpose_y_0, x = var_3977_cast_fp16_0, y = var_3990_0)[name = string("attn_weights_145_cast_fp16")]; fp16 var_3993_to_fp16 = const()[name = string("op_3993_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_147_cast_fp16 = mul(x = attn_weights_145_cast_fp16, y = var_3993_to_fp16)[name = string("attn_weights_147_cast_fp16")]; tensor attn_weights_149_cast_fp16 = add(x = attn_weights_147_cast_fp16, y = attn_mask_1)[name = string("attn_weights_149_cast_fp16")]; int32 var_3997 = const()[name = string("op_3997"), val = int32(-2)]; tensor attn_weights_151_cast_fp16 = softmax(axis = var_3997, x = attn_weights_149_cast_fp16)[name = string("attn_weights_151_cast_fp16")]; bool var_4003_transpose_x_1 = const()[name = string("op_4003_transpose_x_1"), val = bool(true)]; bool var_4003_transpose_y_1 = const()[name = string("op_4003_transpose_y_1"), val = bool(false)]; tensor var_4003_cast_fp16 = matmul(transpose_x = var_4003_transpose_x_1, transpose_y = var_4003_transpose_y_1, x = attn_weights_151_cast_fp16, y = var_3987_cast_fp16_0)[name = string("op_4003_cast_fp16")]; bool attn_weights_153_transpose_x_0 = const()[name = string("attn_weights_153_transpose_x_0"), val = bool(false)]; bool attn_weights_153_transpose_y_0 = const()[name = string("attn_weights_153_transpose_y_0"), val = bool(false)]; tensor attn_weights_153_cast_fp16 = matmul(transpose_x = attn_weights_153_transpose_x_0, transpose_y = attn_weights_153_transpose_y_0, x = var_3977_cast_fp16_1, y = var_3990_1)[name = string("attn_weights_153_cast_fp16")]; fp16 var_4005_to_fp16 = const()[name = string("op_4005_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_155_cast_fp16 = mul(x = attn_weights_153_cast_fp16, y = var_4005_to_fp16)[name = string("attn_weights_155_cast_fp16")]; tensor attn_weights_157_cast_fp16 = add(x = attn_weights_155_cast_fp16, y = attn_mask_1)[name = string("attn_weights_157_cast_fp16")]; int32 var_4009 = const()[name = string("op_4009"), val = int32(-2)]; tensor attn_weights_159_cast_fp16 = softmax(axis = var_4009, x = attn_weights_157_cast_fp16)[name = string("attn_weights_159_cast_fp16")]; bool attn_output_73_transpose_x_1 = const()[name = string("attn_output_73_transpose_x_1"), val = bool(true)]; bool attn_output_73_transpose_y_1 = const()[name = string("attn_output_73_transpose_y_1"), val = bool(false)]; tensor attn_output_73_cast_fp16 = matmul(transpose_x = attn_output_73_transpose_x_1, transpose_y = attn_output_73_transpose_y_1, x = attn_weights_159_cast_fp16, y = var_3987_cast_fp16_1)[name = string("attn_output_73_cast_fp16")]; int32 var_4017 = const()[name = string("op_4017"), val = int32(1)]; bool attn_output_75_interleave_0 = const()[name = string("attn_output_75_interleave_0"), val = bool(false)]; tensor attn_output_75_cast_fp16 = concat(axis = var_4017, interleave = attn_output_75_interleave_0, values = (var_4003_cast_fp16, attn_output_73_cast_fp16))[name = string("attn_output_75_cast_fp16")]; tensor var_4021_perm_0 = const()[name = string("op_4021_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_119x = const()[name = string("concat_119x"), val = tensor([1, 2048, 1, -1])]; tensor var_4021_cast_fp16 = transpose(perm = var_4021_perm_0, x = attn_output_75_cast_fp16)[name = string("transpose_312")]; tensor attn_output_79_cast_fp16 = reshape(shape = concat_119x, x = var_4021_cast_fp16)[name = string("attn_output_79_cast_fp16")]; tensor hidden_states_93_strides_0 = const()[name = string("hidden_states_93_strides_0"), val = tensor([1, 1])]; string hidden_states_93_pad_type_0 = const()[name = string("hidden_states_93_pad_type_0"), val = string("valid")]; tensor hidden_states_93_pad_0 = const()[name = string("hidden_states_93_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_93_dilations_0 = const()[name = string("hidden_states_93_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_93_groups_0 = const()[name = string("hidden_states_93_groups_0"), val = int32(1)]; tensor hidden_states_93_cast_fp16 = conv(dilations = hidden_states_93_dilations_0, groups = hidden_states_93_groups_0, pad = hidden_states_93_pad_0, pad_type = hidden_states_93_pad_type_0, strides = hidden_states_93_strides_0, weight = layers_9_self_attn_o_proj_weight_cast_fp16, x = attn_output_79_cast_fp16)[name = string("hidden_states_93_cast_fp16")]; tensor hidden_states_95_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = hidden_states_93_cast_fp16)[name = string("hidden_states_95_cast_fp16")]; fp16 const_98_promoted_to_fp16 = const()[name = string("const_98_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4054_cast_fp16 = mul(x = hidden_states_95_cast_fp16, y = const_98_promoted_to_fp16)[name = string("op_4054_cast_fp16")]; int32 var_4052 = const()[name = string("op_4052"), val = int32(1)]; bool doubled_77_interleave_0 = const()[name = string("doubled_77_interleave_0"), val = bool(false)]; tensor doubled_77_cast_fp16 = concat(axis = var_4052, interleave = doubled_77_interleave_0, values = (hidden_states_95_cast_fp16, var_4054_cast_fp16))[name = string("doubled_77_cast_fp16")]; tensor out_39_axes_0 = const()[name = string("out_39_axes_0"), val = tensor([1])]; tensor out_39_gamma_0_to_fp16 = const()[name = string("out_39_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405896064)))]; fp16 var_4064_to_fp16 = const()[name = string("op_4064_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_39_cast_fp16 = layer_norm(axes = out_39_axes_0, epsilon = var_4064_to_fp16, gamma = out_39_gamma_0_to_fp16, x = doubled_77_cast_fp16)[name = string("out_39_cast_fp16")]; tensor var_4075_split_sizes_0 = const()[name = string("op_4075_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4075_axis_0 = const()[name = string("op_4075_axis_0"), val = int32(1)]; tensor var_4075_cast_fp16_0, tensor var_4075_cast_fp16_1 = split(axis = var_4075_axis_0, split_sizes = var_4075_split_sizes_0, x = out_39_cast_fp16)[name = string("op_4075_cast_fp16")]; tensor input_19_strides_0 = const()[name = string("input_19_strides_0"), val = tensor([1, 1])]; string input_19_pad_type_0 = const()[name = string("input_19_pad_type_0"), val = string("valid")]; tensor input_19_pad_0 = const()[name = string("input_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_19_dilations_0 = const()[name = string("input_19_dilations_0"), val = tensor([1, 1])]; int32 input_19_groups_0 = const()[name = string("input_19_groups_0"), val = int32(1)]; tensor input_19_cast_fp16 = conv(dilations = input_19_dilations_0, groups = input_19_groups_0, pad = input_19_pad_0, pad_type = input_19_pad_type_0, strides = input_19_strides_0, weight = layers_9_mlp_gate_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("input_19_cast_fp16")]; tensor var_4092_cast_fp16 = silu(x = input_19_cast_fp16)[name = string("op_4092_cast_fp16")]; tensor var_4098_strides_0 = const()[name = string("op_4098_strides_0"), val = tensor([1, 1])]; string var_4098_pad_type_0 = const()[name = string("op_4098_pad_type_0"), val = string("valid")]; tensor var_4098_pad_0 = const()[name = string("op_4098_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4098_dilations_0 = const()[name = string("op_4098_dilations_0"), val = tensor([1, 1])]; int32 var_4098_groups_0 = const()[name = string("op_4098_groups_0"), val = int32(1)]; tensor var_4098_cast_fp16 = conv(dilations = var_4098_dilations_0, groups = var_4098_groups_0, pad = var_4098_pad_0, pad_type = var_4098_pad_type_0, strides = var_4098_strides_0, weight = layers_9_mlp_up_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("op_4098_cast_fp16")]; tensor x_99_cast_fp16 = mul(x = var_4092_cast_fp16, y = var_4098_cast_fp16)[name = string("x_99_cast_fp16")]; tensor hidden_states_97_strides_0 = const()[name = string("hidden_states_97_strides_0"), val = tensor([1, 1])]; string hidden_states_97_pad_type_0 = const()[name = string("hidden_states_97_pad_type_0"), val = string("valid")]; tensor hidden_states_97_pad_0 = const()[name = string("hidden_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_97_dilations_0 = const()[name = string("hidden_states_97_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_97_groups_0 = const()[name = string("hidden_states_97_groups_0"), val = int32(1)]; tensor hidden_states_97_cast_fp16 = conv(dilations = hidden_states_97_dilations_0, groups = hidden_states_97_groups_0, pad = hidden_states_97_pad_0, pad_type = hidden_states_97_pad_type_0, strides = hidden_states_97_strides_0, weight = layers_9_mlp_down_proj_weight_cast_fp16, x = x_99_cast_fp16)[name = string("hidden_states_97_cast_fp16")]; tensor hidden_states_99_cast_fp16 = add(x = hidden_states_95_cast_fp16, y = hidden_states_97_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; fp16 const_100_promoted_to_fp16 = const()[name = string("const_100_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4116_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_100_promoted_to_fp16)[name = string("op_4116_cast_fp16")]; int32 var_4114 = const()[name = string("op_4114"), val = int32(1)]; bool doubled_81_interleave_0 = const()[name = string("doubled_81_interleave_0"), val = bool(false)]; tensor doubled_81_cast_fp16 = concat(axis = var_4114, interleave = doubled_81_interleave_0, values = (hidden_states_99_cast_fp16, var_4116_cast_fp16))[name = string("doubled_81_cast_fp16")]; tensor out_41_axes_0 = const()[name = string("out_41_axes_0"), val = tensor([1])]; tensor out_41_gamma_0_to_fp16 = const()[name = string("out_41_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405904320)))]; fp16 var_4126_to_fp16 = const()[name = string("op_4126_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_41_cast_fp16 = layer_norm(axes = out_41_axes_0, epsilon = var_4126_to_fp16, gamma = out_41_gamma_0_to_fp16, x = doubled_81_cast_fp16)[name = string("out_41_cast_fp16")]; tensor var_4137_split_sizes_0 = const()[name = string("op_4137_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4137_axis_0 = const()[name = string("op_4137_axis_0"), val = int32(1)]; tensor var_4137_cast_fp16_0, tensor var_4137_cast_fp16_1 = split(axis = var_4137_axis_0, split_sizes = var_4137_split_sizes_0, x = out_41_cast_fp16)[name = string("op_4137_cast_fp16")]; tensor layers_10_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405912576)))]; tensor query_states_61_strides_0 = const()[name = string("query_states_61_strides_0"), val = tensor([1, 1])]; string query_states_61_pad_type_0 = const()[name = string("query_states_61_pad_type_0"), val = string("valid")]; tensor query_states_61_pad_0 = const()[name = string("query_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_61_dilations_0 = const()[name = string("query_states_61_dilations_0"), val = tensor([1, 1])]; int32 query_states_61_groups_0 = const()[name = string("query_states_61_groups_0"), val = int32(1)]; tensor query_states_61_cast_fp16 = conv(dilations = query_states_61_dilations_0, groups = query_states_61_groups_0, pad = query_states_61_pad_0, pad_type = query_states_61_pad_type_0, strides = query_states_61_strides_0, weight = layers_10_self_attn_q_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("query_states_61_cast_fp16")]; tensor layers_10_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1414301248)))]; tensor key_states_101_strides_0 = const()[name = string("key_states_101_strides_0"), val = tensor([1, 1])]; string key_states_101_pad_type_0 = const()[name = string("key_states_101_pad_type_0"), val = string("valid")]; tensor key_states_101_pad_0 = const()[name = string("key_states_101_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_101_dilations_0 = const()[name = string("key_states_101_dilations_0"), val = tensor([1, 1])]; int32 key_states_101_groups_0 = const()[name = string("key_states_101_groups_0"), val = int32(1)]; tensor key_states_101_cast_fp16 = conv(dilations = key_states_101_dilations_0, groups = key_states_101_groups_0, pad = key_states_101_pad_0, pad_type = key_states_101_pad_type_0, strides = key_states_101_strides_0, weight = layers_10_self_attn_k_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("key_states_101_cast_fp16")]; tensor value_states_61_strides_0 = const()[name = string("value_states_61_strides_0"), val = tensor([1, 1])]; string value_states_61_pad_type_0 = const()[name = string("value_states_61_pad_type_0"), val = string("valid")]; tensor value_states_61_pad_0 = const()[name = string("value_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_61_dilations_0 = const()[name = string("value_states_61_dilations_0"), val = tensor([1, 1])]; int32 value_states_61_groups_0 = const()[name = string("value_states_61_groups_0"), val = int32(1)]; tensor value_states_61_cast_fp16 = conv(dilations = value_states_61_dilations_0, groups = value_states_61_groups_0, pad = value_states_61_pad_0, pad_type = value_states_61_pad_type_0, strides = value_states_61_strides_0, weight = layers_10_self_attn_v_proj_weight_cast_fp16, x = var_4137_cast_fp16_0)[name = string("value_states_61_cast_fp16")]; tensor concat_120x = const()[name = string("concat_120x"), val = tensor([1, 16, 128, -1])]; tensor x_101_cast_fp16 = reshape(shape = concat_120x, x = query_states_61_cast_fp16)[name = string("x_101_cast_fp16")]; tensor concat_121x = const()[name = string("concat_121x"), val = tensor([1, 2, 128, -1])]; tensor var_4194_cast_fp16 = reshape(shape = concat_121x, x = key_states_101_cast_fp16)[name = string("op_4194_cast_fp16")]; tensor concat_122x = const()[name = string("concat_122x"), val = tensor([1, 2, 128, -1])]; tensor var_4201_cast_fp16 = reshape(shape = concat_122x, x = value_states_61_cast_fp16)[name = string("op_4201_cast_fp16")]; tensor var_4205_cast_fp16 = mul(x = x_101_cast_fp16, y = var_869_cast_fp16)[name = string("op_4205_cast_fp16")]; tensor var_4206_split_sizes_0 = const()[name = string("op_4206_split_sizes_0"), val = tensor([64, 64])]; int32 var_4206_axis_0 = const()[name = string("op_4206_axis_0"), val = int32(-2)]; tensor var_4206_cast_fp16_0, tensor var_4206_cast_fp16_1 = split(axis = var_4206_axis_0, split_sizes = var_4206_split_sizes_0, x = x_101_cast_fp16)[name = string("op_4206_cast_fp16")]; fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4208_cast_fp16 = mul(x = var_4206_cast_fp16_1, y = const_102_promoted_to_fp16)[name = string("op_4208_cast_fp16")]; int32 var_4210 = const()[name = string("op_4210"), val = int32(-2)]; bool var_4211_interleave_0 = const()[name = string("op_4211_interleave_0"), val = bool(false)]; tensor var_4211_cast_fp16 = concat(axis = var_4210, interleave = var_4211_interleave_0, values = (var_4208_cast_fp16, var_4206_cast_fp16_0))[name = string("op_4211_cast_fp16")]; tensor var_4212_cast_fp16 = mul(x = var_4211_cast_fp16, y = var_878_cast_fp16)[name = string("op_4212_cast_fp16")]; tensor query_states_63_cast_fp16 = add(x = var_4205_cast_fp16, y = var_4212_cast_fp16)[name = string("query_states_63_cast_fp16")]; tensor var_4218_cast_fp16 = mul(x = var_4194_cast_fp16, y = var_869_cast_fp16)[name = string("op_4218_cast_fp16")]; tensor var_4219_split_sizes_0 = const()[name = string("op_4219_split_sizes_0"), val = tensor([64, 64])]; int32 var_4219_axis_0 = const()[name = string("op_4219_axis_0"), val = int32(-2)]; tensor var_4219_cast_fp16_0, tensor var_4219_cast_fp16_1 = split(axis = var_4219_axis_0, split_sizes = var_4219_split_sizes_0, x = var_4194_cast_fp16)[name = string("op_4219_cast_fp16")]; fp16 const_103_promoted_to_fp16 = const()[name = string("const_103_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4221_cast_fp16 = mul(x = var_4219_cast_fp16_1, y = const_103_promoted_to_fp16)[name = string("op_4221_cast_fp16")]; int32 var_4223 = const()[name = string("op_4223"), val = int32(-2)]; bool var_4224_interleave_0 = const()[name = string("op_4224_interleave_0"), val = bool(false)]; tensor var_4224_cast_fp16 = concat(axis = var_4223, interleave = var_4224_interleave_0, values = (var_4221_cast_fp16, var_4219_cast_fp16_0))[name = string("op_4224_cast_fp16")]; tensor var_4225_cast_fp16 = mul(x = var_4224_cast_fp16, y = var_878_cast_fp16)[name = string("op_4225_cast_fp16")]; tensor key_states_105_cast_fp16 = add(x = var_4218_cast_fp16, y = var_4225_cast_fp16)[name = string("key_states_105_cast_fp16")]; tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; int32 concat_125_axis_0 = const()[name = string("concat_125_axis_0"), val = int32(0)]; bool concat_125_interleave_0 = const()[name = string("concat_125_interleave_0"), val = bool(false)]; tensor concat_125 = concat(axis = concat_125_axis_0, interleave = concat_125_interleave_0, values = (expand_dims_120, expand_dims_121, position_id, expand_dims_123))[name = string("concat_125")]; tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; tensor concat_126_values1_0 = const()[name = string("concat_126_values1_0"), val = tensor([0])]; tensor concat_126_values3_0 = const()[name = string("concat_126_values3_0"), val = tensor([0])]; int32 concat_126_axis_0 = const()[name = string("concat_126_axis_0"), val = int32(0)]; bool concat_126_interleave_0 = const()[name = string("concat_126_interleave_0"), val = bool(false)]; tensor concat_126 = concat(axis = concat_126_axis_0, interleave = concat_126_interleave_0, values = (expand_dims_124, concat_126_values1_0, cache_position_end, concat_126_values3_0))[name = string("concat_126")]; tensor key_states_107_perm_0 = const()[name = string("key_states_107_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_11_stride_0 = const()[name = string("key_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_107_cast_fp16 = transpose(perm = key_states_107_perm_0, x = key_states_105_cast_fp16)[name = string("transpose_311")]; tensor key_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = key_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = key_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_11_squeeze_mask_0, stride = key_cache_internal_tensor_assign_11_stride_0, update = key_states_107_cast_fp16, x = coreml_update_state_186)[name = string("key_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_11_cast_fp16, input = key_cache)[name = string("coreml_update_state_188_write_state")]; tensor coreml_update_state_188 = read_state(input = key_cache)[name = string("coreml_update_state_188")]; tensor value_states_63_perm_0 = const()[name = string("value_states_63_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_11_stride_0 = const()[name = string("value_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_63_cast_fp16 = transpose(perm = value_states_63_perm_0, x = var_4201_cast_fp16)[name = string("transpose_310")]; tensor value_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = value_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = value_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_11_squeeze_mask_0, stride = value_cache_internal_tensor_assign_11_stride_0, update = value_states_63_cast_fp16, x = coreml_update_state_187)[name = string("value_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_11_cast_fp16, input = value_cache)[name = string("coreml_update_state_189_write_state")]; tensor coreml_update_state_189 = read_state(input = value_cache)[name = string("coreml_update_state_189")]; tensor var_4295_begin_0 = const()[name = string("op_4295_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4295_end_0 = const()[name = string("op_4295_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4295_end_mask_0 = const()[name = string("op_4295_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4295_cast_fp16 = slice_by_index(begin = var_4295_begin_0, end = var_4295_end_0, end_mask = var_4295_end_mask_0, x = coreml_update_state_188)[name = string("op_4295_cast_fp16")]; tensor tile_20 = const()[name = string("tile_20"), val = tensor([1, 1])]; int32 var_4298_axis_0 = const()[name = string("op_4298_axis_0"), val = int32(1)]; tensor var_4298_cast_fp16_0, tensor var_4298_cast_fp16_1 = split(axis = var_4298_axis_0, split_sizes = tile_20, x = var_4295_cast_fp16)[name = string("op_4298_cast_fp16")]; tensor var_4305_begin_0 = const()[name = string("op_4305_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4305_end_0 = const()[name = string("op_4305_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4305_end_mask_0 = const()[name = string("op_4305_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4305_cast_fp16 = slice_by_index(begin = var_4305_begin_0, end = var_4305_end_0, end_mask = var_4305_end_mask_0, x = coreml_update_state_189)[name = string("op_4305_cast_fp16")]; tensor tile_21 = const()[name = string("tile_21"), val = tensor([1, 1])]; int32 var_4308_axis_0 = const()[name = string("op_4308_axis_0"), val = int32(1)]; tensor var_4308_cast_fp16_0, tensor var_4308_cast_fp16_1 = split(axis = var_4308_axis_0, split_sizes = tile_21, x = var_4305_cast_fp16)[name = string("op_4308_cast_fp16")]; tensor var_4311_split_sizes_0 = const()[name = string("op_4311_split_sizes_0"), val = tensor([8, 8])]; int32 var_4311_axis_0 = const()[name = string("op_4311_axis_0"), val = int32(1)]; tensor var_4311_0, tensor var_4311_1 = split(axis = var_4311_axis_0, split_sizes = var_4311_split_sizes_0, x = query_states_63_cast_fp16)[name = string("op_4311")]; bool attn_weights_161_transpose_x_0 = const()[name = string("attn_weights_161_transpose_x_0"), val = bool(false)]; bool attn_weights_161_transpose_y_0 = const()[name = string("attn_weights_161_transpose_y_0"), val = bool(false)]; tensor attn_weights_161_cast_fp16 = matmul(transpose_x = attn_weights_161_transpose_x_0, transpose_y = attn_weights_161_transpose_y_0, x = var_4298_cast_fp16_0, y = var_4311_0)[name = string("attn_weights_161_cast_fp16")]; fp16 var_4314_to_fp16 = const()[name = string("op_4314_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_163_cast_fp16 = mul(x = attn_weights_161_cast_fp16, y = var_4314_to_fp16)[name = string("attn_weights_163_cast_fp16")]; tensor attn_weights_165_cast_fp16 = add(x = attn_weights_163_cast_fp16, y = attn_mask_1)[name = string("attn_weights_165_cast_fp16")]; int32 var_4318 = const()[name = string("op_4318"), val = int32(-2)]; tensor attn_weights_167_cast_fp16 = softmax(axis = var_4318, x = attn_weights_165_cast_fp16)[name = string("attn_weights_167_cast_fp16")]; bool var_4324_transpose_x_1 = const()[name = string("op_4324_transpose_x_1"), val = bool(true)]; bool var_4324_transpose_y_1 = const()[name = string("op_4324_transpose_y_1"), val = bool(false)]; tensor var_4324_cast_fp16 = matmul(transpose_x = var_4324_transpose_x_1, transpose_y = var_4324_transpose_y_1, x = attn_weights_167_cast_fp16, y = var_4308_cast_fp16_0)[name = string("op_4324_cast_fp16")]; bool attn_weights_169_transpose_x_0 = const()[name = string("attn_weights_169_transpose_x_0"), val = bool(false)]; bool attn_weights_169_transpose_y_0 = const()[name = string("attn_weights_169_transpose_y_0"), val = bool(false)]; tensor attn_weights_169_cast_fp16 = matmul(transpose_x = attn_weights_169_transpose_x_0, transpose_y = attn_weights_169_transpose_y_0, x = var_4298_cast_fp16_1, y = var_4311_1)[name = string("attn_weights_169_cast_fp16")]; fp16 var_4326_to_fp16 = const()[name = string("op_4326_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_171_cast_fp16 = mul(x = attn_weights_169_cast_fp16, y = var_4326_to_fp16)[name = string("attn_weights_171_cast_fp16")]; tensor attn_weights_173_cast_fp16 = add(x = attn_weights_171_cast_fp16, y = attn_mask_1)[name = string("attn_weights_173_cast_fp16")]; int32 var_4330 = const()[name = string("op_4330"), val = int32(-2)]; tensor attn_weights_175_cast_fp16 = softmax(axis = var_4330, x = attn_weights_173_cast_fp16)[name = string("attn_weights_175_cast_fp16")]; bool attn_output_81_transpose_x_1 = const()[name = string("attn_output_81_transpose_x_1"), val = bool(true)]; bool attn_output_81_transpose_y_1 = const()[name = string("attn_output_81_transpose_y_1"), val = bool(false)]; tensor attn_output_81_cast_fp16 = matmul(transpose_x = attn_output_81_transpose_x_1, transpose_y = attn_output_81_transpose_y_1, x = attn_weights_175_cast_fp16, y = var_4308_cast_fp16_1)[name = string("attn_output_81_cast_fp16")]; int32 var_4338 = const()[name = string("op_4338"), val = int32(1)]; bool attn_output_83_interleave_0 = const()[name = string("attn_output_83_interleave_0"), val = bool(false)]; tensor attn_output_83_cast_fp16 = concat(axis = var_4338, interleave = attn_output_83_interleave_0, values = (var_4324_cast_fp16, attn_output_81_cast_fp16))[name = string("attn_output_83_cast_fp16")]; tensor var_4342_perm_0 = const()[name = string("op_4342_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_131x = const()[name = string("concat_131x"), val = tensor([1, 2048, 1, -1])]; tensor var_4342_cast_fp16 = transpose(perm = var_4342_perm_0, x = attn_output_83_cast_fp16)[name = string("transpose_309")]; tensor attn_output_87_cast_fp16 = reshape(shape = concat_131x, x = var_4342_cast_fp16)[name = string("attn_output_87_cast_fp16")]; tensor hidden_states_103_strides_0 = const()[name = string("hidden_states_103_strides_0"), val = tensor([1, 1])]; string hidden_states_103_pad_type_0 = const()[name = string("hidden_states_103_pad_type_0"), val = string("valid")]; tensor hidden_states_103_pad_0 = const()[name = string("hidden_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_103_dilations_0 = const()[name = string("hidden_states_103_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_103_groups_0 = const()[name = string("hidden_states_103_groups_0"), val = int32(1)]; tensor hidden_states_103_cast_fp16 = conv(dilations = hidden_states_103_dilations_0, groups = hidden_states_103_groups_0, pad = hidden_states_103_pad_0, pad_type = hidden_states_103_pad_type_0, strides = hidden_states_103_strides_0, weight = layers_10_self_attn_o_proj_weight_cast_fp16, x = attn_output_87_cast_fp16)[name = string("hidden_states_103_cast_fp16")]; tensor hidden_states_105_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = hidden_states_103_cast_fp16)[name = string("hidden_states_105_cast_fp16")]; fp16 const_108_promoted_to_fp16 = const()[name = string("const_108_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4375_cast_fp16 = mul(x = hidden_states_105_cast_fp16, y = const_108_promoted_to_fp16)[name = string("op_4375_cast_fp16")]; int32 var_4373 = const()[name = string("op_4373"), val = int32(1)]; bool doubled_85_interleave_0 = const()[name = string("doubled_85_interleave_0"), val = bool(false)]; tensor doubled_85_cast_fp16 = concat(axis = var_4373, interleave = doubled_85_interleave_0, values = (hidden_states_105_cast_fp16, var_4375_cast_fp16))[name = string("doubled_85_cast_fp16")]; tensor out_43_axes_0 = const()[name = string("out_43_axes_0"), val = tensor([1])]; tensor out_43_gamma_0_to_fp16 = const()[name = string("out_43_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415349888)))]; fp16 var_4385_to_fp16 = const()[name = string("op_4385_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_43_cast_fp16 = layer_norm(axes = out_43_axes_0, epsilon = var_4385_to_fp16, gamma = out_43_gamma_0_to_fp16, x = doubled_85_cast_fp16)[name = string("out_43_cast_fp16")]; tensor var_4396_split_sizes_0 = const()[name = string("op_4396_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4396_axis_0 = const()[name = string("op_4396_axis_0"), val = int32(1)]; tensor var_4396_cast_fp16_0, tensor var_4396_cast_fp16_1 = split(axis = var_4396_axis_0, split_sizes = var_4396_split_sizes_0, x = out_43_cast_fp16)[name = string("op_4396_cast_fp16")]; tensor input_21_strides_0 = const()[name = string("input_21_strides_0"), val = tensor([1, 1])]; string input_21_pad_type_0 = const()[name = string("input_21_pad_type_0"), val = string("valid")]; tensor input_21_pad_0 = const()[name = string("input_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_21_dilations_0 = const()[name = string("input_21_dilations_0"), val = tensor([1, 1])]; int32 input_21_groups_0 = const()[name = string("input_21_groups_0"), val = int32(1)]; tensor input_21_cast_fp16 = conv(dilations = input_21_dilations_0, groups = input_21_groups_0, pad = input_21_pad_0, pad_type = input_21_pad_type_0, strides = input_21_strides_0, weight = layers_10_mlp_gate_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("input_21_cast_fp16")]; tensor var_4413_cast_fp16 = silu(x = input_21_cast_fp16)[name = string("op_4413_cast_fp16")]; tensor var_4419_strides_0 = const()[name = string("op_4419_strides_0"), val = tensor([1, 1])]; string var_4419_pad_type_0 = const()[name = string("op_4419_pad_type_0"), val = string("valid")]; tensor var_4419_pad_0 = const()[name = string("op_4419_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4419_dilations_0 = const()[name = string("op_4419_dilations_0"), val = tensor([1, 1])]; int32 var_4419_groups_0 = const()[name = string("op_4419_groups_0"), val = int32(1)]; tensor var_4419_cast_fp16 = conv(dilations = var_4419_dilations_0, groups = var_4419_groups_0, pad = var_4419_pad_0, pad_type = var_4419_pad_type_0, strides = var_4419_strides_0, weight = layers_10_mlp_up_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("op_4419_cast_fp16")]; tensor x_109_cast_fp16 = mul(x = var_4413_cast_fp16, y = var_4419_cast_fp16)[name = string("x_109_cast_fp16")]; tensor hidden_states_107_strides_0 = const()[name = string("hidden_states_107_strides_0"), val = tensor([1, 1])]; string hidden_states_107_pad_type_0 = const()[name = string("hidden_states_107_pad_type_0"), val = string("valid")]; tensor hidden_states_107_pad_0 = const()[name = string("hidden_states_107_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_107_dilations_0 = const()[name = string("hidden_states_107_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_107_groups_0 = const()[name = string("hidden_states_107_groups_0"), val = int32(1)]; tensor hidden_states_107_cast_fp16 = conv(dilations = hidden_states_107_dilations_0, groups = hidden_states_107_groups_0, pad = hidden_states_107_pad_0, pad_type = hidden_states_107_pad_type_0, strides = hidden_states_107_strides_0, weight = layers_10_mlp_down_proj_weight_cast_fp16, x = x_109_cast_fp16)[name = string("hidden_states_107_cast_fp16")]; tensor hidden_states_109_cast_fp16 = add(x = hidden_states_105_cast_fp16, y = hidden_states_107_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; fp16 const_110_promoted_to_fp16 = const()[name = string("const_110_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4437_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_110_promoted_to_fp16)[name = string("op_4437_cast_fp16")]; int32 var_4435 = const()[name = string("op_4435"), val = int32(1)]; bool doubled_89_interleave_0 = const()[name = string("doubled_89_interleave_0"), val = bool(false)]; tensor doubled_89_cast_fp16 = concat(axis = var_4435, interleave = doubled_89_interleave_0, values = (hidden_states_109_cast_fp16, var_4437_cast_fp16))[name = string("doubled_89_cast_fp16")]; tensor out_45_axes_0 = const()[name = string("out_45_axes_0"), val = tensor([1])]; tensor out_45_gamma_0_to_fp16 = const()[name = string("out_45_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415358144)))]; fp16 var_4447_to_fp16 = const()[name = string("op_4447_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_45_cast_fp16 = layer_norm(axes = out_45_axes_0, epsilon = var_4447_to_fp16, gamma = out_45_gamma_0_to_fp16, x = doubled_89_cast_fp16)[name = string("out_45_cast_fp16")]; tensor var_4458_split_sizes_0 = const()[name = string("op_4458_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4458_axis_0 = const()[name = string("op_4458_axis_0"), val = int32(1)]; tensor var_4458_cast_fp16_0, tensor var_4458_cast_fp16_1 = split(axis = var_4458_axis_0, split_sizes = var_4458_split_sizes_0, x = out_45_cast_fp16)[name = string("op_4458_cast_fp16")]; tensor query_states_67_strides_0 = const()[name = string("query_states_67_strides_0"), val = tensor([1, 1])]; string query_states_67_pad_type_0 = const()[name = string("query_states_67_pad_type_0"), val = string("valid")]; tensor query_states_67_pad_0 = const()[name = string("query_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_67_dilations_0 = const()[name = string("query_states_67_dilations_0"), val = tensor([1, 1])]; int32 query_states_67_groups_0 = const()[name = string("query_states_67_groups_0"), val = int32(1)]; tensor query_states_67_cast_fp16 = conv(dilations = query_states_67_dilations_0, groups = query_states_67_groups_0, pad = query_states_67_pad_0, pad_type = query_states_67_pad_type_0, strides = query_states_67_strides_0, weight = layers_11_self_attn_q_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("query_states_67_cast_fp16")]; tensor key_states_111_strides_0 = const()[name = string("key_states_111_strides_0"), val = tensor([1, 1])]; string key_states_111_pad_type_0 = const()[name = string("key_states_111_pad_type_0"), val = string("valid")]; tensor key_states_111_pad_0 = const()[name = string("key_states_111_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_111_dilations_0 = const()[name = string("key_states_111_dilations_0"), val = tensor([1, 1])]; int32 key_states_111_groups_0 = const()[name = string("key_states_111_groups_0"), val = int32(1)]; tensor key_states_111_cast_fp16 = conv(dilations = key_states_111_dilations_0, groups = key_states_111_groups_0, pad = key_states_111_pad_0, pad_type = key_states_111_pad_type_0, strides = key_states_111_strides_0, weight = layers_11_self_attn_k_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("key_states_111_cast_fp16")]; tensor value_states_67_strides_0 = const()[name = string("value_states_67_strides_0"), val = tensor([1, 1])]; string value_states_67_pad_type_0 = const()[name = string("value_states_67_pad_type_0"), val = string("valid")]; tensor value_states_67_pad_0 = const()[name = string("value_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_67_dilations_0 = const()[name = string("value_states_67_dilations_0"), val = tensor([1, 1])]; int32 value_states_67_groups_0 = const()[name = string("value_states_67_groups_0"), val = int32(1)]; tensor value_states_67_cast_fp16 = conv(dilations = value_states_67_dilations_0, groups = value_states_67_groups_0, pad = value_states_67_pad_0, pad_type = value_states_67_pad_type_0, strides = value_states_67_strides_0, weight = layers_11_self_attn_v_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("value_states_67_cast_fp16")]; tensor concat_132x = const()[name = string("concat_132x"), val = tensor([1, 16, 128, -1])]; tensor x_111_cast_fp16 = reshape(shape = concat_132x, x = query_states_67_cast_fp16)[name = string("x_111_cast_fp16")]; tensor concat_133x = const()[name = string("concat_133x"), val = tensor([1, 2, 128, -1])]; tensor var_4515_cast_fp16 = reshape(shape = concat_133x, x = key_states_111_cast_fp16)[name = string("op_4515_cast_fp16")]; tensor concat_134x = const()[name = string("concat_134x"), val = tensor([1, 2, 128, -1])]; tensor var_4522_cast_fp16 = reshape(shape = concat_134x, x = value_states_67_cast_fp16)[name = string("op_4522_cast_fp16")]; tensor var_4526_cast_fp16 = mul(x = x_111_cast_fp16, y = var_869_cast_fp16)[name = string("op_4526_cast_fp16")]; tensor var_4527_split_sizes_0 = const()[name = string("op_4527_split_sizes_0"), val = tensor([64, 64])]; int32 var_4527_axis_0 = const()[name = string("op_4527_axis_0"), val = int32(-2)]; tensor var_4527_cast_fp16_0, tensor var_4527_cast_fp16_1 = split(axis = var_4527_axis_0, split_sizes = var_4527_split_sizes_0, x = x_111_cast_fp16)[name = string("op_4527_cast_fp16")]; fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4529_cast_fp16 = mul(x = var_4527_cast_fp16_1, y = const_112_promoted_to_fp16)[name = string("op_4529_cast_fp16")]; int32 var_4531 = const()[name = string("op_4531"), val = int32(-2)]; bool var_4532_interleave_0 = const()[name = string("op_4532_interleave_0"), val = bool(false)]; tensor var_4532_cast_fp16 = concat(axis = var_4531, interleave = var_4532_interleave_0, values = (var_4529_cast_fp16, var_4527_cast_fp16_0))[name = string("op_4532_cast_fp16")]; tensor var_4533_cast_fp16 = mul(x = var_4532_cast_fp16, y = var_878_cast_fp16)[name = string("op_4533_cast_fp16")]; tensor query_states_69_cast_fp16 = add(x = var_4526_cast_fp16, y = var_4533_cast_fp16)[name = string("query_states_69_cast_fp16")]; tensor var_4539_cast_fp16 = mul(x = var_4515_cast_fp16, y = var_869_cast_fp16)[name = string("op_4539_cast_fp16")]; tensor var_4540_split_sizes_0 = const()[name = string("op_4540_split_sizes_0"), val = tensor([64, 64])]; int32 var_4540_axis_0 = const()[name = string("op_4540_axis_0"), val = int32(-2)]; tensor var_4540_cast_fp16_0, tensor var_4540_cast_fp16_1 = split(axis = var_4540_axis_0, split_sizes = var_4540_split_sizes_0, x = var_4515_cast_fp16)[name = string("op_4540_cast_fp16")]; fp16 const_113_promoted_to_fp16 = const()[name = string("const_113_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4542_cast_fp16 = mul(x = var_4540_cast_fp16_1, y = const_113_promoted_to_fp16)[name = string("op_4542_cast_fp16")]; int32 var_4544 = const()[name = string("op_4544"), val = int32(-2)]; bool var_4545_interleave_0 = const()[name = string("op_4545_interleave_0"), val = bool(false)]; tensor var_4545_cast_fp16 = concat(axis = var_4544, interleave = var_4545_interleave_0, values = (var_4542_cast_fp16, var_4540_cast_fp16_0))[name = string("op_4545_cast_fp16")]; tensor var_4546_cast_fp16 = mul(x = var_4545_cast_fp16, y = var_878_cast_fp16)[name = string("op_4546_cast_fp16")]; tensor key_states_115_cast_fp16 = add(x = var_4539_cast_fp16, y = var_4546_cast_fp16)[name = string("key_states_115_cast_fp16")]; tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; int32 concat_137_axis_0 = const()[name = string("concat_137_axis_0"), val = int32(0)]; bool concat_137_interleave_0 = const()[name = string("concat_137_interleave_0"), val = bool(false)]; tensor concat_137 = concat(axis = concat_137_axis_0, interleave = concat_137_interleave_0, values = (expand_dims_132, expand_dims_133, position_id, expand_dims_135))[name = string("concat_137")]; tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; tensor concat_138_values1_0 = const()[name = string("concat_138_values1_0"), val = tensor([0])]; tensor concat_138_values3_0 = const()[name = string("concat_138_values3_0"), val = tensor([0])]; int32 concat_138_axis_0 = const()[name = string("concat_138_axis_0"), val = int32(0)]; bool concat_138_interleave_0 = const()[name = string("concat_138_interleave_0"), val = bool(false)]; tensor concat_138 = concat(axis = concat_138_axis_0, interleave = concat_138_interleave_0, values = (expand_dims_136, concat_138_values1_0, cache_position_end, concat_138_values3_0))[name = string("concat_138")]; tensor key_states_117_perm_0 = const()[name = string("key_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_12_stride_0 = const()[name = string("key_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_117_cast_fp16 = transpose(perm = key_states_117_perm_0, x = key_states_115_cast_fp16)[name = string("transpose_308")]; tensor key_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = key_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = key_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_12_squeeze_mask_0, stride = key_cache_internal_tensor_assign_12_stride_0, update = key_states_117_cast_fp16, x = coreml_update_state_188)[name = string("key_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_12_cast_fp16, input = key_cache)[name = string("coreml_update_state_190_write_state")]; tensor coreml_update_state_190 = read_state(input = key_cache)[name = string("coreml_update_state_190")]; tensor value_states_69_perm_0 = const()[name = string("value_states_69_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_12_stride_0 = const()[name = string("value_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_69_cast_fp16 = transpose(perm = value_states_69_perm_0, x = var_4522_cast_fp16)[name = string("transpose_307")]; tensor value_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = value_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = value_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_12_squeeze_mask_0, stride = value_cache_internal_tensor_assign_12_stride_0, update = value_states_69_cast_fp16, x = coreml_update_state_189)[name = string("value_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_12_cast_fp16, input = value_cache)[name = string("coreml_update_state_191_write_state")]; tensor coreml_update_state_191 = read_state(input = value_cache)[name = string("coreml_update_state_191")]; tensor var_4616_begin_0 = const()[name = string("op_4616_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4616_end_0 = const()[name = string("op_4616_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4616_end_mask_0 = const()[name = string("op_4616_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4616_cast_fp16 = slice_by_index(begin = var_4616_begin_0, end = var_4616_end_0, end_mask = var_4616_end_mask_0, x = coreml_update_state_190)[name = string("op_4616_cast_fp16")]; tensor tile_22 = const()[name = string("tile_22"), val = tensor([1, 1])]; int32 var_4619_axis_0 = const()[name = string("op_4619_axis_0"), val = int32(1)]; tensor var_4619_cast_fp16_0, tensor var_4619_cast_fp16_1 = split(axis = var_4619_axis_0, split_sizes = tile_22, x = var_4616_cast_fp16)[name = string("op_4619_cast_fp16")]; tensor var_4626_begin_0 = const()[name = string("op_4626_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4626_end_0 = const()[name = string("op_4626_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4626_end_mask_0 = const()[name = string("op_4626_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4626_cast_fp16 = slice_by_index(begin = var_4626_begin_0, end = var_4626_end_0, end_mask = var_4626_end_mask_0, x = coreml_update_state_191)[name = string("op_4626_cast_fp16")]; tensor tile_23 = const()[name = string("tile_23"), val = tensor([1, 1])]; int32 var_4629_axis_0 = const()[name = string("op_4629_axis_0"), val = int32(1)]; tensor var_4629_cast_fp16_0, tensor var_4629_cast_fp16_1 = split(axis = var_4629_axis_0, split_sizes = tile_23, x = var_4626_cast_fp16)[name = string("op_4629_cast_fp16")]; tensor var_4632_split_sizes_0 = const()[name = string("op_4632_split_sizes_0"), val = tensor([8, 8])]; int32 var_4632_axis_0 = const()[name = string("op_4632_axis_0"), val = int32(1)]; tensor var_4632_0, tensor var_4632_1 = split(axis = var_4632_axis_0, split_sizes = var_4632_split_sizes_0, x = query_states_69_cast_fp16)[name = string("op_4632")]; bool attn_weights_177_transpose_x_0 = const()[name = string("attn_weights_177_transpose_x_0"), val = bool(false)]; bool attn_weights_177_transpose_y_0 = const()[name = string("attn_weights_177_transpose_y_0"), val = bool(false)]; tensor attn_weights_177_cast_fp16 = matmul(transpose_x = attn_weights_177_transpose_x_0, transpose_y = attn_weights_177_transpose_y_0, x = var_4619_cast_fp16_0, y = var_4632_0)[name = string("attn_weights_177_cast_fp16")]; fp16 var_4635_to_fp16 = const()[name = string("op_4635_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_179_cast_fp16 = mul(x = attn_weights_177_cast_fp16, y = var_4635_to_fp16)[name = string("attn_weights_179_cast_fp16")]; tensor attn_weights_181_cast_fp16 = add(x = attn_weights_179_cast_fp16, y = attn_mask_1)[name = string("attn_weights_181_cast_fp16")]; int32 var_4639 = const()[name = string("op_4639"), val = int32(-2)]; tensor attn_weights_183_cast_fp16 = softmax(axis = var_4639, x = attn_weights_181_cast_fp16)[name = string("attn_weights_183_cast_fp16")]; bool var_4645_transpose_x_1 = const()[name = string("op_4645_transpose_x_1"), val = bool(true)]; bool var_4645_transpose_y_1 = const()[name = string("op_4645_transpose_y_1"), val = bool(false)]; tensor var_4645_cast_fp16 = matmul(transpose_x = var_4645_transpose_x_1, transpose_y = var_4645_transpose_y_1, x = attn_weights_183_cast_fp16, y = var_4629_cast_fp16_0)[name = string("op_4645_cast_fp16")]; bool attn_weights_185_transpose_x_0 = const()[name = string("attn_weights_185_transpose_x_0"), val = bool(false)]; bool attn_weights_185_transpose_y_0 = const()[name = string("attn_weights_185_transpose_y_0"), val = bool(false)]; tensor attn_weights_185_cast_fp16 = matmul(transpose_x = attn_weights_185_transpose_x_0, transpose_y = attn_weights_185_transpose_y_0, x = var_4619_cast_fp16_1, y = var_4632_1)[name = string("attn_weights_185_cast_fp16")]; fp16 var_4647_to_fp16 = const()[name = string("op_4647_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_187_cast_fp16 = mul(x = attn_weights_185_cast_fp16, y = var_4647_to_fp16)[name = string("attn_weights_187_cast_fp16")]; tensor attn_weights_189_cast_fp16 = add(x = attn_weights_187_cast_fp16, y = attn_mask_1)[name = string("attn_weights_189_cast_fp16")]; int32 var_4651 = const()[name = string("op_4651"), val = int32(-2)]; tensor attn_weights_191_cast_fp16 = softmax(axis = var_4651, x = attn_weights_189_cast_fp16)[name = string("attn_weights_191_cast_fp16")]; bool attn_output_89_transpose_x_1 = const()[name = string("attn_output_89_transpose_x_1"), val = bool(true)]; bool attn_output_89_transpose_y_1 = const()[name = string("attn_output_89_transpose_y_1"), val = bool(false)]; tensor attn_output_89_cast_fp16 = matmul(transpose_x = attn_output_89_transpose_x_1, transpose_y = attn_output_89_transpose_y_1, x = attn_weights_191_cast_fp16, y = var_4629_cast_fp16_1)[name = string("attn_output_89_cast_fp16")]; int32 var_4659 = const()[name = string("op_4659"), val = int32(1)]; bool attn_output_91_interleave_0 = const()[name = string("attn_output_91_interleave_0"), val = bool(false)]; tensor attn_output_91_cast_fp16 = concat(axis = var_4659, interleave = attn_output_91_interleave_0, values = (var_4645_cast_fp16, attn_output_89_cast_fp16))[name = string("attn_output_91_cast_fp16")]; tensor var_4663_perm_0 = const()[name = string("op_4663_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_143x = const()[name = string("concat_143x"), val = tensor([1, 2048, 1, -1])]; tensor var_4663_cast_fp16 = transpose(perm = var_4663_perm_0, x = attn_output_91_cast_fp16)[name = string("transpose_306")]; tensor attn_output_95_cast_fp16 = reshape(shape = concat_143x, x = var_4663_cast_fp16)[name = string("attn_output_95_cast_fp16")]; tensor hidden_states_113_strides_0 = const()[name = string("hidden_states_113_strides_0"), val = tensor([1, 1])]; string hidden_states_113_pad_type_0 = const()[name = string("hidden_states_113_pad_type_0"), val = string("valid")]; tensor hidden_states_113_pad_0 = const()[name = string("hidden_states_113_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_113_dilations_0 = const()[name = string("hidden_states_113_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_113_groups_0 = const()[name = string("hidden_states_113_groups_0"), val = int32(1)]; tensor hidden_states_113_cast_fp16 = conv(dilations = hidden_states_113_dilations_0, groups = hidden_states_113_groups_0, pad = hidden_states_113_pad_0, pad_type = hidden_states_113_pad_type_0, strides = hidden_states_113_strides_0, weight = layers_11_self_attn_o_proj_weight_cast_fp16, x = attn_output_95_cast_fp16)[name = string("hidden_states_113_cast_fp16")]; tensor hidden_states_115_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = hidden_states_113_cast_fp16)[name = string("hidden_states_115_cast_fp16")]; fp16 const_118_promoted_to_fp16 = const()[name = string("const_118_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4696_cast_fp16 = mul(x = hidden_states_115_cast_fp16, y = const_118_promoted_to_fp16)[name = string("op_4696_cast_fp16")]; int32 var_4694 = const()[name = string("op_4694"), val = int32(1)]; bool doubled_93_interleave_0 = const()[name = string("doubled_93_interleave_0"), val = bool(false)]; tensor doubled_93_cast_fp16 = concat(axis = var_4694, interleave = doubled_93_interleave_0, values = (hidden_states_115_cast_fp16, var_4696_cast_fp16))[name = string("doubled_93_cast_fp16")]; tensor out_47_axes_0 = const()[name = string("out_47_axes_0"), val = tensor([1])]; tensor out_47_gamma_0_to_fp16 = const()[name = string("out_47_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415366400)))]; fp16 var_4706_to_fp16 = const()[name = string("op_4706_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_47_cast_fp16 = layer_norm(axes = out_47_axes_0, epsilon = var_4706_to_fp16, gamma = out_47_gamma_0_to_fp16, x = doubled_93_cast_fp16)[name = string("out_47_cast_fp16")]; tensor var_4717_split_sizes_0 = const()[name = string("op_4717_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4717_axis_0 = const()[name = string("op_4717_axis_0"), val = int32(1)]; tensor var_4717_cast_fp16_0, tensor var_4717_cast_fp16_1 = split(axis = var_4717_axis_0, split_sizes = var_4717_split_sizes_0, x = out_47_cast_fp16)[name = string("op_4717_cast_fp16")]; tensor input_23_strides_0 = const()[name = string("input_23_strides_0"), val = tensor([1, 1])]; string input_23_pad_type_0 = const()[name = string("input_23_pad_type_0"), val = string("valid")]; tensor input_23_pad_0 = const()[name = string("input_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_23_dilations_0 = const()[name = string("input_23_dilations_0"), val = tensor([1, 1])]; int32 input_23_groups_0 = const()[name = string("input_23_groups_0"), val = int32(1)]; tensor input_23_cast_fp16 = conv(dilations = input_23_dilations_0, groups = input_23_groups_0, pad = input_23_pad_0, pad_type = input_23_pad_type_0, strides = input_23_strides_0, weight = layers_11_mlp_gate_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("input_23_cast_fp16")]; tensor var_4734_cast_fp16 = silu(x = input_23_cast_fp16)[name = string("op_4734_cast_fp16")]; tensor var_4740_strides_0 = const()[name = string("op_4740_strides_0"), val = tensor([1, 1])]; string var_4740_pad_type_0 = const()[name = string("op_4740_pad_type_0"), val = string("valid")]; tensor var_4740_pad_0 = const()[name = string("op_4740_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4740_dilations_0 = const()[name = string("op_4740_dilations_0"), val = tensor([1, 1])]; int32 var_4740_groups_0 = const()[name = string("op_4740_groups_0"), val = int32(1)]; tensor var_4740_cast_fp16 = conv(dilations = var_4740_dilations_0, groups = var_4740_groups_0, pad = var_4740_pad_0, pad_type = var_4740_pad_type_0, strides = var_4740_strides_0, weight = layers_11_mlp_up_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("op_4740_cast_fp16")]; tensor x_119_cast_fp16 = mul(x = var_4734_cast_fp16, y = var_4740_cast_fp16)[name = string("x_119_cast_fp16")]; tensor hidden_states_117_strides_0 = const()[name = string("hidden_states_117_strides_0"), val = tensor([1, 1])]; string hidden_states_117_pad_type_0 = const()[name = string("hidden_states_117_pad_type_0"), val = string("valid")]; tensor hidden_states_117_pad_0 = const()[name = string("hidden_states_117_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_117_dilations_0 = const()[name = string("hidden_states_117_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_117_groups_0 = const()[name = string("hidden_states_117_groups_0"), val = int32(1)]; tensor hidden_states_117_cast_fp16 = conv(dilations = hidden_states_117_dilations_0, groups = hidden_states_117_groups_0, pad = hidden_states_117_pad_0, pad_type = hidden_states_117_pad_type_0, strides = hidden_states_117_strides_0, weight = layers_11_mlp_down_proj_weight_cast_fp16, x = x_119_cast_fp16)[name = string("hidden_states_117_cast_fp16")]; tensor hidden_states_119_cast_fp16 = add(x = hidden_states_115_cast_fp16, y = hidden_states_117_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; fp16 const_120_promoted_to_fp16 = const()[name = string("const_120_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4758_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_120_promoted_to_fp16)[name = string("op_4758_cast_fp16")]; int32 var_4756 = const()[name = string("op_4756"), val = int32(1)]; bool doubled_97_interleave_0 = const()[name = string("doubled_97_interleave_0"), val = bool(false)]; tensor doubled_97_cast_fp16 = concat(axis = var_4756, interleave = doubled_97_interleave_0, values = (hidden_states_119_cast_fp16, var_4758_cast_fp16))[name = string("doubled_97_cast_fp16")]; tensor out_49_axes_0 = const()[name = string("out_49_axes_0"), val = tensor([1])]; tensor out_49_gamma_0_to_fp16 = const()[name = string("out_49_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415374656)))]; fp16 var_4768_to_fp16 = const()[name = string("op_4768_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_49_cast_fp16 = layer_norm(axes = out_49_axes_0, epsilon = var_4768_to_fp16, gamma = out_49_gamma_0_to_fp16, x = doubled_97_cast_fp16)[name = string("out_49_cast_fp16")]; tensor var_4779_split_sizes_0 = const()[name = string("op_4779_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4779_axis_0 = const()[name = string("op_4779_axis_0"), val = int32(1)]; tensor var_4779_cast_fp16_0, tensor var_4779_cast_fp16_1 = split(axis = var_4779_axis_0, split_sizes = var_4779_split_sizes_0, x = out_49_cast_fp16)[name = string("op_4779_cast_fp16")]; tensor query_states_73_strides_0 = const()[name = string("query_states_73_strides_0"), val = tensor([1, 1])]; string query_states_73_pad_type_0 = const()[name = string("query_states_73_pad_type_0"), val = string("valid")]; tensor query_states_73_pad_0 = const()[name = string("query_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_73_dilations_0 = const()[name = string("query_states_73_dilations_0"), val = tensor([1, 1])]; int32 query_states_73_groups_0 = const()[name = string("query_states_73_groups_0"), val = int32(1)]; tensor query_states_73_cast_fp16 = conv(dilations = query_states_73_dilations_0, groups = query_states_73_groups_0, pad = query_states_73_pad_0, pad_type = query_states_73_pad_type_0, strides = query_states_73_strides_0, weight = layers_12_self_attn_q_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("query_states_73_cast_fp16")]; tensor key_states_121_strides_0 = const()[name = string("key_states_121_strides_0"), val = tensor([1, 1])]; string key_states_121_pad_type_0 = const()[name = string("key_states_121_pad_type_0"), val = string("valid")]; tensor key_states_121_pad_0 = const()[name = string("key_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_121_dilations_0 = const()[name = string("key_states_121_dilations_0"), val = tensor([1, 1])]; int32 key_states_121_groups_0 = const()[name = string("key_states_121_groups_0"), val = int32(1)]; tensor key_states_121_cast_fp16 = conv(dilations = key_states_121_dilations_0, groups = key_states_121_groups_0, pad = key_states_121_pad_0, pad_type = key_states_121_pad_type_0, strides = key_states_121_strides_0, weight = layers_12_self_attn_k_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("key_states_121_cast_fp16")]; tensor value_states_73_strides_0 = const()[name = string("value_states_73_strides_0"), val = tensor([1, 1])]; string value_states_73_pad_type_0 = const()[name = string("value_states_73_pad_type_0"), val = string("valid")]; tensor value_states_73_pad_0 = const()[name = string("value_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_73_dilations_0 = const()[name = string("value_states_73_dilations_0"), val = tensor([1, 1])]; int32 value_states_73_groups_0 = const()[name = string("value_states_73_groups_0"), val = int32(1)]; tensor value_states_73_cast_fp16 = conv(dilations = value_states_73_dilations_0, groups = value_states_73_groups_0, pad = value_states_73_pad_0, pad_type = value_states_73_pad_type_0, strides = value_states_73_strides_0, weight = layers_12_self_attn_v_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("value_states_73_cast_fp16")]; tensor concat_144x = const()[name = string("concat_144x"), val = tensor([1, 16, 128, -1])]; tensor x_121_cast_fp16 = reshape(shape = concat_144x, x = query_states_73_cast_fp16)[name = string("x_121_cast_fp16")]; tensor concat_145x = const()[name = string("concat_145x"), val = tensor([1, 2, 128, -1])]; tensor var_4836_cast_fp16 = reshape(shape = concat_145x, x = key_states_121_cast_fp16)[name = string("op_4836_cast_fp16")]; tensor concat_146x = const()[name = string("concat_146x"), val = tensor([1, 2, 128, -1])]; tensor var_4843_cast_fp16 = reshape(shape = concat_146x, x = value_states_73_cast_fp16)[name = string("op_4843_cast_fp16")]; tensor var_4847_cast_fp16 = mul(x = x_121_cast_fp16, y = var_869_cast_fp16)[name = string("op_4847_cast_fp16")]; tensor var_4848_split_sizes_0 = const()[name = string("op_4848_split_sizes_0"), val = tensor([64, 64])]; int32 var_4848_axis_0 = const()[name = string("op_4848_axis_0"), val = int32(-2)]; tensor var_4848_cast_fp16_0, tensor var_4848_cast_fp16_1 = split(axis = var_4848_axis_0, split_sizes = var_4848_split_sizes_0, x = x_121_cast_fp16)[name = string("op_4848_cast_fp16")]; fp16 const_122_promoted_to_fp16 = const()[name = string("const_122_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4850_cast_fp16 = mul(x = var_4848_cast_fp16_1, y = const_122_promoted_to_fp16)[name = string("op_4850_cast_fp16")]; int32 var_4852 = const()[name = string("op_4852"), val = int32(-2)]; bool var_4853_interleave_0 = const()[name = string("op_4853_interleave_0"), val = bool(false)]; tensor var_4853_cast_fp16 = concat(axis = var_4852, interleave = var_4853_interleave_0, values = (var_4850_cast_fp16, var_4848_cast_fp16_0))[name = string("op_4853_cast_fp16")]; tensor var_4854_cast_fp16 = mul(x = var_4853_cast_fp16, y = var_878_cast_fp16)[name = string("op_4854_cast_fp16")]; tensor query_states_75_cast_fp16 = add(x = var_4847_cast_fp16, y = var_4854_cast_fp16)[name = string("query_states_75_cast_fp16")]; tensor var_4860_cast_fp16 = mul(x = var_4836_cast_fp16, y = var_869_cast_fp16)[name = string("op_4860_cast_fp16")]; tensor var_4861_split_sizes_0 = const()[name = string("op_4861_split_sizes_0"), val = tensor([64, 64])]; int32 var_4861_axis_0 = const()[name = string("op_4861_axis_0"), val = int32(-2)]; tensor var_4861_cast_fp16_0, tensor var_4861_cast_fp16_1 = split(axis = var_4861_axis_0, split_sizes = var_4861_split_sizes_0, x = var_4836_cast_fp16)[name = string("op_4861_cast_fp16")]; fp16 const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4863_cast_fp16 = mul(x = var_4861_cast_fp16_1, y = const_123_promoted_to_fp16)[name = string("op_4863_cast_fp16")]; int32 var_4865 = const()[name = string("op_4865"), val = int32(-2)]; bool var_4866_interleave_0 = const()[name = string("op_4866_interleave_0"), val = bool(false)]; tensor var_4866_cast_fp16 = concat(axis = var_4865, interleave = var_4866_interleave_0, values = (var_4863_cast_fp16, var_4861_cast_fp16_0))[name = string("op_4866_cast_fp16")]; tensor var_4867_cast_fp16 = mul(x = var_4866_cast_fp16, y = var_878_cast_fp16)[name = string("op_4867_cast_fp16")]; tensor key_states_125_cast_fp16 = add(x = var_4860_cast_fp16, y = var_4867_cast_fp16)[name = string("key_states_125_cast_fp16")]; tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; int32 concat_149_axis_0 = const()[name = string("concat_149_axis_0"), val = int32(0)]; bool concat_149_interleave_0 = const()[name = string("concat_149_interleave_0"), val = bool(false)]; tensor concat_149 = concat(axis = concat_149_axis_0, interleave = concat_149_interleave_0, values = (expand_dims_144, expand_dims_145, position_id, expand_dims_147))[name = string("concat_149")]; tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; tensor concat_150_values1_0 = const()[name = string("concat_150_values1_0"), val = tensor([0])]; tensor concat_150_values3_0 = const()[name = string("concat_150_values3_0"), val = tensor([0])]; int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_148, concat_150_values1_0, cache_position_end, concat_150_values3_0))[name = string("concat_150")]; tensor key_states_127_perm_0 = const()[name = string("key_states_127_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_13_stride_0 = const()[name = string("key_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_127_cast_fp16 = transpose(perm = key_states_127_perm_0, x = key_states_125_cast_fp16)[name = string("transpose_305")]; tensor key_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = key_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = key_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_13_squeeze_mask_0, stride = key_cache_internal_tensor_assign_13_stride_0, update = key_states_127_cast_fp16, x = coreml_update_state_190)[name = string("key_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_13_cast_fp16, input = key_cache)[name = string("coreml_update_state_192_write_state")]; tensor coreml_update_state_192 = read_state(input = key_cache)[name = string("coreml_update_state_192")]; tensor value_states_75_perm_0 = const()[name = string("value_states_75_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_13_stride_0 = const()[name = string("value_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_75_cast_fp16 = transpose(perm = value_states_75_perm_0, x = var_4843_cast_fp16)[name = string("transpose_304")]; tensor value_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = value_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = value_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_13_squeeze_mask_0, stride = value_cache_internal_tensor_assign_13_stride_0, update = value_states_75_cast_fp16, x = coreml_update_state_191)[name = string("value_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_13_cast_fp16, input = value_cache)[name = string("coreml_update_state_193_write_state")]; tensor coreml_update_state_193 = read_state(input = value_cache)[name = string("coreml_update_state_193")]; tensor var_4937_begin_0 = const()[name = string("op_4937_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4937_end_0 = const()[name = string("op_4937_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4937_end_mask_0 = const()[name = string("op_4937_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4937_cast_fp16 = slice_by_index(begin = var_4937_begin_0, end = var_4937_end_0, end_mask = var_4937_end_mask_0, x = coreml_update_state_192)[name = string("op_4937_cast_fp16")]; tensor tile_24 = const()[name = string("tile_24"), val = tensor([1, 1])]; int32 var_4940_axis_0 = const()[name = string("op_4940_axis_0"), val = int32(1)]; tensor var_4940_cast_fp16_0, tensor var_4940_cast_fp16_1 = split(axis = var_4940_axis_0, split_sizes = tile_24, x = var_4937_cast_fp16)[name = string("op_4940_cast_fp16")]; tensor var_4947_begin_0 = const()[name = string("op_4947_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4947_end_0 = const()[name = string("op_4947_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4947_end_mask_0 = const()[name = string("op_4947_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4947_cast_fp16 = slice_by_index(begin = var_4947_begin_0, end = var_4947_end_0, end_mask = var_4947_end_mask_0, x = coreml_update_state_193)[name = string("op_4947_cast_fp16")]; tensor tile_25 = const()[name = string("tile_25"), val = tensor([1, 1])]; int32 var_4950_axis_0 = const()[name = string("op_4950_axis_0"), val = int32(1)]; tensor var_4950_cast_fp16_0, tensor var_4950_cast_fp16_1 = split(axis = var_4950_axis_0, split_sizes = tile_25, x = var_4947_cast_fp16)[name = string("op_4950_cast_fp16")]; tensor var_4953_split_sizes_0 = const()[name = string("op_4953_split_sizes_0"), val = tensor([8, 8])]; int32 var_4953_axis_0 = const()[name = string("op_4953_axis_0"), val = int32(1)]; tensor var_4953_0, tensor var_4953_1 = split(axis = var_4953_axis_0, split_sizes = var_4953_split_sizes_0, x = query_states_75_cast_fp16)[name = string("op_4953")]; bool attn_weights_193_transpose_x_0 = const()[name = string("attn_weights_193_transpose_x_0"), val = bool(false)]; bool attn_weights_193_transpose_y_0 = const()[name = string("attn_weights_193_transpose_y_0"), val = bool(false)]; tensor attn_weights_193_cast_fp16 = matmul(transpose_x = attn_weights_193_transpose_x_0, transpose_y = attn_weights_193_transpose_y_0, x = var_4940_cast_fp16_0, y = var_4953_0)[name = string("attn_weights_193_cast_fp16")]; fp16 var_4956_to_fp16 = const()[name = string("op_4956_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_195_cast_fp16 = mul(x = attn_weights_193_cast_fp16, y = var_4956_to_fp16)[name = string("attn_weights_195_cast_fp16")]; tensor attn_weights_197_cast_fp16 = add(x = attn_weights_195_cast_fp16, y = attn_mask_1)[name = string("attn_weights_197_cast_fp16")]; int32 var_4960 = const()[name = string("op_4960"), val = int32(-2)]; tensor attn_weights_199_cast_fp16 = softmax(axis = var_4960, x = attn_weights_197_cast_fp16)[name = string("attn_weights_199_cast_fp16")]; bool var_4966_transpose_x_1 = const()[name = string("op_4966_transpose_x_1"), val = bool(true)]; bool var_4966_transpose_y_1 = const()[name = string("op_4966_transpose_y_1"), val = bool(false)]; tensor var_4966_cast_fp16 = matmul(transpose_x = var_4966_transpose_x_1, transpose_y = var_4966_transpose_y_1, x = attn_weights_199_cast_fp16, y = var_4950_cast_fp16_0)[name = string("op_4966_cast_fp16")]; bool attn_weights_201_transpose_x_0 = const()[name = string("attn_weights_201_transpose_x_0"), val = bool(false)]; bool attn_weights_201_transpose_y_0 = const()[name = string("attn_weights_201_transpose_y_0"), val = bool(false)]; tensor attn_weights_201_cast_fp16 = matmul(transpose_x = attn_weights_201_transpose_x_0, transpose_y = attn_weights_201_transpose_y_0, x = var_4940_cast_fp16_1, y = var_4953_1)[name = string("attn_weights_201_cast_fp16")]; fp16 var_4968_to_fp16 = const()[name = string("op_4968_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_203_cast_fp16 = mul(x = attn_weights_201_cast_fp16, y = var_4968_to_fp16)[name = string("attn_weights_203_cast_fp16")]; tensor attn_weights_205_cast_fp16 = add(x = attn_weights_203_cast_fp16, y = attn_mask_1)[name = string("attn_weights_205_cast_fp16")]; int32 var_4972 = const()[name = string("op_4972"), val = int32(-2)]; tensor attn_weights_207_cast_fp16 = softmax(axis = var_4972, x = attn_weights_205_cast_fp16)[name = string("attn_weights_207_cast_fp16")]; bool attn_output_97_transpose_x_1 = const()[name = string("attn_output_97_transpose_x_1"), val = bool(true)]; bool attn_output_97_transpose_y_1 = const()[name = string("attn_output_97_transpose_y_1"), val = bool(false)]; tensor attn_output_97_cast_fp16 = matmul(transpose_x = attn_output_97_transpose_x_1, transpose_y = attn_output_97_transpose_y_1, x = attn_weights_207_cast_fp16, y = var_4950_cast_fp16_1)[name = string("attn_output_97_cast_fp16")]; int32 var_4980 = const()[name = string("op_4980"), val = int32(1)]; bool attn_output_99_interleave_0 = const()[name = string("attn_output_99_interleave_0"), val = bool(false)]; tensor attn_output_99_cast_fp16 = concat(axis = var_4980, interleave = attn_output_99_interleave_0, values = (var_4966_cast_fp16, attn_output_97_cast_fp16))[name = string("attn_output_99_cast_fp16")]; tensor var_4984_perm_0 = const()[name = string("op_4984_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_155x = const()[name = string("concat_155x"), val = tensor([1, 2048, 1, -1])]; tensor var_4984_cast_fp16 = transpose(perm = var_4984_perm_0, x = attn_output_99_cast_fp16)[name = string("transpose_303")]; tensor attn_output_103_cast_fp16 = reshape(shape = concat_155x, x = var_4984_cast_fp16)[name = string("attn_output_103_cast_fp16")]; tensor hidden_states_123_strides_0 = const()[name = string("hidden_states_123_strides_0"), val = tensor([1, 1])]; string hidden_states_123_pad_type_0 = const()[name = string("hidden_states_123_pad_type_0"), val = string("valid")]; tensor hidden_states_123_pad_0 = const()[name = string("hidden_states_123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_123_dilations_0 = const()[name = string("hidden_states_123_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_123_groups_0 = const()[name = string("hidden_states_123_groups_0"), val = int32(1)]; tensor hidden_states_123_cast_fp16 = conv(dilations = hidden_states_123_dilations_0, groups = hidden_states_123_groups_0, pad = hidden_states_123_pad_0, pad_type = hidden_states_123_pad_type_0, strides = hidden_states_123_strides_0, weight = layers_12_self_attn_o_proj_weight_cast_fp16, x = attn_output_103_cast_fp16)[name = string("hidden_states_123_cast_fp16")]; tensor hidden_states_125_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = hidden_states_123_cast_fp16)[name = string("hidden_states_125_cast_fp16")]; fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5017_cast_fp16 = mul(x = hidden_states_125_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_5017_cast_fp16")]; int32 var_5015 = const()[name = string("op_5015"), val = int32(1)]; bool doubled_101_interleave_0 = const()[name = string("doubled_101_interleave_0"), val = bool(false)]; tensor doubled_101_cast_fp16 = concat(axis = var_5015, interleave = doubled_101_interleave_0, values = (hidden_states_125_cast_fp16, var_5017_cast_fp16))[name = string("doubled_101_cast_fp16")]; tensor out_51_axes_0 = const()[name = string("out_51_axes_0"), val = tensor([1])]; tensor out_51_gamma_0_to_fp16 = const()[name = string("out_51_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415382912)))]; fp16 var_5027_to_fp16 = const()[name = string("op_5027_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_51_cast_fp16 = layer_norm(axes = out_51_axes_0, epsilon = var_5027_to_fp16, gamma = out_51_gamma_0_to_fp16, x = doubled_101_cast_fp16)[name = string("out_51_cast_fp16")]; tensor var_5038_split_sizes_0 = const()[name = string("op_5038_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5038_axis_0 = const()[name = string("op_5038_axis_0"), val = int32(1)]; tensor var_5038_cast_fp16_0, tensor var_5038_cast_fp16_1 = split(axis = var_5038_axis_0, split_sizes = var_5038_split_sizes_0, x = out_51_cast_fp16)[name = string("op_5038_cast_fp16")]; tensor input_25_strides_0 = const()[name = string("input_25_strides_0"), val = tensor([1, 1])]; string input_25_pad_type_0 = const()[name = string("input_25_pad_type_0"), val = string("valid")]; tensor input_25_pad_0 = const()[name = string("input_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_25_dilations_0 = const()[name = string("input_25_dilations_0"), val = tensor([1, 1])]; int32 input_25_groups_0 = const()[name = string("input_25_groups_0"), val = int32(1)]; tensor input_25_cast_fp16 = conv(dilations = input_25_dilations_0, groups = input_25_groups_0, pad = input_25_pad_0, pad_type = input_25_pad_type_0, strides = input_25_strides_0, weight = layers_12_mlp_gate_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("input_25_cast_fp16")]; tensor var_5055_cast_fp16 = silu(x = input_25_cast_fp16)[name = string("op_5055_cast_fp16")]; tensor var_5061_strides_0 = const()[name = string("op_5061_strides_0"), val = tensor([1, 1])]; string var_5061_pad_type_0 = const()[name = string("op_5061_pad_type_0"), val = string("valid")]; tensor var_5061_pad_0 = const()[name = string("op_5061_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5061_dilations_0 = const()[name = string("op_5061_dilations_0"), val = tensor([1, 1])]; int32 var_5061_groups_0 = const()[name = string("op_5061_groups_0"), val = int32(1)]; tensor var_5061_cast_fp16 = conv(dilations = var_5061_dilations_0, groups = var_5061_groups_0, pad = var_5061_pad_0, pad_type = var_5061_pad_type_0, strides = var_5061_strides_0, weight = layers_12_mlp_up_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("op_5061_cast_fp16")]; tensor x_129_cast_fp16 = mul(x = var_5055_cast_fp16, y = var_5061_cast_fp16)[name = string("x_129_cast_fp16")]; tensor hidden_states_127_strides_0 = const()[name = string("hidden_states_127_strides_0"), val = tensor([1, 1])]; string hidden_states_127_pad_type_0 = const()[name = string("hidden_states_127_pad_type_0"), val = string("valid")]; tensor hidden_states_127_pad_0 = const()[name = string("hidden_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_127_dilations_0 = const()[name = string("hidden_states_127_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_127_groups_0 = const()[name = string("hidden_states_127_groups_0"), val = int32(1)]; tensor hidden_states_127_cast_fp16 = conv(dilations = hidden_states_127_dilations_0, groups = hidden_states_127_groups_0, pad = hidden_states_127_pad_0, pad_type = hidden_states_127_pad_type_0, strides = hidden_states_127_strides_0, weight = layers_12_mlp_down_proj_weight_cast_fp16, x = x_129_cast_fp16)[name = string("hidden_states_127_cast_fp16")]; tensor hidden_states_129_cast_fp16 = add(x = hidden_states_125_cast_fp16, y = hidden_states_127_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5079_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_130_promoted_to_fp16)[name = string("op_5079_cast_fp16")]; int32 var_5077 = const()[name = string("op_5077"), val = int32(1)]; bool doubled_105_interleave_0 = const()[name = string("doubled_105_interleave_0"), val = bool(false)]; tensor doubled_105_cast_fp16 = concat(axis = var_5077, interleave = doubled_105_interleave_0, values = (hidden_states_129_cast_fp16, var_5079_cast_fp16))[name = string("doubled_105_cast_fp16")]; tensor out_53_axes_0 = const()[name = string("out_53_axes_0"), val = tensor([1])]; tensor out_53_gamma_0_to_fp16 = const()[name = string("out_53_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415391168)))]; fp16 var_5089_to_fp16 = const()[name = string("op_5089_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_53_cast_fp16 = layer_norm(axes = out_53_axes_0, epsilon = var_5089_to_fp16, gamma = out_53_gamma_0_to_fp16, x = doubled_105_cast_fp16)[name = string("out_53_cast_fp16")]; tensor var_5100_split_sizes_0 = const()[name = string("op_5100_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5100_axis_0 = const()[name = string("op_5100_axis_0"), val = int32(1)]; tensor var_5100_cast_fp16_0, tensor var_5100_cast_fp16_1 = split(axis = var_5100_axis_0, split_sizes = var_5100_split_sizes_0, x = out_53_cast_fp16)[name = string("op_5100_cast_fp16")]; tensor query_states_79_strides_0 = const()[name = string("query_states_79_strides_0"), val = tensor([1, 1])]; string query_states_79_pad_type_0 = const()[name = string("query_states_79_pad_type_0"), val = string("valid")]; tensor query_states_79_pad_0 = const()[name = string("query_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_79_dilations_0 = const()[name = string("query_states_79_dilations_0"), val = tensor([1, 1])]; int32 query_states_79_groups_0 = const()[name = string("query_states_79_groups_0"), val = int32(1)]; tensor query_states_79_cast_fp16 = conv(dilations = query_states_79_dilations_0, groups = query_states_79_groups_0, pad = query_states_79_pad_0, pad_type = query_states_79_pad_type_0, strides = query_states_79_strides_0, weight = layers_13_self_attn_q_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("query_states_79_cast_fp16")]; tensor key_states_131_strides_0 = const()[name = string("key_states_131_strides_0"), val = tensor([1, 1])]; string key_states_131_pad_type_0 = const()[name = string("key_states_131_pad_type_0"), val = string("valid")]; tensor key_states_131_pad_0 = const()[name = string("key_states_131_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_131_dilations_0 = const()[name = string("key_states_131_dilations_0"), val = tensor([1, 1])]; int32 key_states_131_groups_0 = const()[name = string("key_states_131_groups_0"), val = int32(1)]; tensor key_states_131_cast_fp16 = conv(dilations = key_states_131_dilations_0, groups = key_states_131_groups_0, pad = key_states_131_pad_0, pad_type = key_states_131_pad_type_0, strides = key_states_131_strides_0, weight = layers_13_self_attn_k_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("key_states_131_cast_fp16")]; tensor value_states_79_strides_0 = const()[name = string("value_states_79_strides_0"), val = tensor([1, 1])]; string value_states_79_pad_type_0 = const()[name = string("value_states_79_pad_type_0"), val = string("valid")]; tensor value_states_79_pad_0 = const()[name = string("value_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_79_dilations_0 = const()[name = string("value_states_79_dilations_0"), val = tensor([1, 1])]; int32 value_states_79_groups_0 = const()[name = string("value_states_79_groups_0"), val = int32(1)]; tensor value_states_79_cast_fp16 = conv(dilations = value_states_79_dilations_0, groups = value_states_79_groups_0, pad = value_states_79_pad_0, pad_type = value_states_79_pad_type_0, strides = value_states_79_strides_0, weight = layers_13_self_attn_v_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("value_states_79_cast_fp16")]; tensor concat_156x = const()[name = string("concat_156x"), val = tensor([1, 16, 128, -1])]; tensor x_131_cast_fp16 = reshape(shape = concat_156x, x = query_states_79_cast_fp16)[name = string("x_131_cast_fp16")]; tensor concat_157x = const()[name = string("concat_157x"), val = tensor([1, 2, 128, -1])]; tensor var_5157_cast_fp16 = reshape(shape = concat_157x, x = key_states_131_cast_fp16)[name = string("op_5157_cast_fp16")]; tensor concat_158x = const()[name = string("concat_158x"), val = tensor([1, 2, 128, -1])]; tensor var_5164_cast_fp16 = reshape(shape = concat_158x, x = value_states_79_cast_fp16)[name = string("op_5164_cast_fp16")]; tensor var_5168_cast_fp16 = mul(x = x_131_cast_fp16, y = var_869_cast_fp16)[name = string("op_5168_cast_fp16")]; tensor var_5169_split_sizes_0 = const()[name = string("op_5169_split_sizes_0"), val = tensor([64, 64])]; int32 var_5169_axis_0 = const()[name = string("op_5169_axis_0"), val = int32(-2)]; tensor var_5169_cast_fp16_0, tensor var_5169_cast_fp16_1 = split(axis = var_5169_axis_0, split_sizes = var_5169_split_sizes_0, x = x_131_cast_fp16)[name = string("op_5169_cast_fp16")]; fp16 const_132_promoted_to_fp16 = const()[name = string("const_132_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5171_cast_fp16 = mul(x = var_5169_cast_fp16_1, y = const_132_promoted_to_fp16)[name = string("op_5171_cast_fp16")]; int32 var_5173 = const()[name = string("op_5173"), val = int32(-2)]; bool var_5174_interleave_0 = const()[name = string("op_5174_interleave_0"), val = bool(false)]; tensor var_5174_cast_fp16 = concat(axis = var_5173, interleave = var_5174_interleave_0, values = (var_5171_cast_fp16, var_5169_cast_fp16_0))[name = string("op_5174_cast_fp16")]; tensor var_5175_cast_fp16 = mul(x = var_5174_cast_fp16, y = var_878_cast_fp16)[name = string("op_5175_cast_fp16")]; tensor query_states_81_cast_fp16 = add(x = var_5168_cast_fp16, y = var_5175_cast_fp16)[name = string("query_states_81_cast_fp16")]; tensor var_5181_cast_fp16 = mul(x = var_5157_cast_fp16, y = var_869_cast_fp16)[name = string("op_5181_cast_fp16")]; tensor var_5182_split_sizes_0 = const()[name = string("op_5182_split_sizes_0"), val = tensor([64, 64])]; int32 var_5182_axis_0 = const()[name = string("op_5182_axis_0"), val = int32(-2)]; tensor var_5182_cast_fp16_0, tensor var_5182_cast_fp16_1 = split(axis = var_5182_axis_0, split_sizes = var_5182_split_sizes_0, x = var_5157_cast_fp16)[name = string("op_5182_cast_fp16")]; fp16 const_133_promoted_to_fp16 = const()[name = string("const_133_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5184_cast_fp16 = mul(x = var_5182_cast_fp16_1, y = const_133_promoted_to_fp16)[name = string("op_5184_cast_fp16")]; int32 var_5186 = const()[name = string("op_5186"), val = int32(-2)]; bool var_5187_interleave_0 = const()[name = string("op_5187_interleave_0"), val = bool(false)]; tensor var_5187_cast_fp16 = concat(axis = var_5186, interleave = var_5187_interleave_0, values = (var_5184_cast_fp16, var_5182_cast_fp16_0))[name = string("op_5187_cast_fp16")]; tensor var_5188_cast_fp16 = mul(x = var_5187_cast_fp16, y = var_878_cast_fp16)[name = string("op_5188_cast_fp16")]; tensor key_states_135_cast_fp16 = add(x = var_5181_cast_fp16, y = var_5188_cast_fp16)[name = string("key_states_135_cast_fp16")]; tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; int32 concat_161_axis_0 = const()[name = string("concat_161_axis_0"), val = int32(0)]; bool concat_161_interleave_0 = const()[name = string("concat_161_interleave_0"), val = bool(false)]; tensor concat_161 = concat(axis = concat_161_axis_0, interleave = concat_161_interleave_0, values = (expand_dims_156, expand_dims_157, position_id, expand_dims_159))[name = string("concat_161")]; tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; tensor concat_162_values1_0 = const()[name = string("concat_162_values1_0"), val = tensor([0])]; tensor concat_162_values3_0 = const()[name = string("concat_162_values3_0"), val = tensor([0])]; int32 concat_162_axis_0 = const()[name = string("concat_162_axis_0"), val = int32(0)]; bool concat_162_interleave_0 = const()[name = string("concat_162_interleave_0"), val = bool(false)]; tensor concat_162 = concat(axis = concat_162_axis_0, interleave = concat_162_interleave_0, values = (expand_dims_160, concat_162_values1_0, cache_position_end, concat_162_values3_0))[name = string("concat_162")]; tensor key_states_137_perm_0 = const()[name = string("key_states_137_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_14_stride_0 = const()[name = string("key_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_137_cast_fp16 = transpose(perm = key_states_137_perm_0, x = key_states_135_cast_fp16)[name = string("transpose_302")]; tensor key_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = key_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = key_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_14_squeeze_mask_0, stride = key_cache_internal_tensor_assign_14_stride_0, update = key_states_137_cast_fp16, x = coreml_update_state_192)[name = string("key_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_14_cast_fp16, input = key_cache)[name = string("coreml_update_state_194_write_state")]; tensor coreml_update_state_194 = read_state(input = key_cache)[name = string("coreml_update_state_194")]; tensor value_states_81_perm_0 = const()[name = string("value_states_81_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_14_stride_0 = const()[name = string("value_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_81_cast_fp16 = transpose(perm = value_states_81_perm_0, x = var_5164_cast_fp16)[name = string("transpose_301")]; tensor value_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = value_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = value_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_14_squeeze_mask_0, stride = value_cache_internal_tensor_assign_14_stride_0, update = value_states_81_cast_fp16, x = coreml_update_state_193)[name = string("value_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_14_cast_fp16, input = value_cache)[name = string("coreml_update_state_195_write_state")]; tensor coreml_update_state_195 = read_state(input = value_cache)[name = string("coreml_update_state_195")]; tensor var_5258_begin_0 = const()[name = string("op_5258_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5258_end_0 = const()[name = string("op_5258_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5258_end_mask_0 = const()[name = string("op_5258_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5258_cast_fp16 = slice_by_index(begin = var_5258_begin_0, end = var_5258_end_0, end_mask = var_5258_end_mask_0, x = coreml_update_state_194)[name = string("op_5258_cast_fp16")]; tensor tile_26 = const()[name = string("tile_26"), val = tensor([1, 1])]; int32 var_5261_axis_0 = const()[name = string("op_5261_axis_0"), val = int32(1)]; tensor var_5261_cast_fp16_0, tensor var_5261_cast_fp16_1 = split(axis = var_5261_axis_0, split_sizes = tile_26, x = var_5258_cast_fp16)[name = string("op_5261_cast_fp16")]; tensor var_5268_begin_0 = const()[name = string("op_5268_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5268_end_0 = const()[name = string("op_5268_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5268_end_mask_0 = const()[name = string("op_5268_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5268_cast_fp16 = slice_by_index(begin = var_5268_begin_0, end = var_5268_end_0, end_mask = var_5268_end_mask_0, x = coreml_update_state_195)[name = string("op_5268_cast_fp16")]; tensor tile_27 = const()[name = string("tile_27"), val = tensor([1, 1])]; int32 var_5271_axis_0 = const()[name = string("op_5271_axis_0"), val = int32(1)]; tensor var_5271_cast_fp16_0, tensor var_5271_cast_fp16_1 = split(axis = var_5271_axis_0, split_sizes = tile_27, x = var_5268_cast_fp16)[name = string("op_5271_cast_fp16")]; tensor var_5274_split_sizes_0 = const()[name = string("op_5274_split_sizes_0"), val = tensor([8, 8])]; int32 var_5274_axis_0 = const()[name = string("op_5274_axis_0"), val = int32(1)]; tensor var_5274_0, tensor var_5274_1 = split(axis = var_5274_axis_0, split_sizes = var_5274_split_sizes_0, x = query_states_81_cast_fp16)[name = string("op_5274")]; bool attn_weights_209_transpose_x_0 = const()[name = string("attn_weights_209_transpose_x_0"), val = bool(false)]; bool attn_weights_209_transpose_y_0 = const()[name = string("attn_weights_209_transpose_y_0"), val = bool(false)]; tensor attn_weights_209_cast_fp16 = matmul(transpose_x = attn_weights_209_transpose_x_0, transpose_y = attn_weights_209_transpose_y_0, x = var_5261_cast_fp16_0, y = var_5274_0)[name = string("attn_weights_209_cast_fp16")]; fp16 var_5277_to_fp16 = const()[name = string("op_5277_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_211_cast_fp16 = mul(x = attn_weights_209_cast_fp16, y = var_5277_to_fp16)[name = string("attn_weights_211_cast_fp16")]; tensor attn_weights_213_cast_fp16 = add(x = attn_weights_211_cast_fp16, y = attn_mask_1)[name = string("attn_weights_213_cast_fp16")]; int32 var_5281 = const()[name = string("op_5281"), val = int32(-2)]; tensor attn_weights_215_cast_fp16 = softmax(axis = var_5281, x = attn_weights_213_cast_fp16)[name = string("attn_weights_215_cast_fp16")]; bool var_5287_transpose_x_1 = const()[name = string("op_5287_transpose_x_1"), val = bool(true)]; bool var_5287_transpose_y_1 = const()[name = string("op_5287_transpose_y_1"), val = bool(false)]; tensor var_5287_cast_fp16 = matmul(transpose_x = var_5287_transpose_x_1, transpose_y = var_5287_transpose_y_1, x = attn_weights_215_cast_fp16, y = var_5271_cast_fp16_0)[name = string("op_5287_cast_fp16")]; bool attn_weights_217_transpose_x_0 = const()[name = string("attn_weights_217_transpose_x_0"), val = bool(false)]; bool attn_weights_217_transpose_y_0 = const()[name = string("attn_weights_217_transpose_y_0"), val = bool(false)]; tensor attn_weights_217_cast_fp16 = matmul(transpose_x = attn_weights_217_transpose_x_0, transpose_y = attn_weights_217_transpose_y_0, x = var_5261_cast_fp16_1, y = var_5274_1)[name = string("attn_weights_217_cast_fp16")]; fp16 var_5289_to_fp16 = const()[name = string("op_5289_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_219_cast_fp16 = mul(x = attn_weights_217_cast_fp16, y = var_5289_to_fp16)[name = string("attn_weights_219_cast_fp16")]; tensor attn_weights_221_cast_fp16 = add(x = attn_weights_219_cast_fp16, y = attn_mask_1)[name = string("attn_weights_221_cast_fp16")]; int32 var_5293 = const()[name = string("op_5293"), val = int32(-2)]; tensor attn_weights_223_cast_fp16 = softmax(axis = var_5293, x = attn_weights_221_cast_fp16)[name = string("attn_weights_223_cast_fp16")]; bool attn_output_105_transpose_x_1 = const()[name = string("attn_output_105_transpose_x_1"), val = bool(true)]; bool attn_output_105_transpose_y_1 = const()[name = string("attn_output_105_transpose_y_1"), val = bool(false)]; tensor attn_output_105_cast_fp16 = matmul(transpose_x = attn_output_105_transpose_x_1, transpose_y = attn_output_105_transpose_y_1, x = attn_weights_223_cast_fp16, y = var_5271_cast_fp16_1)[name = string("attn_output_105_cast_fp16")]; int32 var_5301 = const()[name = string("op_5301"), val = int32(1)]; bool attn_output_107_interleave_0 = const()[name = string("attn_output_107_interleave_0"), val = bool(false)]; tensor attn_output_107_cast_fp16 = concat(axis = var_5301, interleave = attn_output_107_interleave_0, values = (var_5287_cast_fp16, attn_output_105_cast_fp16))[name = string("attn_output_107_cast_fp16")]; tensor var_5305_perm_0 = const()[name = string("op_5305_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_167x = const()[name = string("concat_167x"), val = tensor([1, 2048, 1, -1])]; tensor var_5305_cast_fp16 = transpose(perm = var_5305_perm_0, x = attn_output_107_cast_fp16)[name = string("transpose_300")]; tensor attn_output_111_cast_fp16 = reshape(shape = concat_167x, x = var_5305_cast_fp16)[name = string("attn_output_111_cast_fp16")]; tensor hidden_states_133_strides_0 = const()[name = string("hidden_states_133_strides_0"), val = tensor([1, 1])]; string hidden_states_133_pad_type_0 = const()[name = string("hidden_states_133_pad_type_0"), val = string("valid")]; tensor hidden_states_133_pad_0 = const()[name = string("hidden_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_133_dilations_0 = const()[name = string("hidden_states_133_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_133_groups_0 = const()[name = string("hidden_states_133_groups_0"), val = int32(1)]; tensor hidden_states_133_cast_fp16 = conv(dilations = hidden_states_133_dilations_0, groups = hidden_states_133_groups_0, pad = hidden_states_133_pad_0, pad_type = hidden_states_133_pad_type_0, strides = hidden_states_133_strides_0, weight = layers_13_self_attn_o_proj_weight_cast_fp16, x = attn_output_111_cast_fp16)[name = string("hidden_states_133_cast_fp16")]; tensor hidden_states_135_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = hidden_states_133_cast_fp16)[name = string("hidden_states_135_cast_fp16")]; fp16 const_138_promoted_to_fp16 = const()[name = string("const_138_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5338_cast_fp16 = mul(x = hidden_states_135_cast_fp16, y = const_138_promoted_to_fp16)[name = string("op_5338_cast_fp16")]; int32 var_5336 = const()[name = string("op_5336"), val = int32(1)]; bool doubled_109_interleave_0 = const()[name = string("doubled_109_interleave_0"), val = bool(false)]; tensor doubled_109_cast_fp16 = concat(axis = var_5336, interleave = doubled_109_interleave_0, values = (hidden_states_135_cast_fp16, var_5338_cast_fp16))[name = string("doubled_109_cast_fp16")]; tensor out_55_axes_0 = const()[name = string("out_55_axes_0"), val = tensor([1])]; tensor out_55_gamma_0_to_fp16 = const()[name = string("out_55_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415399424)))]; fp16 var_5348_to_fp16 = const()[name = string("op_5348_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_55_cast_fp16 = layer_norm(axes = out_55_axes_0, epsilon = var_5348_to_fp16, gamma = out_55_gamma_0_to_fp16, x = doubled_109_cast_fp16)[name = string("out_55_cast_fp16")]; tensor var_5359_split_sizes_0 = const()[name = string("op_5359_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5359_axis_0 = const()[name = string("op_5359_axis_0"), val = int32(1)]; tensor var_5359_cast_fp16_0, tensor var_5359_cast_fp16_1 = split(axis = var_5359_axis_0, split_sizes = var_5359_split_sizes_0, x = out_55_cast_fp16)[name = string("op_5359_cast_fp16")]; tensor input_27_strides_0 = const()[name = string("input_27_strides_0"), val = tensor([1, 1])]; string input_27_pad_type_0 = const()[name = string("input_27_pad_type_0"), val = string("valid")]; tensor input_27_pad_0 = const()[name = string("input_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_27_dilations_0 = const()[name = string("input_27_dilations_0"), val = tensor([1, 1])]; int32 input_27_groups_0 = const()[name = string("input_27_groups_0"), val = int32(1)]; tensor input_27_cast_fp16 = conv(dilations = input_27_dilations_0, groups = input_27_groups_0, pad = input_27_pad_0, pad_type = input_27_pad_type_0, strides = input_27_strides_0, weight = layers_13_mlp_gate_proj_weight_cast_fp16, x = var_5359_cast_fp16_0)[name = string("input_27_cast_fp16")]; tensor var_5376_cast_fp16 = silu(x = input_27_cast_fp16)[name = string("op_5376_cast_fp16")]; tensor layers_13_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_13_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415407680)))]; tensor var_5382_strides_0 = const()[name = string("op_5382_strides_0"), val = tensor([1, 1])]; string var_5382_pad_type_0 = const()[name = string("op_5382_pad_type_0"), val = string("valid")]; tensor var_5382_pad_0 = const()[name = string("op_5382_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5382_dilations_0 = const()[name = string("op_5382_dilations_0"), val = tensor([1, 1])]; int32 var_5382_groups_0 = const()[name = string("op_5382_groups_0"), val = int32(1)]; tensor var_5382_cast_fp16 = conv(dilations = var_5382_dilations_0, groups = var_5382_groups_0, pad = var_5382_pad_0, pad_type = var_5382_pad_type_0, strides = var_5382_strides_0, weight = layers_13_mlp_up_proj_weight_to_fp16, x = var_5359_cast_fp16_0)[name = string("op_5382_cast_fp16")]; tensor x_139_cast_fp16 = mul(x = var_5376_cast_fp16, y = var_5382_cast_fp16)[name = string("x_139_cast_fp16")]; tensor hidden_states_137_strides_0 = const()[name = string("hidden_states_137_strides_0"), val = tensor([1, 1])]; string hidden_states_137_pad_type_0 = const()[name = string("hidden_states_137_pad_type_0"), val = string("valid")]; tensor hidden_states_137_pad_0 = const()[name = string("hidden_states_137_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_137_dilations_0 = const()[name = string("hidden_states_137_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_137_groups_0 = const()[name = string("hidden_states_137_groups_0"), val = int32(1)]; tensor hidden_states_137_cast_fp16 = conv(dilations = hidden_states_137_dilations_0, groups = hidden_states_137_groups_0, pad = hidden_states_137_pad_0, pad_type = hidden_states_137_pad_type_0, strides = hidden_states_137_strides_0, weight = layers_13_mlp_down_proj_weight_cast_fp16, x = x_139_cast_fp16)[name = string("hidden_states_137_cast_fp16")]; tensor hidden_states_139_cast_fp16 = add(x = hidden_states_135_cast_fp16, y = hidden_states_137_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; fp16 const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5400_cast_fp16 = mul(x = hidden_states_139_cast_fp16, y = const_140_promoted_to_fp16)[name = string("op_5400_cast_fp16")]; int32 var_5398 = const()[name = string("op_5398"), val = int32(1)]; bool doubled_113_interleave_0 = const()[name = string("doubled_113_interleave_0"), val = bool(false)]; tensor doubled_113_cast_fp16 = concat(axis = var_5398, interleave = doubled_113_interleave_0, values = (hidden_states_139_cast_fp16, var_5400_cast_fp16))[name = string("doubled_113_cast_fp16")]; tensor out_57_axes_0 = const()[name = string("out_57_axes_0"), val = tensor([1])]; tensor out_57_gamma_0_to_fp16 = const()[name = string("out_57_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440573568)))]; fp16 var_5410_to_fp16 = const()[name = string("op_5410_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_57_cast_fp16 = layer_norm(axes = out_57_axes_0, epsilon = var_5410_to_fp16, gamma = out_57_gamma_0_to_fp16, x = doubled_113_cast_fp16)[name = string("out_57_cast_fp16")]; tensor var_5421_split_sizes_0 = const()[name = string("op_5421_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5421_axis_0 = const()[name = string("op_5421_axis_0"), val = int32(1)]; tensor var_5421_cast_fp16_0, tensor var_5421_cast_fp16_1 = split(axis = var_5421_axis_0, split_sizes = var_5421_split_sizes_0, x = out_57_cast_fp16)[name = string("op_5421_cast_fp16")]; tensor query_states_85_strides_0 = const()[name = string("query_states_85_strides_0"), val = tensor([1, 1])]; string query_states_85_pad_type_0 = const()[name = string("query_states_85_pad_type_0"), val = string("valid")]; tensor query_states_85_pad_0 = const()[name = string("query_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_85_dilations_0 = const()[name = string("query_states_85_dilations_0"), val = tensor([1, 1])]; int32 query_states_85_groups_0 = const()[name = string("query_states_85_groups_0"), val = int32(1)]; tensor query_states_85_cast_fp16 = conv(dilations = query_states_85_dilations_0, groups = query_states_85_groups_0, pad = query_states_85_pad_0, pad_type = query_states_85_pad_type_0, strides = query_states_85_strides_0, weight = layers_14_self_attn_q_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("query_states_85_cast_fp16")]; tensor layers_14_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_14_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440581824)))]; tensor key_states_141_strides_0 = const()[name = string("key_states_141_strides_0"), val = tensor([1, 1])]; string key_states_141_pad_type_0 = const()[name = string("key_states_141_pad_type_0"), val = string("valid")]; tensor key_states_141_pad_0 = const()[name = string("key_states_141_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_141_dilations_0 = const()[name = string("key_states_141_dilations_0"), val = tensor([1, 1])]; int32 key_states_141_groups_0 = const()[name = string("key_states_141_groups_0"), val = int32(1)]; tensor key_states_141_cast_fp16 = conv(dilations = key_states_141_dilations_0, groups = key_states_141_groups_0, pad = key_states_141_pad_0, pad_type = key_states_141_pad_type_0, strides = key_states_141_strides_0, weight = layers_14_self_attn_k_proj_weight_to_fp16, x = var_5421_cast_fp16_0)[name = string("key_states_141_cast_fp16")]; tensor value_states_85_strides_0 = const()[name = string("value_states_85_strides_0"), val = tensor([1, 1])]; string value_states_85_pad_type_0 = const()[name = string("value_states_85_pad_type_0"), val = string("valid")]; tensor value_states_85_pad_0 = const()[name = string("value_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_85_dilations_0 = const()[name = string("value_states_85_dilations_0"), val = tensor([1, 1])]; int32 value_states_85_groups_0 = const()[name = string("value_states_85_groups_0"), val = int32(1)]; tensor value_states_85_cast_fp16 = conv(dilations = value_states_85_dilations_0, groups = value_states_85_groups_0, pad = value_states_85_pad_0, pad_type = value_states_85_pad_type_0, strides = value_states_85_strides_0, weight = layers_14_self_attn_v_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("value_states_85_cast_fp16")]; tensor concat_168x = const()[name = string("concat_168x"), val = tensor([1, 16, 128, -1])]; tensor x_141_cast_fp16 = reshape(shape = concat_168x, x = query_states_85_cast_fp16)[name = string("x_141_cast_fp16")]; tensor concat_169x = const()[name = string("concat_169x"), val = tensor([1, 2, 128, -1])]; tensor var_5478_cast_fp16 = reshape(shape = concat_169x, x = key_states_141_cast_fp16)[name = string("op_5478_cast_fp16")]; tensor concat_170x = const()[name = string("concat_170x"), val = tensor([1, 2, 128, -1])]; tensor var_5485_cast_fp16 = reshape(shape = concat_170x, x = value_states_85_cast_fp16)[name = string("op_5485_cast_fp16")]; tensor var_5489_cast_fp16 = mul(x = x_141_cast_fp16, y = var_869_cast_fp16)[name = string("op_5489_cast_fp16")]; tensor var_5490_split_sizes_0 = const()[name = string("op_5490_split_sizes_0"), val = tensor([64, 64])]; int32 var_5490_axis_0 = const()[name = string("op_5490_axis_0"), val = int32(-2)]; tensor var_5490_cast_fp16_0, tensor var_5490_cast_fp16_1 = split(axis = var_5490_axis_0, split_sizes = var_5490_split_sizes_0, x = x_141_cast_fp16)[name = string("op_5490_cast_fp16")]; fp16 const_142_promoted_to_fp16 = const()[name = string("const_142_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5492_cast_fp16 = mul(x = var_5490_cast_fp16_1, y = const_142_promoted_to_fp16)[name = string("op_5492_cast_fp16")]; int32 var_5494 = const()[name = string("op_5494"), val = int32(-2)]; bool var_5495_interleave_0 = const()[name = string("op_5495_interleave_0"), val = bool(false)]; tensor var_5495_cast_fp16 = concat(axis = var_5494, interleave = var_5495_interleave_0, values = (var_5492_cast_fp16, var_5490_cast_fp16_0))[name = string("op_5495_cast_fp16")]; tensor var_5496_cast_fp16 = mul(x = var_5495_cast_fp16, y = var_878_cast_fp16)[name = string("op_5496_cast_fp16")]; tensor query_states_87_cast_fp16 = add(x = var_5489_cast_fp16, y = var_5496_cast_fp16)[name = string("query_states_87_cast_fp16")]; tensor var_5502_cast_fp16 = mul(x = var_5478_cast_fp16, y = var_869_cast_fp16)[name = string("op_5502_cast_fp16")]; tensor var_5503_split_sizes_0 = const()[name = string("op_5503_split_sizes_0"), val = tensor([64, 64])]; int32 var_5503_axis_0 = const()[name = string("op_5503_axis_0"), val = int32(-2)]; tensor var_5503_cast_fp16_0, tensor var_5503_cast_fp16_1 = split(axis = var_5503_axis_0, split_sizes = var_5503_split_sizes_0, x = var_5478_cast_fp16)[name = string("op_5503_cast_fp16")]; fp16 const_143_promoted_to_fp16 = const()[name = string("const_143_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5505_cast_fp16 = mul(x = var_5503_cast_fp16_1, y = const_143_promoted_to_fp16)[name = string("op_5505_cast_fp16")]; int32 var_5507 = const()[name = string("op_5507"), val = int32(-2)]; bool var_5508_interleave_0 = const()[name = string("op_5508_interleave_0"), val = bool(false)]; tensor var_5508_cast_fp16 = concat(axis = var_5507, interleave = var_5508_interleave_0, values = (var_5505_cast_fp16, var_5503_cast_fp16_0))[name = string("op_5508_cast_fp16")]; tensor var_5509_cast_fp16 = mul(x = var_5508_cast_fp16, y = var_878_cast_fp16)[name = string("op_5509_cast_fp16")]; tensor key_states_145_cast_fp16 = add(x = var_5502_cast_fp16, y = var_5509_cast_fp16)[name = string("key_states_145_cast_fp16")]; tensor expand_dims_168 = const()[name = string("expand_dims_168"), val = tensor([14])]; tensor expand_dims_169 = const()[name = string("expand_dims_169"), val = tensor([0])]; tensor expand_dims_171 = const()[name = string("expand_dims_171"), val = tensor([0])]; int32 concat_173_axis_0 = const()[name = string("concat_173_axis_0"), val = int32(0)]; bool concat_173_interleave_0 = const()[name = string("concat_173_interleave_0"), val = bool(false)]; tensor concat_173 = concat(axis = concat_173_axis_0, interleave = concat_173_interleave_0, values = (expand_dims_168, expand_dims_169, position_id, expand_dims_171))[name = string("concat_173")]; tensor expand_dims_172 = const()[name = string("expand_dims_172"), val = tensor([15])]; tensor concat_174_values1_0 = const()[name = string("concat_174_values1_0"), val = tensor([0])]; tensor concat_174_values3_0 = const()[name = string("concat_174_values3_0"), val = tensor([0])]; int32 concat_174_axis_0 = const()[name = string("concat_174_axis_0"), val = int32(0)]; bool concat_174_interleave_0 = const()[name = string("concat_174_interleave_0"), val = bool(false)]; tensor concat_174 = concat(axis = concat_174_axis_0, interleave = concat_174_interleave_0, values = (expand_dims_172, concat_174_values1_0, cache_position_end, concat_174_values3_0))[name = string("concat_174")]; tensor key_states_147_perm_0 = const()[name = string("key_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_15_stride_0 = const()[name = string("key_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_147_cast_fp16 = transpose(perm = key_states_147_perm_0, x = key_states_145_cast_fp16)[name = string("transpose_299")]; tensor key_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = key_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = key_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_15_squeeze_mask_0, stride = key_cache_internal_tensor_assign_15_stride_0, update = key_states_147_cast_fp16, x = coreml_update_state_194)[name = string("key_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_15_cast_fp16, input = key_cache)[name = string("coreml_update_state_196_write_state")]; tensor coreml_update_state_196 = read_state(input = key_cache)[name = string("coreml_update_state_196")]; tensor value_states_87_perm_0 = const()[name = string("value_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_15_stride_0 = const()[name = string("value_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_87_cast_fp16 = transpose(perm = value_states_87_perm_0, x = var_5485_cast_fp16)[name = string("transpose_298")]; tensor value_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = value_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = value_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_15_squeeze_mask_0, stride = value_cache_internal_tensor_assign_15_stride_0, update = value_states_87_cast_fp16, x = coreml_update_state_195)[name = string("value_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_15_cast_fp16, input = value_cache)[name = string("coreml_update_state_197_write_state")]; tensor coreml_update_state_197 = read_state(input = value_cache)[name = string("coreml_update_state_197")]; tensor var_5579_begin_0 = const()[name = string("op_5579_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5579_end_0 = const()[name = string("op_5579_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5579_end_mask_0 = const()[name = string("op_5579_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5579_cast_fp16 = slice_by_index(begin = var_5579_begin_0, end = var_5579_end_0, end_mask = var_5579_end_mask_0, x = coreml_update_state_196)[name = string("op_5579_cast_fp16")]; tensor tile_28 = const()[name = string("tile_28"), val = tensor([1, 1])]; int32 var_5582_axis_0 = const()[name = string("op_5582_axis_0"), val = int32(1)]; tensor var_5582_cast_fp16_0, tensor var_5582_cast_fp16_1 = split(axis = var_5582_axis_0, split_sizes = tile_28, x = var_5579_cast_fp16)[name = string("op_5582_cast_fp16")]; tensor var_5589_begin_0 = const()[name = string("op_5589_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5589_end_0 = const()[name = string("op_5589_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5589_end_mask_0 = const()[name = string("op_5589_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5589_cast_fp16 = slice_by_index(begin = var_5589_begin_0, end = var_5589_end_0, end_mask = var_5589_end_mask_0, x = coreml_update_state_197)[name = string("op_5589_cast_fp16")]; tensor tile_29 = const()[name = string("tile_29"), val = tensor([1, 1])]; int32 var_5592_axis_0 = const()[name = string("op_5592_axis_0"), val = int32(1)]; tensor var_5592_cast_fp16_0, tensor var_5592_cast_fp16_1 = split(axis = var_5592_axis_0, split_sizes = tile_29, x = var_5589_cast_fp16)[name = string("op_5592_cast_fp16")]; tensor var_5595_split_sizes_0 = const()[name = string("op_5595_split_sizes_0"), val = tensor([8, 8])]; int32 var_5595_axis_0 = const()[name = string("op_5595_axis_0"), val = int32(1)]; tensor var_5595_0, tensor var_5595_1 = split(axis = var_5595_axis_0, split_sizes = var_5595_split_sizes_0, x = query_states_87_cast_fp16)[name = string("op_5595")]; bool attn_weights_225_transpose_x_0 = const()[name = string("attn_weights_225_transpose_x_0"), val = bool(false)]; bool attn_weights_225_transpose_y_0 = const()[name = string("attn_weights_225_transpose_y_0"), val = bool(false)]; tensor attn_weights_225_cast_fp16 = matmul(transpose_x = attn_weights_225_transpose_x_0, transpose_y = attn_weights_225_transpose_y_0, x = var_5582_cast_fp16_0, y = var_5595_0)[name = string("attn_weights_225_cast_fp16")]; fp16 var_5598_to_fp16 = const()[name = string("op_5598_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_227_cast_fp16 = mul(x = attn_weights_225_cast_fp16, y = var_5598_to_fp16)[name = string("attn_weights_227_cast_fp16")]; tensor attn_weights_229_cast_fp16 = add(x = attn_weights_227_cast_fp16, y = attn_mask_1)[name = string("attn_weights_229_cast_fp16")]; int32 var_5602 = const()[name = string("op_5602"), val = int32(-2)]; tensor attn_weights_231_cast_fp16 = softmax(axis = var_5602, x = attn_weights_229_cast_fp16)[name = string("attn_weights_231_cast_fp16")]; bool var_5608_transpose_x_1 = const()[name = string("op_5608_transpose_x_1"), val = bool(true)]; bool var_5608_transpose_y_1 = const()[name = string("op_5608_transpose_y_1"), val = bool(false)]; tensor var_5608_cast_fp16 = matmul(transpose_x = var_5608_transpose_x_1, transpose_y = var_5608_transpose_y_1, x = attn_weights_231_cast_fp16, y = var_5592_cast_fp16_0)[name = string("op_5608_cast_fp16")]; bool attn_weights_233_transpose_x_0 = const()[name = string("attn_weights_233_transpose_x_0"), val = bool(false)]; bool attn_weights_233_transpose_y_0 = const()[name = string("attn_weights_233_transpose_y_0"), val = bool(false)]; tensor attn_weights_233_cast_fp16 = matmul(transpose_x = attn_weights_233_transpose_x_0, transpose_y = attn_weights_233_transpose_y_0, x = var_5582_cast_fp16_1, y = var_5595_1)[name = string("attn_weights_233_cast_fp16")]; fp16 var_5610_to_fp16 = const()[name = string("op_5610_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_235_cast_fp16 = mul(x = attn_weights_233_cast_fp16, y = var_5610_to_fp16)[name = string("attn_weights_235_cast_fp16")]; tensor attn_weights_237_cast_fp16 = add(x = attn_weights_235_cast_fp16, y = attn_mask_1)[name = string("attn_weights_237_cast_fp16")]; int32 var_5614 = const()[name = string("op_5614"), val = int32(-2)]; tensor attn_weights_239_cast_fp16 = softmax(axis = var_5614, x = attn_weights_237_cast_fp16)[name = string("attn_weights_239_cast_fp16")]; bool attn_output_113_transpose_x_1 = const()[name = string("attn_output_113_transpose_x_1"), val = bool(true)]; bool attn_output_113_transpose_y_1 = const()[name = string("attn_output_113_transpose_y_1"), val = bool(false)]; tensor attn_output_113_cast_fp16 = matmul(transpose_x = attn_output_113_transpose_x_1, transpose_y = attn_output_113_transpose_y_1, x = attn_weights_239_cast_fp16, y = var_5592_cast_fp16_1)[name = string("attn_output_113_cast_fp16")]; int32 var_5622 = const()[name = string("op_5622"), val = int32(1)]; bool attn_output_115_interleave_0 = const()[name = string("attn_output_115_interleave_0"), val = bool(false)]; tensor attn_output_115_cast_fp16 = concat(axis = var_5622, interleave = attn_output_115_interleave_0, values = (var_5608_cast_fp16, attn_output_113_cast_fp16))[name = string("attn_output_115_cast_fp16")]; tensor var_5626_perm_0 = const()[name = string("op_5626_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_179x = const()[name = string("concat_179x"), val = tensor([1, 2048, 1, -1])]; tensor var_5626_cast_fp16 = transpose(perm = var_5626_perm_0, x = attn_output_115_cast_fp16)[name = string("transpose_297")]; tensor attn_output_119_cast_fp16 = reshape(shape = concat_179x, x = var_5626_cast_fp16)[name = string("attn_output_119_cast_fp16")]; tensor hidden_states_143_strides_0 = const()[name = string("hidden_states_143_strides_0"), val = tensor([1, 1])]; string hidden_states_143_pad_type_0 = const()[name = string("hidden_states_143_pad_type_0"), val = string("valid")]; tensor hidden_states_143_pad_0 = const()[name = string("hidden_states_143_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_143_dilations_0 = const()[name = string("hidden_states_143_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_143_groups_0 = const()[name = string("hidden_states_143_groups_0"), val = int32(1)]; tensor hidden_states_143_cast_fp16 = conv(dilations = hidden_states_143_dilations_0, groups = hidden_states_143_groups_0, pad = hidden_states_143_pad_0, pad_type = hidden_states_143_pad_type_0, strides = hidden_states_143_strides_0, weight = layers_14_self_attn_o_proj_weight_cast_fp16, x = attn_output_119_cast_fp16)[name = string("hidden_states_143_cast_fp16")]; tensor hidden_states_145_cast_fp16 = add(x = hidden_states_139_cast_fp16, y = hidden_states_143_cast_fp16)[name = string("hidden_states_145_cast_fp16")]; fp16 const_148_promoted_to_fp16 = const()[name = string("const_148_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5659_cast_fp16 = mul(x = hidden_states_145_cast_fp16, y = const_148_promoted_to_fp16)[name = string("op_5659_cast_fp16")]; int32 var_5657 = const()[name = string("op_5657"), val = int32(1)]; bool doubled_117_interleave_0 = const()[name = string("doubled_117_interleave_0"), val = bool(false)]; tensor doubled_117_cast_fp16 = concat(axis = var_5657, interleave = doubled_117_interleave_0, values = (hidden_states_145_cast_fp16, var_5659_cast_fp16))[name = string("doubled_117_cast_fp16")]; tensor out_59_axes_0 = const()[name = string("out_59_axes_0"), val = tensor([1])]; tensor out_59_gamma_0_to_fp16 = const()[name = string("out_59_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441630464)))]; fp16 var_5669_to_fp16 = const()[name = string("op_5669_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_59_cast_fp16 = layer_norm(axes = out_59_axes_0, epsilon = var_5669_to_fp16, gamma = out_59_gamma_0_to_fp16, x = doubled_117_cast_fp16)[name = string("out_59_cast_fp16")]; tensor var_5680_split_sizes_0 = const()[name = string("op_5680_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5680_axis_0 = const()[name = string("op_5680_axis_0"), val = int32(1)]; tensor var_5680_cast_fp16_0, tensor var_5680_cast_fp16_1 = split(axis = var_5680_axis_0, split_sizes = var_5680_split_sizes_0, x = out_59_cast_fp16)[name = string("op_5680_cast_fp16")]; tensor input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor([1, 1])]; string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; tensor input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor([1, 1])]; int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; tensor input_29_cast_fp16 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_14_mlp_gate_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("input_29_cast_fp16")]; tensor var_5697_cast_fp16 = silu(x = input_29_cast_fp16)[name = string("op_5697_cast_fp16")]; tensor var_5703_strides_0 = const()[name = string("op_5703_strides_0"), val = tensor([1, 1])]; string var_5703_pad_type_0 = const()[name = string("op_5703_pad_type_0"), val = string("valid")]; tensor var_5703_pad_0 = const()[name = string("op_5703_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5703_dilations_0 = const()[name = string("op_5703_dilations_0"), val = tensor([1, 1])]; int32 var_5703_groups_0 = const()[name = string("op_5703_groups_0"), val = int32(1)]; tensor var_5703_cast_fp16 = conv(dilations = var_5703_dilations_0, groups = var_5703_groups_0, pad = var_5703_pad_0, pad_type = var_5703_pad_type_0, strides = var_5703_strides_0, weight = layers_14_mlp_up_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("op_5703_cast_fp16")]; tensor x_149_cast_fp16 = mul(x = var_5697_cast_fp16, y = var_5703_cast_fp16)[name = string("x_149_cast_fp16")]; tensor hidden_states_147_strides_0 = const()[name = string("hidden_states_147_strides_0"), val = tensor([1, 1])]; string hidden_states_147_pad_type_0 = const()[name = string("hidden_states_147_pad_type_0"), val = string("valid")]; tensor hidden_states_147_pad_0 = const()[name = string("hidden_states_147_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_147_dilations_0 = const()[name = string("hidden_states_147_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_147_groups_0 = const()[name = string("hidden_states_147_groups_0"), val = int32(1)]; tensor hidden_states_147_cast_fp16 = conv(dilations = hidden_states_147_dilations_0, groups = hidden_states_147_groups_0, pad = hidden_states_147_pad_0, pad_type = hidden_states_147_pad_type_0, strides = hidden_states_147_strides_0, weight = layers_14_mlp_down_proj_weight_cast_fp16, x = x_149_cast_fp16)[name = string("hidden_states_147_cast_fp16")]; tensor hidden_states_149_cast_fp16 = add(x = hidden_states_145_cast_fp16, y = hidden_states_147_cast_fp16)[name = string("hidden_states_149_cast_fp16")]; fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5721_cast_fp16 = mul(x = hidden_states_149_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_5721_cast_fp16")]; int32 var_5719 = const()[name = string("op_5719"), val = int32(1)]; bool doubled_121_interleave_0 = const()[name = string("doubled_121_interleave_0"), val = bool(false)]; tensor doubled_121_cast_fp16 = concat(axis = var_5719, interleave = doubled_121_interleave_0, values = (hidden_states_149_cast_fp16, var_5721_cast_fp16))[name = string("doubled_121_cast_fp16")]; tensor out_61_axes_0 = const()[name = string("out_61_axes_0"), val = tensor([1])]; tensor out_61_gamma_0_to_fp16 = const()[name = string("out_61_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441638720)))]; fp16 var_5731_to_fp16 = const()[name = string("op_5731_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_61_cast_fp16 = layer_norm(axes = out_61_axes_0, epsilon = var_5731_to_fp16, gamma = out_61_gamma_0_to_fp16, x = doubled_121_cast_fp16)[name = string("out_61_cast_fp16")]; tensor var_5742_split_sizes_0 = const()[name = string("op_5742_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5742_axis_0 = const()[name = string("op_5742_axis_0"), val = int32(1)]; tensor var_5742_cast_fp16_0, tensor var_5742_cast_fp16_1 = split(axis = var_5742_axis_0, split_sizes = var_5742_split_sizes_0, x = out_61_cast_fp16)[name = string("op_5742_cast_fp16")]; tensor query_states_91_strides_0 = const()[name = string("query_states_91_strides_0"), val = tensor([1, 1])]; string query_states_91_pad_type_0 = const()[name = string("query_states_91_pad_type_0"), val = string("valid")]; tensor query_states_91_pad_0 = const()[name = string("query_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_91_dilations_0 = const()[name = string("query_states_91_dilations_0"), val = tensor([1, 1])]; int32 query_states_91_groups_0 = const()[name = string("query_states_91_groups_0"), val = int32(1)]; tensor query_states_91_cast_fp16 = conv(dilations = query_states_91_dilations_0, groups = query_states_91_groups_0, pad = query_states_91_pad_0, pad_type = query_states_91_pad_type_0, strides = query_states_91_strides_0, weight = layers_15_self_attn_q_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("query_states_91_cast_fp16")]; tensor key_states_151_strides_0 = const()[name = string("key_states_151_strides_0"), val = tensor([1, 1])]; string key_states_151_pad_type_0 = const()[name = string("key_states_151_pad_type_0"), val = string("valid")]; tensor key_states_151_pad_0 = const()[name = string("key_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_151_dilations_0 = const()[name = string("key_states_151_dilations_0"), val = tensor([1, 1])]; int32 key_states_151_groups_0 = const()[name = string("key_states_151_groups_0"), val = int32(1)]; tensor key_states_151_cast_fp16 = conv(dilations = key_states_151_dilations_0, groups = key_states_151_groups_0, pad = key_states_151_pad_0, pad_type = key_states_151_pad_type_0, strides = key_states_151_strides_0, weight = layers_15_self_attn_k_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("key_states_151_cast_fp16")]; tensor value_states_91_strides_0 = const()[name = string("value_states_91_strides_0"), val = tensor([1, 1])]; string value_states_91_pad_type_0 = const()[name = string("value_states_91_pad_type_0"), val = string("valid")]; tensor value_states_91_pad_0 = const()[name = string("value_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_91_dilations_0 = const()[name = string("value_states_91_dilations_0"), val = tensor([1, 1])]; int32 value_states_91_groups_0 = const()[name = string("value_states_91_groups_0"), val = int32(1)]; tensor value_states_91_cast_fp16 = conv(dilations = value_states_91_dilations_0, groups = value_states_91_groups_0, pad = value_states_91_pad_0, pad_type = value_states_91_pad_type_0, strides = value_states_91_strides_0, weight = layers_15_self_attn_v_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("value_states_91_cast_fp16")]; tensor concat_180x = const()[name = string("concat_180x"), val = tensor([1, 16, 128, -1])]; tensor x_151_cast_fp16 = reshape(shape = concat_180x, x = query_states_91_cast_fp16)[name = string("x_151_cast_fp16")]; tensor concat_181x = const()[name = string("concat_181x"), val = tensor([1, 2, 128, -1])]; tensor var_5799_cast_fp16 = reshape(shape = concat_181x, x = key_states_151_cast_fp16)[name = string("op_5799_cast_fp16")]; tensor concat_182x = const()[name = string("concat_182x"), val = tensor([1, 2, 128, -1])]; tensor var_5806_cast_fp16 = reshape(shape = concat_182x, x = value_states_91_cast_fp16)[name = string("op_5806_cast_fp16")]; tensor var_5810_cast_fp16 = mul(x = x_151_cast_fp16, y = var_869_cast_fp16)[name = string("op_5810_cast_fp16")]; tensor var_5811_split_sizes_0 = const()[name = string("op_5811_split_sizes_0"), val = tensor([64, 64])]; int32 var_5811_axis_0 = const()[name = string("op_5811_axis_0"), val = int32(-2)]; tensor var_5811_cast_fp16_0, tensor var_5811_cast_fp16_1 = split(axis = var_5811_axis_0, split_sizes = var_5811_split_sizes_0, x = x_151_cast_fp16)[name = string("op_5811_cast_fp16")]; fp16 const_152_promoted_to_fp16 = const()[name = string("const_152_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5813_cast_fp16 = mul(x = var_5811_cast_fp16_1, y = const_152_promoted_to_fp16)[name = string("op_5813_cast_fp16")]; int32 var_5815 = const()[name = string("op_5815"), val = int32(-2)]; bool var_5816_interleave_0 = const()[name = string("op_5816_interleave_0"), val = bool(false)]; tensor var_5816_cast_fp16 = concat(axis = var_5815, interleave = var_5816_interleave_0, values = (var_5813_cast_fp16, var_5811_cast_fp16_0))[name = string("op_5816_cast_fp16")]; tensor var_5817_cast_fp16 = mul(x = var_5816_cast_fp16, y = var_878_cast_fp16)[name = string("op_5817_cast_fp16")]; tensor query_states_93_cast_fp16 = add(x = var_5810_cast_fp16, y = var_5817_cast_fp16)[name = string("query_states_93_cast_fp16")]; tensor var_5823_cast_fp16 = mul(x = var_5799_cast_fp16, y = var_869_cast_fp16)[name = string("op_5823_cast_fp16")]; tensor var_5824_split_sizes_0 = const()[name = string("op_5824_split_sizes_0"), val = tensor([64, 64])]; int32 var_5824_axis_0 = const()[name = string("op_5824_axis_0"), val = int32(-2)]; tensor var_5824_cast_fp16_0, tensor var_5824_cast_fp16_1 = split(axis = var_5824_axis_0, split_sizes = var_5824_split_sizes_0, x = var_5799_cast_fp16)[name = string("op_5824_cast_fp16")]; fp16 const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5826_cast_fp16 = mul(x = var_5824_cast_fp16_1, y = const_153_promoted_to_fp16)[name = string("op_5826_cast_fp16")]; int32 var_5828 = const()[name = string("op_5828"), val = int32(-2)]; bool var_5829_interleave_0 = const()[name = string("op_5829_interleave_0"), val = bool(false)]; tensor var_5829_cast_fp16 = concat(axis = var_5828, interleave = var_5829_interleave_0, values = (var_5826_cast_fp16, var_5824_cast_fp16_0))[name = string("op_5829_cast_fp16")]; tensor var_5830_cast_fp16 = mul(x = var_5829_cast_fp16, y = var_878_cast_fp16)[name = string("op_5830_cast_fp16")]; tensor key_states_155_cast_fp16 = add(x = var_5823_cast_fp16, y = var_5830_cast_fp16)[name = string("key_states_155_cast_fp16")]; tensor expand_dims_180 = const()[name = string("expand_dims_180"), val = tensor([15])]; tensor expand_dims_181 = const()[name = string("expand_dims_181"), val = tensor([0])]; tensor expand_dims_183 = const()[name = string("expand_dims_183"), val = tensor([0])]; int32 concat_185_axis_0 = const()[name = string("concat_185_axis_0"), val = int32(0)]; bool concat_185_interleave_0 = const()[name = string("concat_185_interleave_0"), val = bool(false)]; tensor concat_185 = concat(axis = concat_185_axis_0, interleave = concat_185_interleave_0, values = (expand_dims_180, expand_dims_181, position_id, expand_dims_183))[name = string("concat_185")]; tensor expand_dims_184 = const()[name = string("expand_dims_184"), val = tensor([16])]; tensor concat_186_values1_0 = const()[name = string("concat_186_values1_0"), val = tensor([0])]; tensor concat_186_values3_0 = const()[name = string("concat_186_values3_0"), val = tensor([0])]; int32 concat_186_axis_0 = const()[name = string("concat_186_axis_0"), val = int32(0)]; bool concat_186_interleave_0 = const()[name = string("concat_186_interleave_0"), val = bool(false)]; tensor concat_186 = concat(axis = concat_186_axis_0, interleave = concat_186_interleave_0, values = (expand_dims_184, concat_186_values1_0, cache_position_end, concat_186_values3_0))[name = string("concat_186")]; tensor key_states_157_perm_0 = const()[name = string("key_states_157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_16_stride_0 = const()[name = string("key_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_157_cast_fp16 = transpose(perm = key_states_157_perm_0, x = key_states_155_cast_fp16)[name = string("transpose_296")]; tensor key_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = key_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = key_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_16_squeeze_mask_0, stride = key_cache_internal_tensor_assign_16_stride_0, update = key_states_157_cast_fp16, x = coreml_update_state_196)[name = string("key_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_16_cast_fp16, input = key_cache)[name = string("coreml_update_state_198_write_state")]; tensor coreml_update_state_198 = read_state(input = key_cache)[name = string("coreml_update_state_198")]; tensor value_states_93_perm_0 = const()[name = string("value_states_93_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_16_stride_0 = const()[name = string("value_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_93_cast_fp16 = transpose(perm = value_states_93_perm_0, x = var_5806_cast_fp16)[name = string("transpose_295")]; tensor value_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = value_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = value_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_16_squeeze_mask_0, stride = value_cache_internal_tensor_assign_16_stride_0, update = value_states_93_cast_fp16, x = coreml_update_state_197)[name = string("value_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_16_cast_fp16, input = value_cache)[name = string("coreml_update_state_199_write_state")]; tensor coreml_update_state_199 = read_state(input = value_cache)[name = string("coreml_update_state_199")]; tensor var_5900_begin_0 = const()[name = string("op_5900_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5900_end_0 = const()[name = string("op_5900_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5900_end_mask_0 = const()[name = string("op_5900_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5900_cast_fp16 = slice_by_index(begin = var_5900_begin_0, end = var_5900_end_0, end_mask = var_5900_end_mask_0, x = coreml_update_state_198)[name = string("op_5900_cast_fp16")]; tensor tile_30 = const()[name = string("tile_30"), val = tensor([1, 1])]; int32 var_5903_axis_0 = const()[name = string("op_5903_axis_0"), val = int32(1)]; tensor var_5903_cast_fp16_0, tensor var_5903_cast_fp16_1 = split(axis = var_5903_axis_0, split_sizes = tile_30, x = var_5900_cast_fp16)[name = string("op_5903_cast_fp16")]; tensor var_5910_begin_0 = const()[name = string("op_5910_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5910_end_0 = const()[name = string("op_5910_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5910_end_mask_0 = const()[name = string("op_5910_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5910_cast_fp16 = slice_by_index(begin = var_5910_begin_0, end = var_5910_end_0, end_mask = var_5910_end_mask_0, x = coreml_update_state_199)[name = string("op_5910_cast_fp16")]; tensor tile_31 = const()[name = string("tile_31"), val = tensor([1, 1])]; int32 var_5913_axis_0 = const()[name = string("op_5913_axis_0"), val = int32(1)]; tensor var_5913_cast_fp16_0, tensor var_5913_cast_fp16_1 = split(axis = var_5913_axis_0, split_sizes = tile_31, x = var_5910_cast_fp16)[name = string("op_5913_cast_fp16")]; tensor var_5916_split_sizes_0 = const()[name = string("op_5916_split_sizes_0"), val = tensor([8, 8])]; int32 var_5916_axis_0 = const()[name = string("op_5916_axis_0"), val = int32(1)]; tensor var_5916_0, tensor var_5916_1 = split(axis = var_5916_axis_0, split_sizes = var_5916_split_sizes_0, x = query_states_93_cast_fp16)[name = string("op_5916")]; bool attn_weights_241_transpose_x_0 = const()[name = string("attn_weights_241_transpose_x_0"), val = bool(false)]; bool attn_weights_241_transpose_y_0 = const()[name = string("attn_weights_241_transpose_y_0"), val = bool(false)]; tensor attn_weights_241_cast_fp16 = matmul(transpose_x = attn_weights_241_transpose_x_0, transpose_y = attn_weights_241_transpose_y_0, x = var_5903_cast_fp16_0, y = var_5916_0)[name = string("attn_weights_241_cast_fp16")]; fp16 var_5919_to_fp16 = const()[name = string("op_5919_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_243_cast_fp16 = mul(x = attn_weights_241_cast_fp16, y = var_5919_to_fp16)[name = string("attn_weights_243_cast_fp16")]; tensor attn_weights_245_cast_fp16 = add(x = attn_weights_243_cast_fp16, y = attn_mask_1)[name = string("attn_weights_245_cast_fp16")]; int32 var_5923 = const()[name = string("op_5923"), val = int32(-2)]; tensor attn_weights_247_cast_fp16 = softmax(axis = var_5923, x = attn_weights_245_cast_fp16)[name = string("attn_weights_247_cast_fp16")]; bool var_5929_transpose_x_1 = const()[name = string("op_5929_transpose_x_1"), val = bool(true)]; bool var_5929_transpose_y_1 = const()[name = string("op_5929_transpose_y_1"), val = bool(false)]; tensor var_5929_cast_fp16 = matmul(transpose_x = var_5929_transpose_x_1, transpose_y = var_5929_transpose_y_1, x = attn_weights_247_cast_fp16, y = var_5913_cast_fp16_0)[name = string("op_5929_cast_fp16")]; bool attn_weights_249_transpose_x_0 = const()[name = string("attn_weights_249_transpose_x_0"), val = bool(false)]; bool attn_weights_249_transpose_y_0 = const()[name = string("attn_weights_249_transpose_y_0"), val = bool(false)]; tensor attn_weights_249_cast_fp16 = matmul(transpose_x = attn_weights_249_transpose_x_0, transpose_y = attn_weights_249_transpose_y_0, x = var_5903_cast_fp16_1, y = var_5916_1)[name = string("attn_weights_249_cast_fp16")]; fp16 var_5931_to_fp16 = const()[name = string("op_5931_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_251_cast_fp16 = mul(x = attn_weights_249_cast_fp16, y = var_5931_to_fp16)[name = string("attn_weights_251_cast_fp16")]; tensor attn_weights_253_cast_fp16 = add(x = attn_weights_251_cast_fp16, y = attn_mask_1)[name = string("attn_weights_253_cast_fp16")]; int32 var_5935 = const()[name = string("op_5935"), val = int32(-2)]; tensor attn_weights_255_cast_fp16 = softmax(axis = var_5935, x = attn_weights_253_cast_fp16)[name = string("attn_weights_255_cast_fp16")]; bool attn_output_121_transpose_x_1 = const()[name = string("attn_output_121_transpose_x_1"), val = bool(true)]; bool attn_output_121_transpose_y_1 = const()[name = string("attn_output_121_transpose_y_1"), val = bool(false)]; tensor attn_output_121_cast_fp16 = matmul(transpose_x = attn_output_121_transpose_x_1, transpose_y = attn_output_121_transpose_y_1, x = attn_weights_255_cast_fp16, y = var_5913_cast_fp16_1)[name = string("attn_output_121_cast_fp16")]; int32 var_5943 = const()[name = string("op_5943"), val = int32(1)]; bool attn_output_123_interleave_0 = const()[name = string("attn_output_123_interleave_0"), val = bool(false)]; tensor attn_output_123_cast_fp16 = concat(axis = var_5943, interleave = attn_output_123_interleave_0, values = (var_5929_cast_fp16, attn_output_121_cast_fp16))[name = string("attn_output_123_cast_fp16")]; tensor var_5947_perm_0 = const()[name = string("op_5947_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_191x = const()[name = string("concat_191x"), val = tensor([1, 2048, 1, -1])]; tensor var_5947_cast_fp16 = transpose(perm = var_5947_perm_0, x = attn_output_123_cast_fp16)[name = string("transpose_294")]; tensor attn_output_127_cast_fp16 = reshape(shape = concat_191x, x = var_5947_cast_fp16)[name = string("attn_output_127_cast_fp16")]; tensor hidden_states_153_strides_0 = const()[name = string("hidden_states_153_strides_0"), val = tensor([1, 1])]; string hidden_states_153_pad_type_0 = const()[name = string("hidden_states_153_pad_type_0"), val = string("valid")]; tensor hidden_states_153_pad_0 = const()[name = string("hidden_states_153_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_153_dilations_0 = const()[name = string("hidden_states_153_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_153_groups_0 = const()[name = string("hidden_states_153_groups_0"), val = int32(1)]; tensor hidden_states_153_cast_fp16 = conv(dilations = hidden_states_153_dilations_0, groups = hidden_states_153_groups_0, pad = hidden_states_153_pad_0, pad_type = hidden_states_153_pad_type_0, strides = hidden_states_153_strides_0, weight = layers_15_self_attn_o_proj_weight_cast_fp16, x = attn_output_127_cast_fp16)[name = string("hidden_states_153_cast_fp16")]; tensor hidden_states_155_cast_fp16 = add(x = hidden_states_149_cast_fp16, y = hidden_states_153_cast_fp16)[name = string("hidden_states_155_cast_fp16")]; fp16 const_158_promoted_to_fp16 = const()[name = string("const_158_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5980_cast_fp16 = mul(x = hidden_states_155_cast_fp16, y = const_158_promoted_to_fp16)[name = string("op_5980_cast_fp16")]; int32 var_5978 = const()[name = string("op_5978"), val = int32(1)]; bool doubled_125_interleave_0 = const()[name = string("doubled_125_interleave_0"), val = bool(false)]; tensor doubled_125_cast_fp16 = concat(axis = var_5978, interleave = doubled_125_interleave_0, values = (hidden_states_155_cast_fp16, var_5980_cast_fp16))[name = string("doubled_125_cast_fp16")]; tensor out_63_axes_0 = const()[name = string("out_63_axes_0"), val = tensor([1])]; tensor out_63_gamma_0_to_fp16 = const()[name = string("out_63_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441646976)))]; fp16 var_5990_to_fp16 = const()[name = string("op_5990_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_63_cast_fp16 = layer_norm(axes = out_63_axes_0, epsilon = var_5990_to_fp16, gamma = out_63_gamma_0_to_fp16, x = doubled_125_cast_fp16)[name = string("out_63_cast_fp16")]; tensor var_6001_split_sizes_0 = const()[name = string("op_6001_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6001_axis_0 = const()[name = string("op_6001_axis_0"), val = int32(1)]; tensor var_6001_cast_fp16_0, tensor var_6001_cast_fp16_1 = split(axis = var_6001_axis_0, split_sizes = var_6001_split_sizes_0, x = out_63_cast_fp16)[name = string("op_6001_cast_fp16")]; tensor input_31_strides_0 = const()[name = string("input_31_strides_0"), val = tensor([1, 1])]; string input_31_pad_type_0 = const()[name = string("input_31_pad_type_0"), val = string("valid")]; tensor input_31_pad_0 = const()[name = string("input_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_31_dilations_0 = const()[name = string("input_31_dilations_0"), val = tensor([1, 1])]; int32 input_31_groups_0 = const()[name = string("input_31_groups_0"), val = int32(1)]; tensor input_31_cast_fp16 = conv(dilations = input_31_dilations_0, groups = input_31_groups_0, pad = input_31_pad_0, pad_type = input_31_pad_type_0, strides = input_31_strides_0, weight = layers_15_mlp_gate_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("input_31_cast_fp16")]; tensor var_6018_cast_fp16 = silu(x = input_31_cast_fp16)[name = string("op_6018_cast_fp16")]; tensor var_6024_strides_0 = const()[name = string("op_6024_strides_0"), val = tensor([1, 1])]; string var_6024_pad_type_0 = const()[name = string("op_6024_pad_type_0"), val = string("valid")]; tensor var_6024_pad_0 = const()[name = string("op_6024_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6024_dilations_0 = const()[name = string("op_6024_dilations_0"), val = tensor([1, 1])]; int32 var_6024_groups_0 = const()[name = string("op_6024_groups_0"), val = int32(1)]; tensor var_6024_cast_fp16 = conv(dilations = var_6024_dilations_0, groups = var_6024_groups_0, pad = var_6024_pad_0, pad_type = var_6024_pad_type_0, strides = var_6024_strides_0, weight = layers_15_mlp_up_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("op_6024_cast_fp16")]; tensor x_159_cast_fp16 = mul(x = var_6018_cast_fp16, y = var_6024_cast_fp16)[name = string("x_159_cast_fp16")]; tensor hidden_states_157_strides_0 = const()[name = string("hidden_states_157_strides_0"), val = tensor([1, 1])]; string hidden_states_157_pad_type_0 = const()[name = string("hidden_states_157_pad_type_0"), val = string("valid")]; tensor hidden_states_157_pad_0 = const()[name = string("hidden_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_157_dilations_0 = const()[name = string("hidden_states_157_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_157_groups_0 = const()[name = string("hidden_states_157_groups_0"), val = int32(1)]; tensor hidden_states_157_cast_fp16 = conv(dilations = hidden_states_157_dilations_0, groups = hidden_states_157_groups_0, pad = hidden_states_157_pad_0, pad_type = hidden_states_157_pad_type_0, strides = hidden_states_157_strides_0, weight = layers_15_mlp_down_proj_weight_cast_fp16, x = x_159_cast_fp16)[name = string("hidden_states_157_cast_fp16")]; tensor hidden_states_159_cast_fp16 = add(x = hidden_states_155_cast_fp16, y = hidden_states_157_cast_fp16)[name = string("hidden_states_159_cast_fp16")]; fp16 const_160_promoted_to_fp16 = const()[name = string("const_160_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6042_cast_fp16 = mul(x = hidden_states_159_cast_fp16, y = const_160_promoted_to_fp16)[name = string("op_6042_cast_fp16")]; int32 var_6040 = const()[name = string("op_6040"), val = int32(1)]; bool doubled_129_interleave_0 = const()[name = string("doubled_129_interleave_0"), val = bool(false)]; tensor doubled_129_cast_fp16 = concat(axis = var_6040, interleave = doubled_129_interleave_0, values = (hidden_states_159_cast_fp16, var_6042_cast_fp16))[name = string("doubled_129_cast_fp16")]; tensor out_65_axes_0 = const()[name = string("out_65_axes_0"), val = tensor([1])]; tensor out_65_gamma_0_to_fp16 = const()[name = string("out_65_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441655232)))]; fp16 var_6052_to_fp16 = const()[name = string("op_6052_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_65_cast_fp16 = layer_norm(axes = out_65_axes_0, epsilon = var_6052_to_fp16, gamma = out_65_gamma_0_to_fp16, x = doubled_129_cast_fp16)[name = string("out_65_cast_fp16")]; tensor var_6063_split_sizes_0 = const()[name = string("op_6063_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6063_axis_0 = const()[name = string("op_6063_axis_0"), val = int32(1)]; tensor var_6063_cast_fp16_0, tensor var_6063_cast_fp16_1 = split(axis = var_6063_axis_0, split_sizes = var_6063_split_sizes_0, x = out_65_cast_fp16)[name = string("op_6063_cast_fp16")]; tensor query_states_97_strides_0 = const()[name = string("query_states_97_strides_0"), val = tensor([1, 1])]; string query_states_97_pad_type_0 = const()[name = string("query_states_97_pad_type_0"), val = string("valid")]; tensor query_states_97_pad_0 = const()[name = string("query_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_97_dilations_0 = const()[name = string("query_states_97_dilations_0"), val = tensor([1, 1])]; int32 query_states_97_groups_0 = const()[name = string("query_states_97_groups_0"), val = int32(1)]; tensor query_states_97_cast_fp16 = conv(dilations = query_states_97_dilations_0, groups = query_states_97_groups_0, pad = query_states_97_pad_0, pad_type = query_states_97_pad_type_0, strides = query_states_97_strides_0, weight = layers_16_self_attn_q_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("query_states_97_cast_fp16")]; tensor key_states_161_strides_0 = const()[name = string("key_states_161_strides_0"), val = tensor([1, 1])]; string key_states_161_pad_type_0 = const()[name = string("key_states_161_pad_type_0"), val = string("valid")]; tensor key_states_161_pad_0 = const()[name = string("key_states_161_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_161_dilations_0 = const()[name = string("key_states_161_dilations_0"), val = tensor([1, 1])]; int32 key_states_161_groups_0 = const()[name = string("key_states_161_groups_0"), val = int32(1)]; tensor key_states_161_cast_fp16 = conv(dilations = key_states_161_dilations_0, groups = key_states_161_groups_0, pad = key_states_161_pad_0, pad_type = key_states_161_pad_type_0, strides = key_states_161_strides_0, weight = layers_16_self_attn_k_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("key_states_161_cast_fp16")]; tensor value_states_97_strides_0 = const()[name = string("value_states_97_strides_0"), val = tensor([1, 1])]; string value_states_97_pad_type_0 = const()[name = string("value_states_97_pad_type_0"), val = string("valid")]; tensor value_states_97_pad_0 = const()[name = string("value_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_97_dilations_0 = const()[name = string("value_states_97_dilations_0"), val = tensor([1, 1])]; int32 value_states_97_groups_0 = const()[name = string("value_states_97_groups_0"), val = int32(1)]; tensor value_states_97_cast_fp16 = conv(dilations = value_states_97_dilations_0, groups = value_states_97_groups_0, pad = value_states_97_pad_0, pad_type = value_states_97_pad_type_0, strides = value_states_97_strides_0, weight = layers_16_self_attn_v_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("value_states_97_cast_fp16")]; tensor concat_192x = const()[name = string("concat_192x"), val = tensor([1, 16, 128, -1])]; tensor x_161_cast_fp16 = reshape(shape = concat_192x, x = query_states_97_cast_fp16)[name = string("x_161_cast_fp16")]; tensor concat_193x = const()[name = string("concat_193x"), val = tensor([1, 2, 128, -1])]; tensor var_6120_cast_fp16 = reshape(shape = concat_193x, x = key_states_161_cast_fp16)[name = string("op_6120_cast_fp16")]; tensor concat_194x = const()[name = string("concat_194x"), val = tensor([1, 2, 128, -1])]; tensor var_6127_cast_fp16 = reshape(shape = concat_194x, x = value_states_97_cast_fp16)[name = string("op_6127_cast_fp16")]; tensor var_6131_cast_fp16 = mul(x = x_161_cast_fp16, y = var_869_cast_fp16)[name = string("op_6131_cast_fp16")]; tensor var_6132_split_sizes_0 = const()[name = string("op_6132_split_sizes_0"), val = tensor([64, 64])]; int32 var_6132_axis_0 = const()[name = string("op_6132_axis_0"), val = int32(-2)]; tensor var_6132_cast_fp16_0, tensor var_6132_cast_fp16_1 = split(axis = var_6132_axis_0, split_sizes = var_6132_split_sizes_0, x = x_161_cast_fp16)[name = string("op_6132_cast_fp16")]; fp16 const_162_promoted_to_fp16 = const()[name = string("const_162_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6134_cast_fp16 = mul(x = var_6132_cast_fp16_1, y = const_162_promoted_to_fp16)[name = string("op_6134_cast_fp16")]; int32 var_6136 = const()[name = string("op_6136"), val = int32(-2)]; bool var_6137_interleave_0 = const()[name = string("op_6137_interleave_0"), val = bool(false)]; tensor var_6137_cast_fp16 = concat(axis = var_6136, interleave = var_6137_interleave_0, values = (var_6134_cast_fp16, var_6132_cast_fp16_0))[name = string("op_6137_cast_fp16")]; tensor var_6138_cast_fp16 = mul(x = var_6137_cast_fp16, y = var_878_cast_fp16)[name = string("op_6138_cast_fp16")]; tensor query_states_99_cast_fp16 = add(x = var_6131_cast_fp16, y = var_6138_cast_fp16)[name = string("query_states_99_cast_fp16")]; tensor var_6144_cast_fp16 = mul(x = var_6120_cast_fp16, y = var_869_cast_fp16)[name = string("op_6144_cast_fp16")]; tensor var_6145_split_sizes_0 = const()[name = string("op_6145_split_sizes_0"), val = tensor([64, 64])]; int32 var_6145_axis_0 = const()[name = string("op_6145_axis_0"), val = int32(-2)]; tensor var_6145_cast_fp16_0, tensor var_6145_cast_fp16_1 = split(axis = var_6145_axis_0, split_sizes = var_6145_split_sizes_0, x = var_6120_cast_fp16)[name = string("op_6145_cast_fp16")]; fp16 const_163_promoted_to_fp16 = const()[name = string("const_163_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6147_cast_fp16 = mul(x = var_6145_cast_fp16_1, y = const_163_promoted_to_fp16)[name = string("op_6147_cast_fp16")]; int32 var_6149 = const()[name = string("op_6149"), val = int32(-2)]; bool var_6150_interleave_0 = const()[name = string("op_6150_interleave_0"), val = bool(false)]; tensor var_6150_cast_fp16 = concat(axis = var_6149, interleave = var_6150_interleave_0, values = (var_6147_cast_fp16, var_6145_cast_fp16_0))[name = string("op_6150_cast_fp16")]; tensor var_6151_cast_fp16 = mul(x = var_6150_cast_fp16, y = var_878_cast_fp16)[name = string("op_6151_cast_fp16")]; tensor key_states_165_cast_fp16 = add(x = var_6144_cast_fp16, y = var_6151_cast_fp16)[name = string("key_states_165_cast_fp16")]; tensor expand_dims_192 = const()[name = string("expand_dims_192"), val = tensor([16])]; tensor expand_dims_193 = const()[name = string("expand_dims_193"), val = tensor([0])]; tensor expand_dims_195 = const()[name = string("expand_dims_195"), val = tensor([0])]; int32 concat_197_axis_0 = const()[name = string("concat_197_axis_0"), val = int32(0)]; bool concat_197_interleave_0 = const()[name = string("concat_197_interleave_0"), val = bool(false)]; tensor concat_197 = concat(axis = concat_197_axis_0, interleave = concat_197_interleave_0, values = (expand_dims_192, expand_dims_193, position_id, expand_dims_195))[name = string("concat_197")]; tensor expand_dims_196 = const()[name = string("expand_dims_196"), val = tensor([17])]; tensor concat_198_values1_0 = const()[name = string("concat_198_values1_0"), val = tensor([0])]; tensor concat_198_values3_0 = const()[name = string("concat_198_values3_0"), val = tensor([0])]; int32 concat_198_axis_0 = const()[name = string("concat_198_axis_0"), val = int32(0)]; bool concat_198_interleave_0 = const()[name = string("concat_198_interleave_0"), val = bool(false)]; tensor concat_198 = concat(axis = concat_198_axis_0, interleave = concat_198_interleave_0, values = (expand_dims_196, concat_198_values1_0, cache_position_end, concat_198_values3_0))[name = string("concat_198")]; tensor key_states_167_perm_0 = const()[name = string("key_states_167_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_17_stride_0 = const()[name = string("key_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_167_cast_fp16 = transpose(perm = key_states_167_perm_0, x = key_states_165_cast_fp16)[name = string("transpose_293")]; tensor key_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = key_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = key_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_17_squeeze_mask_0, stride = key_cache_internal_tensor_assign_17_stride_0, update = key_states_167_cast_fp16, x = coreml_update_state_198)[name = string("key_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_17_cast_fp16, input = key_cache)[name = string("coreml_update_state_200_write_state")]; tensor coreml_update_state_200 = read_state(input = key_cache)[name = string("coreml_update_state_200")]; tensor value_states_99_perm_0 = const()[name = string("value_states_99_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_17_stride_0 = const()[name = string("value_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_99_cast_fp16 = transpose(perm = value_states_99_perm_0, x = var_6127_cast_fp16)[name = string("transpose_292")]; tensor value_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = value_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = value_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_17_squeeze_mask_0, stride = value_cache_internal_tensor_assign_17_stride_0, update = value_states_99_cast_fp16, x = coreml_update_state_199)[name = string("value_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_17_cast_fp16, input = value_cache)[name = string("coreml_update_state_201_write_state")]; tensor coreml_update_state_201 = read_state(input = value_cache)[name = string("coreml_update_state_201")]; tensor var_6221_begin_0 = const()[name = string("op_6221_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6221_end_0 = const()[name = string("op_6221_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6221_end_mask_0 = const()[name = string("op_6221_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6221_cast_fp16 = slice_by_index(begin = var_6221_begin_0, end = var_6221_end_0, end_mask = var_6221_end_mask_0, x = coreml_update_state_200)[name = string("op_6221_cast_fp16")]; tensor tile_32 = const()[name = string("tile_32"), val = tensor([1, 1])]; int32 var_6224_axis_0 = const()[name = string("op_6224_axis_0"), val = int32(1)]; tensor var_6224_cast_fp16_0, tensor var_6224_cast_fp16_1 = split(axis = var_6224_axis_0, split_sizes = tile_32, x = var_6221_cast_fp16)[name = string("op_6224_cast_fp16")]; tensor var_6231_begin_0 = const()[name = string("op_6231_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6231_end_0 = const()[name = string("op_6231_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6231_end_mask_0 = const()[name = string("op_6231_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6231_cast_fp16 = slice_by_index(begin = var_6231_begin_0, end = var_6231_end_0, end_mask = var_6231_end_mask_0, x = coreml_update_state_201)[name = string("op_6231_cast_fp16")]; tensor tile_33 = const()[name = string("tile_33"), val = tensor([1, 1])]; int32 var_6234_axis_0 = const()[name = string("op_6234_axis_0"), val = int32(1)]; tensor var_6234_cast_fp16_0, tensor var_6234_cast_fp16_1 = split(axis = var_6234_axis_0, split_sizes = tile_33, x = var_6231_cast_fp16)[name = string("op_6234_cast_fp16")]; tensor var_6237_split_sizes_0 = const()[name = string("op_6237_split_sizes_0"), val = tensor([8, 8])]; int32 var_6237_axis_0 = const()[name = string("op_6237_axis_0"), val = int32(1)]; tensor var_6237_0, tensor var_6237_1 = split(axis = var_6237_axis_0, split_sizes = var_6237_split_sizes_0, x = query_states_99_cast_fp16)[name = string("op_6237")]; bool attn_weights_257_transpose_x_0 = const()[name = string("attn_weights_257_transpose_x_0"), val = bool(false)]; bool attn_weights_257_transpose_y_0 = const()[name = string("attn_weights_257_transpose_y_0"), val = bool(false)]; tensor attn_weights_257_cast_fp16 = matmul(transpose_x = attn_weights_257_transpose_x_0, transpose_y = attn_weights_257_transpose_y_0, x = var_6224_cast_fp16_0, y = var_6237_0)[name = string("attn_weights_257_cast_fp16")]; fp16 var_6240_to_fp16 = const()[name = string("op_6240_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_259_cast_fp16 = mul(x = attn_weights_257_cast_fp16, y = var_6240_to_fp16)[name = string("attn_weights_259_cast_fp16")]; tensor attn_weights_261_cast_fp16 = add(x = attn_weights_259_cast_fp16, y = attn_mask_1)[name = string("attn_weights_261_cast_fp16")]; int32 var_6244 = const()[name = string("op_6244"), val = int32(-2)]; tensor attn_weights_263_cast_fp16 = softmax(axis = var_6244, x = attn_weights_261_cast_fp16)[name = string("attn_weights_263_cast_fp16")]; bool var_6250_transpose_x_1 = const()[name = string("op_6250_transpose_x_1"), val = bool(true)]; bool var_6250_transpose_y_1 = const()[name = string("op_6250_transpose_y_1"), val = bool(false)]; tensor var_6250_cast_fp16 = matmul(transpose_x = var_6250_transpose_x_1, transpose_y = var_6250_transpose_y_1, x = attn_weights_263_cast_fp16, y = var_6234_cast_fp16_0)[name = string("op_6250_cast_fp16")]; bool attn_weights_265_transpose_x_0 = const()[name = string("attn_weights_265_transpose_x_0"), val = bool(false)]; bool attn_weights_265_transpose_y_0 = const()[name = string("attn_weights_265_transpose_y_0"), val = bool(false)]; tensor attn_weights_265_cast_fp16 = matmul(transpose_x = attn_weights_265_transpose_x_0, transpose_y = attn_weights_265_transpose_y_0, x = var_6224_cast_fp16_1, y = var_6237_1)[name = string("attn_weights_265_cast_fp16")]; fp16 var_6252_to_fp16 = const()[name = string("op_6252_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_267_cast_fp16 = mul(x = attn_weights_265_cast_fp16, y = var_6252_to_fp16)[name = string("attn_weights_267_cast_fp16")]; tensor attn_weights_269_cast_fp16 = add(x = attn_weights_267_cast_fp16, y = attn_mask_1)[name = string("attn_weights_269_cast_fp16")]; int32 var_6256 = const()[name = string("op_6256"), val = int32(-2)]; tensor attn_weights_271_cast_fp16 = softmax(axis = var_6256, x = attn_weights_269_cast_fp16)[name = string("attn_weights_271_cast_fp16")]; bool attn_output_129_transpose_x_1 = const()[name = string("attn_output_129_transpose_x_1"), val = bool(true)]; bool attn_output_129_transpose_y_1 = const()[name = string("attn_output_129_transpose_y_1"), val = bool(false)]; tensor attn_output_129_cast_fp16 = matmul(transpose_x = attn_output_129_transpose_x_1, transpose_y = attn_output_129_transpose_y_1, x = attn_weights_271_cast_fp16, y = var_6234_cast_fp16_1)[name = string("attn_output_129_cast_fp16")]; int32 var_6264 = const()[name = string("op_6264"), val = int32(1)]; bool attn_output_131_interleave_0 = const()[name = string("attn_output_131_interleave_0"), val = bool(false)]; tensor attn_output_131_cast_fp16 = concat(axis = var_6264, interleave = attn_output_131_interleave_0, values = (var_6250_cast_fp16, attn_output_129_cast_fp16))[name = string("attn_output_131_cast_fp16")]; tensor var_6268_perm_0 = const()[name = string("op_6268_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_203x = const()[name = string("concat_203x"), val = tensor([1, 2048, 1, -1])]; tensor var_6268_cast_fp16 = transpose(perm = var_6268_perm_0, x = attn_output_131_cast_fp16)[name = string("transpose_291")]; tensor attn_output_135_cast_fp16 = reshape(shape = concat_203x, x = var_6268_cast_fp16)[name = string("attn_output_135_cast_fp16")]; tensor hidden_states_163_strides_0 = const()[name = string("hidden_states_163_strides_0"), val = tensor([1, 1])]; string hidden_states_163_pad_type_0 = const()[name = string("hidden_states_163_pad_type_0"), val = string("valid")]; tensor hidden_states_163_pad_0 = const()[name = string("hidden_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_163_dilations_0 = const()[name = string("hidden_states_163_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_163_groups_0 = const()[name = string("hidden_states_163_groups_0"), val = int32(1)]; tensor hidden_states_163_cast_fp16 = conv(dilations = hidden_states_163_dilations_0, groups = hidden_states_163_groups_0, pad = hidden_states_163_pad_0, pad_type = hidden_states_163_pad_type_0, strides = hidden_states_163_strides_0, weight = layers_16_self_attn_o_proj_weight_cast_fp16, x = attn_output_135_cast_fp16)[name = string("hidden_states_163_cast_fp16")]; tensor hidden_states_165_cast_fp16 = add(x = hidden_states_159_cast_fp16, y = hidden_states_163_cast_fp16)[name = string("hidden_states_165_cast_fp16")]; fp16 const_168_promoted_to_fp16 = const()[name = string("const_168_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6301_cast_fp16 = mul(x = hidden_states_165_cast_fp16, y = const_168_promoted_to_fp16)[name = string("op_6301_cast_fp16")]; int32 var_6299 = const()[name = string("op_6299"), val = int32(1)]; bool doubled_133_interleave_0 = const()[name = string("doubled_133_interleave_0"), val = bool(false)]; tensor doubled_133_cast_fp16 = concat(axis = var_6299, interleave = doubled_133_interleave_0, values = (hidden_states_165_cast_fp16, var_6301_cast_fp16))[name = string("doubled_133_cast_fp16")]; tensor out_67_axes_0 = const()[name = string("out_67_axes_0"), val = tensor([1])]; tensor out_67_gamma_0_to_fp16 = const()[name = string("out_67_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441663488)))]; fp16 var_6311_to_fp16 = const()[name = string("op_6311_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_67_cast_fp16 = layer_norm(axes = out_67_axes_0, epsilon = var_6311_to_fp16, gamma = out_67_gamma_0_to_fp16, x = doubled_133_cast_fp16)[name = string("out_67_cast_fp16")]; tensor var_6322_split_sizes_0 = const()[name = string("op_6322_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6322_axis_0 = const()[name = string("op_6322_axis_0"), val = int32(1)]; tensor var_6322_cast_fp16_0, tensor var_6322_cast_fp16_1 = split(axis = var_6322_axis_0, split_sizes = var_6322_split_sizes_0, x = out_67_cast_fp16)[name = string("op_6322_cast_fp16")]; tensor layers_16_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441671744)))]; tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; tensor input_33_cast_fp16 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = layers_16_mlp_gate_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("input_33_cast_fp16")]; tensor var_6339_cast_fp16 = silu(x = input_33_cast_fp16)[name = string("op_6339_cast_fp16")]; tensor layers_16_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1466837632)))]; tensor var_6345_strides_0 = const()[name = string("op_6345_strides_0"), val = tensor([1, 1])]; string var_6345_pad_type_0 = const()[name = string("op_6345_pad_type_0"), val = string("valid")]; tensor var_6345_pad_0 = const()[name = string("op_6345_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6345_dilations_0 = const()[name = string("op_6345_dilations_0"), val = tensor([1, 1])]; int32 var_6345_groups_0 = const()[name = string("op_6345_groups_0"), val = int32(1)]; tensor var_6345_cast_fp16 = conv(dilations = var_6345_dilations_0, groups = var_6345_groups_0, pad = var_6345_pad_0, pad_type = var_6345_pad_type_0, strides = var_6345_strides_0, weight = layers_16_mlp_up_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("op_6345_cast_fp16")]; tensor x_169_cast_fp16 = mul(x = var_6339_cast_fp16, y = var_6345_cast_fp16)[name = string("x_169_cast_fp16")]; tensor hidden_states_167_strides_0 = const()[name = string("hidden_states_167_strides_0"), val = tensor([1, 1])]; string hidden_states_167_pad_type_0 = const()[name = string("hidden_states_167_pad_type_0"), val = string("valid")]; tensor hidden_states_167_pad_0 = const()[name = string("hidden_states_167_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_167_dilations_0 = const()[name = string("hidden_states_167_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_167_groups_0 = const()[name = string("hidden_states_167_groups_0"), val = int32(1)]; tensor hidden_states_167_cast_fp16 = conv(dilations = hidden_states_167_dilations_0, groups = hidden_states_167_groups_0, pad = hidden_states_167_pad_0, pad_type = hidden_states_167_pad_type_0, strides = hidden_states_167_strides_0, weight = layers_16_mlp_down_proj_weight_cast_fp16, x = x_169_cast_fp16)[name = string("hidden_states_167_cast_fp16")]; tensor hidden_states_169_cast_fp16 = add(x = hidden_states_165_cast_fp16, y = hidden_states_167_cast_fp16)[name = string("hidden_states_169_cast_fp16")]; fp16 const_170_promoted_to_fp16 = const()[name = string("const_170_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6363_cast_fp16 = mul(x = hidden_states_169_cast_fp16, y = const_170_promoted_to_fp16)[name = string("op_6363_cast_fp16")]; int32 var_6361 = const()[name = string("op_6361"), val = int32(1)]; bool doubled_137_interleave_0 = const()[name = string("doubled_137_interleave_0"), val = bool(false)]; tensor doubled_137_cast_fp16 = concat(axis = var_6361, interleave = doubled_137_interleave_0, values = (hidden_states_169_cast_fp16, var_6363_cast_fp16))[name = string("doubled_137_cast_fp16")]; tensor out_69_axes_0 = const()[name = string("out_69_axes_0"), val = tensor([1])]; tensor out_69_gamma_0_to_fp16 = const()[name = string("out_69_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492003520)))]; fp16 var_6373_to_fp16 = const()[name = string("op_6373_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_69_cast_fp16 = layer_norm(axes = out_69_axes_0, epsilon = var_6373_to_fp16, gamma = out_69_gamma_0_to_fp16, x = doubled_137_cast_fp16)[name = string("out_69_cast_fp16")]; tensor var_6384_split_sizes_0 = const()[name = string("op_6384_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6384_axis_0 = const()[name = string("op_6384_axis_0"), val = int32(1)]; tensor var_6384_cast_fp16_0, tensor var_6384_cast_fp16_1 = split(axis = var_6384_axis_0, split_sizes = var_6384_split_sizes_0, x = out_69_cast_fp16)[name = string("op_6384_cast_fp16")]; tensor query_states_103_strides_0 = const()[name = string("query_states_103_strides_0"), val = tensor([1, 1])]; string query_states_103_pad_type_0 = const()[name = string("query_states_103_pad_type_0"), val = string("valid")]; tensor query_states_103_pad_0 = const()[name = string("query_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_103_dilations_0 = const()[name = string("query_states_103_dilations_0"), val = tensor([1, 1])]; int32 query_states_103_groups_0 = const()[name = string("query_states_103_groups_0"), val = int32(1)]; tensor query_states_103_cast_fp16 = conv(dilations = query_states_103_dilations_0, groups = query_states_103_groups_0, pad = query_states_103_pad_0, pad_type = query_states_103_pad_type_0, strides = query_states_103_strides_0, weight = layers_17_self_attn_q_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("query_states_103_cast_fp16")]; tensor key_states_171_strides_0 = const()[name = string("key_states_171_strides_0"), val = tensor([1, 1])]; string key_states_171_pad_type_0 = const()[name = string("key_states_171_pad_type_0"), val = string("valid")]; tensor key_states_171_pad_0 = const()[name = string("key_states_171_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_171_dilations_0 = const()[name = string("key_states_171_dilations_0"), val = tensor([1, 1])]; int32 key_states_171_groups_0 = const()[name = string("key_states_171_groups_0"), val = int32(1)]; tensor key_states_171_cast_fp16 = conv(dilations = key_states_171_dilations_0, groups = key_states_171_groups_0, pad = key_states_171_pad_0, pad_type = key_states_171_pad_type_0, strides = key_states_171_strides_0, weight = layers_17_self_attn_k_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("key_states_171_cast_fp16")]; tensor value_states_103_strides_0 = const()[name = string("value_states_103_strides_0"), val = tensor([1, 1])]; string value_states_103_pad_type_0 = const()[name = string("value_states_103_pad_type_0"), val = string("valid")]; tensor value_states_103_pad_0 = const()[name = string("value_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_103_dilations_0 = const()[name = string("value_states_103_dilations_0"), val = tensor([1, 1])]; int32 value_states_103_groups_0 = const()[name = string("value_states_103_groups_0"), val = int32(1)]; tensor value_states_103_cast_fp16 = conv(dilations = value_states_103_dilations_0, groups = value_states_103_groups_0, pad = value_states_103_pad_0, pad_type = value_states_103_pad_type_0, strides = value_states_103_strides_0, weight = layers_17_self_attn_v_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("value_states_103_cast_fp16")]; tensor concat_204x = const()[name = string("concat_204x"), val = tensor([1, 16, 128, -1])]; tensor x_171_cast_fp16 = reshape(shape = concat_204x, x = query_states_103_cast_fp16)[name = string("x_171_cast_fp16")]; tensor concat_205x = const()[name = string("concat_205x"), val = tensor([1, 2, 128, -1])]; tensor var_6441_cast_fp16 = reshape(shape = concat_205x, x = key_states_171_cast_fp16)[name = string("op_6441_cast_fp16")]; tensor concat_206x = const()[name = string("concat_206x"), val = tensor([1, 2, 128, -1])]; tensor var_6448_cast_fp16 = reshape(shape = concat_206x, x = value_states_103_cast_fp16)[name = string("op_6448_cast_fp16")]; tensor var_6452_cast_fp16 = mul(x = x_171_cast_fp16, y = var_869_cast_fp16)[name = string("op_6452_cast_fp16")]; tensor var_6453_split_sizes_0 = const()[name = string("op_6453_split_sizes_0"), val = tensor([64, 64])]; int32 var_6453_axis_0 = const()[name = string("op_6453_axis_0"), val = int32(-2)]; tensor var_6453_cast_fp16_0, tensor var_6453_cast_fp16_1 = split(axis = var_6453_axis_0, split_sizes = var_6453_split_sizes_0, x = x_171_cast_fp16)[name = string("op_6453_cast_fp16")]; fp16 const_172_promoted_to_fp16 = const()[name = string("const_172_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6455_cast_fp16 = mul(x = var_6453_cast_fp16_1, y = const_172_promoted_to_fp16)[name = string("op_6455_cast_fp16")]; int32 var_6457 = const()[name = string("op_6457"), val = int32(-2)]; bool var_6458_interleave_0 = const()[name = string("op_6458_interleave_0"), val = bool(false)]; tensor var_6458_cast_fp16 = concat(axis = var_6457, interleave = var_6458_interleave_0, values = (var_6455_cast_fp16, var_6453_cast_fp16_0))[name = string("op_6458_cast_fp16")]; tensor var_6459_cast_fp16 = mul(x = var_6458_cast_fp16, y = var_878_cast_fp16)[name = string("op_6459_cast_fp16")]; tensor query_states_105_cast_fp16 = add(x = var_6452_cast_fp16, y = var_6459_cast_fp16)[name = string("query_states_105_cast_fp16")]; tensor var_6465_cast_fp16 = mul(x = var_6441_cast_fp16, y = var_869_cast_fp16)[name = string("op_6465_cast_fp16")]; tensor var_6466_split_sizes_0 = const()[name = string("op_6466_split_sizes_0"), val = tensor([64, 64])]; int32 var_6466_axis_0 = const()[name = string("op_6466_axis_0"), val = int32(-2)]; tensor var_6466_cast_fp16_0, tensor var_6466_cast_fp16_1 = split(axis = var_6466_axis_0, split_sizes = var_6466_split_sizes_0, x = var_6441_cast_fp16)[name = string("op_6466_cast_fp16")]; fp16 const_173_promoted_to_fp16 = const()[name = string("const_173_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6468_cast_fp16 = mul(x = var_6466_cast_fp16_1, y = const_173_promoted_to_fp16)[name = string("op_6468_cast_fp16")]; int32 var_6470 = const()[name = string("op_6470"), val = int32(-2)]; bool var_6471_interleave_0 = const()[name = string("op_6471_interleave_0"), val = bool(false)]; tensor var_6471_cast_fp16 = concat(axis = var_6470, interleave = var_6471_interleave_0, values = (var_6468_cast_fp16, var_6466_cast_fp16_0))[name = string("op_6471_cast_fp16")]; tensor var_6472_cast_fp16 = mul(x = var_6471_cast_fp16, y = var_878_cast_fp16)[name = string("op_6472_cast_fp16")]; tensor key_states_175_cast_fp16 = add(x = var_6465_cast_fp16, y = var_6472_cast_fp16)[name = string("key_states_175_cast_fp16")]; tensor expand_dims_204 = const()[name = string("expand_dims_204"), val = tensor([17])]; tensor expand_dims_205 = const()[name = string("expand_dims_205"), val = tensor([0])]; tensor expand_dims_207 = const()[name = string("expand_dims_207"), val = tensor([0])]; int32 concat_209_axis_0 = const()[name = string("concat_209_axis_0"), val = int32(0)]; bool concat_209_interleave_0 = const()[name = string("concat_209_interleave_0"), val = bool(false)]; tensor concat_209 = concat(axis = concat_209_axis_0, interleave = concat_209_interleave_0, values = (expand_dims_204, expand_dims_205, position_id, expand_dims_207))[name = string("concat_209")]; tensor expand_dims_208 = const()[name = string("expand_dims_208"), val = tensor([18])]; tensor concat_210_values1_0 = const()[name = string("concat_210_values1_0"), val = tensor([0])]; tensor concat_210_values3_0 = const()[name = string("concat_210_values3_0"), val = tensor([0])]; int32 concat_210_axis_0 = const()[name = string("concat_210_axis_0"), val = int32(0)]; bool concat_210_interleave_0 = const()[name = string("concat_210_interleave_0"), val = bool(false)]; tensor concat_210 = concat(axis = concat_210_axis_0, interleave = concat_210_interleave_0, values = (expand_dims_208, concat_210_values1_0, cache_position_end, concat_210_values3_0))[name = string("concat_210")]; tensor key_states_177_perm_0 = const()[name = string("key_states_177_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_18_stride_0 = const()[name = string("key_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_177_cast_fp16 = transpose(perm = key_states_177_perm_0, x = key_states_175_cast_fp16)[name = string("transpose_290")]; tensor key_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = key_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = key_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_18_squeeze_mask_0, stride = key_cache_internal_tensor_assign_18_stride_0, update = key_states_177_cast_fp16, x = coreml_update_state_200)[name = string("key_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_18_cast_fp16, input = key_cache)[name = string("coreml_update_state_202_write_state")]; tensor coreml_update_state_202 = read_state(input = key_cache)[name = string("coreml_update_state_202")]; tensor value_states_105_perm_0 = const()[name = string("value_states_105_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_18_stride_0 = const()[name = string("value_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_105_cast_fp16 = transpose(perm = value_states_105_perm_0, x = var_6448_cast_fp16)[name = string("transpose_289")]; tensor value_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = value_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = value_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_18_squeeze_mask_0, stride = value_cache_internal_tensor_assign_18_stride_0, update = value_states_105_cast_fp16, x = coreml_update_state_201)[name = string("value_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_18_cast_fp16, input = value_cache)[name = string("coreml_update_state_203_write_state")]; tensor coreml_update_state_203 = read_state(input = value_cache)[name = string("coreml_update_state_203")]; tensor var_6542_begin_0 = const()[name = string("op_6542_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6542_end_0 = const()[name = string("op_6542_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6542_end_mask_0 = const()[name = string("op_6542_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6542_cast_fp16 = slice_by_index(begin = var_6542_begin_0, end = var_6542_end_0, end_mask = var_6542_end_mask_0, x = coreml_update_state_202)[name = string("op_6542_cast_fp16")]; tensor tile_34 = const()[name = string("tile_34"), val = tensor([1, 1])]; int32 var_6545_axis_0 = const()[name = string("op_6545_axis_0"), val = int32(1)]; tensor var_6545_cast_fp16_0, tensor var_6545_cast_fp16_1 = split(axis = var_6545_axis_0, split_sizes = tile_34, x = var_6542_cast_fp16)[name = string("op_6545_cast_fp16")]; tensor var_6552_begin_0 = const()[name = string("op_6552_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6552_end_0 = const()[name = string("op_6552_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6552_end_mask_0 = const()[name = string("op_6552_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6552_cast_fp16 = slice_by_index(begin = var_6552_begin_0, end = var_6552_end_0, end_mask = var_6552_end_mask_0, x = coreml_update_state_203)[name = string("op_6552_cast_fp16")]; tensor tile_35 = const()[name = string("tile_35"), val = tensor([1, 1])]; int32 var_6555_axis_0 = const()[name = string("op_6555_axis_0"), val = int32(1)]; tensor var_6555_cast_fp16_0, tensor var_6555_cast_fp16_1 = split(axis = var_6555_axis_0, split_sizes = tile_35, x = var_6552_cast_fp16)[name = string("op_6555_cast_fp16")]; tensor var_6558_split_sizes_0 = const()[name = string("op_6558_split_sizes_0"), val = tensor([8, 8])]; int32 var_6558_axis_0 = const()[name = string("op_6558_axis_0"), val = int32(1)]; tensor var_6558_0, tensor var_6558_1 = split(axis = var_6558_axis_0, split_sizes = var_6558_split_sizes_0, x = query_states_105_cast_fp16)[name = string("op_6558")]; bool attn_weights_273_transpose_x_0 = const()[name = string("attn_weights_273_transpose_x_0"), val = bool(false)]; bool attn_weights_273_transpose_y_0 = const()[name = string("attn_weights_273_transpose_y_0"), val = bool(false)]; tensor attn_weights_273_cast_fp16 = matmul(transpose_x = attn_weights_273_transpose_x_0, transpose_y = attn_weights_273_transpose_y_0, x = var_6545_cast_fp16_0, y = var_6558_0)[name = string("attn_weights_273_cast_fp16")]; fp16 var_6561_to_fp16 = const()[name = string("op_6561_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_275_cast_fp16 = mul(x = attn_weights_273_cast_fp16, y = var_6561_to_fp16)[name = string("attn_weights_275_cast_fp16")]; tensor attn_weights_277_cast_fp16 = add(x = attn_weights_275_cast_fp16, y = attn_mask_1)[name = string("attn_weights_277_cast_fp16")]; int32 var_6565 = const()[name = string("op_6565"), val = int32(-2)]; tensor attn_weights_279_cast_fp16 = softmax(axis = var_6565, x = attn_weights_277_cast_fp16)[name = string("attn_weights_279_cast_fp16")]; bool var_6571_transpose_x_1 = const()[name = string("op_6571_transpose_x_1"), val = bool(true)]; bool var_6571_transpose_y_1 = const()[name = string("op_6571_transpose_y_1"), val = bool(false)]; tensor var_6571_cast_fp16 = matmul(transpose_x = var_6571_transpose_x_1, transpose_y = var_6571_transpose_y_1, x = attn_weights_279_cast_fp16, y = var_6555_cast_fp16_0)[name = string("op_6571_cast_fp16")]; bool attn_weights_281_transpose_x_0 = const()[name = string("attn_weights_281_transpose_x_0"), val = bool(false)]; bool attn_weights_281_transpose_y_0 = const()[name = string("attn_weights_281_transpose_y_0"), val = bool(false)]; tensor attn_weights_281_cast_fp16 = matmul(transpose_x = attn_weights_281_transpose_x_0, transpose_y = attn_weights_281_transpose_y_0, x = var_6545_cast_fp16_1, y = var_6558_1)[name = string("attn_weights_281_cast_fp16")]; fp16 var_6573_to_fp16 = const()[name = string("op_6573_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_283_cast_fp16 = mul(x = attn_weights_281_cast_fp16, y = var_6573_to_fp16)[name = string("attn_weights_283_cast_fp16")]; tensor attn_weights_285_cast_fp16 = add(x = attn_weights_283_cast_fp16, y = attn_mask_1)[name = string("attn_weights_285_cast_fp16")]; int32 var_6577 = const()[name = string("op_6577"), val = int32(-2)]; tensor attn_weights_287_cast_fp16 = softmax(axis = var_6577, x = attn_weights_285_cast_fp16)[name = string("attn_weights_287_cast_fp16")]; bool attn_output_137_transpose_x_1 = const()[name = string("attn_output_137_transpose_x_1"), val = bool(true)]; bool attn_output_137_transpose_y_1 = const()[name = string("attn_output_137_transpose_y_1"), val = bool(false)]; tensor attn_output_137_cast_fp16 = matmul(transpose_x = attn_output_137_transpose_x_1, transpose_y = attn_output_137_transpose_y_1, x = attn_weights_287_cast_fp16, y = var_6555_cast_fp16_1)[name = string("attn_output_137_cast_fp16")]; int32 var_6585 = const()[name = string("op_6585"), val = int32(1)]; bool attn_output_139_interleave_0 = const()[name = string("attn_output_139_interleave_0"), val = bool(false)]; tensor attn_output_139_cast_fp16 = concat(axis = var_6585, interleave = attn_output_139_interleave_0, values = (var_6571_cast_fp16, attn_output_137_cast_fp16))[name = string("attn_output_139_cast_fp16")]; tensor var_6589_perm_0 = const()[name = string("op_6589_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_215x = const()[name = string("concat_215x"), val = tensor([1, 2048, 1, -1])]; tensor var_6589_cast_fp16 = transpose(perm = var_6589_perm_0, x = attn_output_139_cast_fp16)[name = string("transpose_288")]; tensor attn_output_143_cast_fp16 = reshape(shape = concat_215x, x = var_6589_cast_fp16)[name = string("attn_output_143_cast_fp16")]; tensor hidden_states_173_strides_0 = const()[name = string("hidden_states_173_strides_0"), val = tensor([1, 1])]; string hidden_states_173_pad_type_0 = const()[name = string("hidden_states_173_pad_type_0"), val = string("valid")]; tensor hidden_states_173_pad_0 = const()[name = string("hidden_states_173_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_173_dilations_0 = const()[name = string("hidden_states_173_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_173_groups_0 = const()[name = string("hidden_states_173_groups_0"), val = int32(1)]; tensor hidden_states_173_cast_fp16 = conv(dilations = hidden_states_173_dilations_0, groups = hidden_states_173_groups_0, pad = hidden_states_173_pad_0, pad_type = hidden_states_173_pad_type_0, strides = hidden_states_173_strides_0, weight = layers_17_self_attn_o_proj_weight_cast_fp16, x = attn_output_143_cast_fp16)[name = string("hidden_states_173_cast_fp16")]; tensor hidden_states_175_cast_fp16 = add(x = hidden_states_169_cast_fp16, y = hidden_states_173_cast_fp16)[name = string("hidden_states_175_cast_fp16")]; fp16 const_178_promoted_to_fp16 = const()[name = string("const_178_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6622_cast_fp16 = mul(x = hidden_states_175_cast_fp16, y = const_178_promoted_to_fp16)[name = string("op_6622_cast_fp16")]; int32 var_6620 = const()[name = string("op_6620"), val = int32(1)]; bool doubled_141_interleave_0 = const()[name = string("doubled_141_interleave_0"), val = bool(false)]; tensor doubled_141_cast_fp16 = concat(axis = var_6620, interleave = doubled_141_interleave_0, values = (hidden_states_175_cast_fp16, var_6622_cast_fp16))[name = string("doubled_141_cast_fp16")]; tensor out_71_axes_0 = const()[name = string("out_71_axes_0"), val = tensor([1])]; tensor out_71_gamma_0_to_fp16 = const()[name = string("out_71_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492011776)))]; fp16 var_6632_to_fp16 = const()[name = string("op_6632_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_71_cast_fp16 = layer_norm(axes = out_71_axes_0, epsilon = var_6632_to_fp16, gamma = out_71_gamma_0_to_fp16, x = doubled_141_cast_fp16)[name = string("out_71_cast_fp16")]; tensor var_6643_split_sizes_0 = const()[name = string("op_6643_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6643_axis_0 = const()[name = string("op_6643_axis_0"), val = int32(1)]; tensor var_6643_cast_fp16_0, tensor var_6643_cast_fp16_1 = split(axis = var_6643_axis_0, split_sizes = var_6643_split_sizes_0, x = out_71_cast_fp16)[name = string("op_6643_cast_fp16")]; tensor input_35_strides_0 = const()[name = string("input_35_strides_0"), val = tensor([1, 1])]; string input_35_pad_type_0 = const()[name = string("input_35_pad_type_0"), val = string("valid")]; tensor input_35_pad_0 = const()[name = string("input_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_35_dilations_0 = const()[name = string("input_35_dilations_0"), val = tensor([1, 1])]; int32 input_35_groups_0 = const()[name = string("input_35_groups_0"), val = int32(1)]; tensor input_35_cast_fp16 = conv(dilations = input_35_dilations_0, groups = input_35_groups_0, pad = input_35_pad_0, pad_type = input_35_pad_type_0, strides = input_35_strides_0, weight = layers_17_mlp_gate_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("input_35_cast_fp16")]; tensor var_6660_cast_fp16 = silu(x = input_35_cast_fp16)[name = string("op_6660_cast_fp16")]; tensor var_6666_strides_0 = const()[name = string("op_6666_strides_0"), val = tensor([1, 1])]; string var_6666_pad_type_0 = const()[name = string("op_6666_pad_type_0"), val = string("valid")]; tensor var_6666_pad_0 = const()[name = string("op_6666_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6666_dilations_0 = const()[name = string("op_6666_dilations_0"), val = tensor([1, 1])]; int32 var_6666_groups_0 = const()[name = string("op_6666_groups_0"), val = int32(1)]; tensor var_6666_cast_fp16 = conv(dilations = var_6666_dilations_0, groups = var_6666_groups_0, pad = var_6666_pad_0, pad_type = var_6666_pad_type_0, strides = var_6666_strides_0, weight = layers_17_mlp_up_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("op_6666_cast_fp16")]; tensor x_179_cast_fp16 = mul(x = var_6660_cast_fp16, y = var_6666_cast_fp16)[name = string("x_179_cast_fp16")]; tensor hidden_states_177_strides_0 = const()[name = string("hidden_states_177_strides_0"), val = tensor([1, 1])]; string hidden_states_177_pad_type_0 = const()[name = string("hidden_states_177_pad_type_0"), val = string("valid")]; tensor hidden_states_177_pad_0 = const()[name = string("hidden_states_177_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_177_dilations_0 = const()[name = string("hidden_states_177_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_177_groups_0 = const()[name = string("hidden_states_177_groups_0"), val = int32(1)]; tensor hidden_states_177_cast_fp16 = conv(dilations = hidden_states_177_dilations_0, groups = hidden_states_177_groups_0, pad = hidden_states_177_pad_0, pad_type = hidden_states_177_pad_type_0, strides = hidden_states_177_strides_0, weight = layers_17_mlp_down_proj_weight_cast_fp16, x = x_179_cast_fp16)[name = string("hidden_states_177_cast_fp16")]; tensor hidden_states_179_cast_fp16 = add(x = hidden_states_175_cast_fp16, y = hidden_states_177_cast_fp16)[name = string("hidden_states_179_cast_fp16")]; fp16 const_180_promoted_to_fp16 = const()[name = string("const_180_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6684_cast_fp16 = mul(x = hidden_states_179_cast_fp16, y = const_180_promoted_to_fp16)[name = string("op_6684_cast_fp16")]; int32 var_6682 = const()[name = string("op_6682"), val = int32(1)]; bool doubled_145_interleave_0 = const()[name = string("doubled_145_interleave_0"), val = bool(false)]; tensor doubled_145_cast_fp16 = concat(axis = var_6682, interleave = doubled_145_interleave_0, values = (hidden_states_179_cast_fp16, var_6684_cast_fp16))[name = string("doubled_145_cast_fp16")]; tensor out_73_axes_0 = const()[name = string("out_73_axes_0"), val = tensor([1])]; tensor out_73_gamma_0_to_fp16 = const()[name = string("out_73_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492020032)))]; fp16 var_6694_to_fp16 = const()[name = string("op_6694_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_73_cast_fp16 = layer_norm(axes = out_73_axes_0, epsilon = var_6694_to_fp16, gamma = out_73_gamma_0_to_fp16, x = doubled_145_cast_fp16)[name = string("out_73_cast_fp16")]; tensor var_6705_split_sizes_0 = const()[name = string("op_6705_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6705_axis_0 = const()[name = string("op_6705_axis_0"), val = int32(1)]; tensor var_6705_cast_fp16_0, tensor var_6705_cast_fp16_1 = split(axis = var_6705_axis_0, split_sizes = var_6705_split_sizes_0, x = out_73_cast_fp16)[name = string("op_6705_cast_fp16")]; tensor query_states_109_strides_0 = const()[name = string("query_states_109_strides_0"), val = tensor([1, 1])]; string query_states_109_pad_type_0 = const()[name = string("query_states_109_pad_type_0"), val = string("valid")]; tensor query_states_109_pad_0 = const()[name = string("query_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_109_dilations_0 = const()[name = string("query_states_109_dilations_0"), val = tensor([1, 1])]; int32 query_states_109_groups_0 = const()[name = string("query_states_109_groups_0"), val = int32(1)]; tensor query_states_109_cast_fp16 = conv(dilations = query_states_109_dilations_0, groups = query_states_109_groups_0, pad = query_states_109_pad_0, pad_type = query_states_109_pad_type_0, strides = query_states_109_strides_0, weight = layers_18_self_attn_q_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("query_states_109_cast_fp16")]; tensor key_states_181_strides_0 = const()[name = string("key_states_181_strides_0"), val = tensor([1, 1])]; string key_states_181_pad_type_0 = const()[name = string("key_states_181_pad_type_0"), val = string("valid")]; tensor key_states_181_pad_0 = const()[name = string("key_states_181_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_181_dilations_0 = const()[name = string("key_states_181_dilations_0"), val = tensor([1, 1])]; int32 key_states_181_groups_0 = const()[name = string("key_states_181_groups_0"), val = int32(1)]; tensor key_states_181_cast_fp16 = conv(dilations = key_states_181_dilations_0, groups = key_states_181_groups_0, pad = key_states_181_pad_0, pad_type = key_states_181_pad_type_0, strides = key_states_181_strides_0, weight = layers_18_self_attn_k_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("key_states_181_cast_fp16")]; tensor value_states_109_strides_0 = const()[name = string("value_states_109_strides_0"), val = tensor([1, 1])]; string value_states_109_pad_type_0 = const()[name = string("value_states_109_pad_type_0"), val = string("valid")]; tensor value_states_109_pad_0 = const()[name = string("value_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_109_dilations_0 = const()[name = string("value_states_109_dilations_0"), val = tensor([1, 1])]; int32 value_states_109_groups_0 = const()[name = string("value_states_109_groups_0"), val = int32(1)]; tensor value_states_109_cast_fp16 = conv(dilations = value_states_109_dilations_0, groups = value_states_109_groups_0, pad = value_states_109_pad_0, pad_type = value_states_109_pad_type_0, strides = value_states_109_strides_0, weight = layers_18_self_attn_v_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("value_states_109_cast_fp16")]; tensor concat_216x = const()[name = string("concat_216x"), val = tensor([1, 16, 128, -1])]; tensor x_181_cast_fp16 = reshape(shape = concat_216x, x = query_states_109_cast_fp16)[name = string("x_181_cast_fp16")]; tensor concat_217x = const()[name = string("concat_217x"), val = tensor([1, 2, 128, -1])]; tensor var_6762_cast_fp16 = reshape(shape = concat_217x, x = key_states_181_cast_fp16)[name = string("op_6762_cast_fp16")]; tensor concat_218x = const()[name = string("concat_218x"), val = tensor([1, 2, 128, -1])]; tensor var_6769_cast_fp16 = reshape(shape = concat_218x, x = value_states_109_cast_fp16)[name = string("op_6769_cast_fp16")]; tensor var_6773_cast_fp16 = mul(x = x_181_cast_fp16, y = var_869_cast_fp16)[name = string("op_6773_cast_fp16")]; tensor var_6774_split_sizes_0 = const()[name = string("op_6774_split_sizes_0"), val = tensor([64, 64])]; int32 var_6774_axis_0 = const()[name = string("op_6774_axis_0"), val = int32(-2)]; tensor var_6774_cast_fp16_0, tensor var_6774_cast_fp16_1 = split(axis = var_6774_axis_0, split_sizes = var_6774_split_sizes_0, x = x_181_cast_fp16)[name = string("op_6774_cast_fp16")]; fp16 const_182_promoted_to_fp16 = const()[name = string("const_182_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6776_cast_fp16 = mul(x = var_6774_cast_fp16_1, y = const_182_promoted_to_fp16)[name = string("op_6776_cast_fp16")]; int32 var_6778 = const()[name = string("op_6778"), val = int32(-2)]; bool var_6779_interleave_0 = const()[name = string("op_6779_interleave_0"), val = bool(false)]; tensor var_6779_cast_fp16 = concat(axis = var_6778, interleave = var_6779_interleave_0, values = (var_6776_cast_fp16, var_6774_cast_fp16_0))[name = string("op_6779_cast_fp16")]; tensor var_6780_cast_fp16 = mul(x = var_6779_cast_fp16, y = var_878_cast_fp16)[name = string("op_6780_cast_fp16")]; tensor query_states_111_cast_fp16 = add(x = var_6773_cast_fp16, y = var_6780_cast_fp16)[name = string("query_states_111_cast_fp16")]; tensor var_6786_cast_fp16 = mul(x = var_6762_cast_fp16, y = var_869_cast_fp16)[name = string("op_6786_cast_fp16")]; tensor var_6787_split_sizes_0 = const()[name = string("op_6787_split_sizes_0"), val = tensor([64, 64])]; int32 var_6787_axis_0 = const()[name = string("op_6787_axis_0"), val = int32(-2)]; tensor var_6787_cast_fp16_0, tensor var_6787_cast_fp16_1 = split(axis = var_6787_axis_0, split_sizes = var_6787_split_sizes_0, x = var_6762_cast_fp16)[name = string("op_6787_cast_fp16")]; fp16 const_183_promoted_to_fp16 = const()[name = string("const_183_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6789_cast_fp16 = mul(x = var_6787_cast_fp16_1, y = const_183_promoted_to_fp16)[name = string("op_6789_cast_fp16")]; int32 var_6791 = const()[name = string("op_6791"), val = int32(-2)]; bool var_6792_interleave_0 = const()[name = string("op_6792_interleave_0"), val = bool(false)]; tensor var_6792_cast_fp16 = concat(axis = var_6791, interleave = var_6792_interleave_0, values = (var_6789_cast_fp16, var_6787_cast_fp16_0))[name = string("op_6792_cast_fp16")]; tensor var_6793_cast_fp16 = mul(x = var_6792_cast_fp16, y = var_878_cast_fp16)[name = string("op_6793_cast_fp16")]; tensor key_states_185_cast_fp16 = add(x = var_6786_cast_fp16, y = var_6793_cast_fp16)[name = string("key_states_185_cast_fp16")]; tensor expand_dims_216 = const()[name = string("expand_dims_216"), val = tensor([18])]; tensor expand_dims_217 = const()[name = string("expand_dims_217"), val = tensor([0])]; tensor expand_dims_219 = const()[name = string("expand_dims_219"), val = tensor([0])]; int32 concat_221_axis_0 = const()[name = string("concat_221_axis_0"), val = int32(0)]; bool concat_221_interleave_0 = const()[name = string("concat_221_interleave_0"), val = bool(false)]; tensor concat_221 = concat(axis = concat_221_axis_0, interleave = concat_221_interleave_0, values = (expand_dims_216, expand_dims_217, position_id, expand_dims_219))[name = string("concat_221")]; tensor expand_dims_220 = const()[name = string("expand_dims_220"), val = tensor([19])]; tensor concat_222_values1_0 = const()[name = string("concat_222_values1_0"), val = tensor([0])]; tensor concat_222_values3_0 = const()[name = string("concat_222_values3_0"), val = tensor([0])]; int32 concat_222_axis_0 = const()[name = string("concat_222_axis_0"), val = int32(0)]; bool concat_222_interleave_0 = const()[name = string("concat_222_interleave_0"), val = bool(false)]; tensor concat_222 = concat(axis = concat_222_axis_0, interleave = concat_222_interleave_0, values = (expand_dims_220, concat_222_values1_0, cache_position_end, concat_222_values3_0))[name = string("concat_222")]; tensor key_states_187_perm_0 = const()[name = string("key_states_187_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_19_stride_0 = const()[name = string("key_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_187_cast_fp16 = transpose(perm = key_states_187_perm_0, x = key_states_185_cast_fp16)[name = string("transpose_287")]; tensor key_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = key_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = key_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_19_squeeze_mask_0, stride = key_cache_internal_tensor_assign_19_stride_0, update = key_states_187_cast_fp16, x = coreml_update_state_202)[name = string("key_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_19_cast_fp16, input = key_cache)[name = string("coreml_update_state_204_write_state")]; tensor coreml_update_state_204 = read_state(input = key_cache)[name = string("coreml_update_state_204")]; tensor value_states_111_perm_0 = const()[name = string("value_states_111_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_19_stride_0 = const()[name = string("value_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_111_cast_fp16 = transpose(perm = value_states_111_perm_0, x = var_6769_cast_fp16)[name = string("transpose_286")]; tensor value_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = value_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = value_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_19_squeeze_mask_0, stride = value_cache_internal_tensor_assign_19_stride_0, update = value_states_111_cast_fp16, x = coreml_update_state_203)[name = string("value_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_19_cast_fp16, input = value_cache)[name = string("coreml_update_state_205_write_state")]; tensor coreml_update_state_205 = read_state(input = value_cache)[name = string("coreml_update_state_205")]; tensor var_6863_begin_0 = const()[name = string("op_6863_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6863_end_0 = const()[name = string("op_6863_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6863_end_mask_0 = const()[name = string("op_6863_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6863_cast_fp16 = slice_by_index(begin = var_6863_begin_0, end = var_6863_end_0, end_mask = var_6863_end_mask_0, x = coreml_update_state_204)[name = string("op_6863_cast_fp16")]; tensor tile_36 = const()[name = string("tile_36"), val = tensor([1, 1])]; int32 var_6866_axis_0 = const()[name = string("op_6866_axis_0"), val = int32(1)]; tensor var_6866_cast_fp16_0, tensor var_6866_cast_fp16_1 = split(axis = var_6866_axis_0, split_sizes = tile_36, x = var_6863_cast_fp16)[name = string("op_6866_cast_fp16")]; tensor var_6873_begin_0 = const()[name = string("op_6873_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6873_end_0 = const()[name = string("op_6873_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6873_end_mask_0 = const()[name = string("op_6873_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6873_cast_fp16 = slice_by_index(begin = var_6873_begin_0, end = var_6873_end_0, end_mask = var_6873_end_mask_0, x = coreml_update_state_205)[name = string("op_6873_cast_fp16")]; tensor tile_37 = const()[name = string("tile_37"), val = tensor([1, 1])]; int32 var_6876_axis_0 = const()[name = string("op_6876_axis_0"), val = int32(1)]; tensor var_6876_cast_fp16_0, tensor var_6876_cast_fp16_1 = split(axis = var_6876_axis_0, split_sizes = tile_37, x = var_6873_cast_fp16)[name = string("op_6876_cast_fp16")]; tensor var_6879_split_sizes_0 = const()[name = string("op_6879_split_sizes_0"), val = tensor([8, 8])]; int32 var_6879_axis_0 = const()[name = string("op_6879_axis_0"), val = int32(1)]; tensor var_6879_0, tensor var_6879_1 = split(axis = var_6879_axis_0, split_sizes = var_6879_split_sizes_0, x = query_states_111_cast_fp16)[name = string("op_6879")]; bool attn_weights_289_transpose_x_0 = const()[name = string("attn_weights_289_transpose_x_0"), val = bool(false)]; bool attn_weights_289_transpose_y_0 = const()[name = string("attn_weights_289_transpose_y_0"), val = bool(false)]; tensor attn_weights_289_cast_fp16 = matmul(transpose_x = attn_weights_289_transpose_x_0, transpose_y = attn_weights_289_transpose_y_0, x = var_6866_cast_fp16_0, y = var_6879_0)[name = string("attn_weights_289_cast_fp16")]; fp16 var_6882_to_fp16 = const()[name = string("op_6882_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_291_cast_fp16 = mul(x = attn_weights_289_cast_fp16, y = var_6882_to_fp16)[name = string("attn_weights_291_cast_fp16")]; tensor attn_weights_293_cast_fp16 = add(x = attn_weights_291_cast_fp16, y = attn_mask_1)[name = string("attn_weights_293_cast_fp16")]; int32 var_6886 = const()[name = string("op_6886"), val = int32(-2)]; tensor attn_weights_295_cast_fp16 = softmax(axis = var_6886, x = attn_weights_293_cast_fp16)[name = string("attn_weights_295_cast_fp16")]; bool var_6892_transpose_x_1 = const()[name = string("op_6892_transpose_x_1"), val = bool(true)]; bool var_6892_transpose_y_1 = const()[name = string("op_6892_transpose_y_1"), val = bool(false)]; tensor var_6892_cast_fp16 = matmul(transpose_x = var_6892_transpose_x_1, transpose_y = var_6892_transpose_y_1, x = attn_weights_295_cast_fp16, y = var_6876_cast_fp16_0)[name = string("op_6892_cast_fp16")]; bool attn_weights_297_transpose_x_0 = const()[name = string("attn_weights_297_transpose_x_0"), val = bool(false)]; bool attn_weights_297_transpose_y_0 = const()[name = string("attn_weights_297_transpose_y_0"), val = bool(false)]; tensor attn_weights_297_cast_fp16 = matmul(transpose_x = attn_weights_297_transpose_x_0, transpose_y = attn_weights_297_transpose_y_0, x = var_6866_cast_fp16_1, y = var_6879_1)[name = string("attn_weights_297_cast_fp16")]; fp16 var_6894_to_fp16 = const()[name = string("op_6894_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_299_cast_fp16 = mul(x = attn_weights_297_cast_fp16, y = var_6894_to_fp16)[name = string("attn_weights_299_cast_fp16")]; tensor attn_weights_301_cast_fp16 = add(x = attn_weights_299_cast_fp16, y = attn_mask_1)[name = string("attn_weights_301_cast_fp16")]; int32 var_6898 = const()[name = string("op_6898"), val = int32(-2)]; tensor attn_weights_303_cast_fp16 = softmax(axis = var_6898, x = attn_weights_301_cast_fp16)[name = string("attn_weights_303_cast_fp16")]; bool attn_output_145_transpose_x_1 = const()[name = string("attn_output_145_transpose_x_1"), val = bool(true)]; bool attn_output_145_transpose_y_1 = const()[name = string("attn_output_145_transpose_y_1"), val = bool(false)]; tensor attn_output_145_cast_fp16 = matmul(transpose_x = attn_output_145_transpose_x_1, transpose_y = attn_output_145_transpose_y_1, x = attn_weights_303_cast_fp16, y = var_6876_cast_fp16_1)[name = string("attn_output_145_cast_fp16")]; int32 var_6906 = const()[name = string("op_6906"), val = int32(1)]; bool attn_output_147_interleave_0 = const()[name = string("attn_output_147_interleave_0"), val = bool(false)]; tensor attn_output_147_cast_fp16 = concat(axis = var_6906, interleave = attn_output_147_interleave_0, values = (var_6892_cast_fp16, attn_output_145_cast_fp16))[name = string("attn_output_147_cast_fp16")]; tensor var_6910_perm_0 = const()[name = string("op_6910_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_227x = const()[name = string("concat_227x"), val = tensor([1, 2048, 1, -1])]; tensor var_6910_cast_fp16 = transpose(perm = var_6910_perm_0, x = attn_output_147_cast_fp16)[name = string("transpose_285")]; tensor attn_output_151_cast_fp16 = reshape(shape = concat_227x, x = var_6910_cast_fp16)[name = string("attn_output_151_cast_fp16")]; tensor hidden_states_183_strides_0 = const()[name = string("hidden_states_183_strides_0"), val = tensor([1, 1])]; string hidden_states_183_pad_type_0 = const()[name = string("hidden_states_183_pad_type_0"), val = string("valid")]; tensor hidden_states_183_pad_0 = const()[name = string("hidden_states_183_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_183_dilations_0 = const()[name = string("hidden_states_183_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_183_groups_0 = const()[name = string("hidden_states_183_groups_0"), val = int32(1)]; tensor hidden_states_183_cast_fp16 = conv(dilations = hidden_states_183_dilations_0, groups = hidden_states_183_groups_0, pad = hidden_states_183_pad_0, pad_type = hidden_states_183_pad_type_0, strides = hidden_states_183_strides_0, weight = layers_18_self_attn_o_proj_weight_cast_fp16, x = attn_output_151_cast_fp16)[name = string("hidden_states_183_cast_fp16")]; tensor hidden_states_185_cast_fp16 = add(x = hidden_states_179_cast_fp16, y = hidden_states_183_cast_fp16)[name = string("hidden_states_185_cast_fp16")]; fp16 const_188_promoted_to_fp16 = const()[name = string("const_188_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6943_cast_fp16 = mul(x = hidden_states_185_cast_fp16, y = const_188_promoted_to_fp16)[name = string("op_6943_cast_fp16")]; int32 var_6941 = const()[name = string("op_6941"), val = int32(1)]; bool doubled_149_interleave_0 = const()[name = string("doubled_149_interleave_0"), val = bool(false)]; tensor doubled_149_cast_fp16 = concat(axis = var_6941, interleave = doubled_149_interleave_0, values = (hidden_states_185_cast_fp16, var_6943_cast_fp16))[name = string("doubled_149_cast_fp16")]; tensor out_75_axes_0 = const()[name = string("out_75_axes_0"), val = tensor([1])]; tensor out_75_gamma_0_to_fp16 = const()[name = string("out_75_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492028288)))]; fp16 var_6953_to_fp16 = const()[name = string("op_6953_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_75_cast_fp16 = layer_norm(axes = out_75_axes_0, epsilon = var_6953_to_fp16, gamma = out_75_gamma_0_to_fp16, x = doubled_149_cast_fp16)[name = string("out_75_cast_fp16")]; tensor var_6964_split_sizes_0 = const()[name = string("op_6964_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6964_axis_0 = const()[name = string("op_6964_axis_0"), val = int32(1)]; tensor var_6964_cast_fp16_0, tensor var_6964_cast_fp16_1 = split(axis = var_6964_axis_0, split_sizes = var_6964_split_sizes_0, x = out_75_cast_fp16)[name = string("op_6964_cast_fp16")]; tensor input_37_strides_0 = const()[name = string("input_37_strides_0"), val = tensor([1, 1])]; string input_37_pad_type_0 = const()[name = string("input_37_pad_type_0"), val = string("valid")]; tensor input_37_pad_0 = const()[name = string("input_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_37_dilations_0 = const()[name = string("input_37_dilations_0"), val = tensor([1, 1])]; int32 input_37_groups_0 = const()[name = string("input_37_groups_0"), val = int32(1)]; tensor input_37_cast_fp16 = conv(dilations = input_37_dilations_0, groups = input_37_groups_0, pad = input_37_pad_0, pad_type = input_37_pad_type_0, strides = input_37_strides_0, weight = layers_18_mlp_gate_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("input_37_cast_fp16")]; tensor var_6981_cast_fp16 = silu(x = input_37_cast_fp16)[name = string("op_6981_cast_fp16")]; tensor var_6987_strides_0 = const()[name = string("op_6987_strides_0"), val = tensor([1, 1])]; string var_6987_pad_type_0 = const()[name = string("op_6987_pad_type_0"), val = string("valid")]; tensor var_6987_pad_0 = const()[name = string("op_6987_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6987_dilations_0 = const()[name = string("op_6987_dilations_0"), val = tensor([1, 1])]; int32 var_6987_groups_0 = const()[name = string("op_6987_groups_0"), val = int32(1)]; tensor var_6987_cast_fp16 = conv(dilations = var_6987_dilations_0, groups = var_6987_groups_0, pad = var_6987_pad_0, pad_type = var_6987_pad_type_0, strides = var_6987_strides_0, weight = layers_18_mlp_up_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("op_6987_cast_fp16")]; tensor x_189_cast_fp16 = mul(x = var_6981_cast_fp16, y = var_6987_cast_fp16)[name = string("x_189_cast_fp16")]; tensor hidden_states_187_strides_0 = const()[name = string("hidden_states_187_strides_0"), val = tensor([1, 1])]; string hidden_states_187_pad_type_0 = const()[name = string("hidden_states_187_pad_type_0"), val = string("valid")]; tensor hidden_states_187_pad_0 = const()[name = string("hidden_states_187_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_187_dilations_0 = const()[name = string("hidden_states_187_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_187_groups_0 = const()[name = string("hidden_states_187_groups_0"), val = int32(1)]; tensor hidden_states_187_cast_fp16 = conv(dilations = hidden_states_187_dilations_0, groups = hidden_states_187_groups_0, pad = hidden_states_187_pad_0, pad_type = hidden_states_187_pad_type_0, strides = hidden_states_187_strides_0, weight = layers_18_mlp_down_proj_weight_cast_fp16, x = x_189_cast_fp16)[name = string("hidden_states_187_cast_fp16")]; tensor hidden_states_189_cast_fp16 = add(x = hidden_states_185_cast_fp16, y = hidden_states_187_cast_fp16)[name = string("hidden_states_189_cast_fp16")]; fp16 const_190_promoted_to_fp16 = const()[name = string("const_190_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7005_cast_fp16 = mul(x = hidden_states_189_cast_fp16, y = const_190_promoted_to_fp16)[name = string("op_7005_cast_fp16")]; int32 var_7003 = const()[name = string("op_7003"), val = int32(1)]; bool doubled_153_interleave_0 = const()[name = string("doubled_153_interleave_0"), val = bool(false)]; tensor doubled_153_cast_fp16 = concat(axis = var_7003, interleave = doubled_153_interleave_0, values = (hidden_states_189_cast_fp16, var_7005_cast_fp16))[name = string("doubled_153_cast_fp16")]; tensor out_77_axes_0 = const()[name = string("out_77_axes_0"), val = tensor([1])]; tensor out_77_gamma_0_to_fp16 = const()[name = string("out_77_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492036544)))]; fp16 var_7015_to_fp16 = const()[name = string("op_7015_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_77_cast_fp16 = layer_norm(axes = out_77_axes_0, epsilon = var_7015_to_fp16, gamma = out_77_gamma_0_to_fp16, x = doubled_153_cast_fp16)[name = string("out_77_cast_fp16")]; tensor var_7026_split_sizes_0 = const()[name = string("op_7026_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7026_axis_0 = const()[name = string("op_7026_axis_0"), val = int32(1)]; tensor var_7026_cast_fp16_0, tensor var_7026_cast_fp16_1 = split(axis = var_7026_axis_0, split_sizes = var_7026_split_sizes_0, x = out_77_cast_fp16)[name = string("op_7026_cast_fp16")]; tensor query_states_115_strides_0 = const()[name = string("query_states_115_strides_0"), val = tensor([1, 1])]; string query_states_115_pad_type_0 = const()[name = string("query_states_115_pad_type_0"), val = string("valid")]; tensor query_states_115_pad_0 = const()[name = string("query_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_115_dilations_0 = const()[name = string("query_states_115_dilations_0"), val = tensor([1, 1])]; int32 query_states_115_groups_0 = const()[name = string("query_states_115_groups_0"), val = int32(1)]; tensor query_states_115_cast_fp16 = conv(dilations = query_states_115_dilations_0, groups = query_states_115_groups_0, pad = query_states_115_pad_0, pad_type = query_states_115_pad_type_0, strides = query_states_115_strides_0, weight = layers_19_self_attn_q_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("query_states_115_cast_fp16")]; tensor key_states_191_strides_0 = const()[name = string("key_states_191_strides_0"), val = tensor([1, 1])]; string key_states_191_pad_type_0 = const()[name = string("key_states_191_pad_type_0"), val = string("valid")]; tensor key_states_191_pad_0 = const()[name = string("key_states_191_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_191_dilations_0 = const()[name = string("key_states_191_dilations_0"), val = tensor([1, 1])]; int32 key_states_191_groups_0 = const()[name = string("key_states_191_groups_0"), val = int32(1)]; tensor key_states_191_cast_fp16 = conv(dilations = key_states_191_dilations_0, groups = key_states_191_groups_0, pad = key_states_191_pad_0, pad_type = key_states_191_pad_type_0, strides = key_states_191_strides_0, weight = layers_19_self_attn_k_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("key_states_191_cast_fp16")]; tensor layers_19_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492044800)))]; tensor value_states_115_strides_0 = const()[name = string("value_states_115_strides_0"), val = tensor([1, 1])]; string value_states_115_pad_type_0 = const()[name = string("value_states_115_pad_type_0"), val = string("valid")]; tensor value_states_115_pad_0 = const()[name = string("value_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_115_dilations_0 = const()[name = string("value_states_115_dilations_0"), val = tensor([1, 1])]; int32 value_states_115_groups_0 = const()[name = string("value_states_115_groups_0"), val = int32(1)]; tensor value_states_115_cast_fp16 = conv(dilations = value_states_115_dilations_0, groups = value_states_115_groups_0, pad = value_states_115_pad_0, pad_type = value_states_115_pad_type_0, strides = value_states_115_strides_0, weight = layers_19_self_attn_v_proj_weight_to_fp16, x = var_7026_cast_fp16_0)[name = string("value_states_115_cast_fp16")]; tensor concat_228x = const()[name = string("concat_228x"), val = tensor([1, 16, 128, -1])]; tensor x_191_cast_fp16 = reshape(shape = concat_228x, x = query_states_115_cast_fp16)[name = string("x_191_cast_fp16")]; tensor concat_229x = const()[name = string("concat_229x"), val = tensor([1, 2, 128, -1])]; tensor var_7083_cast_fp16 = reshape(shape = concat_229x, x = key_states_191_cast_fp16)[name = string("op_7083_cast_fp16")]; tensor concat_230x = const()[name = string("concat_230x"), val = tensor([1, 2, 128, -1])]; tensor var_7090_cast_fp16 = reshape(shape = concat_230x, x = value_states_115_cast_fp16)[name = string("op_7090_cast_fp16")]; tensor var_7094_cast_fp16 = mul(x = x_191_cast_fp16, y = var_869_cast_fp16)[name = string("op_7094_cast_fp16")]; tensor var_7095_split_sizes_0 = const()[name = string("op_7095_split_sizes_0"), val = tensor([64, 64])]; int32 var_7095_axis_0 = const()[name = string("op_7095_axis_0"), val = int32(-2)]; tensor var_7095_cast_fp16_0, tensor var_7095_cast_fp16_1 = split(axis = var_7095_axis_0, split_sizes = var_7095_split_sizes_0, x = x_191_cast_fp16)[name = string("op_7095_cast_fp16")]; fp16 const_192_promoted_to_fp16 = const()[name = string("const_192_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7097_cast_fp16 = mul(x = var_7095_cast_fp16_1, y = const_192_promoted_to_fp16)[name = string("op_7097_cast_fp16")]; int32 var_7099 = const()[name = string("op_7099"), val = int32(-2)]; bool var_7100_interleave_0 = const()[name = string("op_7100_interleave_0"), val = bool(false)]; tensor var_7100_cast_fp16 = concat(axis = var_7099, interleave = var_7100_interleave_0, values = (var_7097_cast_fp16, var_7095_cast_fp16_0))[name = string("op_7100_cast_fp16")]; tensor var_7101_cast_fp16 = mul(x = var_7100_cast_fp16, y = var_878_cast_fp16)[name = string("op_7101_cast_fp16")]; tensor query_states_117_cast_fp16 = add(x = var_7094_cast_fp16, y = var_7101_cast_fp16)[name = string("query_states_117_cast_fp16")]; tensor var_7107_cast_fp16 = mul(x = var_7083_cast_fp16, y = var_869_cast_fp16)[name = string("op_7107_cast_fp16")]; tensor var_7108_split_sizes_0 = const()[name = string("op_7108_split_sizes_0"), val = tensor([64, 64])]; int32 var_7108_axis_0 = const()[name = string("op_7108_axis_0"), val = int32(-2)]; tensor var_7108_cast_fp16_0, tensor var_7108_cast_fp16_1 = split(axis = var_7108_axis_0, split_sizes = var_7108_split_sizes_0, x = var_7083_cast_fp16)[name = string("op_7108_cast_fp16")]; fp16 const_193_promoted_to_fp16 = const()[name = string("const_193_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7110_cast_fp16 = mul(x = var_7108_cast_fp16_1, y = const_193_promoted_to_fp16)[name = string("op_7110_cast_fp16")]; int32 var_7112 = const()[name = string("op_7112"), val = int32(-2)]; bool var_7113_interleave_0 = const()[name = string("op_7113_interleave_0"), val = bool(false)]; tensor var_7113_cast_fp16 = concat(axis = var_7112, interleave = var_7113_interleave_0, values = (var_7110_cast_fp16, var_7108_cast_fp16_0))[name = string("op_7113_cast_fp16")]; tensor var_7114_cast_fp16 = mul(x = var_7113_cast_fp16, y = var_878_cast_fp16)[name = string("op_7114_cast_fp16")]; tensor key_states_195_cast_fp16 = add(x = var_7107_cast_fp16, y = var_7114_cast_fp16)[name = string("key_states_195_cast_fp16")]; tensor expand_dims_228 = const()[name = string("expand_dims_228"), val = tensor([19])]; tensor expand_dims_229 = const()[name = string("expand_dims_229"), val = tensor([0])]; tensor expand_dims_231 = const()[name = string("expand_dims_231"), val = tensor([0])]; int32 concat_233_axis_0 = const()[name = string("concat_233_axis_0"), val = int32(0)]; bool concat_233_interleave_0 = const()[name = string("concat_233_interleave_0"), val = bool(false)]; tensor concat_233 = concat(axis = concat_233_axis_0, interleave = concat_233_interleave_0, values = (expand_dims_228, expand_dims_229, position_id, expand_dims_231))[name = string("concat_233")]; tensor expand_dims_232 = const()[name = string("expand_dims_232"), val = tensor([20])]; tensor concat_234_values1_0 = const()[name = string("concat_234_values1_0"), val = tensor([0])]; tensor concat_234_values3_0 = const()[name = string("concat_234_values3_0"), val = tensor([0])]; int32 concat_234_axis_0 = const()[name = string("concat_234_axis_0"), val = int32(0)]; bool concat_234_interleave_0 = const()[name = string("concat_234_interleave_0"), val = bool(false)]; tensor concat_234 = concat(axis = concat_234_axis_0, interleave = concat_234_interleave_0, values = (expand_dims_232, concat_234_values1_0, cache_position_end, concat_234_values3_0))[name = string("concat_234")]; tensor key_states_197_perm_0 = const()[name = string("key_states_197_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_20_stride_0 = const()[name = string("key_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_197_cast_fp16 = transpose(perm = key_states_197_perm_0, x = key_states_195_cast_fp16)[name = string("transpose_284")]; tensor key_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = key_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = key_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_20_squeeze_mask_0, stride = key_cache_internal_tensor_assign_20_stride_0, update = key_states_197_cast_fp16, x = coreml_update_state_204)[name = string("key_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_20_cast_fp16, input = key_cache)[name = string("coreml_update_state_206_write_state")]; tensor coreml_update_state_206 = read_state(input = key_cache)[name = string("coreml_update_state_206")]; tensor value_states_117_perm_0 = const()[name = string("value_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_20_stride_0 = const()[name = string("value_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_117_cast_fp16 = transpose(perm = value_states_117_perm_0, x = var_7090_cast_fp16)[name = string("transpose_283")]; tensor value_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = value_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = value_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_20_squeeze_mask_0, stride = value_cache_internal_tensor_assign_20_stride_0, update = value_states_117_cast_fp16, x = coreml_update_state_205)[name = string("value_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_20_cast_fp16, input = value_cache)[name = string("coreml_update_state_207_write_state")]; tensor coreml_update_state_207 = read_state(input = value_cache)[name = string("coreml_update_state_207")]; tensor var_7184_begin_0 = const()[name = string("op_7184_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7184_end_0 = const()[name = string("op_7184_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7184_end_mask_0 = const()[name = string("op_7184_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7184_cast_fp16 = slice_by_index(begin = var_7184_begin_0, end = var_7184_end_0, end_mask = var_7184_end_mask_0, x = coreml_update_state_206)[name = string("op_7184_cast_fp16")]; tensor tile_38 = const()[name = string("tile_38"), val = tensor([1, 1])]; int32 var_7187_axis_0 = const()[name = string("op_7187_axis_0"), val = int32(1)]; tensor var_7187_cast_fp16_0, tensor var_7187_cast_fp16_1 = split(axis = var_7187_axis_0, split_sizes = tile_38, x = var_7184_cast_fp16)[name = string("op_7187_cast_fp16")]; tensor var_7194_begin_0 = const()[name = string("op_7194_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7194_end_0 = const()[name = string("op_7194_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7194_end_mask_0 = const()[name = string("op_7194_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7194_cast_fp16 = slice_by_index(begin = var_7194_begin_0, end = var_7194_end_0, end_mask = var_7194_end_mask_0, x = coreml_update_state_207)[name = string("op_7194_cast_fp16")]; tensor tile_39 = const()[name = string("tile_39"), val = tensor([1, 1])]; int32 var_7197_axis_0 = const()[name = string("op_7197_axis_0"), val = int32(1)]; tensor var_7197_cast_fp16_0, tensor var_7197_cast_fp16_1 = split(axis = var_7197_axis_0, split_sizes = tile_39, x = var_7194_cast_fp16)[name = string("op_7197_cast_fp16")]; tensor var_7200_split_sizes_0 = const()[name = string("op_7200_split_sizes_0"), val = tensor([8, 8])]; int32 var_7200_axis_0 = const()[name = string("op_7200_axis_0"), val = int32(1)]; tensor var_7200_0, tensor var_7200_1 = split(axis = var_7200_axis_0, split_sizes = var_7200_split_sizes_0, x = query_states_117_cast_fp16)[name = string("op_7200")]; bool attn_weights_305_transpose_x_0 = const()[name = string("attn_weights_305_transpose_x_0"), val = bool(false)]; bool attn_weights_305_transpose_y_0 = const()[name = string("attn_weights_305_transpose_y_0"), val = bool(false)]; tensor attn_weights_305_cast_fp16 = matmul(transpose_x = attn_weights_305_transpose_x_0, transpose_y = attn_weights_305_transpose_y_0, x = var_7187_cast_fp16_0, y = var_7200_0)[name = string("attn_weights_305_cast_fp16")]; fp16 var_7203_to_fp16 = const()[name = string("op_7203_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_307_cast_fp16 = mul(x = attn_weights_305_cast_fp16, y = var_7203_to_fp16)[name = string("attn_weights_307_cast_fp16")]; tensor attn_weights_309_cast_fp16 = add(x = attn_weights_307_cast_fp16, y = attn_mask_1)[name = string("attn_weights_309_cast_fp16")]; int32 var_7207 = const()[name = string("op_7207"), val = int32(-2)]; tensor attn_weights_311_cast_fp16 = softmax(axis = var_7207, x = attn_weights_309_cast_fp16)[name = string("attn_weights_311_cast_fp16")]; bool var_7213_transpose_x_1 = const()[name = string("op_7213_transpose_x_1"), val = bool(true)]; bool var_7213_transpose_y_1 = const()[name = string("op_7213_transpose_y_1"), val = bool(false)]; tensor var_7213_cast_fp16 = matmul(transpose_x = var_7213_transpose_x_1, transpose_y = var_7213_transpose_y_1, x = attn_weights_311_cast_fp16, y = var_7197_cast_fp16_0)[name = string("op_7213_cast_fp16")]; bool attn_weights_313_transpose_x_0 = const()[name = string("attn_weights_313_transpose_x_0"), val = bool(false)]; bool attn_weights_313_transpose_y_0 = const()[name = string("attn_weights_313_transpose_y_0"), val = bool(false)]; tensor attn_weights_313_cast_fp16 = matmul(transpose_x = attn_weights_313_transpose_x_0, transpose_y = attn_weights_313_transpose_y_0, x = var_7187_cast_fp16_1, y = var_7200_1)[name = string("attn_weights_313_cast_fp16")]; fp16 var_7215_to_fp16 = const()[name = string("op_7215_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_315_cast_fp16 = mul(x = attn_weights_313_cast_fp16, y = var_7215_to_fp16)[name = string("attn_weights_315_cast_fp16")]; tensor attn_weights_317_cast_fp16 = add(x = attn_weights_315_cast_fp16, y = attn_mask_1)[name = string("attn_weights_317_cast_fp16")]; int32 var_7219 = const()[name = string("op_7219"), val = int32(-2)]; tensor attn_weights_319_cast_fp16 = softmax(axis = var_7219, x = attn_weights_317_cast_fp16)[name = string("attn_weights_319_cast_fp16")]; bool attn_output_153_transpose_x_1 = const()[name = string("attn_output_153_transpose_x_1"), val = bool(true)]; bool attn_output_153_transpose_y_1 = const()[name = string("attn_output_153_transpose_y_1"), val = bool(false)]; tensor attn_output_153_cast_fp16 = matmul(transpose_x = attn_output_153_transpose_x_1, transpose_y = attn_output_153_transpose_y_1, x = attn_weights_319_cast_fp16, y = var_7197_cast_fp16_1)[name = string("attn_output_153_cast_fp16")]; int32 var_7227 = const()[name = string("op_7227"), val = int32(1)]; bool attn_output_155_interleave_0 = const()[name = string("attn_output_155_interleave_0"), val = bool(false)]; tensor attn_output_155_cast_fp16 = concat(axis = var_7227, interleave = attn_output_155_interleave_0, values = (var_7213_cast_fp16, attn_output_153_cast_fp16))[name = string("attn_output_155_cast_fp16")]; tensor var_7231_perm_0 = const()[name = string("op_7231_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_239x = const()[name = string("concat_239x"), val = tensor([1, 2048, 1, -1])]; tensor var_7231_cast_fp16 = transpose(perm = var_7231_perm_0, x = attn_output_155_cast_fp16)[name = string("transpose_282")]; tensor attn_output_159_cast_fp16 = reshape(shape = concat_239x, x = var_7231_cast_fp16)[name = string("attn_output_159_cast_fp16")]; tensor layers_19_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1493093440)))]; tensor hidden_states_193_strides_0 = const()[name = string("hidden_states_193_strides_0"), val = tensor([1, 1])]; string hidden_states_193_pad_type_0 = const()[name = string("hidden_states_193_pad_type_0"), val = string("valid")]; tensor hidden_states_193_pad_0 = const()[name = string("hidden_states_193_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_193_dilations_0 = const()[name = string("hidden_states_193_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_193_groups_0 = const()[name = string("hidden_states_193_groups_0"), val = int32(1)]; tensor hidden_states_193_cast_fp16 = conv(dilations = hidden_states_193_dilations_0, groups = hidden_states_193_groups_0, pad = hidden_states_193_pad_0, pad_type = hidden_states_193_pad_type_0, strides = hidden_states_193_strides_0, weight = layers_19_self_attn_o_proj_weight_to_fp16, x = attn_output_159_cast_fp16)[name = string("hidden_states_193_cast_fp16")]; tensor hidden_states_195_cast_fp16 = add(x = hidden_states_189_cast_fp16, y = hidden_states_193_cast_fp16)[name = string("hidden_states_195_cast_fp16")]; fp16 const_198_promoted_to_fp16 = const()[name = string("const_198_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7264_cast_fp16 = mul(x = hidden_states_195_cast_fp16, y = const_198_promoted_to_fp16)[name = string("op_7264_cast_fp16")]; int32 var_7262 = const()[name = string("op_7262"), val = int32(1)]; bool doubled_157_interleave_0 = const()[name = string("doubled_157_interleave_0"), val = bool(false)]; tensor doubled_157_cast_fp16 = concat(axis = var_7262, interleave = doubled_157_interleave_0, values = (hidden_states_195_cast_fp16, var_7264_cast_fp16))[name = string("doubled_157_cast_fp16")]; tensor out_79_axes_0 = const()[name = string("out_79_axes_0"), val = tensor([1])]; tensor out_79_gamma_0_to_fp16 = const()[name = string("out_79_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501482112)))]; fp16 var_7274_to_fp16 = const()[name = string("op_7274_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_79_cast_fp16 = layer_norm(axes = out_79_axes_0, epsilon = var_7274_to_fp16, gamma = out_79_gamma_0_to_fp16, x = doubled_157_cast_fp16)[name = string("out_79_cast_fp16")]; tensor var_7285_split_sizes_0 = const()[name = string("op_7285_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7285_axis_0 = const()[name = string("op_7285_axis_0"), val = int32(1)]; tensor var_7285_cast_fp16_0, tensor var_7285_cast_fp16_1 = split(axis = var_7285_axis_0, split_sizes = var_7285_split_sizes_0, x = out_79_cast_fp16)[name = string("op_7285_cast_fp16")]; tensor input_39_strides_0 = const()[name = string("input_39_strides_0"), val = tensor([1, 1])]; string input_39_pad_type_0 = const()[name = string("input_39_pad_type_0"), val = string("valid")]; tensor input_39_pad_0 = const()[name = string("input_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_39_dilations_0 = const()[name = string("input_39_dilations_0"), val = tensor([1, 1])]; int32 input_39_groups_0 = const()[name = string("input_39_groups_0"), val = int32(1)]; tensor input_39_cast_fp16 = conv(dilations = input_39_dilations_0, groups = input_39_groups_0, pad = input_39_pad_0, pad_type = input_39_pad_type_0, strides = input_39_strides_0, weight = layers_19_mlp_gate_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("input_39_cast_fp16")]; tensor var_7302_cast_fp16 = silu(x = input_39_cast_fp16)[name = string("op_7302_cast_fp16")]; tensor var_7308_strides_0 = const()[name = string("op_7308_strides_0"), val = tensor([1, 1])]; string var_7308_pad_type_0 = const()[name = string("op_7308_pad_type_0"), val = string("valid")]; tensor var_7308_pad_0 = const()[name = string("op_7308_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7308_dilations_0 = const()[name = string("op_7308_dilations_0"), val = tensor([1, 1])]; int32 var_7308_groups_0 = const()[name = string("op_7308_groups_0"), val = int32(1)]; tensor var_7308_cast_fp16 = conv(dilations = var_7308_dilations_0, groups = var_7308_groups_0, pad = var_7308_pad_0, pad_type = var_7308_pad_type_0, strides = var_7308_strides_0, weight = layers_19_mlp_up_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("op_7308_cast_fp16")]; tensor x_199_cast_fp16 = mul(x = var_7302_cast_fp16, y = var_7308_cast_fp16)[name = string("x_199_cast_fp16")]; tensor hidden_states_197_strides_0 = const()[name = string("hidden_states_197_strides_0"), val = tensor([1, 1])]; string hidden_states_197_pad_type_0 = const()[name = string("hidden_states_197_pad_type_0"), val = string("valid")]; tensor hidden_states_197_pad_0 = const()[name = string("hidden_states_197_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_197_dilations_0 = const()[name = string("hidden_states_197_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_197_groups_0 = const()[name = string("hidden_states_197_groups_0"), val = int32(1)]; tensor hidden_states_197_cast_fp16 = conv(dilations = hidden_states_197_dilations_0, groups = hidden_states_197_groups_0, pad = hidden_states_197_pad_0, pad_type = hidden_states_197_pad_type_0, strides = hidden_states_197_strides_0, weight = layers_19_mlp_down_proj_weight_cast_fp16, x = x_199_cast_fp16)[name = string("hidden_states_197_cast_fp16")]; tensor hidden_states_199_cast_fp16 = add(x = hidden_states_195_cast_fp16, y = hidden_states_197_cast_fp16)[name = string("hidden_states_199_cast_fp16")]; fp16 const_200_promoted_to_fp16 = const()[name = string("const_200_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7326_cast_fp16 = mul(x = hidden_states_199_cast_fp16, y = const_200_promoted_to_fp16)[name = string("op_7326_cast_fp16")]; int32 var_7324 = const()[name = string("op_7324"), val = int32(1)]; bool doubled_161_interleave_0 = const()[name = string("doubled_161_interleave_0"), val = bool(false)]; tensor doubled_161_cast_fp16 = concat(axis = var_7324, interleave = doubled_161_interleave_0, values = (hidden_states_199_cast_fp16, var_7326_cast_fp16))[name = string("doubled_161_cast_fp16")]; tensor out_81_axes_0 = const()[name = string("out_81_axes_0"), val = tensor([1])]; tensor out_81_gamma_0_to_fp16 = const()[name = string("out_81_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501490368)))]; fp16 var_7336_to_fp16 = const()[name = string("op_7336_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_81_cast_fp16 = layer_norm(axes = out_81_axes_0, epsilon = var_7336_to_fp16, gamma = out_81_gamma_0_to_fp16, x = doubled_161_cast_fp16)[name = string("out_81_cast_fp16")]; tensor var_7347_split_sizes_0 = const()[name = string("op_7347_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7347_axis_0 = const()[name = string("op_7347_axis_0"), val = int32(1)]; tensor var_7347_cast_fp16_0, tensor var_7347_cast_fp16_1 = split(axis = var_7347_axis_0, split_sizes = var_7347_split_sizes_0, x = out_81_cast_fp16)[name = string("op_7347_cast_fp16")]; tensor query_states_121_strides_0 = const()[name = string("query_states_121_strides_0"), val = tensor([1, 1])]; string query_states_121_pad_type_0 = const()[name = string("query_states_121_pad_type_0"), val = string("valid")]; tensor query_states_121_pad_0 = const()[name = string("query_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_121_dilations_0 = const()[name = string("query_states_121_dilations_0"), val = tensor([1, 1])]; int32 query_states_121_groups_0 = const()[name = string("query_states_121_groups_0"), val = int32(1)]; tensor query_states_121_cast_fp16 = conv(dilations = query_states_121_dilations_0, groups = query_states_121_groups_0, pad = query_states_121_pad_0, pad_type = query_states_121_pad_type_0, strides = query_states_121_strides_0, weight = layers_20_self_attn_q_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("query_states_121_cast_fp16")]; tensor key_states_201_strides_0 = const()[name = string("key_states_201_strides_0"), val = tensor([1, 1])]; string key_states_201_pad_type_0 = const()[name = string("key_states_201_pad_type_0"), val = string("valid")]; tensor key_states_201_pad_0 = const()[name = string("key_states_201_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_201_dilations_0 = const()[name = string("key_states_201_dilations_0"), val = tensor([1, 1])]; int32 key_states_201_groups_0 = const()[name = string("key_states_201_groups_0"), val = int32(1)]; tensor key_states_201_cast_fp16 = conv(dilations = key_states_201_dilations_0, groups = key_states_201_groups_0, pad = key_states_201_pad_0, pad_type = key_states_201_pad_type_0, strides = key_states_201_strides_0, weight = layers_20_self_attn_k_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("key_states_201_cast_fp16")]; tensor layers_20_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_20_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501498624)))]; tensor value_states_121_strides_0 = const()[name = string("value_states_121_strides_0"), val = tensor([1, 1])]; string value_states_121_pad_type_0 = const()[name = string("value_states_121_pad_type_0"), val = string("valid")]; tensor value_states_121_pad_0 = const()[name = string("value_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_121_dilations_0 = const()[name = string("value_states_121_dilations_0"), val = tensor([1, 1])]; int32 value_states_121_groups_0 = const()[name = string("value_states_121_groups_0"), val = int32(1)]; tensor value_states_121_cast_fp16 = conv(dilations = value_states_121_dilations_0, groups = value_states_121_groups_0, pad = value_states_121_pad_0, pad_type = value_states_121_pad_type_0, strides = value_states_121_strides_0, weight = layers_20_self_attn_v_proj_weight_to_fp16, x = var_7347_cast_fp16_0)[name = string("value_states_121_cast_fp16")]; tensor concat_240x = const()[name = string("concat_240x"), val = tensor([1, 16, 128, -1])]; tensor x_201_cast_fp16 = reshape(shape = concat_240x, x = query_states_121_cast_fp16)[name = string("x_201_cast_fp16")]; tensor concat_241x = const()[name = string("concat_241x"), val = tensor([1, 2, 128, -1])]; tensor var_7404_cast_fp16 = reshape(shape = concat_241x, x = key_states_201_cast_fp16)[name = string("op_7404_cast_fp16")]; tensor concat_242x = const()[name = string("concat_242x"), val = tensor([1, 2, 128, -1])]; tensor var_7411_cast_fp16 = reshape(shape = concat_242x, x = value_states_121_cast_fp16)[name = string("op_7411_cast_fp16")]; tensor var_7415_cast_fp16 = mul(x = x_201_cast_fp16, y = var_869_cast_fp16)[name = string("op_7415_cast_fp16")]; tensor var_7416_split_sizes_0 = const()[name = string("op_7416_split_sizes_0"), val = tensor([64, 64])]; int32 var_7416_axis_0 = const()[name = string("op_7416_axis_0"), val = int32(-2)]; tensor var_7416_cast_fp16_0, tensor var_7416_cast_fp16_1 = split(axis = var_7416_axis_0, split_sizes = var_7416_split_sizes_0, x = x_201_cast_fp16)[name = string("op_7416_cast_fp16")]; fp16 const_202_promoted_to_fp16 = const()[name = string("const_202_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7418_cast_fp16 = mul(x = var_7416_cast_fp16_1, y = const_202_promoted_to_fp16)[name = string("op_7418_cast_fp16")]; int32 var_7420 = const()[name = string("op_7420"), val = int32(-2)]; bool var_7421_interleave_0 = const()[name = string("op_7421_interleave_0"), val = bool(false)]; tensor var_7421_cast_fp16 = concat(axis = var_7420, interleave = var_7421_interleave_0, values = (var_7418_cast_fp16, var_7416_cast_fp16_0))[name = string("op_7421_cast_fp16")]; tensor var_7422_cast_fp16 = mul(x = var_7421_cast_fp16, y = var_878_cast_fp16)[name = string("op_7422_cast_fp16")]; tensor query_states_123_cast_fp16 = add(x = var_7415_cast_fp16, y = var_7422_cast_fp16)[name = string("query_states_123_cast_fp16")]; tensor var_7428_cast_fp16 = mul(x = var_7404_cast_fp16, y = var_869_cast_fp16)[name = string("op_7428_cast_fp16")]; tensor var_7429_split_sizes_0 = const()[name = string("op_7429_split_sizes_0"), val = tensor([64, 64])]; int32 var_7429_axis_0 = const()[name = string("op_7429_axis_0"), val = int32(-2)]; tensor var_7429_cast_fp16_0, tensor var_7429_cast_fp16_1 = split(axis = var_7429_axis_0, split_sizes = var_7429_split_sizes_0, x = var_7404_cast_fp16)[name = string("op_7429_cast_fp16")]; fp16 const_203_promoted_to_fp16 = const()[name = string("const_203_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7431_cast_fp16 = mul(x = var_7429_cast_fp16_1, y = const_203_promoted_to_fp16)[name = string("op_7431_cast_fp16")]; int32 var_7433 = const()[name = string("op_7433"), val = int32(-2)]; bool var_7434_interleave_0 = const()[name = string("op_7434_interleave_0"), val = bool(false)]; tensor var_7434_cast_fp16 = concat(axis = var_7433, interleave = var_7434_interleave_0, values = (var_7431_cast_fp16, var_7429_cast_fp16_0))[name = string("op_7434_cast_fp16")]; tensor var_7435_cast_fp16 = mul(x = var_7434_cast_fp16, y = var_878_cast_fp16)[name = string("op_7435_cast_fp16")]; tensor key_states_205_cast_fp16 = add(x = var_7428_cast_fp16, y = var_7435_cast_fp16)[name = string("key_states_205_cast_fp16")]; tensor expand_dims_240 = const()[name = string("expand_dims_240"), val = tensor([20])]; tensor expand_dims_241 = const()[name = string("expand_dims_241"), val = tensor([0])]; tensor expand_dims_243 = const()[name = string("expand_dims_243"), val = tensor([0])]; int32 concat_245_axis_0 = const()[name = string("concat_245_axis_0"), val = int32(0)]; bool concat_245_interleave_0 = const()[name = string("concat_245_interleave_0"), val = bool(false)]; tensor concat_245 = concat(axis = concat_245_axis_0, interleave = concat_245_interleave_0, values = (expand_dims_240, expand_dims_241, position_id, expand_dims_243))[name = string("concat_245")]; tensor expand_dims_244 = const()[name = string("expand_dims_244"), val = tensor([21])]; tensor concat_246_values1_0 = const()[name = string("concat_246_values1_0"), val = tensor([0])]; tensor concat_246_values3_0 = const()[name = string("concat_246_values3_0"), val = tensor([0])]; int32 concat_246_axis_0 = const()[name = string("concat_246_axis_0"), val = int32(0)]; bool concat_246_interleave_0 = const()[name = string("concat_246_interleave_0"), val = bool(false)]; tensor concat_246 = concat(axis = concat_246_axis_0, interleave = concat_246_interleave_0, values = (expand_dims_244, concat_246_values1_0, cache_position_end, concat_246_values3_0))[name = string("concat_246")]; tensor key_states_207_perm_0 = const()[name = string("key_states_207_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_21_stride_0 = const()[name = string("key_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_207_cast_fp16 = transpose(perm = key_states_207_perm_0, x = key_states_205_cast_fp16)[name = string("transpose_281")]; tensor key_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = key_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = key_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_21_squeeze_mask_0, stride = key_cache_internal_tensor_assign_21_stride_0, update = key_states_207_cast_fp16, x = coreml_update_state_206)[name = string("key_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_21_cast_fp16, input = key_cache)[name = string("coreml_update_state_208_write_state")]; tensor coreml_update_state_208 = read_state(input = key_cache)[name = string("coreml_update_state_208")]; tensor value_states_123_perm_0 = const()[name = string("value_states_123_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_21_stride_0 = const()[name = string("value_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_123_cast_fp16 = transpose(perm = value_states_123_perm_0, x = var_7411_cast_fp16)[name = string("transpose_280")]; tensor value_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = value_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = value_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_21_squeeze_mask_0, stride = value_cache_internal_tensor_assign_21_stride_0, update = value_states_123_cast_fp16, x = coreml_update_state_207)[name = string("value_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_21_cast_fp16, input = value_cache)[name = string("coreml_update_state_209_write_state")]; tensor coreml_update_state_209 = read_state(input = value_cache)[name = string("coreml_update_state_209")]; tensor var_7505_begin_0 = const()[name = string("op_7505_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7505_end_0 = const()[name = string("op_7505_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7505_end_mask_0 = const()[name = string("op_7505_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7505_cast_fp16 = slice_by_index(begin = var_7505_begin_0, end = var_7505_end_0, end_mask = var_7505_end_mask_0, x = coreml_update_state_208)[name = string("op_7505_cast_fp16")]; tensor tile_40 = const()[name = string("tile_40"), val = tensor([1, 1])]; int32 var_7508_axis_0 = const()[name = string("op_7508_axis_0"), val = int32(1)]; tensor var_7508_cast_fp16_0, tensor var_7508_cast_fp16_1 = split(axis = var_7508_axis_0, split_sizes = tile_40, x = var_7505_cast_fp16)[name = string("op_7508_cast_fp16")]; tensor var_7515_begin_0 = const()[name = string("op_7515_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7515_end_0 = const()[name = string("op_7515_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7515_end_mask_0 = const()[name = string("op_7515_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7515_cast_fp16 = slice_by_index(begin = var_7515_begin_0, end = var_7515_end_0, end_mask = var_7515_end_mask_0, x = coreml_update_state_209)[name = string("op_7515_cast_fp16")]; tensor tile_41 = const()[name = string("tile_41"), val = tensor([1, 1])]; int32 var_7518_axis_0 = const()[name = string("op_7518_axis_0"), val = int32(1)]; tensor var_7518_cast_fp16_0, tensor var_7518_cast_fp16_1 = split(axis = var_7518_axis_0, split_sizes = tile_41, x = var_7515_cast_fp16)[name = string("op_7518_cast_fp16")]; tensor var_7521_split_sizes_0 = const()[name = string("op_7521_split_sizes_0"), val = tensor([8, 8])]; int32 var_7521_axis_0 = const()[name = string("op_7521_axis_0"), val = int32(1)]; tensor var_7521_0, tensor var_7521_1 = split(axis = var_7521_axis_0, split_sizes = var_7521_split_sizes_0, x = query_states_123_cast_fp16)[name = string("op_7521")]; bool attn_weights_321_transpose_x_0 = const()[name = string("attn_weights_321_transpose_x_0"), val = bool(false)]; bool attn_weights_321_transpose_y_0 = const()[name = string("attn_weights_321_transpose_y_0"), val = bool(false)]; tensor attn_weights_321_cast_fp16 = matmul(transpose_x = attn_weights_321_transpose_x_0, transpose_y = attn_weights_321_transpose_y_0, x = var_7508_cast_fp16_0, y = var_7521_0)[name = string("attn_weights_321_cast_fp16")]; fp16 var_7524_to_fp16 = const()[name = string("op_7524_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_323_cast_fp16 = mul(x = attn_weights_321_cast_fp16, y = var_7524_to_fp16)[name = string("attn_weights_323_cast_fp16")]; tensor attn_weights_325_cast_fp16 = add(x = attn_weights_323_cast_fp16, y = attn_mask_1)[name = string("attn_weights_325_cast_fp16")]; int32 var_7528 = const()[name = string("op_7528"), val = int32(-2)]; tensor attn_weights_327_cast_fp16 = softmax(axis = var_7528, x = attn_weights_325_cast_fp16)[name = string("attn_weights_327_cast_fp16")]; bool var_7534_transpose_x_1 = const()[name = string("op_7534_transpose_x_1"), val = bool(true)]; bool var_7534_transpose_y_1 = const()[name = string("op_7534_transpose_y_1"), val = bool(false)]; tensor var_7534_cast_fp16 = matmul(transpose_x = var_7534_transpose_x_1, transpose_y = var_7534_transpose_y_1, x = attn_weights_327_cast_fp16, y = var_7518_cast_fp16_0)[name = string("op_7534_cast_fp16")]; bool attn_weights_329_transpose_x_0 = const()[name = string("attn_weights_329_transpose_x_0"), val = bool(false)]; bool attn_weights_329_transpose_y_0 = const()[name = string("attn_weights_329_transpose_y_0"), val = bool(false)]; tensor attn_weights_329_cast_fp16 = matmul(transpose_x = attn_weights_329_transpose_x_0, transpose_y = attn_weights_329_transpose_y_0, x = var_7508_cast_fp16_1, y = var_7521_1)[name = string("attn_weights_329_cast_fp16")]; fp16 var_7536_to_fp16 = const()[name = string("op_7536_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_331_cast_fp16 = mul(x = attn_weights_329_cast_fp16, y = var_7536_to_fp16)[name = string("attn_weights_331_cast_fp16")]; tensor attn_weights_333_cast_fp16 = add(x = attn_weights_331_cast_fp16, y = attn_mask_1)[name = string("attn_weights_333_cast_fp16")]; int32 var_7540 = const()[name = string("op_7540"), val = int32(-2)]; tensor attn_weights_335_cast_fp16 = softmax(axis = var_7540, x = attn_weights_333_cast_fp16)[name = string("attn_weights_335_cast_fp16")]; bool attn_output_161_transpose_x_1 = const()[name = string("attn_output_161_transpose_x_1"), val = bool(true)]; bool attn_output_161_transpose_y_1 = const()[name = string("attn_output_161_transpose_y_1"), val = bool(false)]; tensor attn_output_161_cast_fp16 = matmul(transpose_x = attn_output_161_transpose_x_1, transpose_y = attn_output_161_transpose_y_1, x = attn_weights_335_cast_fp16, y = var_7518_cast_fp16_1)[name = string("attn_output_161_cast_fp16")]; int32 var_7548 = const()[name = string("op_7548"), val = int32(1)]; bool attn_output_163_interleave_0 = const()[name = string("attn_output_163_interleave_0"), val = bool(false)]; tensor attn_output_163_cast_fp16 = concat(axis = var_7548, interleave = attn_output_163_interleave_0, values = (var_7534_cast_fp16, attn_output_161_cast_fp16))[name = string("attn_output_163_cast_fp16")]; tensor var_7552_perm_0 = const()[name = string("op_7552_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_251x = const()[name = string("concat_251x"), val = tensor([1, 2048, 1, -1])]; tensor var_7552_cast_fp16 = transpose(perm = var_7552_perm_0, x = attn_output_163_cast_fp16)[name = string("transpose_279")]; tensor attn_output_167_cast_fp16 = reshape(shape = concat_251x, x = var_7552_cast_fp16)[name = string("attn_output_167_cast_fp16")]; tensor hidden_states_203_strides_0 = const()[name = string("hidden_states_203_strides_0"), val = tensor([1, 1])]; string hidden_states_203_pad_type_0 = const()[name = string("hidden_states_203_pad_type_0"), val = string("valid")]; tensor hidden_states_203_pad_0 = const()[name = string("hidden_states_203_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_203_dilations_0 = const()[name = string("hidden_states_203_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_203_groups_0 = const()[name = string("hidden_states_203_groups_0"), val = int32(1)]; tensor hidden_states_203_cast_fp16 = conv(dilations = hidden_states_203_dilations_0, groups = hidden_states_203_groups_0, pad = hidden_states_203_pad_0, pad_type = hidden_states_203_pad_type_0, strides = hidden_states_203_strides_0, weight = layers_20_self_attn_o_proj_weight_cast_fp16, x = attn_output_167_cast_fp16)[name = string("hidden_states_203_cast_fp16")]; tensor hidden_states_205_cast_fp16 = add(x = hidden_states_199_cast_fp16, y = hidden_states_203_cast_fp16)[name = string("hidden_states_205_cast_fp16")]; fp16 const_208_promoted_to_fp16 = const()[name = string("const_208_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7585_cast_fp16 = mul(x = hidden_states_205_cast_fp16, y = const_208_promoted_to_fp16)[name = string("op_7585_cast_fp16")]; int32 var_7583 = const()[name = string("op_7583"), val = int32(1)]; bool doubled_165_interleave_0 = const()[name = string("doubled_165_interleave_0"), val = bool(false)]; tensor doubled_165_cast_fp16 = concat(axis = var_7583, interleave = doubled_165_interleave_0, values = (hidden_states_205_cast_fp16, var_7585_cast_fp16))[name = string("doubled_165_cast_fp16")]; tensor out_83_axes_0 = const()[name = string("out_83_axes_0"), val = tensor([1])]; tensor out_83_gamma_0_to_fp16 = const()[name = string("out_83_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502547264)))]; fp16 var_7595_to_fp16 = const()[name = string("op_7595_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_83_cast_fp16 = layer_norm(axes = out_83_axes_0, epsilon = var_7595_to_fp16, gamma = out_83_gamma_0_to_fp16, x = doubled_165_cast_fp16)[name = string("out_83_cast_fp16")]; tensor var_7606_split_sizes_0 = const()[name = string("op_7606_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7606_axis_0 = const()[name = string("op_7606_axis_0"), val = int32(1)]; tensor var_7606_cast_fp16_0, tensor var_7606_cast_fp16_1 = split(axis = var_7606_axis_0, split_sizes = var_7606_split_sizes_0, x = out_83_cast_fp16)[name = string("op_7606_cast_fp16")]; tensor input_41_strides_0 = const()[name = string("input_41_strides_0"), val = tensor([1, 1])]; string input_41_pad_type_0 = const()[name = string("input_41_pad_type_0"), val = string("valid")]; tensor input_41_pad_0 = const()[name = string("input_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_41_dilations_0 = const()[name = string("input_41_dilations_0"), val = tensor([1, 1])]; int32 input_41_groups_0 = const()[name = string("input_41_groups_0"), val = int32(1)]; tensor input_41_cast_fp16 = conv(dilations = input_41_dilations_0, groups = input_41_groups_0, pad = input_41_pad_0, pad_type = input_41_pad_type_0, strides = input_41_strides_0, weight = layers_20_mlp_gate_proj_weight_cast_fp16, x = var_7606_cast_fp16_0)[name = string("input_41_cast_fp16")]; tensor var_7623_cast_fp16 = silu(x = input_41_cast_fp16)[name = string("op_7623_cast_fp16")]; tensor layers_20_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_20_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502555520)))]; tensor var_7629_strides_0 = const()[name = string("op_7629_strides_0"), val = tensor([1, 1])]; string var_7629_pad_type_0 = const()[name = string("op_7629_pad_type_0"), val = string("valid")]; tensor var_7629_pad_0 = const()[name = string("op_7629_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7629_dilations_0 = const()[name = string("op_7629_dilations_0"), val = tensor([1, 1])]; int32 var_7629_groups_0 = const()[name = string("op_7629_groups_0"), val = int32(1)]; tensor var_7629_cast_fp16 = conv(dilations = var_7629_dilations_0, groups = var_7629_groups_0, pad = var_7629_pad_0, pad_type = var_7629_pad_type_0, strides = var_7629_strides_0, weight = layers_20_mlp_up_proj_weight_to_fp16, x = var_7606_cast_fp16_0)[name = string("op_7629_cast_fp16")]; tensor x_209_cast_fp16 = mul(x = var_7623_cast_fp16, y = var_7629_cast_fp16)[name = string("x_209_cast_fp16")]; tensor hidden_states_207_strides_0 = const()[name = string("hidden_states_207_strides_0"), val = tensor([1, 1])]; string hidden_states_207_pad_type_0 = const()[name = string("hidden_states_207_pad_type_0"), val = string("valid")]; tensor hidden_states_207_pad_0 = const()[name = string("hidden_states_207_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_207_dilations_0 = const()[name = string("hidden_states_207_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_207_groups_0 = const()[name = string("hidden_states_207_groups_0"), val = int32(1)]; tensor hidden_states_207_cast_fp16 = conv(dilations = hidden_states_207_dilations_0, groups = hidden_states_207_groups_0, pad = hidden_states_207_pad_0, pad_type = hidden_states_207_pad_type_0, strides = hidden_states_207_strides_0, weight = layers_20_mlp_down_proj_weight_cast_fp16, x = x_209_cast_fp16)[name = string("hidden_states_207_cast_fp16")]; tensor hidden_states_209_cast_fp16 = add(x = hidden_states_205_cast_fp16, y = hidden_states_207_cast_fp16)[name = string("hidden_states_209_cast_fp16")]; fp16 const_210_promoted_to_fp16 = const()[name = string("const_210_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7647_cast_fp16 = mul(x = hidden_states_209_cast_fp16, y = const_210_promoted_to_fp16)[name = string("op_7647_cast_fp16")]; int32 var_7645 = const()[name = string("op_7645"), val = int32(1)]; bool doubled_169_interleave_0 = const()[name = string("doubled_169_interleave_0"), val = bool(false)]; tensor doubled_169_cast_fp16 = concat(axis = var_7645, interleave = doubled_169_interleave_0, values = (hidden_states_209_cast_fp16, var_7647_cast_fp16))[name = string("doubled_169_cast_fp16")]; tensor out_85_axes_0 = const()[name = string("out_85_axes_0"), val = tensor([1])]; tensor out_85_gamma_0_to_fp16 = const()[name = string("out_85_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527721408)))]; fp16 var_7657_to_fp16 = const()[name = string("op_7657_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_85_cast_fp16 = layer_norm(axes = out_85_axes_0, epsilon = var_7657_to_fp16, gamma = out_85_gamma_0_to_fp16, x = doubled_169_cast_fp16)[name = string("out_85_cast_fp16")]; tensor var_7668_split_sizes_0 = const()[name = string("op_7668_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7668_axis_0 = const()[name = string("op_7668_axis_0"), val = int32(1)]; tensor var_7668_cast_fp16_0, tensor var_7668_cast_fp16_1 = split(axis = var_7668_axis_0, split_sizes = var_7668_split_sizes_0, x = out_85_cast_fp16)[name = string("op_7668_cast_fp16")]; tensor query_states_127_strides_0 = const()[name = string("query_states_127_strides_0"), val = tensor([1, 1])]; string query_states_127_pad_type_0 = const()[name = string("query_states_127_pad_type_0"), val = string("valid")]; tensor query_states_127_pad_0 = const()[name = string("query_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_127_dilations_0 = const()[name = string("query_states_127_dilations_0"), val = tensor([1, 1])]; int32 query_states_127_groups_0 = const()[name = string("query_states_127_groups_0"), val = int32(1)]; tensor query_states_127_cast_fp16 = conv(dilations = query_states_127_dilations_0, groups = query_states_127_groups_0, pad = query_states_127_pad_0, pad_type = query_states_127_pad_type_0, strides = query_states_127_strides_0, weight = layers_21_self_attn_q_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("query_states_127_cast_fp16")]; tensor key_states_211_strides_0 = const()[name = string("key_states_211_strides_0"), val = tensor([1, 1])]; string key_states_211_pad_type_0 = const()[name = string("key_states_211_pad_type_0"), val = string("valid")]; tensor key_states_211_pad_0 = const()[name = string("key_states_211_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_211_dilations_0 = const()[name = string("key_states_211_dilations_0"), val = tensor([1, 1])]; int32 key_states_211_groups_0 = const()[name = string("key_states_211_groups_0"), val = int32(1)]; tensor key_states_211_cast_fp16 = conv(dilations = key_states_211_dilations_0, groups = key_states_211_groups_0, pad = key_states_211_pad_0, pad_type = key_states_211_pad_type_0, strides = key_states_211_strides_0, weight = layers_21_self_attn_k_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("key_states_211_cast_fp16")]; tensor layers_21_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_21_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527729664)))]; tensor value_states_127_strides_0 = const()[name = string("value_states_127_strides_0"), val = tensor([1, 1])]; string value_states_127_pad_type_0 = const()[name = string("value_states_127_pad_type_0"), val = string("valid")]; tensor value_states_127_pad_0 = const()[name = string("value_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_127_dilations_0 = const()[name = string("value_states_127_dilations_0"), val = tensor([1, 1])]; int32 value_states_127_groups_0 = const()[name = string("value_states_127_groups_0"), val = int32(1)]; tensor value_states_127_cast_fp16 = conv(dilations = value_states_127_dilations_0, groups = value_states_127_groups_0, pad = value_states_127_pad_0, pad_type = value_states_127_pad_type_0, strides = value_states_127_strides_0, weight = layers_21_self_attn_v_proj_weight_to_fp16, x = var_7668_cast_fp16_0)[name = string("value_states_127_cast_fp16")]; tensor concat_252x = const()[name = string("concat_252x"), val = tensor([1, 16, 128, -1])]; tensor x_211_cast_fp16 = reshape(shape = concat_252x, x = query_states_127_cast_fp16)[name = string("x_211_cast_fp16")]; tensor concat_253x = const()[name = string("concat_253x"), val = tensor([1, 2, 128, -1])]; tensor var_7725_cast_fp16 = reshape(shape = concat_253x, x = key_states_211_cast_fp16)[name = string("op_7725_cast_fp16")]; tensor concat_254x = const()[name = string("concat_254x"), val = tensor([1, 2, 128, -1])]; tensor var_7732_cast_fp16 = reshape(shape = concat_254x, x = value_states_127_cast_fp16)[name = string("op_7732_cast_fp16")]; tensor var_7736_cast_fp16 = mul(x = x_211_cast_fp16, y = var_869_cast_fp16)[name = string("op_7736_cast_fp16")]; tensor var_7737_split_sizes_0 = const()[name = string("op_7737_split_sizes_0"), val = tensor([64, 64])]; int32 var_7737_axis_0 = const()[name = string("op_7737_axis_0"), val = int32(-2)]; tensor var_7737_cast_fp16_0, tensor var_7737_cast_fp16_1 = split(axis = var_7737_axis_0, split_sizes = var_7737_split_sizes_0, x = x_211_cast_fp16)[name = string("op_7737_cast_fp16")]; fp16 const_212_promoted_to_fp16 = const()[name = string("const_212_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7739_cast_fp16 = mul(x = var_7737_cast_fp16_1, y = const_212_promoted_to_fp16)[name = string("op_7739_cast_fp16")]; int32 var_7741 = const()[name = string("op_7741"), val = int32(-2)]; bool var_7742_interleave_0 = const()[name = string("op_7742_interleave_0"), val = bool(false)]; tensor var_7742_cast_fp16 = concat(axis = var_7741, interleave = var_7742_interleave_0, values = (var_7739_cast_fp16, var_7737_cast_fp16_0))[name = string("op_7742_cast_fp16")]; tensor var_7743_cast_fp16 = mul(x = var_7742_cast_fp16, y = var_878_cast_fp16)[name = string("op_7743_cast_fp16")]; tensor query_states_129_cast_fp16 = add(x = var_7736_cast_fp16, y = var_7743_cast_fp16)[name = string("query_states_129_cast_fp16")]; tensor var_7749_cast_fp16 = mul(x = var_7725_cast_fp16, y = var_869_cast_fp16)[name = string("op_7749_cast_fp16")]; tensor var_7750_split_sizes_0 = const()[name = string("op_7750_split_sizes_0"), val = tensor([64, 64])]; int32 var_7750_axis_0 = const()[name = string("op_7750_axis_0"), val = int32(-2)]; tensor var_7750_cast_fp16_0, tensor var_7750_cast_fp16_1 = split(axis = var_7750_axis_0, split_sizes = var_7750_split_sizes_0, x = var_7725_cast_fp16)[name = string("op_7750_cast_fp16")]; fp16 const_213_promoted_to_fp16 = const()[name = string("const_213_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7752_cast_fp16 = mul(x = var_7750_cast_fp16_1, y = const_213_promoted_to_fp16)[name = string("op_7752_cast_fp16")]; int32 var_7754 = const()[name = string("op_7754"), val = int32(-2)]; bool var_7755_interleave_0 = const()[name = string("op_7755_interleave_0"), val = bool(false)]; tensor var_7755_cast_fp16 = concat(axis = var_7754, interleave = var_7755_interleave_0, values = (var_7752_cast_fp16, var_7750_cast_fp16_0))[name = string("op_7755_cast_fp16")]; tensor var_7756_cast_fp16 = mul(x = var_7755_cast_fp16, y = var_878_cast_fp16)[name = string("op_7756_cast_fp16")]; tensor key_states_215_cast_fp16 = add(x = var_7749_cast_fp16, y = var_7756_cast_fp16)[name = string("key_states_215_cast_fp16")]; tensor expand_dims_252 = const()[name = string("expand_dims_252"), val = tensor([21])]; tensor expand_dims_253 = const()[name = string("expand_dims_253"), val = tensor([0])]; tensor expand_dims_255 = const()[name = string("expand_dims_255"), val = tensor([0])]; int32 concat_257_axis_0 = const()[name = string("concat_257_axis_0"), val = int32(0)]; bool concat_257_interleave_0 = const()[name = string("concat_257_interleave_0"), val = bool(false)]; tensor concat_257 = concat(axis = concat_257_axis_0, interleave = concat_257_interleave_0, values = (expand_dims_252, expand_dims_253, position_id, expand_dims_255))[name = string("concat_257")]; tensor expand_dims_256 = const()[name = string("expand_dims_256"), val = tensor([22])]; tensor concat_258_values1_0 = const()[name = string("concat_258_values1_0"), val = tensor([0])]; tensor concat_258_values3_0 = const()[name = string("concat_258_values3_0"), val = tensor([0])]; int32 concat_258_axis_0 = const()[name = string("concat_258_axis_0"), val = int32(0)]; bool concat_258_interleave_0 = const()[name = string("concat_258_interleave_0"), val = bool(false)]; tensor concat_258 = concat(axis = concat_258_axis_0, interleave = concat_258_interleave_0, values = (expand_dims_256, concat_258_values1_0, cache_position_end, concat_258_values3_0))[name = string("concat_258")]; tensor key_states_217_perm_0 = const()[name = string("key_states_217_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_22_stride_0 = const()[name = string("key_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_217_cast_fp16 = transpose(perm = key_states_217_perm_0, x = key_states_215_cast_fp16)[name = string("transpose_278")]; tensor key_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = key_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = key_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_22_squeeze_mask_0, stride = key_cache_internal_tensor_assign_22_stride_0, update = key_states_217_cast_fp16, x = coreml_update_state_208)[name = string("key_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_22_cast_fp16, input = key_cache)[name = string("coreml_update_state_210_write_state")]; tensor coreml_update_state_210 = read_state(input = key_cache)[name = string("coreml_update_state_210")]; tensor value_states_129_perm_0 = const()[name = string("value_states_129_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_22_stride_0 = const()[name = string("value_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_129_cast_fp16 = transpose(perm = value_states_129_perm_0, x = var_7732_cast_fp16)[name = string("transpose_277")]; tensor value_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = value_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = value_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_22_squeeze_mask_0, stride = value_cache_internal_tensor_assign_22_stride_0, update = value_states_129_cast_fp16, x = coreml_update_state_209)[name = string("value_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_22_cast_fp16, input = value_cache)[name = string("coreml_update_state_211_write_state")]; tensor coreml_update_state_211 = read_state(input = value_cache)[name = string("coreml_update_state_211")]; tensor var_7826_begin_0 = const()[name = string("op_7826_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7826_end_0 = const()[name = string("op_7826_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7826_end_mask_0 = const()[name = string("op_7826_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7826_cast_fp16 = slice_by_index(begin = var_7826_begin_0, end = var_7826_end_0, end_mask = var_7826_end_mask_0, x = coreml_update_state_210)[name = string("op_7826_cast_fp16")]; tensor tile_42 = const()[name = string("tile_42"), val = tensor([1, 1])]; int32 var_7829_axis_0 = const()[name = string("op_7829_axis_0"), val = int32(1)]; tensor var_7829_cast_fp16_0, tensor var_7829_cast_fp16_1 = split(axis = var_7829_axis_0, split_sizes = tile_42, x = var_7826_cast_fp16)[name = string("op_7829_cast_fp16")]; tensor var_7836_begin_0 = const()[name = string("op_7836_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7836_end_0 = const()[name = string("op_7836_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7836_end_mask_0 = const()[name = string("op_7836_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7836_cast_fp16 = slice_by_index(begin = var_7836_begin_0, end = var_7836_end_0, end_mask = var_7836_end_mask_0, x = coreml_update_state_211)[name = string("op_7836_cast_fp16")]; tensor tile_43 = const()[name = string("tile_43"), val = tensor([1, 1])]; int32 var_7839_axis_0 = const()[name = string("op_7839_axis_0"), val = int32(1)]; tensor var_7839_cast_fp16_0, tensor var_7839_cast_fp16_1 = split(axis = var_7839_axis_0, split_sizes = tile_43, x = var_7836_cast_fp16)[name = string("op_7839_cast_fp16")]; tensor var_7842_split_sizes_0 = const()[name = string("op_7842_split_sizes_0"), val = tensor([8, 8])]; int32 var_7842_axis_0 = const()[name = string("op_7842_axis_0"), val = int32(1)]; tensor var_7842_0, tensor var_7842_1 = split(axis = var_7842_axis_0, split_sizes = var_7842_split_sizes_0, x = query_states_129_cast_fp16)[name = string("op_7842")]; bool attn_weights_337_transpose_x_0 = const()[name = string("attn_weights_337_transpose_x_0"), val = bool(false)]; bool attn_weights_337_transpose_y_0 = const()[name = string("attn_weights_337_transpose_y_0"), val = bool(false)]; tensor attn_weights_337_cast_fp16 = matmul(transpose_x = attn_weights_337_transpose_x_0, transpose_y = attn_weights_337_transpose_y_0, x = var_7829_cast_fp16_0, y = var_7842_0)[name = string("attn_weights_337_cast_fp16")]; fp16 var_7845_to_fp16 = const()[name = string("op_7845_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_339_cast_fp16 = mul(x = attn_weights_337_cast_fp16, y = var_7845_to_fp16)[name = string("attn_weights_339_cast_fp16")]; tensor attn_weights_341_cast_fp16 = add(x = attn_weights_339_cast_fp16, y = attn_mask_1)[name = string("attn_weights_341_cast_fp16")]; int32 var_7849 = const()[name = string("op_7849"), val = int32(-2)]; tensor attn_weights_343_cast_fp16 = softmax(axis = var_7849, x = attn_weights_341_cast_fp16)[name = string("attn_weights_343_cast_fp16")]; bool var_7855_transpose_x_1 = const()[name = string("op_7855_transpose_x_1"), val = bool(true)]; bool var_7855_transpose_y_1 = const()[name = string("op_7855_transpose_y_1"), val = bool(false)]; tensor var_7855_cast_fp16 = matmul(transpose_x = var_7855_transpose_x_1, transpose_y = var_7855_transpose_y_1, x = attn_weights_343_cast_fp16, y = var_7839_cast_fp16_0)[name = string("op_7855_cast_fp16")]; bool attn_weights_345_transpose_x_0 = const()[name = string("attn_weights_345_transpose_x_0"), val = bool(false)]; bool attn_weights_345_transpose_y_0 = const()[name = string("attn_weights_345_transpose_y_0"), val = bool(false)]; tensor attn_weights_345_cast_fp16 = matmul(transpose_x = attn_weights_345_transpose_x_0, transpose_y = attn_weights_345_transpose_y_0, x = var_7829_cast_fp16_1, y = var_7842_1)[name = string("attn_weights_345_cast_fp16")]; fp16 var_7857_to_fp16 = const()[name = string("op_7857_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_347_cast_fp16 = mul(x = attn_weights_345_cast_fp16, y = var_7857_to_fp16)[name = string("attn_weights_347_cast_fp16")]; tensor attn_weights_349_cast_fp16 = add(x = attn_weights_347_cast_fp16, y = attn_mask_1)[name = string("attn_weights_349_cast_fp16")]; int32 var_7861 = const()[name = string("op_7861"), val = int32(-2)]; tensor attn_weights_351_cast_fp16 = softmax(axis = var_7861, x = attn_weights_349_cast_fp16)[name = string("attn_weights_351_cast_fp16")]; bool attn_output_169_transpose_x_1 = const()[name = string("attn_output_169_transpose_x_1"), val = bool(true)]; bool attn_output_169_transpose_y_1 = const()[name = string("attn_output_169_transpose_y_1"), val = bool(false)]; tensor attn_output_169_cast_fp16 = matmul(transpose_x = attn_output_169_transpose_x_1, transpose_y = attn_output_169_transpose_y_1, x = attn_weights_351_cast_fp16, y = var_7839_cast_fp16_1)[name = string("attn_output_169_cast_fp16")]; int32 var_7869 = const()[name = string("op_7869"), val = int32(1)]; bool attn_output_171_interleave_0 = const()[name = string("attn_output_171_interleave_0"), val = bool(false)]; tensor attn_output_171_cast_fp16 = concat(axis = var_7869, interleave = attn_output_171_interleave_0, values = (var_7855_cast_fp16, attn_output_169_cast_fp16))[name = string("attn_output_171_cast_fp16")]; tensor var_7873_perm_0 = const()[name = string("op_7873_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_263x = const()[name = string("concat_263x"), val = tensor([1, 2048, 1, -1])]; tensor var_7873_cast_fp16 = transpose(perm = var_7873_perm_0, x = attn_output_171_cast_fp16)[name = string("transpose_276")]; tensor attn_output_175_cast_fp16 = reshape(shape = concat_263x, x = var_7873_cast_fp16)[name = string("attn_output_175_cast_fp16")]; tensor hidden_states_213_strides_0 = const()[name = string("hidden_states_213_strides_0"), val = tensor([1, 1])]; string hidden_states_213_pad_type_0 = const()[name = string("hidden_states_213_pad_type_0"), val = string("valid")]; tensor hidden_states_213_pad_0 = const()[name = string("hidden_states_213_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_213_dilations_0 = const()[name = string("hidden_states_213_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_213_groups_0 = const()[name = string("hidden_states_213_groups_0"), val = int32(1)]; tensor hidden_states_213_cast_fp16 = conv(dilations = hidden_states_213_dilations_0, groups = hidden_states_213_groups_0, pad = hidden_states_213_pad_0, pad_type = hidden_states_213_pad_type_0, strides = hidden_states_213_strides_0, weight = layers_21_self_attn_o_proj_weight_cast_fp16, x = attn_output_175_cast_fp16)[name = string("hidden_states_213_cast_fp16")]; tensor hidden_states_215_cast_fp16 = add(x = hidden_states_209_cast_fp16, y = hidden_states_213_cast_fp16)[name = string("hidden_states_215_cast_fp16")]; fp16 const_218_promoted_to_fp16 = const()[name = string("const_218_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7906_cast_fp16 = mul(x = hidden_states_215_cast_fp16, y = const_218_promoted_to_fp16)[name = string("op_7906_cast_fp16")]; int32 var_7904 = const()[name = string("op_7904"), val = int32(1)]; bool doubled_173_interleave_0 = const()[name = string("doubled_173_interleave_0"), val = bool(false)]; tensor doubled_173_cast_fp16 = concat(axis = var_7904, interleave = doubled_173_interleave_0, values = (hidden_states_215_cast_fp16, var_7906_cast_fp16))[name = string("doubled_173_cast_fp16")]; tensor out_87_axes_0 = const()[name = string("out_87_axes_0"), val = tensor([1])]; tensor out_87_gamma_0_to_fp16 = const()[name = string("out_87_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528778304)))]; fp16 var_7916_to_fp16 = const()[name = string("op_7916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_87_cast_fp16 = layer_norm(axes = out_87_axes_0, epsilon = var_7916_to_fp16, gamma = out_87_gamma_0_to_fp16, x = doubled_173_cast_fp16)[name = string("out_87_cast_fp16")]; tensor var_7927_split_sizes_0 = const()[name = string("op_7927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7927_axis_0 = const()[name = string("op_7927_axis_0"), val = int32(1)]; tensor var_7927_cast_fp16_0, tensor var_7927_cast_fp16_1 = split(axis = var_7927_axis_0, split_sizes = var_7927_split_sizes_0, x = out_87_cast_fp16)[name = string("op_7927_cast_fp16")]; tensor input_43_strides_0 = const()[name = string("input_43_strides_0"), val = tensor([1, 1])]; string input_43_pad_type_0 = const()[name = string("input_43_pad_type_0"), val = string("valid")]; tensor input_43_pad_0 = const()[name = string("input_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_43_dilations_0 = const()[name = string("input_43_dilations_0"), val = tensor([1, 1])]; int32 input_43_groups_0 = const()[name = string("input_43_groups_0"), val = int32(1)]; tensor input_43_cast_fp16 = conv(dilations = input_43_dilations_0, groups = input_43_groups_0, pad = input_43_pad_0, pad_type = input_43_pad_type_0, strides = input_43_strides_0, weight = layers_21_mlp_gate_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("input_43_cast_fp16")]; tensor var_7944_cast_fp16 = silu(x = input_43_cast_fp16)[name = string("op_7944_cast_fp16")]; tensor var_7950_strides_0 = const()[name = string("op_7950_strides_0"), val = tensor([1, 1])]; string var_7950_pad_type_0 = const()[name = string("op_7950_pad_type_0"), val = string("valid")]; tensor var_7950_pad_0 = const()[name = string("op_7950_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7950_dilations_0 = const()[name = string("op_7950_dilations_0"), val = tensor([1, 1])]; int32 var_7950_groups_0 = const()[name = string("op_7950_groups_0"), val = int32(1)]; tensor var_7950_cast_fp16 = conv(dilations = var_7950_dilations_0, groups = var_7950_groups_0, pad = var_7950_pad_0, pad_type = var_7950_pad_type_0, strides = var_7950_strides_0, weight = layers_21_mlp_up_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("op_7950_cast_fp16")]; tensor x_219_cast_fp16 = mul(x = var_7944_cast_fp16, y = var_7950_cast_fp16)[name = string("x_219_cast_fp16")]; tensor hidden_states_217_strides_0 = const()[name = string("hidden_states_217_strides_0"), val = tensor([1, 1])]; string hidden_states_217_pad_type_0 = const()[name = string("hidden_states_217_pad_type_0"), val = string("valid")]; tensor hidden_states_217_pad_0 = const()[name = string("hidden_states_217_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_217_dilations_0 = const()[name = string("hidden_states_217_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_217_groups_0 = const()[name = string("hidden_states_217_groups_0"), val = int32(1)]; tensor hidden_states_217_cast_fp16 = conv(dilations = hidden_states_217_dilations_0, groups = hidden_states_217_groups_0, pad = hidden_states_217_pad_0, pad_type = hidden_states_217_pad_type_0, strides = hidden_states_217_strides_0, weight = layers_21_mlp_down_proj_weight_cast_fp16, x = x_219_cast_fp16)[name = string("hidden_states_217_cast_fp16")]; tensor hidden_states_219_cast_fp16 = add(x = hidden_states_215_cast_fp16, y = hidden_states_217_cast_fp16)[name = string("hidden_states_219_cast_fp16")]; fp16 const_220_promoted_to_fp16 = const()[name = string("const_220_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7968_cast_fp16 = mul(x = hidden_states_219_cast_fp16, y = const_220_promoted_to_fp16)[name = string("op_7968_cast_fp16")]; int32 var_7966 = const()[name = string("op_7966"), val = int32(1)]; bool doubled_177_interleave_0 = const()[name = string("doubled_177_interleave_0"), val = bool(false)]; tensor doubled_177_cast_fp16 = concat(axis = var_7966, interleave = doubled_177_interleave_0, values = (hidden_states_219_cast_fp16, var_7968_cast_fp16))[name = string("doubled_177_cast_fp16")]; tensor out_89_axes_0 = const()[name = string("out_89_axes_0"), val = tensor([1])]; tensor out_89_gamma_0_to_fp16 = const()[name = string("out_89_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528786560)))]; fp16 var_7978_to_fp16 = const()[name = string("op_7978_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_89_cast_fp16 = layer_norm(axes = out_89_axes_0, epsilon = var_7978_to_fp16, gamma = out_89_gamma_0_to_fp16, x = doubled_177_cast_fp16)[name = string("out_89_cast_fp16")]; tensor var_7989_split_sizes_0 = const()[name = string("op_7989_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7989_axis_0 = const()[name = string("op_7989_axis_0"), val = int32(1)]; tensor var_7989_cast_fp16_0, tensor var_7989_cast_fp16_1 = split(axis = var_7989_axis_0, split_sizes = var_7989_split_sizes_0, x = out_89_cast_fp16)[name = string("op_7989_cast_fp16")]; tensor query_states_133_strides_0 = const()[name = string("query_states_133_strides_0"), val = tensor([1, 1])]; string query_states_133_pad_type_0 = const()[name = string("query_states_133_pad_type_0"), val = string("valid")]; tensor query_states_133_pad_0 = const()[name = string("query_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_133_dilations_0 = const()[name = string("query_states_133_dilations_0"), val = tensor([1, 1])]; int32 query_states_133_groups_0 = const()[name = string("query_states_133_groups_0"), val = int32(1)]; tensor query_states_133_cast_fp16 = conv(dilations = query_states_133_dilations_0, groups = query_states_133_groups_0, pad = query_states_133_pad_0, pad_type = query_states_133_pad_type_0, strides = query_states_133_strides_0, weight = layers_22_self_attn_q_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("query_states_133_cast_fp16")]; tensor key_states_221_strides_0 = const()[name = string("key_states_221_strides_0"), val = tensor([1, 1])]; string key_states_221_pad_type_0 = const()[name = string("key_states_221_pad_type_0"), val = string("valid")]; tensor key_states_221_pad_0 = const()[name = string("key_states_221_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_221_dilations_0 = const()[name = string("key_states_221_dilations_0"), val = tensor([1, 1])]; int32 key_states_221_groups_0 = const()[name = string("key_states_221_groups_0"), val = int32(1)]; tensor key_states_221_cast_fp16 = conv(dilations = key_states_221_dilations_0, groups = key_states_221_groups_0, pad = key_states_221_pad_0, pad_type = key_states_221_pad_type_0, strides = key_states_221_strides_0, weight = layers_22_self_attn_k_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("key_states_221_cast_fp16")]; tensor layers_22_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528794816)))]; tensor value_states_133_strides_0 = const()[name = string("value_states_133_strides_0"), val = tensor([1, 1])]; string value_states_133_pad_type_0 = const()[name = string("value_states_133_pad_type_0"), val = string("valid")]; tensor value_states_133_pad_0 = const()[name = string("value_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_133_dilations_0 = const()[name = string("value_states_133_dilations_0"), val = tensor([1, 1])]; int32 value_states_133_groups_0 = const()[name = string("value_states_133_groups_0"), val = int32(1)]; tensor value_states_133_cast_fp16 = conv(dilations = value_states_133_dilations_0, groups = value_states_133_groups_0, pad = value_states_133_pad_0, pad_type = value_states_133_pad_type_0, strides = value_states_133_strides_0, weight = layers_22_self_attn_v_proj_weight_to_fp16, x = var_7989_cast_fp16_0)[name = string("value_states_133_cast_fp16")]; tensor concat_264x = const()[name = string("concat_264x"), val = tensor([1, 16, 128, -1])]; tensor x_221_cast_fp16 = reshape(shape = concat_264x, x = query_states_133_cast_fp16)[name = string("x_221_cast_fp16")]; tensor concat_265x = const()[name = string("concat_265x"), val = tensor([1, 2, 128, -1])]; tensor var_8046_cast_fp16 = reshape(shape = concat_265x, x = key_states_221_cast_fp16)[name = string("op_8046_cast_fp16")]; tensor concat_266x = const()[name = string("concat_266x"), val = tensor([1, 2, 128, -1])]; tensor var_8053_cast_fp16 = reshape(shape = concat_266x, x = value_states_133_cast_fp16)[name = string("op_8053_cast_fp16")]; tensor var_8057_cast_fp16 = mul(x = x_221_cast_fp16, y = var_869_cast_fp16)[name = string("op_8057_cast_fp16")]; tensor var_8058_split_sizes_0 = const()[name = string("op_8058_split_sizes_0"), val = tensor([64, 64])]; int32 var_8058_axis_0 = const()[name = string("op_8058_axis_0"), val = int32(-2)]; tensor var_8058_cast_fp16_0, tensor var_8058_cast_fp16_1 = split(axis = var_8058_axis_0, split_sizes = var_8058_split_sizes_0, x = x_221_cast_fp16)[name = string("op_8058_cast_fp16")]; fp16 const_222_promoted_to_fp16 = const()[name = string("const_222_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8060_cast_fp16 = mul(x = var_8058_cast_fp16_1, y = const_222_promoted_to_fp16)[name = string("op_8060_cast_fp16")]; int32 var_8062 = const()[name = string("op_8062"), val = int32(-2)]; bool var_8063_interleave_0 = const()[name = string("op_8063_interleave_0"), val = bool(false)]; tensor var_8063_cast_fp16 = concat(axis = var_8062, interleave = var_8063_interleave_0, values = (var_8060_cast_fp16, var_8058_cast_fp16_0))[name = string("op_8063_cast_fp16")]; tensor var_8064_cast_fp16 = mul(x = var_8063_cast_fp16, y = var_878_cast_fp16)[name = string("op_8064_cast_fp16")]; tensor query_states_135_cast_fp16 = add(x = var_8057_cast_fp16, y = var_8064_cast_fp16)[name = string("query_states_135_cast_fp16")]; tensor var_8070_cast_fp16 = mul(x = var_8046_cast_fp16, y = var_869_cast_fp16)[name = string("op_8070_cast_fp16")]; tensor var_8071_split_sizes_0 = const()[name = string("op_8071_split_sizes_0"), val = tensor([64, 64])]; int32 var_8071_axis_0 = const()[name = string("op_8071_axis_0"), val = int32(-2)]; tensor var_8071_cast_fp16_0, tensor var_8071_cast_fp16_1 = split(axis = var_8071_axis_0, split_sizes = var_8071_split_sizes_0, x = var_8046_cast_fp16)[name = string("op_8071_cast_fp16")]; fp16 const_223_promoted_to_fp16 = const()[name = string("const_223_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8073_cast_fp16 = mul(x = var_8071_cast_fp16_1, y = const_223_promoted_to_fp16)[name = string("op_8073_cast_fp16")]; int32 var_8075 = const()[name = string("op_8075"), val = int32(-2)]; bool var_8076_interleave_0 = const()[name = string("op_8076_interleave_0"), val = bool(false)]; tensor var_8076_cast_fp16 = concat(axis = var_8075, interleave = var_8076_interleave_0, values = (var_8073_cast_fp16, var_8071_cast_fp16_0))[name = string("op_8076_cast_fp16")]; tensor var_8077_cast_fp16 = mul(x = var_8076_cast_fp16, y = var_878_cast_fp16)[name = string("op_8077_cast_fp16")]; tensor key_states_225_cast_fp16 = add(x = var_8070_cast_fp16, y = var_8077_cast_fp16)[name = string("key_states_225_cast_fp16")]; tensor expand_dims_264 = const()[name = string("expand_dims_264"), val = tensor([22])]; tensor expand_dims_265 = const()[name = string("expand_dims_265"), val = tensor([0])]; tensor expand_dims_267 = const()[name = string("expand_dims_267"), val = tensor([0])]; int32 concat_269_axis_0 = const()[name = string("concat_269_axis_0"), val = int32(0)]; bool concat_269_interleave_0 = const()[name = string("concat_269_interleave_0"), val = bool(false)]; tensor concat_269 = concat(axis = concat_269_axis_0, interleave = concat_269_interleave_0, values = (expand_dims_264, expand_dims_265, position_id, expand_dims_267))[name = string("concat_269")]; tensor expand_dims_268 = const()[name = string("expand_dims_268"), val = tensor([23])]; tensor concat_270_values1_0 = const()[name = string("concat_270_values1_0"), val = tensor([0])]; tensor concat_270_values3_0 = const()[name = string("concat_270_values3_0"), val = tensor([0])]; int32 concat_270_axis_0 = const()[name = string("concat_270_axis_0"), val = int32(0)]; bool concat_270_interleave_0 = const()[name = string("concat_270_interleave_0"), val = bool(false)]; tensor concat_270 = concat(axis = concat_270_axis_0, interleave = concat_270_interleave_0, values = (expand_dims_268, concat_270_values1_0, cache_position_end, concat_270_values3_0))[name = string("concat_270")]; tensor key_states_227_perm_0 = const()[name = string("key_states_227_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_23_stride_0 = const()[name = string("key_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_227_cast_fp16 = transpose(perm = key_states_227_perm_0, x = key_states_225_cast_fp16)[name = string("transpose_275")]; tensor key_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = key_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = key_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_23_squeeze_mask_0, stride = key_cache_internal_tensor_assign_23_stride_0, update = key_states_227_cast_fp16, x = coreml_update_state_210)[name = string("key_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_23_cast_fp16, input = key_cache)[name = string("coreml_update_state_212_write_state")]; tensor coreml_update_state_212 = read_state(input = key_cache)[name = string("coreml_update_state_212")]; tensor value_states_135_perm_0 = const()[name = string("value_states_135_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_23_stride_0 = const()[name = string("value_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_135_cast_fp16 = transpose(perm = value_states_135_perm_0, x = var_8053_cast_fp16)[name = string("transpose_274")]; tensor value_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = value_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = value_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_23_squeeze_mask_0, stride = value_cache_internal_tensor_assign_23_stride_0, update = value_states_135_cast_fp16, x = coreml_update_state_211)[name = string("value_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_23_cast_fp16, input = value_cache)[name = string("coreml_update_state_213_write_state")]; tensor coreml_update_state_213 = read_state(input = value_cache)[name = string("coreml_update_state_213")]; tensor var_8147_begin_0 = const()[name = string("op_8147_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8147_end_0 = const()[name = string("op_8147_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8147_end_mask_0 = const()[name = string("op_8147_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8147_cast_fp16 = slice_by_index(begin = var_8147_begin_0, end = var_8147_end_0, end_mask = var_8147_end_mask_0, x = coreml_update_state_212)[name = string("op_8147_cast_fp16")]; tensor tile_44 = const()[name = string("tile_44"), val = tensor([1, 1])]; int32 var_8150_axis_0 = const()[name = string("op_8150_axis_0"), val = int32(1)]; tensor var_8150_cast_fp16_0, tensor var_8150_cast_fp16_1 = split(axis = var_8150_axis_0, split_sizes = tile_44, x = var_8147_cast_fp16)[name = string("op_8150_cast_fp16")]; tensor var_8157_begin_0 = const()[name = string("op_8157_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8157_end_0 = const()[name = string("op_8157_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8157_end_mask_0 = const()[name = string("op_8157_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8157_cast_fp16 = slice_by_index(begin = var_8157_begin_0, end = var_8157_end_0, end_mask = var_8157_end_mask_0, x = coreml_update_state_213)[name = string("op_8157_cast_fp16")]; tensor tile_45 = const()[name = string("tile_45"), val = tensor([1, 1])]; int32 var_8160_axis_0 = const()[name = string("op_8160_axis_0"), val = int32(1)]; tensor var_8160_cast_fp16_0, tensor var_8160_cast_fp16_1 = split(axis = var_8160_axis_0, split_sizes = tile_45, x = var_8157_cast_fp16)[name = string("op_8160_cast_fp16")]; tensor var_8163_split_sizes_0 = const()[name = string("op_8163_split_sizes_0"), val = tensor([8, 8])]; int32 var_8163_axis_0 = const()[name = string("op_8163_axis_0"), val = int32(1)]; tensor var_8163_0, tensor var_8163_1 = split(axis = var_8163_axis_0, split_sizes = var_8163_split_sizes_0, x = query_states_135_cast_fp16)[name = string("op_8163")]; bool attn_weights_353_transpose_x_0 = const()[name = string("attn_weights_353_transpose_x_0"), val = bool(false)]; bool attn_weights_353_transpose_y_0 = const()[name = string("attn_weights_353_transpose_y_0"), val = bool(false)]; tensor attn_weights_353_cast_fp16 = matmul(transpose_x = attn_weights_353_transpose_x_0, transpose_y = attn_weights_353_transpose_y_0, x = var_8150_cast_fp16_0, y = var_8163_0)[name = string("attn_weights_353_cast_fp16")]; fp16 var_8166_to_fp16 = const()[name = string("op_8166_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_355_cast_fp16 = mul(x = attn_weights_353_cast_fp16, y = var_8166_to_fp16)[name = string("attn_weights_355_cast_fp16")]; tensor attn_weights_357_cast_fp16 = add(x = attn_weights_355_cast_fp16, y = attn_mask_1)[name = string("attn_weights_357_cast_fp16")]; int32 var_8170 = const()[name = string("op_8170"), val = int32(-2)]; tensor attn_weights_359_cast_fp16 = softmax(axis = var_8170, x = attn_weights_357_cast_fp16)[name = string("attn_weights_359_cast_fp16")]; bool var_8176_transpose_x_1 = const()[name = string("op_8176_transpose_x_1"), val = bool(true)]; bool var_8176_transpose_y_1 = const()[name = string("op_8176_transpose_y_1"), val = bool(false)]; tensor var_8176_cast_fp16 = matmul(transpose_x = var_8176_transpose_x_1, transpose_y = var_8176_transpose_y_1, x = attn_weights_359_cast_fp16, y = var_8160_cast_fp16_0)[name = string("op_8176_cast_fp16")]; bool attn_weights_361_transpose_x_0 = const()[name = string("attn_weights_361_transpose_x_0"), val = bool(false)]; bool attn_weights_361_transpose_y_0 = const()[name = string("attn_weights_361_transpose_y_0"), val = bool(false)]; tensor attn_weights_361_cast_fp16 = matmul(transpose_x = attn_weights_361_transpose_x_0, transpose_y = attn_weights_361_transpose_y_0, x = var_8150_cast_fp16_1, y = var_8163_1)[name = string("attn_weights_361_cast_fp16")]; fp16 var_8178_to_fp16 = const()[name = string("op_8178_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_363_cast_fp16 = mul(x = attn_weights_361_cast_fp16, y = var_8178_to_fp16)[name = string("attn_weights_363_cast_fp16")]; tensor attn_weights_365_cast_fp16 = add(x = attn_weights_363_cast_fp16, y = attn_mask_1)[name = string("attn_weights_365_cast_fp16")]; int32 var_8182 = const()[name = string("op_8182"), val = int32(-2)]; tensor attn_weights_367_cast_fp16 = softmax(axis = var_8182, x = attn_weights_365_cast_fp16)[name = string("attn_weights_367_cast_fp16")]; bool attn_output_177_transpose_x_1 = const()[name = string("attn_output_177_transpose_x_1"), val = bool(true)]; bool attn_output_177_transpose_y_1 = const()[name = string("attn_output_177_transpose_y_1"), val = bool(false)]; tensor attn_output_177_cast_fp16 = matmul(transpose_x = attn_output_177_transpose_x_1, transpose_y = attn_output_177_transpose_y_1, x = attn_weights_367_cast_fp16, y = var_8160_cast_fp16_1)[name = string("attn_output_177_cast_fp16")]; int32 var_8190 = const()[name = string("op_8190"), val = int32(1)]; bool attn_output_179_interleave_0 = const()[name = string("attn_output_179_interleave_0"), val = bool(false)]; tensor attn_output_179_cast_fp16 = concat(axis = var_8190, interleave = attn_output_179_interleave_0, values = (var_8176_cast_fp16, attn_output_177_cast_fp16))[name = string("attn_output_179_cast_fp16")]; tensor var_8194_perm_0 = const()[name = string("op_8194_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_275x = const()[name = string("concat_275x"), val = tensor([1, 2048, 1, -1])]; tensor var_8194_cast_fp16 = transpose(perm = var_8194_perm_0, x = attn_output_179_cast_fp16)[name = string("transpose_273")]; tensor attn_output_183_cast_fp16 = reshape(shape = concat_275x, x = var_8194_cast_fp16)[name = string("attn_output_183_cast_fp16")]; tensor layers_22_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1529843456)))]; tensor hidden_states_223_strides_0 = const()[name = string("hidden_states_223_strides_0"), val = tensor([1, 1])]; string hidden_states_223_pad_type_0 = const()[name = string("hidden_states_223_pad_type_0"), val = string("valid")]; tensor hidden_states_223_pad_0 = const()[name = string("hidden_states_223_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_223_dilations_0 = const()[name = string("hidden_states_223_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_223_groups_0 = const()[name = string("hidden_states_223_groups_0"), val = int32(1)]; tensor hidden_states_223_cast_fp16 = conv(dilations = hidden_states_223_dilations_0, groups = hidden_states_223_groups_0, pad = hidden_states_223_pad_0, pad_type = hidden_states_223_pad_type_0, strides = hidden_states_223_strides_0, weight = layers_22_self_attn_o_proj_weight_to_fp16, x = attn_output_183_cast_fp16)[name = string("hidden_states_223_cast_fp16")]; tensor hidden_states_225_cast_fp16 = add(x = hidden_states_219_cast_fp16, y = hidden_states_223_cast_fp16)[name = string("hidden_states_225_cast_fp16")]; fp16 const_228_promoted_to_fp16 = const()[name = string("const_228_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8227_cast_fp16 = mul(x = hidden_states_225_cast_fp16, y = const_228_promoted_to_fp16)[name = string("op_8227_cast_fp16")]; int32 var_8225 = const()[name = string("op_8225"), val = int32(1)]; bool doubled_181_interleave_0 = const()[name = string("doubled_181_interleave_0"), val = bool(false)]; tensor doubled_181_cast_fp16 = concat(axis = var_8225, interleave = doubled_181_interleave_0, values = (hidden_states_225_cast_fp16, var_8227_cast_fp16))[name = string("doubled_181_cast_fp16")]; tensor out_91_axes_0 = const()[name = string("out_91_axes_0"), val = tensor([1])]; tensor out_91_gamma_0_to_fp16 = const()[name = string("out_91_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538232128)))]; fp16 var_8237_to_fp16 = const()[name = string("op_8237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_91_cast_fp16 = layer_norm(axes = out_91_axes_0, epsilon = var_8237_to_fp16, gamma = out_91_gamma_0_to_fp16, x = doubled_181_cast_fp16)[name = string("out_91_cast_fp16")]; tensor var_8248_split_sizes_0 = const()[name = string("op_8248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8248_axis_0 = const()[name = string("op_8248_axis_0"), val = int32(1)]; tensor var_8248_cast_fp16_0, tensor var_8248_cast_fp16_1 = split(axis = var_8248_axis_0, split_sizes = var_8248_split_sizes_0, x = out_91_cast_fp16)[name = string("op_8248_cast_fp16")]; tensor input_45_strides_0 = const()[name = string("input_45_strides_0"), val = tensor([1, 1])]; string input_45_pad_type_0 = const()[name = string("input_45_pad_type_0"), val = string("valid")]; tensor input_45_pad_0 = const()[name = string("input_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_45_dilations_0 = const()[name = string("input_45_dilations_0"), val = tensor([1, 1])]; int32 input_45_groups_0 = const()[name = string("input_45_groups_0"), val = int32(1)]; tensor input_45_cast_fp16 = conv(dilations = input_45_dilations_0, groups = input_45_groups_0, pad = input_45_pad_0, pad_type = input_45_pad_type_0, strides = input_45_strides_0, weight = layers_22_mlp_gate_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("input_45_cast_fp16")]; tensor var_8265_cast_fp16 = silu(x = input_45_cast_fp16)[name = string("op_8265_cast_fp16")]; tensor var_8271_strides_0 = const()[name = string("op_8271_strides_0"), val = tensor([1, 1])]; string var_8271_pad_type_0 = const()[name = string("op_8271_pad_type_0"), val = string("valid")]; tensor var_8271_pad_0 = const()[name = string("op_8271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8271_dilations_0 = const()[name = string("op_8271_dilations_0"), val = tensor([1, 1])]; int32 var_8271_groups_0 = const()[name = string("op_8271_groups_0"), val = int32(1)]; tensor var_8271_cast_fp16 = conv(dilations = var_8271_dilations_0, groups = var_8271_groups_0, pad = var_8271_pad_0, pad_type = var_8271_pad_type_0, strides = var_8271_strides_0, weight = layers_22_mlp_up_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("op_8271_cast_fp16")]; tensor x_229_cast_fp16 = mul(x = var_8265_cast_fp16, y = var_8271_cast_fp16)[name = string("x_229_cast_fp16")]; tensor hidden_states_227_strides_0 = const()[name = string("hidden_states_227_strides_0"), val = tensor([1, 1])]; string hidden_states_227_pad_type_0 = const()[name = string("hidden_states_227_pad_type_0"), val = string("valid")]; tensor hidden_states_227_pad_0 = const()[name = string("hidden_states_227_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_227_dilations_0 = const()[name = string("hidden_states_227_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_227_groups_0 = const()[name = string("hidden_states_227_groups_0"), val = int32(1)]; tensor hidden_states_227_cast_fp16 = conv(dilations = hidden_states_227_dilations_0, groups = hidden_states_227_groups_0, pad = hidden_states_227_pad_0, pad_type = hidden_states_227_pad_type_0, strides = hidden_states_227_strides_0, weight = layers_22_mlp_down_proj_weight_cast_fp16, x = x_229_cast_fp16)[name = string("hidden_states_227_cast_fp16")]; tensor hidden_states_229_cast_fp16 = add(x = hidden_states_225_cast_fp16, y = hidden_states_227_cast_fp16)[name = string("hidden_states_229_cast_fp16")]; fp16 const_230_promoted_to_fp16 = const()[name = string("const_230_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8289_cast_fp16 = mul(x = hidden_states_229_cast_fp16, y = const_230_promoted_to_fp16)[name = string("op_8289_cast_fp16")]; int32 var_8287 = const()[name = string("op_8287"), val = int32(1)]; bool doubled_185_interleave_0 = const()[name = string("doubled_185_interleave_0"), val = bool(false)]; tensor doubled_185_cast_fp16 = concat(axis = var_8287, interleave = doubled_185_interleave_0, values = (hidden_states_229_cast_fp16, var_8289_cast_fp16))[name = string("doubled_185_cast_fp16")]; tensor out_93_axes_0 = const()[name = string("out_93_axes_0"), val = tensor([1])]; tensor out_93_gamma_0_to_fp16 = const()[name = string("out_93_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538240384)))]; fp16 var_8299_to_fp16 = const()[name = string("op_8299_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_93_cast_fp16 = layer_norm(axes = out_93_axes_0, epsilon = var_8299_to_fp16, gamma = out_93_gamma_0_to_fp16, x = doubled_185_cast_fp16)[name = string("out_93_cast_fp16")]; tensor var_8310_split_sizes_0 = const()[name = string("op_8310_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8310_axis_0 = const()[name = string("op_8310_axis_0"), val = int32(1)]; tensor var_8310_cast_fp16_0, tensor var_8310_cast_fp16_1 = split(axis = var_8310_axis_0, split_sizes = var_8310_split_sizes_0, x = out_93_cast_fp16)[name = string("op_8310_cast_fp16")]; tensor query_states_139_strides_0 = const()[name = string("query_states_139_strides_0"), val = tensor([1, 1])]; string query_states_139_pad_type_0 = const()[name = string("query_states_139_pad_type_0"), val = string("valid")]; tensor query_states_139_pad_0 = const()[name = string("query_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_139_dilations_0 = const()[name = string("query_states_139_dilations_0"), val = tensor([1, 1])]; int32 query_states_139_groups_0 = const()[name = string("query_states_139_groups_0"), val = int32(1)]; tensor query_states_139_cast_fp16 = conv(dilations = query_states_139_dilations_0, groups = query_states_139_groups_0, pad = query_states_139_pad_0, pad_type = query_states_139_pad_type_0, strides = query_states_139_strides_0, weight = layers_23_self_attn_q_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("query_states_139_cast_fp16")]; tensor key_states_231_strides_0 = const()[name = string("key_states_231_strides_0"), val = tensor([1, 1])]; string key_states_231_pad_type_0 = const()[name = string("key_states_231_pad_type_0"), val = string("valid")]; tensor key_states_231_pad_0 = const()[name = string("key_states_231_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_231_dilations_0 = const()[name = string("key_states_231_dilations_0"), val = tensor([1, 1])]; int32 key_states_231_groups_0 = const()[name = string("key_states_231_groups_0"), val = int32(1)]; tensor key_states_231_cast_fp16 = conv(dilations = key_states_231_dilations_0, groups = key_states_231_groups_0, pad = key_states_231_pad_0, pad_type = key_states_231_pad_type_0, strides = key_states_231_strides_0, weight = layers_23_self_attn_k_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("key_states_231_cast_fp16")]; tensor layers_23_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_23_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538248640)))]; tensor value_states_139_strides_0 = const()[name = string("value_states_139_strides_0"), val = tensor([1, 1])]; string value_states_139_pad_type_0 = const()[name = string("value_states_139_pad_type_0"), val = string("valid")]; tensor value_states_139_pad_0 = const()[name = string("value_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_139_dilations_0 = const()[name = string("value_states_139_dilations_0"), val = tensor([1, 1])]; int32 value_states_139_groups_0 = const()[name = string("value_states_139_groups_0"), val = int32(1)]; tensor value_states_139_cast_fp16 = conv(dilations = value_states_139_dilations_0, groups = value_states_139_groups_0, pad = value_states_139_pad_0, pad_type = value_states_139_pad_type_0, strides = value_states_139_strides_0, weight = layers_23_self_attn_v_proj_weight_to_fp16, x = var_8310_cast_fp16_0)[name = string("value_states_139_cast_fp16")]; tensor concat_276x = const()[name = string("concat_276x"), val = tensor([1, 16, 128, -1])]; tensor x_231_cast_fp16 = reshape(shape = concat_276x, x = query_states_139_cast_fp16)[name = string("x_231_cast_fp16")]; tensor concat_277x = const()[name = string("concat_277x"), val = tensor([1, 2, 128, -1])]; tensor var_8367_cast_fp16 = reshape(shape = concat_277x, x = key_states_231_cast_fp16)[name = string("op_8367_cast_fp16")]; tensor concat_278x = const()[name = string("concat_278x"), val = tensor([1, 2, 128, -1])]; tensor var_8374_cast_fp16 = reshape(shape = concat_278x, x = value_states_139_cast_fp16)[name = string("op_8374_cast_fp16")]; tensor var_8378_cast_fp16 = mul(x = x_231_cast_fp16, y = var_869_cast_fp16)[name = string("op_8378_cast_fp16")]; tensor var_8379_split_sizes_0 = const()[name = string("op_8379_split_sizes_0"), val = tensor([64, 64])]; int32 var_8379_axis_0 = const()[name = string("op_8379_axis_0"), val = int32(-2)]; tensor var_8379_cast_fp16_0, tensor var_8379_cast_fp16_1 = split(axis = var_8379_axis_0, split_sizes = var_8379_split_sizes_0, x = x_231_cast_fp16)[name = string("op_8379_cast_fp16")]; fp16 const_232_promoted_to_fp16 = const()[name = string("const_232_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8381_cast_fp16 = mul(x = var_8379_cast_fp16_1, y = const_232_promoted_to_fp16)[name = string("op_8381_cast_fp16")]; int32 var_8383 = const()[name = string("op_8383"), val = int32(-2)]; bool var_8384_interleave_0 = const()[name = string("op_8384_interleave_0"), val = bool(false)]; tensor var_8384_cast_fp16 = concat(axis = var_8383, interleave = var_8384_interleave_0, values = (var_8381_cast_fp16, var_8379_cast_fp16_0))[name = string("op_8384_cast_fp16")]; tensor var_8385_cast_fp16 = mul(x = var_8384_cast_fp16, y = var_878_cast_fp16)[name = string("op_8385_cast_fp16")]; tensor query_states_141_cast_fp16 = add(x = var_8378_cast_fp16, y = var_8385_cast_fp16)[name = string("query_states_141_cast_fp16")]; tensor var_8391_cast_fp16 = mul(x = var_8367_cast_fp16, y = var_869_cast_fp16)[name = string("op_8391_cast_fp16")]; tensor var_8392_split_sizes_0 = const()[name = string("op_8392_split_sizes_0"), val = tensor([64, 64])]; int32 var_8392_axis_0 = const()[name = string("op_8392_axis_0"), val = int32(-2)]; tensor var_8392_cast_fp16_0, tensor var_8392_cast_fp16_1 = split(axis = var_8392_axis_0, split_sizes = var_8392_split_sizes_0, x = var_8367_cast_fp16)[name = string("op_8392_cast_fp16")]; fp16 const_233_promoted_to_fp16 = const()[name = string("const_233_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8394_cast_fp16 = mul(x = var_8392_cast_fp16_1, y = const_233_promoted_to_fp16)[name = string("op_8394_cast_fp16")]; int32 var_8396 = const()[name = string("op_8396"), val = int32(-2)]; bool var_8397_interleave_0 = const()[name = string("op_8397_interleave_0"), val = bool(false)]; tensor var_8397_cast_fp16 = concat(axis = var_8396, interleave = var_8397_interleave_0, values = (var_8394_cast_fp16, var_8392_cast_fp16_0))[name = string("op_8397_cast_fp16")]; tensor var_8398_cast_fp16 = mul(x = var_8397_cast_fp16, y = var_878_cast_fp16)[name = string("op_8398_cast_fp16")]; tensor key_states_235_cast_fp16 = add(x = var_8391_cast_fp16, y = var_8398_cast_fp16)[name = string("key_states_235_cast_fp16")]; tensor expand_dims_276 = const()[name = string("expand_dims_276"), val = tensor([23])]; tensor expand_dims_277 = const()[name = string("expand_dims_277"), val = tensor([0])]; tensor expand_dims_279 = const()[name = string("expand_dims_279"), val = tensor([0])]; int32 concat_281_axis_0 = const()[name = string("concat_281_axis_0"), val = int32(0)]; bool concat_281_interleave_0 = const()[name = string("concat_281_interleave_0"), val = bool(false)]; tensor concat_281 = concat(axis = concat_281_axis_0, interleave = concat_281_interleave_0, values = (expand_dims_276, expand_dims_277, position_id, expand_dims_279))[name = string("concat_281")]; tensor expand_dims_280 = const()[name = string("expand_dims_280"), val = tensor([24])]; tensor concat_282_values1_0 = const()[name = string("concat_282_values1_0"), val = tensor([0])]; tensor concat_282_values3_0 = const()[name = string("concat_282_values3_0"), val = tensor([0])]; int32 concat_282_axis_0 = const()[name = string("concat_282_axis_0"), val = int32(0)]; bool concat_282_interleave_0 = const()[name = string("concat_282_interleave_0"), val = bool(false)]; tensor concat_282 = concat(axis = concat_282_axis_0, interleave = concat_282_interleave_0, values = (expand_dims_280, concat_282_values1_0, cache_position_end, concat_282_values3_0))[name = string("concat_282")]; tensor key_states_237_perm_0 = const()[name = string("key_states_237_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_24_stride_0 = const()[name = string("key_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_237_cast_fp16 = transpose(perm = key_states_237_perm_0, x = key_states_235_cast_fp16)[name = string("transpose_272")]; tensor key_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = key_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = key_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_24_squeeze_mask_0, stride = key_cache_internal_tensor_assign_24_stride_0, update = key_states_237_cast_fp16, x = coreml_update_state_212)[name = string("key_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_24_cast_fp16, input = key_cache)[name = string("coreml_update_state_214_write_state")]; tensor coreml_update_state_214 = read_state(input = key_cache)[name = string("coreml_update_state_214")]; tensor value_states_141_perm_0 = const()[name = string("value_states_141_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_24_stride_0 = const()[name = string("value_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_141_cast_fp16 = transpose(perm = value_states_141_perm_0, x = var_8374_cast_fp16)[name = string("transpose_271")]; tensor value_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = value_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = value_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_24_squeeze_mask_0, stride = value_cache_internal_tensor_assign_24_stride_0, update = value_states_141_cast_fp16, x = coreml_update_state_213)[name = string("value_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_24_cast_fp16, input = value_cache)[name = string("coreml_update_state_215_write_state")]; tensor coreml_update_state_215 = read_state(input = value_cache)[name = string("coreml_update_state_215")]; tensor var_8468_begin_0 = const()[name = string("op_8468_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8468_end_0 = const()[name = string("op_8468_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8468_end_mask_0 = const()[name = string("op_8468_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8468_cast_fp16 = slice_by_index(begin = var_8468_begin_0, end = var_8468_end_0, end_mask = var_8468_end_mask_0, x = coreml_update_state_214)[name = string("op_8468_cast_fp16")]; tensor tile_46 = const()[name = string("tile_46"), val = tensor([1, 1])]; int32 var_8471_axis_0 = const()[name = string("op_8471_axis_0"), val = int32(1)]; tensor var_8471_cast_fp16_0, tensor var_8471_cast_fp16_1 = split(axis = var_8471_axis_0, split_sizes = tile_46, x = var_8468_cast_fp16)[name = string("op_8471_cast_fp16")]; tensor var_8478_begin_0 = const()[name = string("op_8478_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8478_end_0 = const()[name = string("op_8478_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8478_end_mask_0 = const()[name = string("op_8478_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8478_cast_fp16 = slice_by_index(begin = var_8478_begin_0, end = var_8478_end_0, end_mask = var_8478_end_mask_0, x = coreml_update_state_215)[name = string("op_8478_cast_fp16")]; tensor tile_47 = const()[name = string("tile_47"), val = tensor([1, 1])]; int32 var_8481_axis_0 = const()[name = string("op_8481_axis_0"), val = int32(1)]; tensor var_8481_cast_fp16_0, tensor var_8481_cast_fp16_1 = split(axis = var_8481_axis_0, split_sizes = tile_47, x = var_8478_cast_fp16)[name = string("op_8481_cast_fp16")]; tensor var_8484_split_sizes_0 = const()[name = string("op_8484_split_sizes_0"), val = tensor([8, 8])]; int32 var_8484_axis_0 = const()[name = string("op_8484_axis_0"), val = int32(1)]; tensor var_8484_0, tensor var_8484_1 = split(axis = var_8484_axis_0, split_sizes = var_8484_split_sizes_0, x = query_states_141_cast_fp16)[name = string("op_8484")]; bool attn_weights_369_transpose_x_0 = const()[name = string("attn_weights_369_transpose_x_0"), val = bool(false)]; bool attn_weights_369_transpose_y_0 = const()[name = string("attn_weights_369_transpose_y_0"), val = bool(false)]; tensor attn_weights_369_cast_fp16 = matmul(transpose_x = attn_weights_369_transpose_x_0, transpose_y = attn_weights_369_transpose_y_0, x = var_8471_cast_fp16_0, y = var_8484_0)[name = string("attn_weights_369_cast_fp16")]; fp16 var_8487_to_fp16 = const()[name = string("op_8487_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_371_cast_fp16 = mul(x = attn_weights_369_cast_fp16, y = var_8487_to_fp16)[name = string("attn_weights_371_cast_fp16")]; tensor attn_weights_373_cast_fp16 = add(x = attn_weights_371_cast_fp16, y = attn_mask_1)[name = string("attn_weights_373_cast_fp16")]; int32 var_8491 = const()[name = string("op_8491"), val = int32(-2)]; tensor attn_weights_375_cast_fp16 = softmax(axis = var_8491, x = attn_weights_373_cast_fp16)[name = string("attn_weights_375_cast_fp16")]; bool var_8497_transpose_x_1 = const()[name = string("op_8497_transpose_x_1"), val = bool(true)]; bool var_8497_transpose_y_1 = const()[name = string("op_8497_transpose_y_1"), val = bool(false)]; tensor var_8497_cast_fp16 = matmul(transpose_x = var_8497_transpose_x_1, transpose_y = var_8497_transpose_y_1, x = attn_weights_375_cast_fp16, y = var_8481_cast_fp16_0)[name = string("op_8497_cast_fp16")]; bool attn_weights_377_transpose_x_0 = const()[name = string("attn_weights_377_transpose_x_0"), val = bool(false)]; bool attn_weights_377_transpose_y_0 = const()[name = string("attn_weights_377_transpose_y_0"), val = bool(false)]; tensor attn_weights_377_cast_fp16 = matmul(transpose_x = attn_weights_377_transpose_x_0, transpose_y = attn_weights_377_transpose_y_0, x = var_8471_cast_fp16_1, y = var_8484_1)[name = string("attn_weights_377_cast_fp16")]; fp16 var_8499_to_fp16 = const()[name = string("op_8499_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_379_cast_fp16 = mul(x = attn_weights_377_cast_fp16, y = var_8499_to_fp16)[name = string("attn_weights_379_cast_fp16")]; tensor attn_weights_381_cast_fp16 = add(x = attn_weights_379_cast_fp16, y = attn_mask_1)[name = string("attn_weights_381_cast_fp16")]; int32 var_8503 = const()[name = string("op_8503"), val = int32(-2)]; tensor attn_weights_383_cast_fp16 = softmax(axis = var_8503, x = attn_weights_381_cast_fp16)[name = string("attn_weights_383_cast_fp16")]; bool attn_output_185_transpose_x_1 = const()[name = string("attn_output_185_transpose_x_1"), val = bool(true)]; bool attn_output_185_transpose_y_1 = const()[name = string("attn_output_185_transpose_y_1"), val = bool(false)]; tensor attn_output_185_cast_fp16 = matmul(transpose_x = attn_output_185_transpose_x_1, transpose_y = attn_output_185_transpose_y_1, x = attn_weights_383_cast_fp16, y = var_8481_cast_fp16_1)[name = string("attn_output_185_cast_fp16")]; int32 var_8511 = const()[name = string("op_8511"), val = int32(1)]; bool attn_output_187_interleave_0 = const()[name = string("attn_output_187_interleave_0"), val = bool(false)]; tensor attn_output_187_cast_fp16 = concat(axis = var_8511, interleave = attn_output_187_interleave_0, values = (var_8497_cast_fp16, attn_output_185_cast_fp16))[name = string("attn_output_187_cast_fp16")]; tensor var_8515_perm_0 = const()[name = string("op_8515_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_287x = const()[name = string("concat_287x"), val = tensor([1, 2048, 1, -1])]; tensor var_8515_cast_fp16 = transpose(perm = var_8515_perm_0, x = attn_output_187_cast_fp16)[name = string("transpose_270")]; tensor attn_output_191_cast_fp16 = reshape(shape = concat_287x, x = var_8515_cast_fp16)[name = string("attn_output_191_cast_fp16")]; tensor hidden_states_233_strides_0 = const()[name = string("hidden_states_233_strides_0"), val = tensor([1, 1])]; string hidden_states_233_pad_type_0 = const()[name = string("hidden_states_233_pad_type_0"), val = string("valid")]; tensor hidden_states_233_pad_0 = const()[name = string("hidden_states_233_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_233_dilations_0 = const()[name = string("hidden_states_233_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_233_groups_0 = const()[name = string("hidden_states_233_groups_0"), val = int32(1)]; tensor hidden_states_233_cast_fp16 = conv(dilations = hidden_states_233_dilations_0, groups = hidden_states_233_groups_0, pad = hidden_states_233_pad_0, pad_type = hidden_states_233_pad_type_0, strides = hidden_states_233_strides_0, weight = layers_23_self_attn_o_proj_weight_cast_fp16, x = attn_output_191_cast_fp16)[name = string("hidden_states_233_cast_fp16")]; tensor hidden_states_235_cast_fp16 = add(x = hidden_states_229_cast_fp16, y = hidden_states_233_cast_fp16)[name = string("hidden_states_235_cast_fp16")]; fp16 const_238_promoted_to_fp16 = const()[name = string("const_238_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8548_cast_fp16 = mul(x = hidden_states_235_cast_fp16, y = const_238_promoted_to_fp16)[name = string("op_8548_cast_fp16")]; int32 var_8546 = const()[name = string("op_8546"), val = int32(1)]; bool doubled_189_interleave_0 = const()[name = string("doubled_189_interleave_0"), val = bool(false)]; tensor doubled_189_cast_fp16 = concat(axis = var_8546, interleave = doubled_189_interleave_0, values = (hidden_states_235_cast_fp16, var_8548_cast_fp16))[name = string("doubled_189_cast_fp16")]; tensor out_95_axes_0 = const()[name = string("out_95_axes_0"), val = tensor([1])]; tensor out_95_gamma_0_to_fp16 = const()[name = string("out_95_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539297280)))]; fp16 var_8558_to_fp16 = const()[name = string("op_8558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_95_cast_fp16 = layer_norm(axes = out_95_axes_0, epsilon = var_8558_to_fp16, gamma = out_95_gamma_0_to_fp16, x = doubled_189_cast_fp16)[name = string("out_95_cast_fp16")]; tensor var_8569_split_sizes_0 = const()[name = string("op_8569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8569_axis_0 = const()[name = string("op_8569_axis_0"), val = int32(1)]; tensor var_8569_cast_fp16_0, tensor var_8569_cast_fp16_1 = split(axis = var_8569_axis_0, split_sizes = var_8569_split_sizes_0, x = out_95_cast_fp16)[name = string("op_8569_cast_fp16")]; tensor input_47_strides_0 = const()[name = string("input_47_strides_0"), val = tensor([1, 1])]; string input_47_pad_type_0 = const()[name = string("input_47_pad_type_0"), val = string("valid")]; tensor input_47_pad_0 = const()[name = string("input_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_47_dilations_0 = const()[name = string("input_47_dilations_0"), val = tensor([1, 1])]; int32 input_47_groups_0 = const()[name = string("input_47_groups_0"), val = int32(1)]; tensor input_47_cast_fp16 = conv(dilations = input_47_dilations_0, groups = input_47_groups_0, pad = input_47_pad_0, pad_type = input_47_pad_type_0, strides = input_47_strides_0, weight = layers_23_mlp_gate_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("input_47_cast_fp16")]; tensor var_8586_cast_fp16 = silu(x = input_47_cast_fp16)[name = string("op_8586_cast_fp16")]; tensor var_8592_strides_0 = const()[name = string("op_8592_strides_0"), val = tensor([1, 1])]; string var_8592_pad_type_0 = const()[name = string("op_8592_pad_type_0"), val = string("valid")]; tensor var_8592_pad_0 = const()[name = string("op_8592_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8592_dilations_0 = const()[name = string("op_8592_dilations_0"), val = tensor([1, 1])]; int32 var_8592_groups_0 = const()[name = string("op_8592_groups_0"), val = int32(1)]; tensor var_8592_cast_fp16 = conv(dilations = var_8592_dilations_0, groups = var_8592_groups_0, pad = var_8592_pad_0, pad_type = var_8592_pad_type_0, strides = var_8592_strides_0, weight = layers_23_mlp_up_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("op_8592_cast_fp16")]; tensor x_239_cast_fp16 = mul(x = var_8586_cast_fp16, y = var_8592_cast_fp16)[name = string("x_239_cast_fp16")]; tensor hidden_states_237_strides_0 = const()[name = string("hidden_states_237_strides_0"), val = tensor([1, 1])]; string hidden_states_237_pad_type_0 = const()[name = string("hidden_states_237_pad_type_0"), val = string("valid")]; tensor hidden_states_237_pad_0 = const()[name = string("hidden_states_237_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_237_dilations_0 = const()[name = string("hidden_states_237_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_237_groups_0 = const()[name = string("hidden_states_237_groups_0"), val = int32(1)]; tensor hidden_states_237_cast_fp16 = conv(dilations = hidden_states_237_dilations_0, groups = hidden_states_237_groups_0, pad = hidden_states_237_pad_0, pad_type = hidden_states_237_pad_type_0, strides = hidden_states_237_strides_0, weight = layers_23_mlp_down_proj_weight_cast_fp16, x = x_239_cast_fp16)[name = string("hidden_states_237_cast_fp16")]; tensor hidden_states_239_cast_fp16 = add(x = hidden_states_235_cast_fp16, y = hidden_states_237_cast_fp16)[name = string("hidden_states_239_cast_fp16")]; fp16 const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8610_cast_fp16 = mul(x = hidden_states_239_cast_fp16, y = const_240_promoted_to_fp16)[name = string("op_8610_cast_fp16")]; int32 var_8608 = const()[name = string("op_8608"), val = int32(1)]; bool doubled_193_interleave_0 = const()[name = string("doubled_193_interleave_0"), val = bool(false)]; tensor doubled_193_cast_fp16 = concat(axis = var_8608, interleave = doubled_193_interleave_0, values = (hidden_states_239_cast_fp16, var_8610_cast_fp16))[name = string("doubled_193_cast_fp16")]; tensor out_97_axes_0 = const()[name = string("out_97_axes_0"), val = tensor([1])]; tensor out_97_gamma_0_to_fp16 = const()[name = string("out_97_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539305536)))]; fp16 var_8620_to_fp16 = const()[name = string("op_8620_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_97_cast_fp16 = layer_norm(axes = out_97_axes_0, epsilon = var_8620_to_fp16, gamma = out_97_gamma_0_to_fp16, x = doubled_193_cast_fp16)[name = string("out_97_cast_fp16")]; tensor var_8631_split_sizes_0 = const()[name = string("op_8631_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8631_axis_0 = const()[name = string("op_8631_axis_0"), val = int32(1)]; tensor var_8631_cast_fp16_0, tensor var_8631_cast_fp16_1 = split(axis = var_8631_axis_0, split_sizes = var_8631_split_sizes_0, x = out_97_cast_fp16)[name = string("op_8631_cast_fp16")]; tensor query_states_145_strides_0 = const()[name = string("query_states_145_strides_0"), val = tensor([1, 1])]; string query_states_145_pad_type_0 = const()[name = string("query_states_145_pad_type_0"), val = string("valid")]; tensor query_states_145_pad_0 = const()[name = string("query_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_145_dilations_0 = const()[name = string("query_states_145_dilations_0"), val = tensor([1, 1])]; int32 query_states_145_groups_0 = const()[name = string("query_states_145_groups_0"), val = int32(1)]; tensor query_states_145_cast_fp16 = conv(dilations = query_states_145_dilations_0, groups = query_states_145_groups_0, pad = query_states_145_pad_0, pad_type = query_states_145_pad_type_0, strides = query_states_145_strides_0, weight = layers_24_self_attn_q_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("query_states_145_cast_fp16")]; tensor key_states_241_strides_0 = const()[name = string("key_states_241_strides_0"), val = tensor([1, 1])]; string key_states_241_pad_type_0 = const()[name = string("key_states_241_pad_type_0"), val = string("valid")]; tensor key_states_241_pad_0 = const()[name = string("key_states_241_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_241_dilations_0 = const()[name = string("key_states_241_dilations_0"), val = tensor([1, 1])]; int32 key_states_241_groups_0 = const()[name = string("key_states_241_groups_0"), val = int32(1)]; tensor key_states_241_cast_fp16 = conv(dilations = key_states_241_dilations_0, groups = key_states_241_groups_0, pad = key_states_241_pad_0, pad_type = key_states_241_pad_type_0, strides = key_states_241_strides_0, weight = layers_24_self_attn_k_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("key_states_241_cast_fp16")]; tensor layers_24_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_24_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539313792)))]; tensor value_states_145_strides_0 = const()[name = string("value_states_145_strides_0"), val = tensor([1, 1])]; string value_states_145_pad_type_0 = const()[name = string("value_states_145_pad_type_0"), val = string("valid")]; tensor value_states_145_pad_0 = const()[name = string("value_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_145_dilations_0 = const()[name = string("value_states_145_dilations_0"), val = tensor([1, 1])]; int32 value_states_145_groups_0 = const()[name = string("value_states_145_groups_0"), val = int32(1)]; tensor value_states_145_cast_fp16 = conv(dilations = value_states_145_dilations_0, groups = value_states_145_groups_0, pad = value_states_145_pad_0, pad_type = value_states_145_pad_type_0, strides = value_states_145_strides_0, weight = layers_24_self_attn_v_proj_weight_to_fp16, x = var_8631_cast_fp16_0)[name = string("value_states_145_cast_fp16")]; tensor concat_288x = const()[name = string("concat_288x"), val = tensor([1, 16, 128, -1])]; tensor x_241_cast_fp16 = reshape(shape = concat_288x, x = query_states_145_cast_fp16)[name = string("x_241_cast_fp16")]; tensor concat_289x = const()[name = string("concat_289x"), val = tensor([1, 2, 128, -1])]; tensor var_8688_cast_fp16 = reshape(shape = concat_289x, x = key_states_241_cast_fp16)[name = string("op_8688_cast_fp16")]; tensor concat_290x = const()[name = string("concat_290x"), val = tensor([1, 2, 128, -1])]; tensor var_8695_cast_fp16 = reshape(shape = concat_290x, x = value_states_145_cast_fp16)[name = string("op_8695_cast_fp16")]; tensor var_8699_cast_fp16 = mul(x = x_241_cast_fp16, y = var_869_cast_fp16)[name = string("op_8699_cast_fp16")]; tensor var_8700_split_sizes_0 = const()[name = string("op_8700_split_sizes_0"), val = tensor([64, 64])]; int32 var_8700_axis_0 = const()[name = string("op_8700_axis_0"), val = int32(-2)]; tensor var_8700_cast_fp16_0, tensor var_8700_cast_fp16_1 = split(axis = var_8700_axis_0, split_sizes = var_8700_split_sizes_0, x = x_241_cast_fp16)[name = string("op_8700_cast_fp16")]; fp16 const_242_promoted_to_fp16 = const()[name = string("const_242_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8702_cast_fp16 = mul(x = var_8700_cast_fp16_1, y = const_242_promoted_to_fp16)[name = string("op_8702_cast_fp16")]; int32 var_8704 = const()[name = string("op_8704"), val = int32(-2)]; bool var_8705_interleave_0 = const()[name = string("op_8705_interleave_0"), val = bool(false)]; tensor var_8705_cast_fp16 = concat(axis = var_8704, interleave = var_8705_interleave_0, values = (var_8702_cast_fp16, var_8700_cast_fp16_0))[name = string("op_8705_cast_fp16")]; tensor var_8706_cast_fp16 = mul(x = var_8705_cast_fp16, y = var_878_cast_fp16)[name = string("op_8706_cast_fp16")]; tensor query_states_147_cast_fp16 = add(x = var_8699_cast_fp16, y = var_8706_cast_fp16)[name = string("query_states_147_cast_fp16")]; tensor var_8712_cast_fp16 = mul(x = var_8688_cast_fp16, y = var_869_cast_fp16)[name = string("op_8712_cast_fp16")]; tensor var_8713_split_sizes_0 = const()[name = string("op_8713_split_sizes_0"), val = tensor([64, 64])]; int32 var_8713_axis_0 = const()[name = string("op_8713_axis_0"), val = int32(-2)]; tensor var_8713_cast_fp16_0, tensor var_8713_cast_fp16_1 = split(axis = var_8713_axis_0, split_sizes = var_8713_split_sizes_0, x = var_8688_cast_fp16)[name = string("op_8713_cast_fp16")]; fp16 const_243_promoted_to_fp16 = const()[name = string("const_243_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8715_cast_fp16 = mul(x = var_8713_cast_fp16_1, y = const_243_promoted_to_fp16)[name = string("op_8715_cast_fp16")]; int32 var_8717 = const()[name = string("op_8717"), val = int32(-2)]; bool var_8718_interleave_0 = const()[name = string("op_8718_interleave_0"), val = bool(false)]; tensor var_8718_cast_fp16 = concat(axis = var_8717, interleave = var_8718_interleave_0, values = (var_8715_cast_fp16, var_8713_cast_fp16_0))[name = string("op_8718_cast_fp16")]; tensor var_8719_cast_fp16 = mul(x = var_8718_cast_fp16, y = var_878_cast_fp16)[name = string("op_8719_cast_fp16")]; tensor key_states_245_cast_fp16 = add(x = var_8712_cast_fp16, y = var_8719_cast_fp16)[name = string("key_states_245_cast_fp16")]; tensor expand_dims_288 = const()[name = string("expand_dims_288"), val = tensor([24])]; tensor expand_dims_289 = const()[name = string("expand_dims_289"), val = tensor([0])]; tensor expand_dims_291 = const()[name = string("expand_dims_291"), val = tensor([0])]; int32 concat_293_axis_0 = const()[name = string("concat_293_axis_0"), val = int32(0)]; bool concat_293_interleave_0 = const()[name = string("concat_293_interleave_0"), val = bool(false)]; tensor concat_293 = concat(axis = concat_293_axis_0, interleave = concat_293_interleave_0, values = (expand_dims_288, expand_dims_289, position_id, expand_dims_291))[name = string("concat_293")]; tensor expand_dims_292 = const()[name = string("expand_dims_292"), val = tensor([25])]; tensor concat_294_values1_0 = const()[name = string("concat_294_values1_0"), val = tensor([0])]; tensor concat_294_values3_0 = const()[name = string("concat_294_values3_0"), val = tensor([0])]; int32 concat_294_axis_0 = const()[name = string("concat_294_axis_0"), val = int32(0)]; bool concat_294_interleave_0 = const()[name = string("concat_294_interleave_0"), val = bool(false)]; tensor concat_294 = concat(axis = concat_294_axis_0, interleave = concat_294_interleave_0, values = (expand_dims_292, concat_294_values1_0, cache_position_end, concat_294_values3_0))[name = string("concat_294")]; tensor key_states_247_perm_0 = const()[name = string("key_states_247_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_25_stride_0 = const()[name = string("key_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_247_cast_fp16 = transpose(perm = key_states_247_perm_0, x = key_states_245_cast_fp16)[name = string("transpose_269")]; tensor key_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = key_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = key_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_25_squeeze_mask_0, stride = key_cache_internal_tensor_assign_25_stride_0, update = key_states_247_cast_fp16, x = coreml_update_state_214)[name = string("key_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_25_cast_fp16, input = key_cache)[name = string("coreml_update_state_216_write_state")]; tensor coreml_update_state_216 = read_state(input = key_cache)[name = string("coreml_update_state_216")]; tensor value_states_147_perm_0 = const()[name = string("value_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_25_stride_0 = const()[name = string("value_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_147_cast_fp16 = transpose(perm = value_states_147_perm_0, x = var_8695_cast_fp16)[name = string("transpose_268")]; tensor value_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = value_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = value_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_25_squeeze_mask_0, stride = value_cache_internal_tensor_assign_25_stride_0, update = value_states_147_cast_fp16, x = coreml_update_state_215)[name = string("value_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_25_cast_fp16, input = value_cache)[name = string("coreml_update_state_217_write_state")]; tensor coreml_update_state_217 = read_state(input = value_cache)[name = string("coreml_update_state_217")]; tensor var_8789_begin_0 = const()[name = string("op_8789_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8789_end_0 = const()[name = string("op_8789_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8789_end_mask_0 = const()[name = string("op_8789_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8789_cast_fp16 = slice_by_index(begin = var_8789_begin_0, end = var_8789_end_0, end_mask = var_8789_end_mask_0, x = coreml_update_state_216)[name = string("op_8789_cast_fp16")]; tensor tile_48 = const()[name = string("tile_48"), val = tensor([1, 1])]; int32 var_8792_axis_0 = const()[name = string("op_8792_axis_0"), val = int32(1)]; tensor var_8792_cast_fp16_0, tensor var_8792_cast_fp16_1 = split(axis = var_8792_axis_0, split_sizes = tile_48, x = var_8789_cast_fp16)[name = string("op_8792_cast_fp16")]; tensor var_8799_begin_0 = const()[name = string("op_8799_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8799_end_0 = const()[name = string("op_8799_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8799_end_mask_0 = const()[name = string("op_8799_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8799_cast_fp16 = slice_by_index(begin = var_8799_begin_0, end = var_8799_end_0, end_mask = var_8799_end_mask_0, x = coreml_update_state_217)[name = string("op_8799_cast_fp16")]; tensor tile_49 = const()[name = string("tile_49"), val = tensor([1, 1])]; int32 var_8802_axis_0 = const()[name = string("op_8802_axis_0"), val = int32(1)]; tensor var_8802_cast_fp16_0, tensor var_8802_cast_fp16_1 = split(axis = var_8802_axis_0, split_sizes = tile_49, x = var_8799_cast_fp16)[name = string("op_8802_cast_fp16")]; tensor var_8805_split_sizes_0 = const()[name = string("op_8805_split_sizes_0"), val = tensor([8, 8])]; int32 var_8805_axis_0 = const()[name = string("op_8805_axis_0"), val = int32(1)]; tensor var_8805_0, tensor var_8805_1 = split(axis = var_8805_axis_0, split_sizes = var_8805_split_sizes_0, x = query_states_147_cast_fp16)[name = string("op_8805")]; bool attn_weights_385_transpose_x_0 = const()[name = string("attn_weights_385_transpose_x_0"), val = bool(false)]; bool attn_weights_385_transpose_y_0 = const()[name = string("attn_weights_385_transpose_y_0"), val = bool(false)]; tensor attn_weights_385_cast_fp16 = matmul(transpose_x = attn_weights_385_transpose_x_0, transpose_y = attn_weights_385_transpose_y_0, x = var_8792_cast_fp16_0, y = var_8805_0)[name = string("attn_weights_385_cast_fp16")]; fp16 var_8808_to_fp16 = const()[name = string("op_8808_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_387_cast_fp16 = mul(x = attn_weights_385_cast_fp16, y = var_8808_to_fp16)[name = string("attn_weights_387_cast_fp16")]; tensor attn_weights_389_cast_fp16 = add(x = attn_weights_387_cast_fp16, y = attn_mask_1)[name = string("attn_weights_389_cast_fp16")]; int32 var_8812 = const()[name = string("op_8812"), val = int32(-2)]; tensor attn_weights_391_cast_fp16 = softmax(axis = var_8812, x = attn_weights_389_cast_fp16)[name = string("attn_weights_391_cast_fp16")]; bool var_8818_transpose_x_1 = const()[name = string("op_8818_transpose_x_1"), val = bool(true)]; bool var_8818_transpose_y_1 = const()[name = string("op_8818_transpose_y_1"), val = bool(false)]; tensor var_8818_cast_fp16 = matmul(transpose_x = var_8818_transpose_x_1, transpose_y = var_8818_transpose_y_1, x = attn_weights_391_cast_fp16, y = var_8802_cast_fp16_0)[name = string("op_8818_cast_fp16")]; bool attn_weights_393_transpose_x_0 = const()[name = string("attn_weights_393_transpose_x_0"), val = bool(false)]; bool attn_weights_393_transpose_y_0 = const()[name = string("attn_weights_393_transpose_y_0"), val = bool(false)]; tensor attn_weights_393_cast_fp16 = matmul(transpose_x = attn_weights_393_transpose_x_0, transpose_y = attn_weights_393_transpose_y_0, x = var_8792_cast_fp16_1, y = var_8805_1)[name = string("attn_weights_393_cast_fp16")]; fp16 var_8820_to_fp16 = const()[name = string("op_8820_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_395_cast_fp16 = mul(x = attn_weights_393_cast_fp16, y = var_8820_to_fp16)[name = string("attn_weights_395_cast_fp16")]; tensor attn_weights_397_cast_fp16 = add(x = attn_weights_395_cast_fp16, y = attn_mask_1)[name = string("attn_weights_397_cast_fp16")]; int32 var_8824 = const()[name = string("op_8824"), val = int32(-2)]; tensor attn_weights_399_cast_fp16 = softmax(axis = var_8824, x = attn_weights_397_cast_fp16)[name = string("attn_weights_399_cast_fp16")]; bool attn_output_193_transpose_x_1 = const()[name = string("attn_output_193_transpose_x_1"), val = bool(true)]; bool attn_output_193_transpose_y_1 = const()[name = string("attn_output_193_transpose_y_1"), val = bool(false)]; tensor attn_output_193_cast_fp16 = matmul(transpose_x = attn_output_193_transpose_x_1, transpose_y = attn_output_193_transpose_y_1, x = attn_weights_399_cast_fp16, y = var_8802_cast_fp16_1)[name = string("attn_output_193_cast_fp16")]; int32 var_8832 = const()[name = string("op_8832"), val = int32(1)]; bool attn_output_195_interleave_0 = const()[name = string("attn_output_195_interleave_0"), val = bool(false)]; tensor attn_output_195_cast_fp16 = concat(axis = var_8832, interleave = attn_output_195_interleave_0, values = (var_8818_cast_fp16, attn_output_193_cast_fp16))[name = string("attn_output_195_cast_fp16")]; tensor var_8836_perm_0 = const()[name = string("op_8836_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_299x = const()[name = string("concat_299x"), val = tensor([1, 2048, 1, -1])]; tensor var_8836_cast_fp16 = transpose(perm = var_8836_perm_0, x = attn_output_195_cast_fp16)[name = string("transpose_267")]; tensor attn_output_199_cast_fp16 = reshape(shape = concat_299x, x = var_8836_cast_fp16)[name = string("attn_output_199_cast_fp16")]; tensor hidden_states_243_strides_0 = const()[name = string("hidden_states_243_strides_0"), val = tensor([1, 1])]; string hidden_states_243_pad_type_0 = const()[name = string("hidden_states_243_pad_type_0"), val = string("valid")]; tensor hidden_states_243_pad_0 = const()[name = string("hidden_states_243_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_243_dilations_0 = const()[name = string("hidden_states_243_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_243_groups_0 = const()[name = string("hidden_states_243_groups_0"), val = int32(1)]; tensor hidden_states_243_cast_fp16 = conv(dilations = hidden_states_243_dilations_0, groups = hidden_states_243_groups_0, pad = hidden_states_243_pad_0, pad_type = hidden_states_243_pad_type_0, strides = hidden_states_243_strides_0, weight = layers_24_self_attn_o_proj_weight_cast_fp16, x = attn_output_199_cast_fp16)[name = string("hidden_states_243_cast_fp16")]; tensor hidden_states_245_cast_fp16 = add(x = hidden_states_239_cast_fp16, y = hidden_states_243_cast_fp16)[name = string("hidden_states_245_cast_fp16")]; fp16 const_248_promoted_to_fp16 = const()[name = string("const_248_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8869_cast_fp16 = mul(x = hidden_states_245_cast_fp16, y = const_248_promoted_to_fp16)[name = string("op_8869_cast_fp16")]; int32 var_8867 = const()[name = string("op_8867"), val = int32(1)]; bool doubled_197_interleave_0 = const()[name = string("doubled_197_interleave_0"), val = bool(false)]; tensor doubled_197_cast_fp16 = concat(axis = var_8867, interleave = doubled_197_interleave_0, values = (hidden_states_245_cast_fp16, var_8869_cast_fp16))[name = string("doubled_197_cast_fp16")]; tensor out_99_axes_0 = const()[name = string("out_99_axes_0"), val = tensor([1])]; tensor out_99_gamma_0_to_fp16 = const()[name = string("out_99_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540362432)))]; fp16 var_8879_to_fp16 = const()[name = string("op_8879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_99_cast_fp16 = layer_norm(axes = out_99_axes_0, epsilon = var_8879_to_fp16, gamma = out_99_gamma_0_to_fp16, x = doubled_197_cast_fp16)[name = string("out_99_cast_fp16")]; tensor var_8890_split_sizes_0 = const()[name = string("op_8890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8890_axis_0 = const()[name = string("op_8890_axis_0"), val = int32(1)]; tensor var_8890_cast_fp16_0, tensor var_8890_cast_fp16_1 = split(axis = var_8890_axis_0, split_sizes = var_8890_split_sizes_0, x = out_99_cast_fp16)[name = string("op_8890_cast_fp16")]; tensor input_49_strides_0 = const()[name = string("input_49_strides_0"), val = tensor([1, 1])]; string input_49_pad_type_0 = const()[name = string("input_49_pad_type_0"), val = string("valid")]; tensor input_49_pad_0 = const()[name = string("input_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_49_dilations_0 = const()[name = string("input_49_dilations_0"), val = tensor([1, 1])]; int32 input_49_groups_0 = const()[name = string("input_49_groups_0"), val = int32(1)]; tensor input_49_cast_fp16 = conv(dilations = input_49_dilations_0, groups = input_49_groups_0, pad = input_49_pad_0, pad_type = input_49_pad_type_0, strides = input_49_strides_0, weight = layers_24_mlp_gate_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("input_49_cast_fp16")]; tensor var_8907_cast_fp16 = silu(x = input_49_cast_fp16)[name = string("op_8907_cast_fp16")]; tensor var_8913_strides_0 = const()[name = string("op_8913_strides_0"), val = tensor([1, 1])]; string var_8913_pad_type_0 = const()[name = string("op_8913_pad_type_0"), val = string("valid")]; tensor var_8913_pad_0 = const()[name = string("op_8913_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8913_dilations_0 = const()[name = string("op_8913_dilations_0"), val = tensor([1, 1])]; int32 var_8913_groups_0 = const()[name = string("op_8913_groups_0"), val = int32(1)]; tensor var_8913_cast_fp16 = conv(dilations = var_8913_dilations_0, groups = var_8913_groups_0, pad = var_8913_pad_0, pad_type = var_8913_pad_type_0, strides = var_8913_strides_0, weight = layers_24_mlp_up_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("op_8913_cast_fp16")]; tensor x_249_cast_fp16 = mul(x = var_8907_cast_fp16, y = var_8913_cast_fp16)[name = string("x_249_cast_fp16")]; tensor hidden_states_247_strides_0 = const()[name = string("hidden_states_247_strides_0"), val = tensor([1, 1])]; string hidden_states_247_pad_type_0 = const()[name = string("hidden_states_247_pad_type_0"), val = string("valid")]; tensor hidden_states_247_pad_0 = const()[name = string("hidden_states_247_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_247_dilations_0 = const()[name = string("hidden_states_247_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_247_groups_0 = const()[name = string("hidden_states_247_groups_0"), val = int32(1)]; tensor hidden_states_247_cast_fp16 = conv(dilations = hidden_states_247_dilations_0, groups = hidden_states_247_groups_0, pad = hidden_states_247_pad_0, pad_type = hidden_states_247_pad_type_0, strides = hidden_states_247_strides_0, weight = layers_24_mlp_down_proj_weight_cast_fp16, x = x_249_cast_fp16)[name = string("hidden_states_247_cast_fp16")]; tensor hidden_states_249_cast_fp16 = add(x = hidden_states_245_cast_fp16, y = hidden_states_247_cast_fp16)[name = string("hidden_states_249_cast_fp16")]; fp16 const_250_promoted_to_fp16 = const()[name = string("const_250_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8931_cast_fp16 = mul(x = hidden_states_249_cast_fp16, y = const_250_promoted_to_fp16)[name = string("op_8931_cast_fp16")]; int32 var_8929 = const()[name = string("op_8929"), val = int32(1)]; bool doubled_201_interleave_0 = const()[name = string("doubled_201_interleave_0"), val = bool(false)]; tensor doubled_201_cast_fp16 = concat(axis = var_8929, interleave = doubled_201_interleave_0, values = (hidden_states_249_cast_fp16, var_8931_cast_fp16))[name = string("doubled_201_cast_fp16")]; tensor out_101_axes_0 = const()[name = string("out_101_axes_0"), val = tensor([1])]; tensor out_101_gamma_0_to_fp16 = const()[name = string("out_101_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540370688)))]; fp16 var_8941_to_fp16 = const()[name = string("op_8941_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_101_cast_fp16 = layer_norm(axes = out_101_axes_0, epsilon = var_8941_to_fp16, gamma = out_101_gamma_0_to_fp16, x = doubled_201_cast_fp16)[name = string("out_101_cast_fp16")]; tensor var_8952_split_sizes_0 = const()[name = string("op_8952_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8952_axis_0 = const()[name = string("op_8952_axis_0"), val = int32(1)]; tensor var_8952_cast_fp16_0, tensor var_8952_cast_fp16_1 = split(axis = var_8952_axis_0, split_sizes = var_8952_split_sizes_0, x = out_101_cast_fp16)[name = string("op_8952_cast_fp16")]; tensor query_states_151_strides_0 = const()[name = string("query_states_151_strides_0"), val = tensor([1, 1])]; string query_states_151_pad_type_0 = const()[name = string("query_states_151_pad_type_0"), val = string("valid")]; tensor query_states_151_pad_0 = const()[name = string("query_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_151_dilations_0 = const()[name = string("query_states_151_dilations_0"), val = tensor([1, 1])]; int32 query_states_151_groups_0 = const()[name = string("query_states_151_groups_0"), val = int32(1)]; tensor query_states_151_cast_fp16 = conv(dilations = query_states_151_dilations_0, groups = query_states_151_groups_0, pad = query_states_151_pad_0, pad_type = query_states_151_pad_type_0, strides = query_states_151_strides_0, weight = layers_25_self_attn_q_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("query_states_151_cast_fp16")]; tensor key_states_251_strides_0 = const()[name = string("key_states_251_strides_0"), val = tensor([1, 1])]; string key_states_251_pad_type_0 = const()[name = string("key_states_251_pad_type_0"), val = string("valid")]; tensor key_states_251_pad_0 = const()[name = string("key_states_251_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_251_dilations_0 = const()[name = string("key_states_251_dilations_0"), val = tensor([1, 1])]; int32 key_states_251_groups_0 = const()[name = string("key_states_251_groups_0"), val = int32(1)]; tensor key_states_251_cast_fp16 = conv(dilations = key_states_251_dilations_0, groups = key_states_251_groups_0, pad = key_states_251_pad_0, pad_type = key_states_251_pad_type_0, strides = key_states_251_strides_0, weight = layers_25_self_attn_k_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("key_states_251_cast_fp16")]; tensor layers_25_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_25_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540378944)))]; tensor value_states_151_strides_0 = const()[name = string("value_states_151_strides_0"), val = tensor([1, 1])]; string value_states_151_pad_type_0 = const()[name = string("value_states_151_pad_type_0"), val = string("valid")]; tensor value_states_151_pad_0 = const()[name = string("value_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_151_dilations_0 = const()[name = string("value_states_151_dilations_0"), val = tensor([1, 1])]; int32 value_states_151_groups_0 = const()[name = string("value_states_151_groups_0"), val = int32(1)]; tensor value_states_151_cast_fp16 = conv(dilations = value_states_151_dilations_0, groups = value_states_151_groups_0, pad = value_states_151_pad_0, pad_type = value_states_151_pad_type_0, strides = value_states_151_strides_0, weight = layers_25_self_attn_v_proj_weight_to_fp16, x = var_8952_cast_fp16_0)[name = string("value_states_151_cast_fp16")]; tensor concat_300x = const()[name = string("concat_300x"), val = tensor([1, 16, 128, -1])]; tensor x_251_cast_fp16 = reshape(shape = concat_300x, x = query_states_151_cast_fp16)[name = string("x_251_cast_fp16")]; tensor concat_301x = const()[name = string("concat_301x"), val = tensor([1, 2, 128, -1])]; tensor var_9009_cast_fp16 = reshape(shape = concat_301x, x = key_states_251_cast_fp16)[name = string("op_9009_cast_fp16")]; tensor concat_302x = const()[name = string("concat_302x"), val = tensor([1, 2, 128, -1])]; tensor var_9016_cast_fp16 = reshape(shape = concat_302x, x = value_states_151_cast_fp16)[name = string("op_9016_cast_fp16")]; tensor var_9020_cast_fp16 = mul(x = x_251_cast_fp16, y = var_869_cast_fp16)[name = string("op_9020_cast_fp16")]; tensor var_9021_split_sizes_0 = const()[name = string("op_9021_split_sizes_0"), val = tensor([64, 64])]; int32 var_9021_axis_0 = const()[name = string("op_9021_axis_0"), val = int32(-2)]; tensor var_9021_cast_fp16_0, tensor var_9021_cast_fp16_1 = split(axis = var_9021_axis_0, split_sizes = var_9021_split_sizes_0, x = x_251_cast_fp16)[name = string("op_9021_cast_fp16")]; fp16 const_252_promoted_to_fp16 = const()[name = string("const_252_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9023_cast_fp16 = mul(x = var_9021_cast_fp16_1, y = const_252_promoted_to_fp16)[name = string("op_9023_cast_fp16")]; int32 var_9025 = const()[name = string("op_9025"), val = int32(-2)]; bool var_9026_interleave_0 = const()[name = string("op_9026_interleave_0"), val = bool(false)]; tensor var_9026_cast_fp16 = concat(axis = var_9025, interleave = var_9026_interleave_0, values = (var_9023_cast_fp16, var_9021_cast_fp16_0))[name = string("op_9026_cast_fp16")]; tensor var_9027_cast_fp16 = mul(x = var_9026_cast_fp16, y = var_878_cast_fp16)[name = string("op_9027_cast_fp16")]; tensor query_states_153_cast_fp16 = add(x = var_9020_cast_fp16, y = var_9027_cast_fp16)[name = string("query_states_153_cast_fp16")]; tensor var_9033_cast_fp16 = mul(x = var_9009_cast_fp16, y = var_869_cast_fp16)[name = string("op_9033_cast_fp16")]; tensor var_9034_split_sizes_0 = const()[name = string("op_9034_split_sizes_0"), val = tensor([64, 64])]; int32 var_9034_axis_0 = const()[name = string("op_9034_axis_0"), val = int32(-2)]; tensor var_9034_cast_fp16_0, tensor var_9034_cast_fp16_1 = split(axis = var_9034_axis_0, split_sizes = var_9034_split_sizes_0, x = var_9009_cast_fp16)[name = string("op_9034_cast_fp16")]; fp16 const_253_promoted_to_fp16 = const()[name = string("const_253_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9036_cast_fp16 = mul(x = var_9034_cast_fp16_1, y = const_253_promoted_to_fp16)[name = string("op_9036_cast_fp16")]; int32 var_9038 = const()[name = string("op_9038"), val = int32(-2)]; bool var_9039_interleave_0 = const()[name = string("op_9039_interleave_0"), val = bool(false)]; tensor var_9039_cast_fp16 = concat(axis = var_9038, interleave = var_9039_interleave_0, values = (var_9036_cast_fp16, var_9034_cast_fp16_0))[name = string("op_9039_cast_fp16")]; tensor var_9040_cast_fp16 = mul(x = var_9039_cast_fp16, y = var_878_cast_fp16)[name = string("op_9040_cast_fp16")]; tensor key_states_255_cast_fp16 = add(x = var_9033_cast_fp16, y = var_9040_cast_fp16)[name = string("key_states_255_cast_fp16")]; tensor expand_dims_300 = const()[name = string("expand_dims_300"), val = tensor([25])]; tensor expand_dims_301 = const()[name = string("expand_dims_301"), val = tensor([0])]; tensor expand_dims_303 = const()[name = string("expand_dims_303"), val = tensor([0])]; int32 concat_305_axis_0 = const()[name = string("concat_305_axis_0"), val = int32(0)]; bool concat_305_interleave_0 = const()[name = string("concat_305_interleave_0"), val = bool(false)]; tensor concat_305 = concat(axis = concat_305_axis_0, interleave = concat_305_interleave_0, values = (expand_dims_300, expand_dims_301, position_id, expand_dims_303))[name = string("concat_305")]; tensor expand_dims_304 = const()[name = string("expand_dims_304"), val = tensor([26])]; tensor concat_306_values1_0 = const()[name = string("concat_306_values1_0"), val = tensor([0])]; tensor concat_306_values3_0 = const()[name = string("concat_306_values3_0"), val = tensor([0])]; int32 concat_306_axis_0 = const()[name = string("concat_306_axis_0"), val = int32(0)]; bool concat_306_interleave_0 = const()[name = string("concat_306_interleave_0"), val = bool(false)]; tensor concat_306 = concat(axis = concat_306_axis_0, interleave = concat_306_interleave_0, values = (expand_dims_304, concat_306_values1_0, cache_position_end, concat_306_values3_0))[name = string("concat_306")]; tensor key_states_257_perm_0 = const()[name = string("key_states_257_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_26_stride_0 = const()[name = string("key_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_257_cast_fp16 = transpose(perm = key_states_257_perm_0, x = key_states_255_cast_fp16)[name = string("transpose_266")]; tensor key_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = key_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = key_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_26_squeeze_mask_0, stride = key_cache_internal_tensor_assign_26_stride_0, update = key_states_257_cast_fp16, x = coreml_update_state_216)[name = string("key_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_26_cast_fp16, input = key_cache)[name = string("coreml_update_state_218_write_state")]; tensor coreml_update_state_218 = read_state(input = key_cache)[name = string("coreml_update_state_218")]; tensor value_states_153_perm_0 = const()[name = string("value_states_153_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_26_stride_0 = const()[name = string("value_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_153_cast_fp16 = transpose(perm = value_states_153_perm_0, x = var_9016_cast_fp16)[name = string("transpose_265")]; tensor value_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = value_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = value_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_26_squeeze_mask_0, stride = value_cache_internal_tensor_assign_26_stride_0, update = value_states_153_cast_fp16, x = coreml_update_state_217)[name = string("value_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_26_cast_fp16, input = value_cache)[name = string("coreml_update_state_219_write_state")]; tensor coreml_update_state_219 = read_state(input = value_cache)[name = string("coreml_update_state_219")]; tensor var_9110_begin_0 = const()[name = string("op_9110_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9110_end_0 = const()[name = string("op_9110_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9110_end_mask_0 = const()[name = string("op_9110_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9110_cast_fp16 = slice_by_index(begin = var_9110_begin_0, end = var_9110_end_0, end_mask = var_9110_end_mask_0, x = coreml_update_state_218)[name = string("op_9110_cast_fp16")]; tensor tile_50 = const()[name = string("tile_50"), val = tensor([1, 1])]; int32 var_9113_axis_0 = const()[name = string("op_9113_axis_0"), val = int32(1)]; tensor var_9113_cast_fp16_0, tensor var_9113_cast_fp16_1 = split(axis = var_9113_axis_0, split_sizes = tile_50, x = var_9110_cast_fp16)[name = string("op_9113_cast_fp16")]; tensor var_9120_begin_0 = const()[name = string("op_9120_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9120_end_0 = const()[name = string("op_9120_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9120_end_mask_0 = const()[name = string("op_9120_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9120_cast_fp16 = slice_by_index(begin = var_9120_begin_0, end = var_9120_end_0, end_mask = var_9120_end_mask_0, x = coreml_update_state_219)[name = string("op_9120_cast_fp16")]; tensor tile_51 = const()[name = string("tile_51"), val = tensor([1, 1])]; int32 var_9123_axis_0 = const()[name = string("op_9123_axis_0"), val = int32(1)]; tensor var_9123_cast_fp16_0, tensor var_9123_cast_fp16_1 = split(axis = var_9123_axis_0, split_sizes = tile_51, x = var_9120_cast_fp16)[name = string("op_9123_cast_fp16")]; tensor var_9126_split_sizes_0 = const()[name = string("op_9126_split_sizes_0"), val = tensor([8, 8])]; int32 var_9126_axis_0 = const()[name = string("op_9126_axis_0"), val = int32(1)]; tensor var_9126_0, tensor var_9126_1 = split(axis = var_9126_axis_0, split_sizes = var_9126_split_sizes_0, x = query_states_153_cast_fp16)[name = string("op_9126")]; bool attn_weights_401_transpose_x_0 = const()[name = string("attn_weights_401_transpose_x_0"), val = bool(false)]; bool attn_weights_401_transpose_y_0 = const()[name = string("attn_weights_401_transpose_y_0"), val = bool(false)]; tensor attn_weights_401_cast_fp16 = matmul(transpose_x = attn_weights_401_transpose_x_0, transpose_y = attn_weights_401_transpose_y_0, x = var_9113_cast_fp16_0, y = var_9126_0)[name = string("attn_weights_401_cast_fp16")]; fp16 var_9129_to_fp16 = const()[name = string("op_9129_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_403_cast_fp16 = mul(x = attn_weights_401_cast_fp16, y = var_9129_to_fp16)[name = string("attn_weights_403_cast_fp16")]; tensor attn_weights_405_cast_fp16 = add(x = attn_weights_403_cast_fp16, y = attn_mask_1)[name = string("attn_weights_405_cast_fp16")]; int32 var_9133 = const()[name = string("op_9133"), val = int32(-2)]; tensor attn_weights_407_cast_fp16 = softmax(axis = var_9133, x = attn_weights_405_cast_fp16)[name = string("attn_weights_407_cast_fp16")]; bool var_9139_transpose_x_1 = const()[name = string("op_9139_transpose_x_1"), val = bool(true)]; bool var_9139_transpose_y_1 = const()[name = string("op_9139_transpose_y_1"), val = bool(false)]; tensor var_9139_cast_fp16 = matmul(transpose_x = var_9139_transpose_x_1, transpose_y = var_9139_transpose_y_1, x = attn_weights_407_cast_fp16, y = var_9123_cast_fp16_0)[name = string("op_9139_cast_fp16")]; bool attn_weights_409_transpose_x_0 = const()[name = string("attn_weights_409_transpose_x_0"), val = bool(false)]; bool attn_weights_409_transpose_y_0 = const()[name = string("attn_weights_409_transpose_y_0"), val = bool(false)]; tensor attn_weights_409_cast_fp16 = matmul(transpose_x = attn_weights_409_transpose_x_0, transpose_y = attn_weights_409_transpose_y_0, x = var_9113_cast_fp16_1, y = var_9126_1)[name = string("attn_weights_409_cast_fp16")]; fp16 var_9141_to_fp16 = const()[name = string("op_9141_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_411_cast_fp16 = mul(x = attn_weights_409_cast_fp16, y = var_9141_to_fp16)[name = string("attn_weights_411_cast_fp16")]; tensor attn_weights_413_cast_fp16 = add(x = attn_weights_411_cast_fp16, y = attn_mask_1)[name = string("attn_weights_413_cast_fp16")]; int32 var_9145 = const()[name = string("op_9145"), val = int32(-2)]; tensor attn_weights_415_cast_fp16 = softmax(axis = var_9145, x = attn_weights_413_cast_fp16)[name = string("attn_weights_415_cast_fp16")]; bool attn_output_201_transpose_x_1 = const()[name = string("attn_output_201_transpose_x_1"), val = bool(true)]; bool attn_output_201_transpose_y_1 = const()[name = string("attn_output_201_transpose_y_1"), val = bool(false)]; tensor attn_output_201_cast_fp16 = matmul(transpose_x = attn_output_201_transpose_x_1, transpose_y = attn_output_201_transpose_y_1, x = attn_weights_415_cast_fp16, y = var_9123_cast_fp16_1)[name = string("attn_output_201_cast_fp16")]; int32 var_9153 = const()[name = string("op_9153"), val = int32(1)]; bool attn_output_203_interleave_0 = const()[name = string("attn_output_203_interleave_0"), val = bool(false)]; tensor attn_output_203_cast_fp16 = concat(axis = var_9153, interleave = attn_output_203_interleave_0, values = (var_9139_cast_fp16, attn_output_201_cast_fp16))[name = string("attn_output_203_cast_fp16")]; tensor var_9157_perm_0 = const()[name = string("op_9157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_311x = const()[name = string("concat_311x"), val = tensor([1, 2048, 1, -1])]; tensor var_9157_cast_fp16 = transpose(perm = var_9157_perm_0, x = attn_output_203_cast_fp16)[name = string("transpose_264")]; tensor attn_output_207_cast_fp16 = reshape(shape = concat_311x, x = var_9157_cast_fp16)[name = string("attn_output_207_cast_fp16")]; tensor hidden_states_253_strides_0 = const()[name = string("hidden_states_253_strides_0"), val = tensor([1, 1])]; string hidden_states_253_pad_type_0 = const()[name = string("hidden_states_253_pad_type_0"), val = string("valid")]; tensor hidden_states_253_pad_0 = const()[name = string("hidden_states_253_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_253_dilations_0 = const()[name = string("hidden_states_253_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_253_groups_0 = const()[name = string("hidden_states_253_groups_0"), val = int32(1)]; tensor hidden_states_253_cast_fp16 = conv(dilations = hidden_states_253_dilations_0, groups = hidden_states_253_groups_0, pad = hidden_states_253_pad_0, pad_type = hidden_states_253_pad_type_0, strides = hidden_states_253_strides_0, weight = layers_25_self_attn_o_proj_weight_cast_fp16, x = attn_output_207_cast_fp16)[name = string("hidden_states_253_cast_fp16")]; tensor hidden_states_255_cast_fp16 = add(x = hidden_states_249_cast_fp16, y = hidden_states_253_cast_fp16)[name = string("hidden_states_255_cast_fp16")]; fp16 const_258_promoted_to_fp16 = const()[name = string("const_258_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9190_cast_fp16 = mul(x = hidden_states_255_cast_fp16, y = const_258_promoted_to_fp16)[name = string("op_9190_cast_fp16")]; int32 var_9188 = const()[name = string("op_9188"), val = int32(1)]; bool doubled_205_interleave_0 = const()[name = string("doubled_205_interleave_0"), val = bool(false)]; tensor doubled_205_cast_fp16 = concat(axis = var_9188, interleave = doubled_205_interleave_0, values = (hidden_states_255_cast_fp16, var_9190_cast_fp16))[name = string("doubled_205_cast_fp16")]; tensor out_103_axes_0 = const()[name = string("out_103_axes_0"), val = tensor([1])]; tensor out_103_gamma_0_to_fp16 = const()[name = string("out_103_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541427584)))]; fp16 var_9200_to_fp16 = const()[name = string("op_9200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_103_cast_fp16 = layer_norm(axes = out_103_axes_0, epsilon = var_9200_to_fp16, gamma = out_103_gamma_0_to_fp16, x = doubled_205_cast_fp16)[name = string("out_103_cast_fp16")]; tensor var_9211_split_sizes_0 = const()[name = string("op_9211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9211_axis_0 = const()[name = string("op_9211_axis_0"), val = int32(1)]; tensor var_9211_cast_fp16_0, tensor var_9211_cast_fp16_1 = split(axis = var_9211_axis_0, split_sizes = var_9211_split_sizes_0, x = out_103_cast_fp16)[name = string("op_9211_cast_fp16")]; tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; tensor input_51_cast_fp16 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = layers_25_mlp_gate_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("input_51_cast_fp16")]; tensor var_9228_cast_fp16 = silu(x = input_51_cast_fp16)[name = string("op_9228_cast_fp16")]; tensor var_9234_strides_0 = const()[name = string("op_9234_strides_0"), val = tensor([1, 1])]; string var_9234_pad_type_0 = const()[name = string("op_9234_pad_type_0"), val = string("valid")]; tensor var_9234_pad_0 = const()[name = string("op_9234_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9234_dilations_0 = const()[name = string("op_9234_dilations_0"), val = tensor([1, 1])]; int32 var_9234_groups_0 = const()[name = string("op_9234_groups_0"), val = int32(1)]; tensor var_9234_cast_fp16 = conv(dilations = var_9234_dilations_0, groups = var_9234_groups_0, pad = var_9234_pad_0, pad_type = var_9234_pad_type_0, strides = var_9234_strides_0, weight = layers_25_mlp_up_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("op_9234_cast_fp16")]; tensor x_259_cast_fp16 = mul(x = var_9228_cast_fp16, y = var_9234_cast_fp16)[name = string("x_259_cast_fp16")]; tensor layers_25_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_25_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541435840)))]; tensor hidden_states_257_strides_0 = const()[name = string("hidden_states_257_strides_0"), val = tensor([1, 1])]; string hidden_states_257_pad_type_0 = const()[name = string("hidden_states_257_pad_type_0"), val = string("valid")]; tensor hidden_states_257_pad_0 = const()[name = string("hidden_states_257_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_257_dilations_0 = const()[name = string("hidden_states_257_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_257_groups_0 = const()[name = string("hidden_states_257_groups_0"), val = int32(1)]; tensor hidden_states_257_cast_fp16 = conv(dilations = hidden_states_257_dilations_0, groups = hidden_states_257_groups_0, pad = hidden_states_257_pad_0, pad_type = hidden_states_257_pad_type_0, strides = hidden_states_257_strides_0, weight = layers_25_mlp_down_proj_weight_to_fp16, x = x_259_cast_fp16)[name = string("hidden_states_257_cast_fp16")]; tensor hidden_states_259_cast_fp16 = add(x = hidden_states_255_cast_fp16, y = hidden_states_257_cast_fp16)[name = string("hidden_states_259_cast_fp16")]; fp16 const_260_promoted_to_fp16 = const()[name = string("const_260_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9252_cast_fp16 = mul(x = hidden_states_259_cast_fp16, y = const_260_promoted_to_fp16)[name = string("op_9252_cast_fp16")]; int32 var_9250 = const()[name = string("op_9250"), val = int32(1)]; bool doubled_209_interleave_0 = const()[name = string("doubled_209_interleave_0"), val = bool(false)]; tensor doubled_209_cast_fp16 = concat(axis = var_9250, interleave = doubled_209_interleave_0, values = (hidden_states_259_cast_fp16, var_9252_cast_fp16))[name = string("doubled_209_cast_fp16")]; tensor out_105_axes_0 = const()[name = string("out_105_axes_0"), val = tensor([1])]; tensor out_105_gamma_0_to_fp16 = const()[name = string("out_105_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566601728)))]; fp16 var_9262_to_fp16 = const()[name = string("op_9262_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_105_cast_fp16 = layer_norm(axes = out_105_axes_0, epsilon = var_9262_to_fp16, gamma = out_105_gamma_0_to_fp16, x = doubled_209_cast_fp16)[name = string("out_105_cast_fp16")]; tensor var_9273_split_sizes_0 = const()[name = string("op_9273_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9273_axis_0 = const()[name = string("op_9273_axis_0"), val = int32(1)]; tensor var_9273_cast_fp16_0, tensor var_9273_cast_fp16_1 = split(axis = var_9273_axis_0, split_sizes = var_9273_split_sizes_0, x = out_105_cast_fp16)[name = string("op_9273_cast_fp16")]; tensor query_states_157_strides_0 = const()[name = string("query_states_157_strides_0"), val = tensor([1, 1])]; string query_states_157_pad_type_0 = const()[name = string("query_states_157_pad_type_0"), val = string("valid")]; tensor query_states_157_pad_0 = const()[name = string("query_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_157_dilations_0 = const()[name = string("query_states_157_dilations_0"), val = tensor([1, 1])]; int32 query_states_157_groups_0 = const()[name = string("query_states_157_groups_0"), val = int32(1)]; tensor query_states_157_cast_fp16 = conv(dilations = query_states_157_dilations_0, groups = query_states_157_groups_0, pad = query_states_157_pad_0, pad_type = query_states_157_pad_type_0, strides = query_states_157_strides_0, weight = layers_26_self_attn_q_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("query_states_157_cast_fp16")]; tensor key_states_261_strides_0 = const()[name = string("key_states_261_strides_0"), val = tensor([1, 1])]; string key_states_261_pad_type_0 = const()[name = string("key_states_261_pad_type_0"), val = string("valid")]; tensor key_states_261_pad_0 = const()[name = string("key_states_261_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_261_dilations_0 = const()[name = string("key_states_261_dilations_0"), val = tensor([1, 1])]; int32 key_states_261_groups_0 = const()[name = string("key_states_261_groups_0"), val = int32(1)]; tensor key_states_261_cast_fp16 = conv(dilations = key_states_261_dilations_0, groups = key_states_261_groups_0, pad = key_states_261_pad_0, pad_type = key_states_261_pad_type_0, strides = key_states_261_strides_0, weight = layers_26_self_attn_k_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("key_states_261_cast_fp16")]; tensor layers_26_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_26_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566609984)))]; tensor value_states_157_strides_0 = const()[name = string("value_states_157_strides_0"), val = tensor([1, 1])]; string value_states_157_pad_type_0 = const()[name = string("value_states_157_pad_type_0"), val = string("valid")]; tensor value_states_157_pad_0 = const()[name = string("value_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_157_dilations_0 = const()[name = string("value_states_157_dilations_0"), val = tensor([1, 1])]; int32 value_states_157_groups_0 = const()[name = string("value_states_157_groups_0"), val = int32(1)]; tensor value_states_157_cast_fp16 = conv(dilations = value_states_157_dilations_0, groups = value_states_157_groups_0, pad = value_states_157_pad_0, pad_type = value_states_157_pad_type_0, strides = value_states_157_strides_0, weight = layers_26_self_attn_v_proj_weight_to_fp16, x = var_9273_cast_fp16_0)[name = string("value_states_157_cast_fp16")]; tensor concat_312x = const()[name = string("concat_312x"), val = tensor([1, 16, 128, -1])]; tensor x_261_cast_fp16 = reshape(shape = concat_312x, x = query_states_157_cast_fp16)[name = string("x_261_cast_fp16")]; tensor concat_313x = const()[name = string("concat_313x"), val = tensor([1, 2, 128, -1])]; tensor var_9330_cast_fp16 = reshape(shape = concat_313x, x = key_states_261_cast_fp16)[name = string("op_9330_cast_fp16")]; tensor concat_314x = const()[name = string("concat_314x"), val = tensor([1, 2, 128, -1])]; tensor var_9337_cast_fp16 = reshape(shape = concat_314x, x = value_states_157_cast_fp16)[name = string("op_9337_cast_fp16")]; tensor var_9341_cast_fp16 = mul(x = x_261_cast_fp16, y = var_869_cast_fp16)[name = string("op_9341_cast_fp16")]; tensor var_9342_split_sizes_0 = const()[name = string("op_9342_split_sizes_0"), val = tensor([64, 64])]; int32 var_9342_axis_0 = const()[name = string("op_9342_axis_0"), val = int32(-2)]; tensor var_9342_cast_fp16_0, tensor var_9342_cast_fp16_1 = split(axis = var_9342_axis_0, split_sizes = var_9342_split_sizes_0, x = x_261_cast_fp16)[name = string("op_9342_cast_fp16")]; fp16 const_262_promoted_to_fp16 = const()[name = string("const_262_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9344_cast_fp16 = mul(x = var_9342_cast_fp16_1, y = const_262_promoted_to_fp16)[name = string("op_9344_cast_fp16")]; int32 var_9346 = const()[name = string("op_9346"), val = int32(-2)]; bool var_9347_interleave_0 = const()[name = string("op_9347_interleave_0"), val = bool(false)]; tensor var_9347_cast_fp16 = concat(axis = var_9346, interleave = var_9347_interleave_0, values = (var_9344_cast_fp16, var_9342_cast_fp16_0))[name = string("op_9347_cast_fp16")]; tensor var_9348_cast_fp16 = mul(x = var_9347_cast_fp16, y = var_878_cast_fp16)[name = string("op_9348_cast_fp16")]; tensor query_states_159_cast_fp16 = add(x = var_9341_cast_fp16, y = var_9348_cast_fp16)[name = string("query_states_159_cast_fp16")]; tensor var_9354_cast_fp16 = mul(x = var_9330_cast_fp16, y = var_869_cast_fp16)[name = string("op_9354_cast_fp16")]; tensor var_9355_split_sizes_0 = const()[name = string("op_9355_split_sizes_0"), val = tensor([64, 64])]; int32 var_9355_axis_0 = const()[name = string("op_9355_axis_0"), val = int32(-2)]; tensor var_9355_cast_fp16_0, tensor var_9355_cast_fp16_1 = split(axis = var_9355_axis_0, split_sizes = var_9355_split_sizes_0, x = var_9330_cast_fp16)[name = string("op_9355_cast_fp16")]; fp16 const_263_promoted_to_fp16 = const()[name = string("const_263_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9357_cast_fp16 = mul(x = var_9355_cast_fp16_1, y = const_263_promoted_to_fp16)[name = string("op_9357_cast_fp16")]; int32 var_9359 = const()[name = string("op_9359"), val = int32(-2)]; bool var_9360_interleave_0 = const()[name = string("op_9360_interleave_0"), val = bool(false)]; tensor var_9360_cast_fp16 = concat(axis = var_9359, interleave = var_9360_interleave_0, values = (var_9357_cast_fp16, var_9355_cast_fp16_0))[name = string("op_9360_cast_fp16")]; tensor var_9361_cast_fp16 = mul(x = var_9360_cast_fp16, y = var_878_cast_fp16)[name = string("op_9361_cast_fp16")]; tensor key_states_265_cast_fp16 = add(x = var_9354_cast_fp16, y = var_9361_cast_fp16)[name = string("key_states_265_cast_fp16")]; tensor expand_dims_312 = const()[name = string("expand_dims_312"), val = tensor([26])]; tensor expand_dims_313 = const()[name = string("expand_dims_313"), val = tensor([0])]; tensor expand_dims_315 = const()[name = string("expand_dims_315"), val = tensor([0])]; int32 concat_317_axis_0 = const()[name = string("concat_317_axis_0"), val = int32(0)]; bool concat_317_interleave_0 = const()[name = string("concat_317_interleave_0"), val = bool(false)]; tensor concat_317 = concat(axis = concat_317_axis_0, interleave = concat_317_interleave_0, values = (expand_dims_312, expand_dims_313, position_id, expand_dims_315))[name = string("concat_317")]; tensor expand_dims_316 = const()[name = string("expand_dims_316"), val = tensor([27])]; tensor concat_318_values1_0 = const()[name = string("concat_318_values1_0"), val = tensor([0])]; tensor concat_318_values3_0 = const()[name = string("concat_318_values3_0"), val = tensor([0])]; int32 concat_318_axis_0 = const()[name = string("concat_318_axis_0"), val = int32(0)]; bool concat_318_interleave_0 = const()[name = string("concat_318_interleave_0"), val = bool(false)]; tensor concat_318 = concat(axis = concat_318_axis_0, interleave = concat_318_interleave_0, values = (expand_dims_316, concat_318_values1_0, cache_position_end, concat_318_values3_0))[name = string("concat_318")]; tensor key_states_267_perm_0 = const()[name = string("key_states_267_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_27_stride_0 = const()[name = string("key_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_267_cast_fp16 = transpose(perm = key_states_267_perm_0, x = key_states_265_cast_fp16)[name = string("transpose_263")]; tensor key_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = key_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = key_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_27_squeeze_mask_0, stride = key_cache_internal_tensor_assign_27_stride_0, update = key_states_267_cast_fp16, x = coreml_update_state_218)[name = string("key_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_27_cast_fp16, input = key_cache)[name = string("coreml_update_state_220_write_state")]; tensor coreml_update_state_220 = read_state(input = key_cache)[name = string("coreml_update_state_220")]; tensor value_states_159_perm_0 = const()[name = string("value_states_159_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_27_stride_0 = const()[name = string("value_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_159_cast_fp16 = transpose(perm = value_states_159_perm_0, x = var_9337_cast_fp16)[name = string("transpose_262")]; tensor value_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = value_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = value_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_27_squeeze_mask_0, stride = value_cache_internal_tensor_assign_27_stride_0, update = value_states_159_cast_fp16, x = coreml_update_state_219)[name = string("value_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_27_cast_fp16, input = value_cache)[name = string("coreml_update_state_221_write_state")]; tensor coreml_update_state_221 = read_state(input = value_cache)[name = string("coreml_update_state_221")]; tensor var_9431_begin_0 = const()[name = string("op_9431_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9431_end_0 = const()[name = string("op_9431_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9431_end_mask_0 = const()[name = string("op_9431_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9431_cast_fp16 = slice_by_index(begin = var_9431_begin_0, end = var_9431_end_0, end_mask = var_9431_end_mask_0, x = coreml_update_state_220)[name = string("op_9431_cast_fp16")]; tensor tile_52 = const()[name = string("tile_52"), val = tensor([1, 1])]; int32 var_9434_axis_0 = const()[name = string("op_9434_axis_0"), val = int32(1)]; tensor var_9434_cast_fp16_0, tensor var_9434_cast_fp16_1 = split(axis = var_9434_axis_0, split_sizes = tile_52, x = var_9431_cast_fp16)[name = string("op_9434_cast_fp16")]; tensor var_9441_begin_0 = const()[name = string("op_9441_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9441_end_0 = const()[name = string("op_9441_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9441_end_mask_0 = const()[name = string("op_9441_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9441_cast_fp16 = slice_by_index(begin = var_9441_begin_0, end = var_9441_end_0, end_mask = var_9441_end_mask_0, x = coreml_update_state_221)[name = string("op_9441_cast_fp16")]; tensor tile_53 = const()[name = string("tile_53"), val = tensor([1, 1])]; int32 var_9444_axis_0 = const()[name = string("op_9444_axis_0"), val = int32(1)]; tensor var_9444_cast_fp16_0, tensor var_9444_cast_fp16_1 = split(axis = var_9444_axis_0, split_sizes = tile_53, x = var_9441_cast_fp16)[name = string("op_9444_cast_fp16")]; tensor var_9447_split_sizes_0 = const()[name = string("op_9447_split_sizes_0"), val = tensor([8, 8])]; int32 var_9447_axis_0 = const()[name = string("op_9447_axis_0"), val = int32(1)]; tensor var_9447_0, tensor var_9447_1 = split(axis = var_9447_axis_0, split_sizes = var_9447_split_sizes_0, x = query_states_159_cast_fp16)[name = string("op_9447")]; bool attn_weights_417_transpose_x_0 = const()[name = string("attn_weights_417_transpose_x_0"), val = bool(false)]; bool attn_weights_417_transpose_y_0 = const()[name = string("attn_weights_417_transpose_y_0"), val = bool(false)]; tensor attn_weights_417_cast_fp16 = matmul(transpose_x = attn_weights_417_transpose_x_0, transpose_y = attn_weights_417_transpose_y_0, x = var_9434_cast_fp16_0, y = var_9447_0)[name = string("attn_weights_417_cast_fp16")]; fp16 var_9450_to_fp16 = const()[name = string("op_9450_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_419_cast_fp16 = mul(x = attn_weights_417_cast_fp16, y = var_9450_to_fp16)[name = string("attn_weights_419_cast_fp16")]; tensor attn_weights_421_cast_fp16 = add(x = attn_weights_419_cast_fp16, y = attn_mask_1)[name = string("attn_weights_421_cast_fp16")]; int32 var_9454 = const()[name = string("op_9454"), val = int32(-2)]; tensor attn_weights_423_cast_fp16 = softmax(axis = var_9454, x = attn_weights_421_cast_fp16)[name = string("attn_weights_423_cast_fp16")]; bool var_9460_transpose_x_1 = const()[name = string("op_9460_transpose_x_1"), val = bool(true)]; bool var_9460_transpose_y_1 = const()[name = string("op_9460_transpose_y_1"), val = bool(false)]; tensor var_9460_cast_fp16 = matmul(transpose_x = var_9460_transpose_x_1, transpose_y = var_9460_transpose_y_1, x = attn_weights_423_cast_fp16, y = var_9444_cast_fp16_0)[name = string("op_9460_cast_fp16")]; bool attn_weights_425_transpose_x_0 = const()[name = string("attn_weights_425_transpose_x_0"), val = bool(false)]; bool attn_weights_425_transpose_y_0 = const()[name = string("attn_weights_425_transpose_y_0"), val = bool(false)]; tensor attn_weights_425_cast_fp16 = matmul(transpose_x = attn_weights_425_transpose_x_0, transpose_y = attn_weights_425_transpose_y_0, x = var_9434_cast_fp16_1, y = var_9447_1)[name = string("attn_weights_425_cast_fp16")]; fp16 var_9462_to_fp16 = const()[name = string("op_9462_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_427_cast_fp16 = mul(x = attn_weights_425_cast_fp16, y = var_9462_to_fp16)[name = string("attn_weights_427_cast_fp16")]; tensor attn_weights_429_cast_fp16 = add(x = attn_weights_427_cast_fp16, y = attn_mask_1)[name = string("attn_weights_429_cast_fp16")]; int32 var_9466 = const()[name = string("op_9466"), val = int32(-2)]; tensor attn_weights_431_cast_fp16 = softmax(axis = var_9466, x = attn_weights_429_cast_fp16)[name = string("attn_weights_431_cast_fp16")]; bool attn_output_209_transpose_x_1 = const()[name = string("attn_output_209_transpose_x_1"), val = bool(true)]; bool attn_output_209_transpose_y_1 = const()[name = string("attn_output_209_transpose_y_1"), val = bool(false)]; tensor attn_output_209_cast_fp16 = matmul(transpose_x = attn_output_209_transpose_x_1, transpose_y = attn_output_209_transpose_y_1, x = attn_weights_431_cast_fp16, y = var_9444_cast_fp16_1)[name = string("attn_output_209_cast_fp16")]; int32 var_9474 = const()[name = string("op_9474"), val = int32(1)]; bool attn_output_211_interleave_0 = const()[name = string("attn_output_211_interleave_0"), val = bool(false)]; tensor attn_output_211_cast_fp16 = concat(axis = var_9474, interleave = attn_output_211_interleave_0, values = (var_9460_cast_fp16, attn_output_209_cast_fp16))[name = string("attn_output_211_cast_fp16")]; tensor var_9478_perm_0 = const()[name = string("op_9478_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_323x = const()[name = string("concat_323x"), val = tensor([1, 2048, 1, -1])]; tensor var_9478_cast_fp16 = transpose(perm = var_9478_perm_0, x = attn_output_211_cast_fp16)[name = string("transpose_261")]; tensor attn_output_215_cast_fp16 = reshape(shape = concat_323x, x = var_9478_cast_fp16)[name = string("attn_output_215_cast_fp16")]; tensor hidden_states_263_strides_0 = const()[name = string("hidden_states_263_strides_0"), val = tensor([1, 1])]; string hidden_states_263_pad_type_0 = const()[name = string("hidden_states_263_pad_type_0"), val = string("valid")]; tensor hidden_states_263_pad_0 = const()[name = string("hidden_states_263_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_263_dilations_0 = const()[name = string("hidden_states_263_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_263_groups_0 = const()[name = string("hidden_states_263_groups_0"), val = int32(1)]; tensor hidden_states_263_cast_fp16 = conv(dilations = hidden_states_263_dilations_0, groups = hidden_states_263_groups_0, pad = hidden_states_263_pad_0, pad_type = hidden_states_263_pad_type_0, strides = hidden_states_263_strides_0, weight = layers_26_self_attn_o_proj_weight_cast_fp16, x = attn_output_215_cast_fp16)[name = string("hidden_states_263_cast_fp16")]; tensor hidden_states_265_cast_fp16 = add(x = hidden_states_259_cast_fp16, y = hidden_states_263_cast_fp16)[name = string("hidden_states_265_cast_fp16")]; fp16 const_268_promoted_to_fp16 = const()[name = string("const_268_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9511_cast_fp16 = mul(x = hidden_states_265_cast_fp16, y = const_268_promoted_to_fp16)[name = string("op_9511_cast_fp16")]; int32 var_9509 = const()[name = string("op_9509"), val = int32(1)]; bool doubled_213_interleave_0 = const()[name = string("doubled_213_interleave_0"), val = bool(false)]; tensor doubled_213_cast_fp16 = concat(axis = var_9509, interleave = doubled_213_interleave_0, values = (hidden_states_265_cast_fp16, var_9511_cast_fp16))[name = string("doubled_213_cast_fp16")]; tensor out_107_axes_0 = const()[name = string("out_107_axes_0"), val = tensor([1])]; tensor out_107_gamma_0_to_fp16 = const()[name = string("out_107_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567658624)))]; fp16 var_9521_to_fp16 = const()[name = string("op_9521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_107_cast_fp16 = layer_norm(axes = out_107_axes_0, epsilon = var_9521_to_fp16, gamma = out_107_gamma_0_to_fp16, x = doubled_213_cast_fp16)[name = string("out_107_cast_fp16")]; tensor var_9532_split_sizes_0 = const()[name = string("op_9532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9532_axis_0 = const()[name = string("op_9532_axis_0"), val = int32(1)]; tensor var_9532_cast_fp16_0, tensor var_9532_cast_fp16_1 = split(axis = var_9532_axis_0, split_sizes = var_9532_split_sizes_0, x = out_107_cast_fp16)[name = string("op_9532_cast_fp16")]; tensor input_53_strides_0 = const()[name = string("input_53_strides_0"), val = tensor([1, 1])]; string input_53_pad_type_0 = const()[name = string("input_53_pad_type_0"), val = string("valid")]; tensor input_53_pad_0 = const()[name = string("input_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_53_dilations_0 = const()[name = string("input_53_dilations_0"), val = tensor([1, 1])]; int32 input_53_groups_0 = const()[name = string("input_53_groups_0"), val = int32(1)]; tensor input_53_cast_fp16 = conv(dilations = input_53_dilations_0, groups = input_53_groups_0, pad = input_53_pad_0, pad_type = input_53_pad_type_0, strides = input_53_strides_0, weight = layers_26_mlp_gate_proj_weight_cast_fp16, x = var_9532_cast_fp16_0)[name = string("input_53_cast_fp16")]; tensor var_9549_cast_fp16 = silu(x = input_53_cast_fp16)[name = string("op_9549_cast_fp16")]; tensor layers_26_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567666880)))]; tensor var_9555_strides_0 = const()[name = string("op_9555_strides_0"), val = tensor([1, 1])]; string var_9555_pad_type_0 = const()[name = string("op_9555_pad_type_0"), val = string("valid")]; tensor var_9555_pad_0 = const()[name = string("op_9555_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9555_dilations_0 = const()[name = string("op_9555_dilations_0"), val = tensor([1, 1])]; int32 var_9555_groups_0 = const()[name = string("op_9555_groups_0"), val = int32(1)]; tensor var_9555_cast_fp16 = conv(dilations = var_9555_dilations_0, groups = var_9555_groups_0, pad = var_9555_pad_0, pad_type = var_9555_pad_type_0, strides = var_9555_strides_0, weight = layers_26_mlp_up_proj_weight_to_fp16, x = var_9532_cast_fp16_0)[name = string("op_9555_cast_fp16")]; tensor x_269_cast_fp16 = mul(x = var_9549_cast_fp16, y = var_9555_cast_fp16)[name = string("x_269_cast_fp16")]; tensor layers_26_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1592832768)))]; tensor hidden_states_267_strides_0 = const()[name = string("hidden_states_267_strides_0"), val = tensor([1, 1])]; string hidden_states_267_pad_type_0 = const()[name = string("hidden_states_267_pad_type_0"), val = string("valid")]; tensor hidden_states_267_pad_0 = const()[name = string("hidden_states_267_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_267_dilations_0 = const()[name = string("hidden_states_267_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_267_groups_0 = const()[name = string("hidden_states_267_groups_0"), val = int32(1)]; tensor hidden_states_267_cast_fp16 = conv(dilations = hidden_states_267_dilations_0, groups = hidden_states_267_groups_0, pad = hidden_states_267_pad_0, pad_type = hidden_states_267_pad_type_0, strides = hidden_states_267_strides_0, weight = layers_26_mlp_down_proj_weight_to_fp16, x = x_269_cast_fp16)[name = string("hidden_states_267_cast_fp16")]; tensor hidden_states_269_cast_fp16 = add(x = hidden_states_265_cast_fp16, y = hidden_states_267_cast_fp16)[name = string("hidden_states_269_cast_fp16")]; fp16 const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9573_cast_fp16 = mul(x = hidden_states_269_cast_fp16, y = const_270_promoted_to_fp16)[name = string("op_9573_cast_fp16")]; int32 var_9571 = const()[name = string("op_9571"), val = int32(1)]; bool doubled_217_interleave_0 = const()[name = string("doubled_217_interleave_0"), val = bool(false)]; tensor doubled_217_cast_fp16 = concat(axis = var_9571, interleave = doubled_217_interleave_0, values = (hidden_states_269_cast_fp16, var_9573_cast_fp16))[name = string("doubled_217_cast_fp16")]; tensor out_109_axes_0 = const()[name = string("out_109_axes_0"), val = tensor([1])]; tensor out_109_gamma_0_to_fp16 = const()[name = string("out_109_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1617998656)))]; fp16 var_9583_to_fp16 = const()[name = string("op_9583_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_109_cast_fp16 = layer_norm(axes = out_109_axes_0, epsilon = var_9583_to_fp16, gamma = out_109_gamma_0_to_fp16, x = doubled_217_cast_fp16)[name = string("out_109_cast_fp16")]; tensor var_9594_split_sizes_0 = const()[name = string("op_9594_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9594_axis_0 = const()[name = string("op_9594_axis_0"), val = int32(1)]; tensor var_9594_cast_fp16_0, tensor var_9594_cast_fp16_1 = split(axis = var_9594_axis_0, split_sizes = var_9594_split_sizes_0, x = out_109_cast_fp16)[name = string("op_9594_cast_fp16")]; tensor layers_27_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1618006912)))]; tensor query_states_163_strides_0 = const()[name = string("query_states_163_strides_0"), val = tensor([1, 1])]; string query_states_163_pad_type_0 = const()[name = string("query_states_163_pad_type_0"), val = string("valid")]; tensor query_states_163_pad_0 = const()[name = string("query_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_163_dilations_0 = const()[name = string("query_states_163_dilations_0"), val = tensor([1, 1])]; int32 query_states_163_groups_0 = const()[name = string("query_states_163_groups_0"), val = int32(1)]; tensor query_states_163_cast_fp16 = conv(dilations = query_states_163_dilations_0, groups = query_states_163_groups_0, pad = query_states_163_pad_0, pad_type = query_states_163_pad_type_0, strides = query_states_163_strides_0, weight = layers_27_self_attn_q_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("query_states_163_cast_fp16")]; tensor layers_27_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1626395584)))]; tensor key_states_271_strides_0 = const()[name = string("key_states_271_strides_0"), val = tensor([1, 1])]; string key_states_271_pad_type_0 = const()[name = string("key_states_271_pad_type_0"), val = string("valid")]; tensor key_states_271_pad_0 = const()[name = string("key_states_271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_271_dilations_0 = const()[name = string("key_states_271_dilations_0"), val = tensor([1, 1])]; int32 key_states_271_groups_0 = const()[name = string("key_states_271_groups_0"), val = int32(1)]; tensor key_states_271_cast_fp16 = conv(dilations = key_states_271_dilations_0, groups = key_states_271_groups_0, pad = key_states_271_pad_0, pad_type = key_states_271_pad_type_0, strides = key_states_271_strides_0, weight = layers_27_self_attn_k_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("key_states_271_cast_fp16")]; tensor layers_27_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1627444224)))]; tensor value_states_163_strides_0 = const()[name = string("value_states_163_strides_0"), val = tensor([1, 1])]; string value_states_163_pad_type_0 = const()[name = string("value_states_163_pad_type_0"), val = string("valid")]; tensor value_states_163_pad_0 = const()[name = string("value_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_163_dilations_0 = const()[name = string("value_states_163_dilations_0"), val = tensor([1, 1])]; int32 value_states_163_groups_0 = const()[name = string("value_states_163_groups_0"), val = int32(1)]; tensor value_states_163_cast_fp16 = conv(dilations = value_states_163_dilations_0, groups = value_states_163_groups_0, pad = value_states_163_pad_0, pad_type = value_states_163_pad_type_0, strides = value_states_163_strides_0, weight = layers_27_self_attn_v_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("value_states_163_cast_fp16")]; tensor concat_324x = const()[name = string("concat_324x"), val = tensor([1, 16, 128, -1])]; tensor x_271_cast_fp16 = reshape(shape = concat_324x, x = query_states_163_cast_fp16)[name = string("x_271_cast_fp16")]; tensor concat_325x = const()[name = string("concat_325x"), val = tensor([1, 2, 128, -1])]; tensor var_9651_cast_fp16 = reshape(shape = concat_325x, x = key_states_271_cast_fp16)[name = string("op_9651_cast_fp16")]; tensor concat_326x = const()[name = string("concat_326x"), val = tensor([1, 2, 128, -1])]; tensor var_9658_cast_fp16 = reshape(shape = concat_326x, x = value_states_163_cast_fp16)[name = string("op_9658_cast_fp16")]; tensor var_9662_cast_fp16 = mul(x = x_271_cast_fp16, y = var_869_cast_fp16)[name = string("op_9662_cast_fp16")]; tensor var_9663_split_sizes_0 = const()[name = string("op_9663_split_sizes_0"), val = tensor([64, 64])]; int32 var_9663_axis_0 = const()[name = string("op_9663_axis_0"), val = int32(-2)]; tensor var_9663_cast_fp16_0, tensor var_9663_cast_fp16_1 = split(axis = var_9663_axis_0, split_sizes = var_9663_split_sizes_0, x = x_271_cast_fp16)[name = string("op_9663_cast_fp16")]; fp16 const_272_promoted_to_fp16 = const()[name = string("const_272_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9665_cast_fp16 = mul(x = var_9663_cast_fp16_1, y = const_272_promoted_to_fp16)[name = string("op_9665_cast_fp16")]; int32 var_9667 = const()[name = string("op_9667"), val = int32(-2)]; bool var_9668_interleave_0 = const()[name = string("op_9668_interleave_0"), val = bool(false)]; tensor var_9668_cast_fp16 = concat(axis = var_9667, interleave = var_9668_interleave_0, values = (var_9665_cast_fp16, var_9663_cast_fp16_0))[name = string("op_9668_cast_fp16")]; tensor var_9669_cast_fp16 = mul(x = var_9668_cast_fp16, y = var_878_cast_fp16)[name = string("op_9669_cast_fp16")]; tensor query_states_165_cast_fp16 = add(x = var_9662_cast_fp16, y = var_9669_cast_fp16)[name = string("query_states_165_cast_fp16")]; tensor var_9675_cast_fp16 = mul(x = var_9651_cast_fp16, y = var_869_cast_fp16)[name = string("op_9675_cast_fp16")]; tensor var_9676_split_sizes_0 = const()[name = string("op_9676_split_sizes_0"), val = tensor([64, 64])]; int32 var_9676_axis_0 = const()[name = string("op_9676_axis_0"), val = int32(-2)]; tensor var_9676_cast_fp16_0, tensor var_9676_cast_fp16_1 = split(axis = var_9676_axis_0, split_sizes = var_9676_split_sizes_0, x = var_9651_cast_fp16)[name = string("op_9676_cast_fp16")]; fp16 const_273_promoted_to_fp16 = const()[name = string("const_273_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9678_cast_fp16 = mul(x = var_9676_cast_fp16_1, y = const_273_promoted_to_fp16)[name = string("op_9678_cast_fp16")]; int32 var_9680 = const()[name = string("op_9680"), val = int32(-2)]; bool var_9681_interleave_0 = const()[name = string("op_9681_interleave_0"), val = bool(false)]; tensor var_9681_cast_fp16 = concat(axis = var_9680, interleave = var_9681_interleave_0, values = (var_9678_cast_fp16, var_9676_cast_fp16_0))[name = string("op_9681_cast_fp16")]; tensor var_9682_cast_fp16 = mul(x = var_9681_cast_fp16, y = var_878_cast_fp16)[name = string("op_9682_cast_fp16")]; tensor key_states_275_cast_fp16 = add(x = var_9675_cast_fp16, y = var_9682_cast_fp16)[name = string("key_states_275_cast_fp16")]; tensor expand_dims_324 = const()[name = string("expand_dims_324"), val = tensor([27])]; tensor expand_dims_325 = const()[name = string("expand_dims_325"), val = tensor([0])]; tensor expand_dims_327 = const()[name = string("expand_dims_327"), val = tensor([0])]; int32 concat_329_axis_0 = const()[name = string("concat_329_axis_0"), val = int32(0)]; bool concat_329_interleave_0 = const()[name = string("concat_329_interleave_0"), val = bool(false)]; tensor concat_329 = concat(axis = concat_329_axis_0, interleave = concat_329_interleave_0, values = (expand_dims_324, expand_dims_325, position_id, expand_dims_327))[name = string("concat_329")]; tensor expand_dims_328 = const()[name = string("expand_dims_328"), val = tensor([28])]; tensor concat_330_values1_0 = const()[name = string("concat_330_values1_0"), val = tensor([0])]; tensor concat_330_values3_0 = const()[name = string("concat_330_values3_0"), val = tensor([0])]; int32 concat_330_axis_0 = const()[name = string("concat_330_axis_0"), val = int32(0)]; bool concat_330_interleave_0 = const()[name = string("concat_330_interleave_0"), val = bool(false)]; tensor concat_330 = concat(axis = concat_330_axis_0, interleave = concat_330_interleave_0, values = (expand_dims_328, concat_330_values1_0, cache_position_end, concat_330_values3_0))[name = string("concat_330")]; tensor key_states_277_perm_0 = const()[name = string("key_states_277_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_28_stride_0 = const()[name = string("key_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_277_cast_fp16 = transpose(perm = key_states_277_perm_0, x = key_states_275_cast_fp16)[name = string("transpose_260")]; tensor key_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = key_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = key_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_28_squeeze_mask_0, stride = key_cache_internal_tensor_assign_28_stride_0, update = key_states_277_cast_fp16, x = coreml_update_state_220)[name = string("key_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_28_cast_fp16, input = key_cache)[name = string("coreml_update_state_222_write_state")]; tensor coreml_update_state_222 = read_state(input = key_cache)[name = string("coreml_update_state_222")]; tensor value_states_165_perm_0 = const()[name = string("value_states_165_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_28_stride_0 = const()[name = string("value_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_165_cast_fp16 = transpose(perm = value_states_165_perm_0, x = var_9658_cast_fp16)[name = string("transpose_259")]; tensor value_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = value_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = value_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_28_squeeze_mask_0, stride = value_cache_internal_tensor_assign_28_stride_0, update = value_states_165_cast_fp16, x = coreml_update_state_221)[name = string("value_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_28_cast_fp16, input = value_cache)[name = string("coreml_update_state_223_write_state")]; tensor coreml_update_state_223 = read_state(input = value_cache)[name = string("coreml_update_state_223")]; tensor var_9752_begin_0 = const()[name = string("op_9752_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9752_end_0 = const()[name = string("op_9752_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9752_end_mask_0 = const()[name = string("op_9752_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9752_cast_fp16 = slice_by_index(begin = var_9752_begin_0, end = var_9752_end_0, end_mask = var_9752_end_mask_0, x = coreml_update_state_222)[name = string("op_9752_cast_fp16")]; tensor tile_54 = const()[name = string("tile_54"), val = tensor([1, 1])]; int32 var_9755_axis_0 = const()[name = string("op_9755_axis_0"), val = int32(1)]; tensor var_9755_cast_fp16_0, tensor var_9755_cast_fp16_1 = split(axis = var_9755_axis_0, split_sizes = tile_54, x = var_9752_cast_fp16)[name = string("op_9755_cast_fp16")]; tensor var_9762_begin_0 = const()[name = string("op_9762_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9762_end_0 = const()[name = string("op_9762_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9762_end_mask_0 = const()[name = string("op_9762_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9762_cast_fp16 = slice_by_index(begin = var_9762_begin_0, end = var_9762_end_0, end_mask = var_9762_end_mask_0, x = coreml_update_state_223)[name = string("op_9762_cast_fp16")]; tensor tile_55 = const()[name = string("tile_55"), val = tensor([1, 1])]; int32 var_9765_axis_0 = const()[name = string("op_9765_axis_0"), val = int32(1)]; tensor var_9765_cast_fp16_0, tensor var_9765_cast_fp16_1 = split(axis = var_9765_axis_0, split_sizes = tile_55, x = var_9762_cast_fp16)[name = string("op_9765_cast_fp16")]; tensor var_9768_split_sizes_0 = const()[name = string("op_9768_split_sizes_0"), val = tensor([8, 8])]; int32 var_9768_axis_0 = const()[name = string("op_9768_axis_0"), val = int32(1)]; tensor var_9768_0, tensor var_9768_1 = split(axis = var_9768_axis_0, split_sizes = var_9768_split_sizes_0, x = query_states_165_cast_fp16)[name = string("op_9768")]; bool attn_weights_433_transpose_x_0 = const()[name = string("attn_weights_433_transpose_x_0"), val = bool(false)]; bool attn_weights_433_transpose_y_0 = const()[name = string("attn_weights_433_transpose_y_0"), val = bool(false)]; tensor attn_weights_433_cast_fp16 = matmul(transpose_x = attn_weights_433_transpose_x_0, transpose_y = attn_weights_433_transpose_y_0, x = var_9755_cast_fp16_0, y = var_9768_0)[name = string("attn_weights_433_cast_fp16")]; fp16 var_9771_to_fp16 = const()[name = string("op_9771_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_435_cast_fp16 = mul(x = attn_weights_433_cast_fp16, y = var_9771_to_fp16)[name = string("attn_weights_435_cast_fp16")]; tensor attn_weights_437_cast_fp16 = add(x = attn_weights_435_cast_fp16, y = attn_mask_1)[name = string("attn_weights_437_cast_fp16")]; int32 var_9775 = const()[name = string("op_9775"), val = int32(-2)]; tensor attn_weights_439_cast_fp16 = softmax(axis = var_9775, x = attn_weights_437_cast_fp16)[name = string("attn_weights_439_cast_fp16")]; bool var_9781_transpose_x_1 = const()[name = string("op_9781_transpose_x_1"), val = bool(true)]; bool var_9781_transpose_y_1 = const()[name = string("op_9781_transpose_y_1"), val = bool(false)]; tensor var_9781_cast_fp16 = matmul(transpose_x = var_9781_transpose_x_1, transpose_y = var_9781_transpose_y_1, x = attn_weights_439_cast_fp16, y = var_9765_cast_fp16_0)[name = string("op_9781_cast_fp16")]; bool attn_weights_441_transpose_x_0 = const()[name = string("attn_weights_441_transpose_x_0"), val = bool(false)]; bool attn_weights_441_transpose_y_0 = const()[name = string("attn_weights_441_transpose_y_0"), val = bool(false)]; tensor attn_weights_441_cast_fp16 = matmul(transpose_x = attn_weights_441_transpose_x_0, transpose_y = attn_weights_441_transpose_y_0, x = var_9755_cast_fp16_1, y = var_9768_1)[name = string("attn_weights_441_cast_fp16")]; fp16 var_9783_to_fp16 = const()[name = string("op_9783_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_443_cast_fp16 = mul(x = attn_weights_441_cast_fp16, y = var_9783_to_fp16)[name = string("attn_weights_443_cast_fp16")]; tensor attn_weights_445_cast_fp16 = add(x = attn_weights_443_cast_fp16, y = attn_mask_1)[name = string("attn_weights_445_cast_fp16")]; int32 var_9787 = const()[name = string("op_9787"), val = int32(-2)]; tensor attn_weights_cast_fp16 = softmax(axis = var_9787, x = attn_weights_445_cast_fp16)[name = string("attn_weights_cast_fp16")]; bool attn_output_217_transpose_x_1 = const()[name = string("attn_output_217_transpose_x_1"), val = bool(true)]; bool attn_output_217_transpose_y_1 = const()[name = string("attn_output_217_transpose_y_1"), val = bool(false)]; tensor attn_output_217_cast_fp16 = matmul(transpose_x = attn_output_217_transpose_x_1, transpose_y = attn_output_217_transpose_y_1, x = attn_weights_cast_fp16, y = var_9765_cast_fp16_1)[name = string("attn_output_217_cast_fp16")]; int32 var_9795 = const()[name = string("op_9795"), val = int32(1)]; bool attn_output_219_interleave_0 = const()[name = string("attn_output_219_interleave_0"), val = bool(false)]; tensor attn_output_219_cast_fp16 = concat(axis = var_9795, interleave = attn_output_219_interleave_0, values = (var_9781_cast_fp16, attn_output_217_cast_fp16))[name = string("attn_output_219_cast_fp16")]; tensor var_9799_perm_0 = const()[name = string("op_9799_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_335x = const()[name = string("concat_335x"), val = tensor([1, 2048, 1, -1])]; tensor var_9799_cast_fp16 = transpose(perm = var_9799_perm_0, x = attn_output_219_cast_fp16)[name = string("transpose_258")]; tensor attn_output_cast_fp16 = reshape(shape = concat_335x, x = var_9799_cast_fp16)[name = string("attn_output_cast_fp16")]; tensor layers_27_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1628492864)))]; tensor hidden_states_273_strides_0 = const()[name = string("hidden_states_273_strides_0"), val = tensor([1, 1])]; string hidden_states_273_pad_type_0 = const()[name = string("hidden_states_273_pad_type_0"), val = string("valid")]; tensor hidden_states_273_pad_0 = const()[name = string("hidden_states_273_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_273_dilations_0 = const()[name = string("hidden_states_273_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_273_groups_0 = const()[name = string("hidden_states_273_groups_0"), val = int32(1)]; tensor hidden_states_273_cast_fp16 = conv(dilations = hidden_states_273_dilations_0, groups = hidden_states_273_groups_0, pad = hidden_states_273_pad_0, pad_type = hidden_states_273_pad_type_0, strides = hidden_states_273_strides_0, weight = layers_27_self_attn_o_proj_weight_to_fp16, x = attn_output_cast_fp16)[name = string("hidden_states_273_cast_fp16")]; tensor hidden_states_275_cast_fp16 = add(x = hidden_states_269_cast_fp16, y = hidden_states_273_cast_fp16)[name = string("hidden_states_275_cast_fp16")]; fp16 const_278_promoted_to_fp16 = const()[name = string("const_278_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9832_cast_fp16 = mul(x = hidden_states_275_cast_fp16, y = const_278_promoted_to_fp16)[name = string("op_9832_cast_fp16")]; int32 var_9830 = const()[name = string("op_9830"), val = int32(1)]; bool doubled_221_interleave_0 = const()[name = string("doubled_221_interleave_0"), val = bool(false)]; tensor doubled_221_cast_fp16 = concat(axis = var_9830, interleave = doubled_221_interleave_0, values = (hidden_states_275_cast_fp16, var_9832_cast_fp16))[name = string("doubled_221_cast_fp16")]; tensor out_111_axes_0 = const()[name = string("out_111_axes_0"), val = tensor([1])]; tensor out_111_gamma_0_to_fp16 = const()[name = string("out_111_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636881536)))]; fp16 var_9842_to_fp16 = const()[name = string("op_9842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_111_cast_fp16 = layer_norm(axes = out_111_axes_0, epsilon = var_9842_to_fp16, gamma = out_111_gamma_0_to_fp16, x = doubled_221_cast_fp16)[name = string("out_111_cast_fp16")]; tensor var_9853_split_sizes_0 = const()[name = string("op_9853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9853_axis_0 = const()[name = string("op_9853_axis_0"), val = int32(1)]; tensor var_9853_cast_fp16_0, tensor var_9853_cast_fp16_1 = split(axis = var_9853_axis_0, split_sizes = var_9853_split_sizes_0, x = out_111_cast_fp16)[name = string("op_9853_cast_fp16")]; tensor layers_27_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636889792)))]; tensor input_strides_0 = const()[name = string("input_strides_0"), val = tensor([1, 1])]; string input_pad_type_0 = const()[name = string("input_pad_type_0"), val = string("valid")]; tensor input_pad_0 = const()[name = string("input_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_dilations_0 = const()[name = string("input_dilations_0"), val = tensor([1, 1])]; int32 input_groups_0 = const()[name = string("input_groups_0"), val = int32(1)]; tensor input_cast_fp16 = conv(dilations = input_dilations_0, groups = input_groups_0, pad = input_pad_0, pad_type = input_pad_type_0, strides = input_strides_0, weight = layers_27_mlp_gate_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("input_cast_fp16")]; tensor var_9870_cast_fp16 = silu(x = input_cast_fp16)[name = string("op_9870_cast_fp16")]; tensor layers_27_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1662055680)))]; tensor var_9876_strides_0 = const()[name = string("op_9876_strides_0"), val = tensor([1, 1])]; string var_9876_pad_type_0 = const()[name = string("op_9876_pad_type_0"), val = string("valid")]; tensor var_9876_pad_0 = const()[name = string("op_9876_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9876_dilations_0 = const()[name = string("op_9876_dilations_0"), val = tensor([1, 1])]; int32 var_9876_groups_0 = const()[name = string("op_9876_groups_0"), val = int32(1)]; tensor var_9876_cast_fp16 = conv(dilations = var_9876_dilations_0, groups = var_9876_groups_0, pad = var_9876_pad_0, pad_type = var_9876_pad_type_0, strides = var_9876_strides_0, weight = layers_27_mlp_up_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("op_9876_cast_fp16")]; tensor x_cast_fp16 = mul(x = var_9870_cast_fp16, y = var_9876_cast_fp16)[name = string("x_cast_fp16")]; tensor layers_27_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1687221568)))]; tensor hidden_states_277_strides_0 = const()[name = string("hidden_states_277_strides_0"), val = tensor([1, 1])]; string hidden_states_277_pad_type_0 = const()[name = string("hidden_states_277_pad_type_0"), val = string("valid")]; tensor hidden_states_277_pad_0 = const()[name = string("hidden_states_277_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_277_dilations_0 = const()[name = string("hidden_states_277_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_277_groups_0 = const()[name = string("hidden_states_277_groups_0"), val = int32(1)]; tensor hidden_states_277_cast_fp16 = conv(dilations = hidden_states_277_dilations_0, groups = hidden_states_277_groups_0, pad = hidden_states_277_pad_0, pad_type = hidden_states_277_pad_type_0, strides = hidden_states_277_strides_0, weight = layers_27_mlp_down_proj_weight_to_fp16, x = x_cast_fp16)[name = string("hidden_states_277_cast_fp16")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_275_cast_fp16, y = hidden_states_277_cast_fp16)[name = string("hidden_states_cast_fp16")]; fp16 const_280_promoted_to_fp16 = const()[name = string("const_280_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9894_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_280_promoted_to_fp16)[name = string("op_9894_cast_fp16")]; int32 var_9892 = const()[name = string("op_9892"), val = int32(1)]; bool doubled_225_interleave_0 = const()[name = string("doubled_225_interleave_0"), val = bool(false)]; tensor doubled_225_cast_fp16 = concat(axis = var_9892, interleave = doubled_225_interleave_0, values = (hidden_states_cast_fp16, var_9894_cast_fp16))[name = string("doubled_225_cast_fp16")]; tensor out_axes_0 = const()[name = string("out_axes_0"), val = tensor([1])]; tensor out_gamma_0_to_fp16 = const()[name = string("out_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1712387456)))]; fp16 var_9904_to_fp16 = const()[name = string("op_9904_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_cast_fp16 = layer_norm(axes = out_axes_0, epsilon = var_9904_to_fp16, gamma = out_gamma_0_to_fp16, x = doubled_225_cast_fp16)[name = string("out_cast_fp16")]; tensor var_9915_split_sizes_0 = const()[name = string("op_9915_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9915_axis_0 = const()[name = string("op_9915_axis_0"), val = int32(1)]; tensor hidden_states, tensor var_9915_cast_fp16_1 = split(axis = var_9915_axis_0, split_sizes = var_9915_split_sizes_0, x = out_cast_fp16)[name = string("op_9915_cast_fp16")]; } -> (hidden_states); func length_256(tensor inputs_embeds, state> key_cache, tensor position_id, tensor position_index_seed, state> value_cache) { tensor layers_1_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524992))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524416))))[name = string("layers_1_self_attn_v_proj_weight_cast_fp16")]; tensor layers_1_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(525312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13120640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13108288))))[name = string("layers_1_mlp_up_proj_weight_cast_fp16")]; tensor layers_2_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13126848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651776))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651200))))[name = string("layers_2_self_attn_v_proj_weight_cast_fp16")]; tensor layers_2_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13652096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26247424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26235072))))[name = string("layers_2_mlp_up_proj_weight_cast_fp16")]; tensor layers_3_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26253632))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778560))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26777984))))[name = string("layers_3_self_attn_v_proj_weight_cast_fp16")]; tensor layers_3_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30977408))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30973248))))[name = string("layers_3_self_attn_o_proj_weight_cast_fp16")]; tensor layers_3_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30979520))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43566656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43562496))))[name = string("layers_3_mlp_down_proj_weight_cast_fp16")]; tensor layers_4_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43568768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093696))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093120))))[name = string("layers_4_self_attn_v_proj_weight_cast_fp16")]; tensor layers_4_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44094016))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48292544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48288384))))[name = string("layers_4_self_attn_o_proj_weight_cast_fp16")]; tensor layers_4_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48294656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60877632))))[name = string("layers_4_mlp_gate_proj_weight_cast_fp16")]; tensor layers_4_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60896192))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73491520))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73479168))))[name = string("layers_4_mlp_up_proj_weight_cast_fp16")]; tensor layers_4_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73497728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86084864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86080704))))[name = string("layers_4_mlp_down_proj_weight_cast_fp16")]; tensor layers_5_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86086976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611328))))[name = string("layers_5_self_attn_v_proj_weight_cast_fp16")]; tensor layers_5_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86612224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90810752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90806592))))[name = string("layers_5_self_attn_o_proj_weight_cast_fp16")]; tensor layers_5_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90812864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103408192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103395840))))[name = string("layers_5_mlp_up_proj_weight_cast_fp16")]; tensor layers_5_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103414400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116001536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115997376))))[name = string("layers_5_mlp_down_proj_weight_cast_fp16")]; tensor layers_6_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116003648))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528576))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528000))))[name = string("layers_6_self_attn_v_proj_weight_cast_fp16")]; tensor layers_6_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528896))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120727424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120723264))))[name = string("layers_6_self_attn_o_proj_weight_cast_fp16")]; tensor layers_6_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120729536))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133324864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133312512))))[name = string("layers_6_mlp_gate_proj_weight_cast_fp16")]; tensor layers_6_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133331072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145926400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145914048))))[name = string("layers_6_mlp_up_proj_weight_cast_fp16")]; tensor layers_6_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145932608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158519744))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158515584))))[name = string("layers_6_mlp_down_proj_weight_cast_fp16")]; tensor layers_7_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158521856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046208))))[name = string("layers_7_self_attn_v_proj_weight_cast_fp16")]; tensor layers_7_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159047104))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163245632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163241472))))[name = string("layers_7_self_attn_o_proj_weight_cast_fp16")]; tensor layers_7_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163247744))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175843072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175830720))))[name = string("layers_7_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175849280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374208))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176373632))))[name = string("layers_8_self_attn_v_proj_weight_cast_fp16")]; tensor layers_8_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374528))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180573056))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180568896))))[name = string("layers_8_self_attn_o_proj_weight_cast_fp16")]; tensor layers_8_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180575168))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193170496))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193158144))))[name = string("layers_8_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193176704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205772032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205759680))))[name = string("layers_8_mlp_up_proj_weight_cast_fp16")]; tensor layers_8_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205778240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218365376))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218361216))))[name = string("layers_8_mlp_down_proj_weight_cast_fp16")]; tensor layers_9_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218367488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218891840))))[name = string("layers_9_self_attn_v_proj_weight_cast_fp16")]; tensor layers_9_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223091264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223087104))))[name = string("layers_9_self_attn_o_proj_weight_cast_fp16")]; tensor layers_9_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223093376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235688704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235676352))))[name = string("layers_9_mlp_gate_proj_weight_cast_fp16")]; tensor layers_9_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235694912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248290240))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248277888))))[name = string("layers_9_mlp_up_proj_weight_cast_fp16")]; tensor layers_9_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248296448))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260883584))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260879424))))[name = string("layers_9_mlp_down_proj_weight_cast_fp16")]; tensor layers_10_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260885696))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410048))))[name = string("layers_10_self_attn_v_proj_weight_cast_fp16")]; tensor layers_10_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410944))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265609472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265605312))))[name = string("layers_10_self_attn_o_proj_weight_cast_fp16")]; tensor layers_10_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265611584))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278206912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278194560))))[name = string("layers_10_mlp_gate_proj_weight_cast_fp16")]; tensor layers_10_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278213120))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290808448))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290796096))))[name = string("layers_10_mlp_up_proj_weight_cast_fp16")]; tensor layers_10_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290814656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303401792))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303397632))))[name = string("layers_10_mlp_down_proj_weight_cast_fp16")]; tensor layers_11_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303403904))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307602432))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307598272))))[name = string("layers_11_self_attn_q_proj_weight_cast_fp16")]; tensor layers_11_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307604544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308128896))))[name = string("layers_11_self_attn_k_proj_weight_cast_fp16")]; tensor layers_11_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129792))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654720))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654144))))[name = string("layers_11_self_attn_v_proj_weight_cast_fp16")]; tensor layers_11_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308655040))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312853568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312849408))))[name = string("layers_11_self_attn_o_proj_weight_cast_fp16")]; tensor layers_11_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312855680))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325451008))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325438656))))[name = string("layers_11_mlp_gate_proj_weight_cast_fp16")]; tensor layers_11_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325457216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338052544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338040192))))[name = string("layers_11_mlp_up_proj_weight_cast_fp16")]; tensor layers_11_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338058752))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350645888))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350641728))))[name = string("layers_11_mlp_down_proj_weight_cast_fp16")]; tensor layers_12_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350648000))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354846528))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354842368))))[name = string("layers_12_self_attn_q_proj_weight_cast_fp16")]; tensor layers_12_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354848640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355372992))))[name = string("layers_12_self_attn_k_proj_weight_cast_fp16")]; tensor layers_12_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373888))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898816))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898240))))[name = string("layers_12_self_attn_v_proj_weight_cast_fp16")]; tensor layers_12_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355899136))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360097664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360093504))))[name = string("layers_12_self_attn_o_proj_weight_cast_fp16")]; tensor layers_12_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360099776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372695104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372682752))))[name = string("layers_12_mlp_gate_proj_weight_cast_fp16")]; tensor layers_12_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372701312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385296640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385284288))))[name = string("layers_12_mlp_up_proj_weight_cast_fp16")]; tensor layers_12_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385302848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397885824))))[name = string("layers_12_mlp_down_proj_weight_cast_fp16")]; tensor layers_13_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397892096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402090624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402086464))))[name = string("layers_13_self_attn_q_proj_weight_cast_fp16")]; tensor layers_13_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402092736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617088))))[name = string("layers_13_self_attn_k_proj_weight_cast_fp16")]; tensor layers_13_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617984))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142336))))[name = string("layers_13_self_attn_v_proj_weight_cast_fp16")]; tensor layers_13_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403143232))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407341760))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407337600))))[name = string("layers_13_self_attn_o_proj_weight_cast_fp16")]; tensor layers_13_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407343872))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419939200))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419926848))))[name = string("layers_13_mlp_gate_proj_weight_cast_fp16")]; tensor layers_13_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419945408))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432532544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432528384))))[name = string("layers_13_mlp_down_proj_weight_cast_fp16")]; tensor layers_14_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432534656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436733184))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436729024))))[name = string("layers_14_self_attn_q_proj_weight_cast_fp16")]; tensor layers_14_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436735296))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437259648))))[name = string("layers_14_self_attn_v_proj_weight_cast_fp16")]; tensor layers_14_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441459072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441454912))))[name = string("layers_14_self_attn_o_proj_weight_cast_fp16")]; tensor layers_14_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441461184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454056512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454044160))))[name = string("layers_14_mlp_gate_proj_weight_cast_fp16")]; tensor layers_14_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454062720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466658048))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466645696))))[name = string("layers_14_mlp_up_proj_weight_cast_fp16")]; tensor layers_14_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466664256))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479251392))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479247232))))[name = string("layers_14_mlp_down_proj_weight_cast_fp16")]; tensor layers_15_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479253504))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483452032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483447872))))[name = string("layers_15_self_attn_q_proj_weight_cast_fp16")]; tensor layers_15_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483454144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483978496))))[name = string("layers_15_self_attn_k_proj_weight_cast_fp16")]; tensor layers_15_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504320))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484503744))))[name = string("layers_15_self_attn_v_proj_weight_cast_fp16")]; tensor layers_15_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488703168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488699008))))[name = string("layers_15_self_attn_o_proj_weight_cast_fp16")]; tensor layers_15_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488705280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501300608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501288256))))[name = string("layers_15_mlp_gate_proj_weight_cast_fp16")]; tensor layers_15_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501306816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513902144))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513889792))))[name = string("layers_15_mlp_up_proj_weight_cast_fp16")]; tensor layers_15_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513908352))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526495488))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526491328))))[name = string("layers_15_mlp_down_proj_weight_cast_fp16")]; tensor layers_16_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526497600))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530696128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530691968))))[name = string("layers_16_self_attn_q_proj_weight_cast_fp16")]; tensor layers_16_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530698240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531222592))))[name = string("layers_16_self_attn_k_proj_weight_cast_fp16")]; tensor layers_16_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531747840))))[name = string("layers_16_self_attn_v_proj_weight_cast_fp16")]; tensor layers_16_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535947264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535943104))))[name = string("layers_16_self_attn_o_proj_weight_cast_fp16")]; tensor layers_16_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535949376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548536512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548532352))))[name = string("layers_16_mlp_down_proj_weight_cast_fp16")]; tensor layers_17_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548538624))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552737152))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552732992))))[name = string("layers_17_self_attn_q_proj_weight_cast_fp16")]; tensor layers_17_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552739264))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553263616))))[name = string("layers_17_self_attn_k_proj_weight_cast_fp16")]; tensor layers_17_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264512))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553788864))))[name = string("layers_17_self_attn_v_proj_weight_cast_fp16")]; tensor layers_17_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789760))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557988288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557984128))))[name = string("layers_17_self_attn_o_proj_weight_cast_fp16")]; tensor layers_17_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557990400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570585728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570573376))))[name = string("layers_17_mlp_gate_proj_weight_cast_fp16")]; tensor layers_17_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570591936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583187264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583174912))))[name = string("layers_17_mlp_up_proj_weight_cast_fp16")]; tensor layers_17_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583193472))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595780608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595776448))))[name = string("layers_17_mlp_down_proj_weight_cast_fp16")]; tensor layers_18_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595782720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599981248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599977088))))[name = string("layers_18_self_attn_q_proj_weight_cast_fp16")]; tensor layers_18_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599983360))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600507712))))[name = string("layers_18_self_attn_k_proj_weight_cast_fp16")]; tensor layers_18_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601032960))))[name = string("layers_18_self_attn_v_proj_weight_cast_fp16")]; tensor layers_18_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605232384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605228224))))[name = string("layers_18_self_attn_o_proj_weight_cast_fp16")]; tensor layers_18_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605234496))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617829824))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617817472))))[name = string("layers_18_mlp_gate_proj_weight_cast_fp16")]; tensor layers_18_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617836032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630431360))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630419008))))[name = string("layers_18_mlp_up_proj_weight_cast_fp16")]; tensor layers_18_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630437568))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643024704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643020544))))[name = string("layers_18_mlp_down_proj_weight_cast_fp16")]; tensor layers_19_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643026816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647225344))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647221184))))[name = string("layers_19_self_attn_q_proj_weight_cast_fp16")]; tensor layers_19_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647227456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647751808))))[name = string("layers_19_self_attn_k_proj_weight_cast_fp16")]; tensor layers_19_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660348032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660335680))))[name = string("layers_19_mlp_gate_proj_weight_cast_fp16")]; tensor layers_19_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660354240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672949568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672937216))))[name = string("layers_19_mlp_up_proj_weight_cast_fp16")]; tensor layers_19_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672955776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685542912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685538752))))[name = string("layers_19_mlp_down_proj_weight_cast_fp16")]; tensor layers_20_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685545024))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689743552))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689739392))))[name = string("layers_20_self_attn_q_proj_weight_cast_fp16")]; tensor layers_20_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689745664))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270592))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270016))))[name = string("layers_20_self_attn_k_proj_weight_cast_fp16")]; tensor layers_20_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694469440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694465280))))[name = string("layers_20_self_attn_o_proj_weight_cast_fp16")]; tensor layers_20_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694471552))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707066880))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707054528))))[name = string("layers_20_mlp_gate_proj_weight_cast_fp16")]; tensor layers_20_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707073088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719660224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719656064))))[name = string("layers_20_mlp_down_proj_weight_cast_fp16")]; tensor layers_21_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719662336))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723860864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723856704))))[name = string("layers_21_self_attn_q_proj_weight_cast_fp16")]; tensor layers_21_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723862976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387328))))[name = string("layers_21_self_attn_k_proj_weight_cast_fp16")]; tensor layers_21_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724388224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728586752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728582592))))[name = string("layers_21_self_attn_o_proj_weight_cast_fp16")]; tensor layers_21_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728588864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741184192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741171840))))[name = string("layers_21_mlp_gate_proj_weight_cast_fp16")]; tensor layers_21_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741190400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753785728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753773376))))[name = string("layers_21_mlp_up_proj_weight_cast_fp16")]; tensor layers_21_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753791936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766379072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766374912))))[name = string("layers_21_mlp_down_proj_weight_cast_fp16")]; tensor layers_22_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766381184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770579712))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770575552))))[name = string("layers_22_self_attn_q_proj_weight_cast_fp16")]; tensor layers_22_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770581824))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106176))))[name = string("layers_22_self_attn_k_proj_weight_cast_fp16")]; tensor layers_22_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771107072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783702400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783690048))))[name = string("layers_22_mlp_gate_proj_weight_cast_fp16")]; tensor layers_22_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783708608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796303936))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796291584))))[name = string("layers_22_mlp_up_proj_weight_cast_fp16")]; tensor layers_22_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796310144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808897280))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808893120))))[name = string("layers_22_mlp_down_proj_weight_cast_fp16")]; tensor layers_23_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808899392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813097920))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813093760))))[name = string("layers_23_self_attn_q_proj_weight_cast_fp16")]; tensor layers_23_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813100032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624960))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624384))))[name = string("layers_23_self_attn_k_proj_weight_cast_fp16")]; tensor layers_23_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813625280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817823808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817819648))))[name = string("layers_23_self_attn_o_proj_weight_cast_fp16")]; tensor layers_23_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817825920))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830421248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830408896))))[name = string("layers_23_mlp_gate_proj_weight_cast_fp16")]; tensor layers_23_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830427456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843022784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843010432))))[name = string("layers_23_mlp_up_proj_weight_cast_fp16")]; tensor layers_23_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843028992))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855616128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855611968))))[name = string("layers_23_mlp_down_proj_weight_cast_fp16")]; tensor layers_24_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855618240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859816768))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859812608))))[name = string("layers_24_self_attn_q_proj_weight_cast_fp16")]; tensor layers_24_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859818880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343232))))[name = string("layers_24_self_attn_k_proj_weight_cast_fp16")]; tensor layers_24_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860344128))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864542656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864538496))))[name = string("layers_24_self_attn_o_proj_weight_cast_fp16")]; tensor layers_24_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864544768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877140096))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877127744))))[name = string("layers_24_mlp_gate_proj_weight_cast_fp16")]; tensor layers_24_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877146304))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889741632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889729280))))[name = string("layers_24_mlp_up_proj_weight_cast_fp16")]; tensor layers_24_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889747840))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902334976))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902330816))))[name = string("layers_24_mlp_down_proj_weight_cast_fp16")]; tensor layers_25_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902337088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906535616))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906531456))))[name = string("layers_25_self_attn_q_proj_weight_cast_fp16")]; tensor layers_25_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906537728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062080))))[name = string("layers_25_self_attn_k_proj_weight_cast_fp16")]; tensor layers_25_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911261504))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911257344))))[name = string("layers_25_self_attn_o_proj_weight_cast_fp16")]; tensor layers_25_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911263616))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923858944))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923846592))))[name = string("layers_25_mlp_gate_proj_weight_cast_fp16")]; tensor layers_25_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923865152))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936460480))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936448128))))[name = string("layers_25_mlp_up_proj_weight_cast_fp16")]; tensor layers_26_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936466688))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940665216))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940661056))))[name = string("layers_26_self_attn_q_proj_weight_cast_fp16")]; tensor layers_26_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940667328))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192256))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941191680))))[name = string("layers_26_self_attn_k_proj_weight_cast_fp16")]; tensor layers_26_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192576))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945391104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945386944))))[name = string("layers_26_self_attn_o_proj_weight_cast_fp16")]; tensor layers_26_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945393216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957988544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957976192))))[name = string("layers_26_mlp_gate_proj_weight_cast_fp16")]; int32 var_765 = const()[name = string("op_765"), val = int32(0)]; tensor var_766 = mul(x = position_index_seed, y = var_765)[name = string("op_766")]; int32 var_768 = const()[name = string("op_768"), val = int32(1)]; tensor ones = add(x = var_766, y = var_768)[name = string("ones")]; int32 var_770 = const()[name = string("op_770"), val = int32(0)]; bool var_772_exclusive_0 = const()[name = string("op_772_exclusive_0"), val = bool(false)]; bool var_772_reverse_0 = const()[name = string("op_772_reverse_0"), val = bool(false)]; tensor var_772 = cumsum(axis = var_770, exclusive = var_772_exclusive_0, reverse = var_772_reverse_0, x = ones)[name = string("op_772")]; int32 var_774 = const()[name = string("op_774"), val = int32(1)]; tensor position_offsets = sub(x = var_772, y = var_774)[name = string("position_offsets")]; tensor position_ids_1 = add(x = position_offsets, y = position_id)[name = string("position_ids_1")]; bool var_784_keep_dims_0 = const()[name = string("op_784_keep_dims_0"), val = bool(false)]; int32 var_784 = reduce_sum(keep_dims = var_784_keep_dims_0, x = ones)[name = string("op_784")]; int32 var_786 = const()[name = string("op_786"), val = int32(1)]; int32 offset = sub(x = var_784, y = var_786)[name = string("offset")]; tensor var_789 = add(x = position_id, y = offset)[name = string("op_789")]; int32 var_791 = const()[name = string("op_791"), val = int32(1)]; tensor cache_position_end = add(x = var_789, y = var_791)[name = string("cache_position_end")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor greater_equal_0 = greater_equal(x = position_ids_1, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(32768)]; tensor add_0 = add(x = position_ids_1, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = position_ids_1, b = add_0, cond = greater_equal_0)[name = string("select_0")]; tensor rope_emb_cos_cached_to_fp16 = const()[name = string("rope_emb_cos_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957994752)))]; int32 cos_1_batch_dims_0 = const()[name = string("cos_1_batch_dims_0"), val = int32(0)]; bool cos_1_validate_indices_0 = const()[name = string("cos_1_validate_indices_0"), val = bool(false)]; int32 greater_equal_14_y_0 = const()[name = string("greater_equal_14_y_0"), val = int32(0)]; tensor greater_equal_14 = greater_equal(x = select_0, y = greater_equal_14_y_0)[name = string("greater_equal_14")]; int32 slice_by_index_14 = const()[name = string("slice_by_index_14"), val = int32(32768)]; tensor add_14 = add(x = select_0, y = slice_by_index_14)[name = string("add_14")]; tensor select_14 = select(a = select_0, b = add_14, cond = greater_equal_14)[name = string("select_14")]; int32 cos_1_cast_fp16_axis_7 = const()[name = string("cos_1_cast_fp16_axis_7"), val = int32(0)]; tensor cos_1_cast_fp16 = gather(axis = cos_1_cast_fp16_axis_7, batch_dims = cos_1_batch_dims_0, indices = select_14, validate_indices = cos_1_validate_indices_0, x = rope_emb_cos_cached_to_fp16)[name = string("cos_1_cast_fp16")]; tensor rope_emb_sin_cached_to_fp16 = const()[name = string("rope_emb_sin_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(966383424)))]; int32 sin_1_batch_dims_0 = const()[name = string("sin_1_batch_dims_0"), val = int32(0)]; bool sin_1_validate_indices_0 = const()[name = string("sin_1_validate_indices_0"), val = bool(false)]; int32 sin_1_cast_fp16_axis_7 = const()[name = string("sin_1_cast_fp16_axis_7"), val = int32(0)]; tensor sin_1_cast_fp16 = gather(axis = sin_1_cast_fp16_axis_7, batch_dims = sin_1_batch_dims_0, indices = select_14, validate_indices = sin_1_validate_indices_0, x = rope_emb_sin_cached_to_fp16)[name = string("sin_1_cast_fp16")]; tensor var_865_perm_0 = const()[name = string("op_865_perm_0"), val = tensor([-1, -2])]; tensor var_867_axes_0 = const()[name = string("op_867_axes_0"), val = tensor([0])]; tensor var_865_cast_fp16 = transpose(perm = var_865_perm_0, x = cos_1_cast_fp16)[name = string("transpose_687")]; tensor var_867_cast_fp16 = expand_dims(axes = var_867_axes_0, x = var_865_cast_fp16)[name = string("op_867_cast_fp16")]; tensor var_869_axes_0 = const()[name = string("op_869_axes_0"), val = tensor([0])]; tensor var_869_cast_fp16 = expand_dims(axes = var_869_axes_0, x = var_867_cast_fp16)[name = string("op_869_cast_fp16")]; tensor var_874_perm_0 = const()[name = string("op_874_perm_0"), val = tensor([-1, -2])]; tensor var_876_axes_0 = const()[name = string("op_876_axes_0"), val = tensor([0])]; tensor var_874_cast_fp16 = transpose(perm = var_874_perm_0, x = sin_1_cast_fp16)[name = string("transpose_686")]; tensor var_876_cast_fp16 = expand_dims(axes = var_876_axes_0, x = var_874_cast_fp16)[name = string("op_876_cast_fp16")]; tensor var_878_axes_0 = const()[name = string("op_878_axes_0"), val = tensor([0])]; tensor var_878_cast_fp16 = expand_dims(axes = var_878_axes_0, x = var_876_cast_fp16)[name = string("op_878_cast_fp16")]; string position_ids_1_to_uint16_dtype_0 = const()[name = string("position_ids_1_to_uint16_dtype_0"), val = string("uint16")]; tensor causal_mask = const()[name = string("causal_mask"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(974772096)))]; int32 mask_axis_0 = const()[name = string("mask_axis_0"), val = int32(1)]; int32 mask_batch_dims_0 = const()[name = string("mask_batch_dims_0"), val = int32(0)]; bool mask_validate_indices_0 = const()[name = string("mask_validate_indices_0"), val = bool(false)]; tensor position_ids_1_to_uint16 = cast(dtype = position_ids_1_to_uint16_dtype_0, x = position_ids_1)[name = string("cast_15")]; tensor mask_cast_uint16 = gather(axis = mask_axis_0, batch_dims = mask_batch_dims_0, indices = position_ids_1_to_uint16, validate_indices = mask_validate_indices_0, x = causal_mask)[name = string("mask_cast_uint16")]; tensor var_895_axes_0 = const()[name = string("op_895_axes_0"), val = tensor([0])]; tensor var_895 = expand_dims(axes = var_895_axes_0, x = mask_cast_uint16)[name = string("op_895")]; tensor attn_mask_1_axes_0 = const()[name = string("attn_mask_1_axes_0"), val = tensor([0])]; tensor attn_mask_1 = expand_dims(axes = attn_mask_1_axes_0, x = var_895)[name = string("attn_mask_1")]; string inputs_embeds_to_fp16_dtype_0 = const()[name = string("inputs_embeds_to_fp16_dtype_0"), val = string("fp16")]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor inputs_embeds_to_fp16 = cast(dtype = inputs_embeds_to_fp16_dtype_0, x = inputs_embeds)[name = string("cast_14")]; tensor var_906_cast_fp16 = mul(x = inputs_embeds_to_fp16, y = const_0_promoted_to_fp16)[name = string("op_906_cast_fp16")]; int32 var_904 = const()[name = string("op_904"), val = int32(1)]; bool doubled_1_interleave_0 = const()[name = string("doubled_1_interleave_0"), val = bool(false)]; tensor doubled_1_cast_fp16 = concat(axis = var_904, interleave = doubled_1_interleave_0, values = (inputs_embeds_to_fp16, var_906_cast_fp16))[name = string("doubled_1_cast_fp16")]; tensor out_1_axes_0 = const()[name = string("out_1_axes_0"), val = tensor([1])]; tensor out_1_gamma_0_to_fp16 = const()[name = string("out_1_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983160768)))]; fp16 var_916_to_fp16 = const()[name = string("op_916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_1_cast_fp16 = layer_norm(axes = out_1_axes_0, epsilon = var_916_to_fp16, gamma = out_1_gamma_0_to_fp16, x = doubled_1_cast_fp16)[name = string("out_1_cast_fp16")]; tensor var_927_split_sizes_0 = const()[name = string("op_927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_927_axis_0 = const()[name = string("op_927_axis_0"), val = int32(1)]; tensor var_927_cast_fp16_0, tensor var_927_cast_fp16_1 = split(axis = var_927_axis_0, split_sizes = var_927_split_sizes_0, x = out_1_cast_fp16)[name = string("op_927_cast_fp16")]; tensor layers_0_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983169024)))]; tensor query_states_1_strides_0 = const()[name = string("query_states_1_strides_0"), val = tensor([1, 1])]; string query_states_1_pad_type_0 = const()[name = string("query_states_1_pad_type_0"), val = string("valid")]; tensor query_states_1_pad_0 = const()[name = string("query_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_1_dilations_0 = const()[name = string("query_states_1_dilations_0"), val = tensor([1, 1])]; int32 query_states_1_groups_0 = const()[name = string("query_states_1_groups_0"), val = int32(1)]; tensor query_states_1_cast_fp16 = conv(dilations = query_states_1_dilations_0, groups = query_states_1_groups_0, pad = query_states_1_pad_0, pad_type = query_states_1_pad_type_0, strides = query_states_1_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("query_states_1_cast_fp16")]; tensor layers_0_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(991557696)))]; tensor key_states_1_strides_0 = const()[name = string("key_states_1_strides_0"), val = tensor([1, 1])]; string key_states_1_pad_type_0 = const()[name = string("key_states_1_pad_type_0"), val = string("valid")]; tensor key_states_1_pad_0 = const()[name = string("key_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_1_dilations_0 = const()[name = string("key_states_1_dilations_0"), val = tensor([1, 1])]; int32 key_states_1_groups_0 = const()[name = string("key_states_1_groups_0"), val = int32(1)]; tensor key_states_1_cast_fp16 = conv(dilations = key_states_1_dilations_0, groups = key_states_1_groups_0, pad = key_states_1_pad_0, pad_type = key_states_1_pad_type_0, strides = key_states_1_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("key_states_1_cast_fp16")]; tensor layers_0_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(992606336)))]; tensor value_states_1_strides_0 = const()[name = string("value_states_1_strides_0"), val = tensor([1, 1])]; string value_states_1_pad_type_0 = const()[name = string("value_states_1_pad_type_0"), val = string("valid")]; tensor value_states_1_pad_0 = const()[name = string("value_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_1_dilations_0 = const()[name = string("value_states_1_dilations_0"), val = tensor([1, 1])]; int32 value_states_1_groups_0 = const()[name = string("value_states_1_groups_0"), val = int32(1)]; tensor value_states_1_cast_fp16 = conv(dilations = value_states_1_dilations_0, groups = value_states_1_groups_0, pad = value_states_1_pad_0, pad_type = value_states_1_pad_type_0, strides = value_states_1_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("value_states_1_cast_fp16")]; tensor concat_0x = const()[name = string("concat_0x"), val = tensor([1, 16, 128, -1])]; tensor x_1_cast_fp16 = reshape(shape = concat_0x, x = query_states_1_cast_fp16)[name = string("x_1_cast_fp16")]; tensor concat_1x = const()[name = string("concat_1x"), val = tensor([1, 2, 128, -1])]; tensor var_984_cast_fp16 = reshape(shape = concat_1x, x = key_states_1_cast_fp16)[name = string("op_984_cast_fp16")]; tensor concat_2x = const()[name = string("concat_2x"), val = tensor([1, 2, 128, -1])]; tensor var_991_cast_fp16 = reshape(shape = concat_2x, x = value_states_1_cast_fp16)[name = string("op_991_cast_fp16")]; tensor var_995_cast_fp16 = mul(x = x_1_cast_fp16, y = var_869_cast_fp16)[name = string("op_995_cast_fp16")]; tensor var_996_split_sizes_0 = const()[name = string("op_996_split_sizes_0"), val = tensor([64, 64])]; int32 var_996_axis_0 = const()[name = string("op_996_axis_0"), val = int32(-2)]; tensor var_996_cast_fp16_0, tensor var_996_cast_fp16_1 = split(axis = var_996_axis_0, split_sizes = var_996_split_sizes_0, x = x_1_cast_fp16)[name = string("op_996_cast_fp16")]; fp16 const_2_promoted_to_fp16 = const()[name = string("const_2_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_998_cast_fp16 = mul(x = var_996_cast_fp16_1, y = const_2_promoted_to_fp16)[name = string("op_998_cast_fp16")]; int32 var_1000 = const()[name = string("op_1000"), val = int32(-2)]; bool var_1001_interleave_0 = const()[name = string("op_1001_interleave_0"), val = bool(false)]; tensor var_1001_cast_fp16 = concat(axis = var_1000, interleave = var_1001_interleave_0, values = (var_998_cast_fp16, var_996_cast_fp16_0))[name = string("op_1001_cast_fp16")]; tensor var_1002_cast_fp16 = mul(x = var_1001_cast_fp16, y = var_878_cast_fp16)[name = string("op_1002_cast_fp16")]; tensor query_states_3_cast_fp16 = add(x = var_995_cast_fp16, y = var_1002_cast_fp16)[name = string("query_states_3_cast_fp16")]; tensor var_1008_cast_fp16 = mul(x = var_984_cast_fp16, y = var_869_cast_fp16)[name = string("op_1008_cast_fp16")]; tensor var_1009_split_sizes_0 = const()[name = string("op_1009_split_sizes_0"), val = tensor([64, 64])]; int32 var_1009_axis_0 = const()[name = string("op_1009_axis_0"), val = int32(-2)]; tensor var_1009_cast_fp16_0, tensor var_1009_cast_fp16_1 = split(axis = var_1009_axis_0, split_sizes = var_1009_split_sizes_0, x = var_984_cast_fp16)[name = string("op_1009_cast_fp16")]; fp16 const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1011_cast_fp16 = mul(x = var_1009_cast_fp16_1, y = const_3_promoted_to_fp16)[name = string("op_1011_cast_fp16")]; int32 var_1013 = const()[name = string("op_1013"), val = int32(-2)]; bool var_1014_interleave_0 = const()[name = string("op_1014_interleave_0"), val = bool(false)]; tensor var_1014_cast_fp16 = concat(axis = var_1013, interleave = var_1014_interleave_0, values = (var_1011_cast_fp16, var_1009_cast_fp16_0))[name = string("op_1014_cast_fp16")]; tensor var_1015_cast_fp16 = mul(x = var_1014_cast_fp16, y = var_878_cast_fp16)[name = string("op_1015_cast_fp16")]; tensor key_states_5_cast_fp16 = add(x = var_1008_cast_fp16, y = var_1015_cast_fp16)[name = string("key_states_5_cast_fp16")]; tensor read_state_0 = read_state(input = key_cache)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; int32 concat_5_axis_0 = const()[name = string("concat_5_axis_0"), val = int32(0)]; bool concat_5_interleave_0 = const()[name = string("concat_5_interleave_0"), val = bool(false)]; tensor concat_5 = concat(axis = concat_5_axis_0, interleave = concat_5_interleave_0, values = (expand_dims_0, expand_dims_1, position_id, expand_dims_3))[name = string("concat_5")]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; tensor concat_6_values1_0 = const()[name = string("concat_6_values1_0"), val = tensor([0])]; tensor concat_6_values3_0 = const()[name = string("concat_6_values3_0"), val = tensor([0])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_4, concat_6_values1_0, cache_position_end, concat_6_values3_0))[name = string("concat_6")]; tensor key_states_7_perm_0 = const()[name = string("key_states_7_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_1_stride_0 = const()[name = string("key_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_7_cast_fp16 = transpose(perm = key_states_7_perm_0, x = key_states_5_cast_fp16)[name = string("transpose_685")]; tensor key_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = key_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = key_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_1_squeeze_mask_0, stride = key_cache_internal_tensor_assign_1_stride_0, update = key_states_7_cast_fp16, x = read_state_0)[name = string("key_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_1_cast_fp16, input = key_cache)[name = string("coreml_update_state_392_write_state")]; tensor coreml_update_state_392 = read_state(input = key_cache)[name = string("coreml_update_state_392")]; tensor read_state_1 = read_state(input = value_cache)[name = string("read_state_1")]; tensor value_states_3_perm_0 = const()[name = string("value_states_3_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_1_stride_0 = const()[name = string("value_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_3_cast_fp16 = transpose(perm = value_states_3_perm_0, x = var_991_cast_fp16)[name = string("transpose_684")]; tensor value_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = value_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = value_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_1_squeeze_mask_0, stride = value_cache_internal_tensor_assign_1_stride_0, update = value_states_3_cast_fp16, x = read_state_1)[name = string("value_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_1_cast_fp16, input = value_cache)[name = string("coreml_update_state_393_write_state")]; tensor coreml_update_state_393 = read_state(input = value_cache)[name = string("coreml_update_state_393")]; tensor var_1085_begin_0 = const()[name = string("op_1085_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1085_end_0 = const()[name = string("op_1085_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1085_end_mask_0 = const()[name = string("op_1085_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1085_cast_fp16 = slice_by_index(begin = var_1085_begin_0, end = var_1085_end_0, end_mask = var_1085_end_mask_0, x = coreml_update_state_392)[name = string("op_1085_cast_fp16")]; tensor tile_0 = const()[name = string("tile_0"), val = tensor([1, 1])]; int32 var_1088_axis_0 = const()[name = string("op_1088_axis_0"), val = int32(1)]; tensor var_1088_cast_fp16_0, tensor var_1088_cast_fp16_1 = split(axis = var_1088_axis_0, split_sizes = tile_0, x = var_1085_cast_fp16)[name = string("op_1088_cast_fp16")]; tensor var_1095_begin_0 = const()[name = string("op_1095_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1095_end_0 = const()[name = string("op_1095_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1095_end_mask_0 = const()[name = string("op_1095_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1095_cast_fp16 = slice_by_index(begin = var_1095_begin_0, end = var_1095_end_0, end_mask = var_1095_end_mask_0, x = coreml_update_state_393)[name = string("op_1095_cast_fp16")]; tensor tile_1 = const()[name = string("tile_1"), val = tensor([1, 1])]; int32 var_1098_axis_0 = const()[name = string("op_1098_axis_0"), val = int32(1)]; tensor var_1098_cast_fp16_0, tensor var_1098_cast_fp16_1 = split(axis = var_1098_axis_0, split_sizes = tile_1, x = var_1095_cast_fp16)[name = string("op_1098_cast_fp16")]; tensor var_1101_split_sizes_0 = const()[name = string("op_1101_split_sizes_0"), val = tensor([8, 8])]; int32 var_1101_axis_0 = const()[name = string("op_1101_axis_0"), val = int32(1)]; tensor var_1101_0, tensor var_1101_1 = split(axis = var_1101_axis_0, split_sizes = var_1101_split_sizes_0, x = query_states_3_cast_fp16)[name = string("op_1101")]; bool attn_weights_1_transpose_x_0 = const()[name = string("attn_weights_1_transpose_x_0"), val = bool(false)]; bool attn_weights_1_transpose_y_0 = const()[name = string("attn_weights_1_transpose_y_0"), val = bool(false)]; tensor attn_weights_1_cast_fp16 = matmul(transpose_x = attn_weights_1_transpose_x_0, transpose_y = attn_weights_1_transpose_y_0, x = var_1088_cast_fp16_0, y = var_1101_0)[name = string("attn_weights_1_cast_fp16")]; fp16 var_1104_to_fp16 = const()[name = string("op_1104_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_3_cast_fp16 = mul(x = attn_weights_1_cast_fp16, y = var_1104_to_fp16)[name = string("attn_weights_3_cast_fp16")]; tensor attn_weights_5_cast_fp16 = add(x = attn_weights_3_cast_fp16, y = attn_mask_1)[name = string("attn_weights_5_cast_fp16")]; int32 var_1108 = const()[name = string("op_1108"), val = int32(-2)]; tensor attn_weights_7_cast_fp16 = softmax(axis = var_1108, x = attn_weights_5_cast_fp16)[name = string("attn_weights_7_cast_fp16")]; bool var_1114_transpose_x_1 = const()[name = string("op_1114_transpose_x_1"), val = bool(true)]; bool var_1114_transpose_y_1 = const()[name = string("op_1114_transpose_y_1"), val = bool(false)]; tensor var_1114_cast_fp16 = matmul(transpose_x = var_1114_transpose_x_1, transpose_y = var_1114_transpose_y_1, x = attn_weights_7_cast_fp16, y = var_1098_cast_fp16_0)[name = string("op_1114_cast_fp16")]; bool attn_weights_9_transpose_x_0 = const()[name = string("attn_weights_9_transpose_x_0"), val = bool(false)]; bool attn_weights_9_transpose_y_0 = const()[name = string("attn_weights_9_transpose_y_0"), val = bool(false)]; tensor attn_weights_9_cast_fp16 = matmul(transpose_x = attn_weights_9_transpose_x_0, transpose_y = attn_weights_9_transpose_y_0, x = var_1088_cast_fp16_1, y = var_1101_1)[name = string("attn_weights_9_cast_fp16")]; fp16 var_1116_to_fp16 = const()[name = string("op_1116_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_11_cast_fp16 = mul(x = attn_weights_9_cast_fp16, y = var_1116_to_fp16)[name = string("attn_weights_11_cast_fp16")]; tensor attn_weights_13_cast_fp16 = add(x = attn_weights_11_cast_fp16, y = attn_mask_1)[name = string("attn_weights_13_cast_fp16")]; int32 var_1120 = const()[name = string("op_1120"), val = int32(-2)]; tensor attn_weights_15_cast_fp16 = softmax(axis = var_1120, x = attn_weights_13_cast_fp16)[name = string("attn_weights_15_cast_fp16")]; bool attn_output_1_transpose_x_1 = const()[name = string("attn_output_1_transpose_x_1"), val = bool(true)]; bool attn_output_1_transpose_y_1 = const()[name = string("attn_output_1_transpose_y_1"), val = bool(false)]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_1, transpose_y = attn_output_1_transpose_y_1, x = attn_weights_15_cast_fp16, y = var_1098_cast_fp16_1)[name = string("attn_output_1_cast_fp16")]; int32 var_1128 = const()[name = string("op_1128"), val = int32(1)]; bool attn_output_3_interleave_0 = const()[name = string("attn_output_3_interleave_0"), val = bool(false)]; tensor attn_output_3_cast_fp16 = concat(axis = var_1128, interleave = attn_output_3_interleave_0, values = (var_1114_cast_fp16, attn_output_1_cast_fp16))[name = string("attn_output_3_cast_fp16")]; tensor var_1132_perm_0 = const()[name = string("op_1132_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_11x = const()[name = string("concat_11x"), val = tensor([1, 2048, 1, -1])]; tensor var_1132_cast_fp16 = transpose(perm = var_1132_perm_0, x = attn_output_3_cast_fp16)[name = string("transpose_683")]; tensor attn_output_7_cast_fp16 = reshape(shape = concat_11x, x = var_1132_cast_fp16)[name = string("attn_output_7_cast_fp16")]; tensor layers_0_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(993654976)))]; tensor hidden_states_3_strides_0 = const()[name = string("hidden_states_3_strides_0"), val = tensor([1, 1])]; string hidden_states_3_pad_type_0 = const()[name = string("hidden_states_3_pad_type_0"), val = string("valid")]; tensor hidden_states_3_pad_0 = const()[name = string("hidden_states_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_3_dilations_0 = const()[name = string("hidden_states_3_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_3_groups_0 = const()[name = string("hidden_states_3_groups_0"), val = int32(1)]; tensor hidden_states_3_cast_fp16 = conv(dilations = hidden_states_3_dilations_0, groups = hidden_states_3_groups_0, pad = hidden_states_3_pad_0, pad_type = hidden_states_3_pad_type_0, strides = hidden_states_3_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16, x = attn_output_7_cast_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor hidden_states_5_cast_fp16 = add(x = inputs_embeds_to_fp16, y = hidden_states_3_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1165_cast_fp16 = mul(x = hidden_states_5_cast_fp16, y = const_8_promoted_to_fp16)[name = string("op_1165_cast_fp16")]; int32 var_1163 = const()[name = string("op_1163"), val = int32(1)]; bool doubled_5_interleave_0 = const()[name = string("doubled_5_interleave_0"), val = bool(false)]; tensor doubled_5_cast_fp16 = concat(axis = var_1163, interleave = doubled_5_interleave_0, values = (hidden_states_5_cast_fp16, var_1165_cast_fp16))[name = string("doubled_5_cast_fp16")]; tensor out_3_axes_0 = const()[name = string("out_3_axes_0"), val = tensor([1])]; tensor out_3_gamma_0_to_fp16 = const()[name = string("out_3_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002043648)))]; fp16 var_1175_to_fp16 = const()[name = string("op_1175_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_3_cast_fp16 = layer_norm(axes = out_3_axes_0, epsilon = var_1175_to_fp16, gamma = out_3_gamma_0_to_fp16, x = doubled_5_cast_fp16)[name = string("out_3_cast_fp16")]; tensor var_1186_split_sizes_0 = const()[name = string("op_1186_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1186_axis_0 = const()[name = string("op_1186_axis_0"), val = int32(1)]; tensor var_1186_cast_fp16_0, tensor var_1186_cast_fp16_1 = split(axis = var_1186_axis_0, split_sizes = var_1186_split_sizes_0, x = out_3_cast_fp16)[name = string("op_1186_cast_fp16")]; tensor layers_0_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002051904)))]; tensor input_1_strides_0 = const()[name = string("input_1_strides_0"), val = tensor([1, 1])]; string input_1_pad_type_0 = const()[name = string("input_1_pad_type_0"), val = string("valid")]; tensor input_1_pad_0 = const()[name = string("input_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_1_dilations_0 = const()[name = string("input_1_dilations_0"), val = tensor([1, 1])]; int32 input_1_groups_0 = const()[name = string("input_1_groups_0"), val = int32(1)]; tensor input_1_cast_fp16 = conv(dilations = input_1_dilations_0, groups = input_1_groups_0, pad = input_1_pad_0, pad_type = input_1_pad_type_0, strides = input_1_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("input_1_cast_fp16")]; tensor var_1203_cast_fp16 = silu(x = input_1_cast_fp16)[name = string("op_1203_cast_fp16")]; tensor layers_0_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1027217792)))]; tensor var_1209_strides_0 = const()[name = string("op_1209_strides_0"), val = tensor([1, 1])]; string var_1209_pad_type_0 = const()[name = string("op_1209_pad_type_0"), val = string("valid")]; tensor var_1209_pad_0 = const()[name = string("op_1209_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1209_dilations_0 = const()[name = string("op_1209_dilations_0"), val = tensor([1, 1])]; int32 var_1209_groups_0 = const()[name = string("op_1209_groups_0"), val = int32(1)]; tensor var_1209_cast_fp16 = conv(dilations = var_1209_dilations_0, groups = var_1209_groups_0, pad = var_1209_pad_0, pad_type = var_1209_pad_type_0, strides = var_1209_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("op_1209_cast_fp16")]; tensor x_9_cast_fp16 = mul(x = var_1203_cast_fp16, y = var_1209_cast_fp16)[name = string("x_9_cast_fp16")]; tensor layers_0_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1052383680)))]; tensor hidden_states_7_strides_0 = const()[name = string("hidden_states_7_strides_0"), val = tensor([1, 1])]; string hidden_states_7_pad_type_0 = const()[name = string("hidden_states_7_pad_type_0"), val = string("valid")]; tensor hidden_states_7_pad_0 = const()[name = string("hidden_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_7_dilations_0 = const()[name = string("hidden_states_7_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_7_groups_0 = const()[name = string("hidden_states_7_groups_0"), val = int32(1)]; tensor hidden_states_7_cast_fp16 = conv(dilations = hidden_states_7_dilations_0, groups = hidden_states_7_groups_0, pad = hidden_states_7_pad_0, pad_type = hidden_states_7_pad_type_0, strides = hidden_states_7_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16, x = x_9_cast_fp16)[name = string("hidden_states_7_cast_fp16")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = hidden_states_7_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; fp16 const_10_promoted_to_fp16 = const()[name = string("const_10_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1227_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_10_promoted_to_fp16)[name = string("op_1227_cast_fp16")]; int32 var_1225 = const()[name = string("op_1225"), val = int32(1)]; bool doubled_9_interleave_0 = const()[name = string("doubled_9_interleave_0"), val = bool(false)]; tensor doubled_9_cast_fp16 = concat(axis = var_1225, interleave = doubled_9_interleave_0, values = (hidden_states_9_cast_fp16, var_1227_cast_fp16))[name = string("doubled_9_cast_fp16")]; tensor out_5_axes_0 = const()[name = string("out_5_axes_0"), val = tensor([1])]; tensor out_5_gamma_0_to_fp16 = const()[name = string("out_5_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077549568)))]; fp16 var_1237_to_fp16 = const()[name = string("op_1237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_5_cast_fp16 = layer_norm(axes = out_5_axes_0, epsilon = var_1237_to_fp16, gamma = out_5_gamma_0_to_fp16, x = doubled_9_cast_fp16)[name = string("out_5_cast_fp16")]; tensor var_1248_split_sizes_0 = const()[name = string("op_1248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1248_axis_0 = const()[name = string("op_1248_axis_0"), val = int32(1)]; tensor var_1248_cast_fp16_0, tensor var_1248_cast_fp16_1 = split(axis = var_1248_axis_0, split_sizes = var_1248_split_sizes_0, x = out_5_cast_fp16)[name = string("op_1248_cast_fp16")]; tensor layers_1_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077557824)))]; tensor query_states_7_strides_0 = const()[name = string("query_states_7_strides_0"), val = tensor([1, 1])]; string query_states_7_pad_type_0 = const()[name = string("query_states_7_pad_type_0"), val = string("valid")]; tensor query_states_7_pad_0 = const()[name = string("query_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_7_dilations_0 = const()[name = string("query_states_7_dilations_0"), val = tensor([1, 1])]; int32 query_states_7_groups_0 = const()[name = string("query_states_7_groups_0"), val = int32(1)]; tensor query_states_7_cast_fp16 = conv(dilations = query_states_7_dilations_0, groups = query_states_7_groups_0, pad = query_states_7_pad_0, pad_type = query_states_7_pad_type_0, strides = query_states_7_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("query_states_7_cast_fp16")]; tensor layers_1_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1085946496)))]; tensor key_states_11_strides_0 = const()[name = string("key_states_11_strides_0"), val = tensor([1, 1])]; string key_states_11_pad_type_0 = const()[name = string("key_states_11_pad_type_0"), val = string("valid")]; tensor key_states_11_pad_0 = const()[name = string("key_states_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_11_dilations_0 = const()[name = string("key_states_11_dilations_0"), val = tensor([1, 1])]; int32 key_states_11_groups_0 = const()[name = string("key_states_11_groups_0"), val = int32(1)]; tensor key_states_11_cast_fp16 = conv(dilations = key_states_11_dilations_0, groups = key_states_11_groups_0, pad = key_states_11_pad_0, pad_type = key_states_11_pad_type_0, strides = key_states_11_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("key_states_11_cast_fp16")]; tensor value_states_7_strides_0 = const()[name = string("value_states_7_strides_0"), val = tensor([1, 1])]; string value_states_7_pad_type_0 = const()[name = string("value_states_7_pad_type_0"), val = string("valid")]; tensor value_states_7_pad_0 = const()[name = string("value_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_7_dilations_0 = const()[name = string("value_states_7_dilations_0"), val = tensor([1, 1])]; int32 value_states_7_groups_0 = const()[name = string("value_states_7_groups_0"), val = int32(1)]; tensor value_states_7_cast_fp16 = conv(dilations = value_states_7_dilations_0, groups = value_states_7_groups_0, pad = value_states_7_pad_0, pad_type = value_states_7_pad_type_0, strides = value_states_7_strides_0, weight = layers_1_self_attn_v_proj_weight_cast_fp16, x = var_1248_cast_fp16_0)[name = string("value_states_7_cast_fp16")]; tensor concat_12x = const()[name = string("concat_12x"), val = tensor([1, 16, 128, -1])]; tensor x_11_cast_fp16 = reshape(shape = concat_12x, x = query_states_7_cast_fp16)[name = string("x_11_cast_fp16")]; tensor concat_13x = const()[name = string("concat_13x"), val = tensor([1, 2, 128, -1])]; tensor var_1305_cast_fp16 = reshape(shape = concat_13x, x = key_states_11_cast_fp16)[name = string("op_1305_cast_fp16")]; tensor concat_14x = const()[name = string("concat_14x"), val = tensor([1, 2, 128, -1])]; tensor var_1312_cast_fp16 = reshape(shape = concat_14x, x = value_states_7_cast_fp16)[name = string("op_1312_cast_fp16")]; tensor var_1316_cast_fp16 = mul(x = x_11_cast_fp16, y = var_869_cast_fp16)[name = string("op_1316_cast_fp16")]; tensor var_1317_split_sizes_0 = const()[name = string("op_1317_split_sizes_0"), val = tensor([64, 64])]; int32 var_1317_axis_0 = const()[name = string("op_1317_axis_0"), val = int32(-2)]; tensor var_1317_cast_fp16_0, tensor var_1317_cast_fp16_1 = split(axis = var_1317_axis_0, split_sizes = var_1317_split_sizes_0, x = x_11_cast_fp16)[name = string("op_1317_cast_fp16")]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1319_cast_fp16 = mul(x = var_1317_cast_fp16_1, y = const_12_promoted_to_fp16)[name = string("op_1319_cast_fp16")]; int32 var_1321 = const()[name = string("op_1321"), val = int32(-2)]; bool var_1322_interleave_0 = const()[name = string("op_1322_interleave_0"), val = bool(false)]; tensor var_1322_cast_fp16 = concat(axis = var_1321, interleave = var_1322_interleave_0, values = (var_1319_cast_fp16, var_1317_cast_fp16_0))[name = string("op_1322_cast_fp16")]; tensor var_1323_cast_fp16 = mul(x = var_1322_cast_fp16, y = var_878_cast_fp16)[name = string("op_1323_cast_fp16")]; tensor query_states_9_cast_fp16 = add(x = var_1316_cast_fp16, y = var_1323_cast_fp16)[name = string("query_states_9_cast_fp16")]; tensor var_1329_cast_fp16 = mul(x = var_1305_cast_fp16, y = var_869_cast_fp16)[name = string("op_1329_cast_fp16")]; tensor var_1330_split_sizes_0 = const()[name = string("op_1330_split_sizes_0"), val = tensor([64, 64])]; int32 var_1330_axis_0 = const()[name = string("op_1330_axis_0"), val = int32(-2)]; tensor var_1330_cast_fp16_0, tensor var_1330_cast_fp16_1 = split(axis = var_1330_axis_0, split_sizes = var_1330_split_sizes_0, x = var_1305_cast_fp16)[name = string("op_1330_cast_fp16")]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1332_cast_fp16 = mul(x = var_1330_cast_fp16_1, y = const_13_promoted_to_fp16)[name = string("op_1332_cast_fp16")]; int32 var_1334 = const()[name = string("op_1334"), val = int32(-2)]; bool var_1335_interleave_0 = const()[name = string("op_1335_interleave_0"), val = bool(false)]; tensor var_1335_cast_fp16 = concat(axis = var_1334, interleave = var_1335_interleave_0, values = (var_1332_cast_fp16, var_1330_cast_fp16_0))[name = string("op_1335_cast_fp16")]; tensor var_1336_cast_fp16 = mul(x = var_1335_cast_fp16, y = var_878_cast_fp16)[name = string("op_1336_cast_fp16")]; tensor key_states_15_cast_fp16 = add(x = var_1329_cast_fp16, y = var_1336_cast_fp16)[name = string("key_states_15_cast_fp16")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; int32 concat_17_axis_0 = const()[name = string("concat_17_axis_0"), val = int32(0)]; bool concat_17_interleave_0 = const()[name = string("concat_17_interleave_0"), val = bool(false)]; tensor concat_17 = concat(axis = concat_17_axis_0, interleave = concat_17_interleave_0, values = (expand_dims_12, expand_dims_13, position_id, expand_dims_15))[name = string("concat_17")]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; tensor concat_18_values1_0 = const()[name = string("concat_18_values1_0"), val = tensor([0])]; tensor concat_18_values3_0 = const()[name = string("concat_18_values3_0"), val = tensor([0])]; int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_16, concat_18_values1_0, cache_position_end, concat_18_values3_0))[name = string("concat_18")]; tensor key_states_17_perm_0 = const()[name = string("key_states_17_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_2_stride_0 = const()[name = string("key_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_17_cast_fp16 = transpose(perm = key_states_17_perm_0, x = key_states_15_cast_fp16)[name = string("transpose_682")]; tensor key_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = key_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = key_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_2_squeeze_mask_0, stride = key_cache_internal_tensor_assign_2_stride_0, update = key_states_17_cast_fp16, x = coreml_update_state_392)[name = string("key_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_2_cast_fp16, input = key_cache)[name = string("coreml_update_state_394_write_state")]; tensor coreml_update_state_394 = read_state(input = key_cache)[name = string("coreml_update_state_394")]; tensor value_states_9_perm_0 = const()[name = string("value_states_9_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_2_stride_0 = const()[name = string("value_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_9_cast_fp16 = transpose(perm = value_states_9_perm_0, x = var_1312_cast_fp16)[name = string("transpose_681")]; tensor value_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = value_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = value_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_2_squeeze_mask_0, stride = value_cache_internal_tensor_assign_2_stride_0, update = value_states_9_cast_fp16, x = coreml_update_state_393)[name = string("value_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_2_cast_fp16, input = value_cache)[name = string("coreml_update_state_395_write_state")]; tensor coreml_update_state_395 = read_state(input = value_cache)[name = string("coreml_update_state_395")]; tensor var_1406_begin_0 = const()[name = string("op_1406_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1406_end_0 = const()[name = string("op_1406_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1406_end_mask_0 = const()[name = string("op_1406_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1406_cast_fp16 = slice_by_index(begin = var_1406_begin_0, end = var_1406_end_0, end_mask = var_1406_end_mask_0, x = coreml_update_state_394)[name = string("op_1406_cast_fp16")]; tensor tile_2 = const()[name = string("tile_2"), val = tensor([1, 1])]; int32 var_1409_axis_0 = const()[name = string("op_1409_axis_0"), val = int32(1)]; tensor var_1409_cast_fp16_0, tensor var_1409_cast_fp16_1 = split(axis = var_1409_axis_0, split_sizes = tile_2, x = var_1406_cast_fp16)[name = string("op_1409_cast_fp16")]; tensor var_1416_begin_0 = const()[name = string("op_1416_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1416_end_0 = const()[name = string("op_1416_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1416_end_mask_0 = const()[name = string("op_1416_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1416_cast_fp16 = slice_by_index(begin = var_1416_begin_0, end = var_1416_end_0, end_mask = var_1416_end_mask_0, x = coreml_update_state_395)[name = string("op_1416_cast_fp16")]; tensor tile_3 = const()[name = string("tile_3"), val = tensor([1, 1])]; int32 var_1419_axis_0 = const()[name = string("op_1419_axis_0"), val = int32(1)]; tensor var_1419_cast_fp16_0, tensor var_1419_cast_fp16_1 = split(axis = var_1419_axis_0, split_sizes = tile_3, x = var_1416_cast_fp16)[name = string("op_1419_cast_fp16")]; tensor var_1422_split_sizes_0 = const()[name = string("op_1422_split_sizes_0"), val = tensor([8, 8])]; int32 var_1422_axis_0 = const()[name = string("op_1422_axis_0"), val = int32(1)]; tensor var_1422_0, tensor var_1422_1 = split(axis = var_1422_axis_0, split_sizes = var_1422_split_sizes_0, x = query_states_9_cast_fp16)[name = string("op_1422")]; bool attn_weights_17_transpose_x_0 = const()[name = string("attn_weights_17_transpose_x_0"), val = bool(false)]; bool attn_weights_17_transpose_y_0 = const()[name = string("attn_weights_17_transpose_y_0"), val = bool(false)]; tensor attn_weights_17_cast_fp16 = matmul(transpose_x = attn_weights_17_transpose_x_0, transpose_y = attn_weights_17_transpose_y_0, x = var_1409_cast_fp16_0, y = var_1422_0)[name = string("attn_weights_17_cast_fp16")]; fp16 var_1425_to_fp16 = const()[name = string("op_1425_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_19_cast_fp16 = mul(x = attn_weights_17_cast_fp16, y = var_1425_to_fp16)[name = string("attn_weights_19_cast_fp16")]; tensor attn_weights_21_cast_fp16 = add(x = attn_weights_19_cast_fp16, y = attn_mask_1)[name = string("attn_weights_21_cast_fp16")]; int32 var_1429 = const()[name = string("op_1429"), val = int32(-2)]; tensor attn_weights_23_cast_fp16 = softmax(axis = var_1429, x = attn_weights_21_cast_fp16)[name = string("attn_weights_23_cast_fp16")]; bool var_1435_transpose_x_1 = const()[name = string("op_1435_transpose_x_1"), val = bool(true)]; bool var_1435_transpose_y_1 = const()[name = string("op_1435_transpose_y_1"), val = bool(false)]; tensor var_1435_cast_fp16 = matmul(transpose_x = var_1435_transpose_x_1, transpose_y = var_1435_transpose_y_1, x = attn_weights_23_cast_fp16, y = var_1419_cast_fp16_0)[name = string("op_1435_cast_fp16")]; bool attn_weights_25_transpose_x_0 = const()[name = string("attn_weights_25_transpose_x_0"), val = bool(false)]; bool attn_weights_25_transpose_y_0 = const()[name = string("attn_weights_25_transpose_y_0"), val = bool(false)]; tensor attn_weights_25_cast_fp16 = matmul(transpose_x = attn_weights_25_transpose_x_0, transpose_y = attn_weights_25_transpose_y_0, x = var_1409_cast_fp16_1, y = var_1422_1)[name = string("attn_weights_25_cast_fp16")]; fp16 var_1437_to_fp16 = const()[name = string("op_1437_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_27_cast_fp16 = mul(x = attn_weights_25_cast_fp16, y = var_1437_to_fp16)[name = string("attn_weights_27_cast_fp16")]; tensor attn_weights_29_cast_fp16 = add(x = attn_weights_27_cast_fp16, y = attn_mask_1)[name = string("attn_weights_29_cast_fp16")]; int32 var_1441 = const()[name = string("op_1441"), val = int32(-2)]; tensor attn_weights_31_cast_fp16 = softmax(axis = var_1441, x = attn_weights_29_cast_fp16)[name = string("attn_weights_31_cast_fp16")]; bool attn_output_9_transpose_x_1 = const()[name = string("attn_output_9_transpose_x_1"), val = bool(true)]; bool attn_output_9_transpose_y_1 = const()[name = string("attn_output_9_transpose_y_1"), val = bool(false)]; tensor attn_output_9_cast_fp16 = matmul(transpose_x = attn_output_9_transpose_x_1, transpose_y = attn_output_9_transpose_y_1, x = attn_weights_31_cast_fp16, y = var_1419_cast_fp16_1)[name = string("attn_output_9_cast_fp16")]; int32 var_1449 = const()[name = string("op_1449"), val = int32(1)]; bool attn_output_11_interleave_0 = const()[name = string("attn_output_11_interleave_0"), val = bool(false)]; tensor attn_output_11_cast_fp16 = concat(axis = var_1449, interleave = attn_output_11_interleave_0, values = (var_1435_cast_fp16, attn_output_9_cast_fp16))[name = string("attn_output_11_cast_fp16")]; tensor var_1453_perm_0 = const()[name = string("op_1453_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_23x = const()[name = string("concat_23x"), val = tensor([1, 2048, 1, -1])]; tensor var_1453_cast_fp16 = transpose(perm = var_1453_perm_0, x = attn_output_11_cast_fp16)[name = string("transpose_680")]; tensor attn_output_15_cast_fp16 = reshape(shape = concat_23x, x = var_1453_cast_fp16)[name = string("attn_output_15_cast_fp16")]; tensor layers_1_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086995136)))]; tensor hidden_states_13_strides_0 = const()[name = string("hidden_states_13_strides_0"), val = tensor([1, 1])]; string hidden_states_13_pad_type_0 = const()[name = string("hidden_states_13_pad_type_0"), val = string("valid")]; tensor hidden_states_13_pad_0 = const()[name = string("hidden_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_13_dilations_0 = const()[name = string("hidden_states_13_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_13_groups_0 = const()[name = string("hidden_states_13_groups_0"), val = int32(1)]; tensor hidden_states_13_cast_fp16 = conv(dilations = hidden_states_13_dilations_0, groups = hidden_states_13_groups_0, pad = hidden_states_13_pad_0, pad_type = hidden_states_13_pad_type_0, strides = hidden_states_13_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16, x = attn_output_15_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; tensor hidden_states_15_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = hidden_states_13_cast_fp16)[name = string("hidden_states_15_cast_fp16")]; fp16 const_18_promoted_to_fp16 = const()[name = string("const_18_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1486_cast_fp16 = mul(x = hidden_states_15_cast_fp16, y = const_18_promoted_to_fp16)[name = string("op_1486_cast_fp16")]; int32 var_1484 = const()[name = string("op_1484"), val = int32(1)]; bool doubled_13_interleave_0 = const()[name = string("doubled_13_interleave_0"), val = bool(false)]; tensor doubled_13_cast_fp16 = concat(axis = var_1484, interleave = doubled_13_interleave_0, values = (hidden_states_15_cast_fp16, var_1486_cast_fp16))[name = string("doubled_13_cast_fp16")]; tensor out_7_axes_0 = const()[name = string("out_7_axes_0"), val = tensor([1])]; tensor out_7_gamma_0_to_fp16 = const()[name = string("out_7_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095383808)))]; fp16 var_1496_to_fp16 = const()[name = string("op_1496_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_7_cast_fp16 = layer_norm(axes = out_7_axes_0, epsilon = var_1496_to_fp16, gamma = out_7_gamma_0_to_fp16, x = doubled_13_cast_fp16)[name = string("out_7_cast_fp16")]; tensor var_1507_split_sizes_0 = const()[name = string("op_1507_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1507_axis_0 = const()[name = string("op_1507_axis_0"), val = int32(1)]; tensor var_1507_cast_fp16_0, tensor var_1507_cast_fp16_1 = split(axis = var_1507_axis_0, split_sizes = var_1507_split_sizes_0, x = out_7_cast_fp16)[name = string("op_1507_cast_fp16")]; tensor layers_1_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095392064)))]; tensor input_3_strides_0 = const()[name = string("input_3_strides_0"), val = tensor([1, 1])]; string input_3_pad_type_0 = const()[name = string("input_3_pad_type_0"), val = string("valid")]; tensor input_3_pad_0 = const()[name = string("input_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_3_dilations_0 = const()[name = string("input_3_dilations_0"), val = tensor([1, 1])]; int32 input_3_groups_0 = const()[name = string("input_3_groups_0"), val = int32(1)]; tensor input_3_cast_fp16 = conv(dilations = input_3_dilations_0, groups = input_3_groups_0, pad = input_3_pad_0, pad_type = input_3_pad_type_0, strides = input_3_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16, x = var_1507_cast_fp16_0)[name = string("input_3_cast_fp16")]; tensor var_1524_cast_fp16 = silu(x = input_3_cast_fp16)[name = string("op_1524_cast_fp16")]; tensor var_1530_strides_0 = const()[name = string("op_1530_strides_0"), val = tensor([1, 1])]; string var_1530_pad_type_0 = const()[name = string("op_1530_pad_type_0"), val = string("valid")]; tensor var_1530_pad_0 = const()[name = string("op_1530_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1530_dilations_0 = const()[name = string("op_1530_dilations_0"), val = tensor([1, 1])]; int32 var_1530_groups_0 = const()[name = string("op_1530_groups_0"), val = int32(1)]; tensor var_1530_cast_fp16 = conv(dilations = var_1530_dilations_0, groups = var_1530_groups_0, pad = var_1530_pad_0, pad_type = var_1530_pad_type_0, strides = var_1530_strides_0, weight = layers_1_mlp_up_proj_weight_cast_fp16, x = var_1507_cast_fp16_0)[name = string("op_1530_cast_fp16")]; tensor x_19_cast_fp16 = mul(x = var_1524_cast_fp16, y = var_1530_cast_fp16)[name = string("x_19_cast_fp16")]; tensor layers_1_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1120557952)))]; tensor hidden_states_17_strides_0 = const()[name = string("hidden_states_17_strides_0"), val = tensor([1, 1])]; string hidden_states_17_pad_type_0 = const()[name = string("hidden_states_17_pad_type_0"), val = string("valid")]; tensor hidden_states_17_pad_0 = const()[name = string("hidden_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_17_dilations_0 = const()[name = string("hidden_states_17_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_17_groups_0 = const()[name = string("hidden_states_17_groups_0"), val = int32(1)]; tensor hidden_states_17_cast_fp16 = conv(dilations = hidden_states_17_dilations_0, groups = hidden_states_17_groups_0, pad = hidden_states_17_pad_0, pad_type = hidden_states_17_pad_type_0, strides = hidden_states_17_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16, x = x_19_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; tensor hidden_states_19_cast_fp16 = add(x = hidden_states_15_cast_fp16, y = hidden_states_17_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1548_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_1548_cast_fp16")]; int32 var_1546 = const()[name = string("op_1546"), val = int32(1)]; bool doubled_17_interleave_0 = const()[name = string("doubled_17_interleave_0"), val = bool(false)]; tensor doubled_17_cast_fp16 = concat(axis = var_1546, interleave = doubled_17_interleave_0, values = (hidden_states_19_cast_fp16, var_1548_cast_fp16))[name = string("doubled_17_cast_fp16")]; tensor out_9_axes_0 = const()[name = string("out_9_axes_0"), val = tensor([1])]; tensor out_9_gamma_0_to_fp16 = const()[name = string("out_9_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145723840)))]; fp16 var_1558_to_fp16 = const()[name = string("op_1558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_9_cast_fp16 = layer_norm(axes = out_9_axes_0, epsilon = var_1558_to_fp16, gamma = out_9_gamma_0_to_fp16, x = doubled_17_cast_fp16)[name = string("out_9_cast_fp16")]; tensor var_1569_split_sizes_0 = const()[name = string("op_1569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1569_axis_0 = const()[name = string("op_1569_axis_0"), val = int32(1)]; tensor var_1569_cast_fp16_0, tensor var_1569_cast_fp16_1 = split(axis = var_1569_axis_0, split_sizes = var_1569_split_sizes_0, x = out_9_cast_fp16)[name = string("op_1569_cast_fp16")]; tensor layers_2_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145732096)))]; tensor query_states_13_strides_0 = const()[name = string("query_states_13_strides_0"), val = tensor([1, 1])]; string query_states_13_pad_type_0 = const()[name = string("query_states_13_pad_type_0"), val = string("valid")]; tensor query_states_13_pad_0 = const()[name = string("query_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_13_dilations_0 = const()[name = string("query_states_13_dilations_0"), val = tensor([1, 1])]; int32 query_states_13_groups_0 = const()[name = string("query_states_13_groups_0"), val = int32(1)]; tensor query_states_13_cast_fp16 = conv(dilations = query_states_13_dilations_0, groups = query_states_13_groups_0, pad = query_states_13_pad_0, pad_type = query_states_13_pad_type_0, strides = query_states_13_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("query_states_13_cast_fp16")]; tensor layers_2_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1154120768)))]; tensor key_states_21_strides_0 = const()[name = string("key_states_21_strides_0"), val = tensor([1, 1])]; string key_states_21_pad_type_0 = const()[name = string("key_states_21_pad_type_0"), val = string("valid")]; tensor key_states_21_pad_0 = const()[name = string("key_states_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_21_dilations_0 = const()[name = string("key_states_21_dilations_0"), val = tensor([1, 1])]; int32 key_states_21_groups_0 = const()[name = string("key_states_21_groups_0"), val = int32(1)]; tensor key_states_21_cast_fp16 = conv(dilations = key_states_21_dilations_0, groups = key_states_21_groups_0, pad = key_states_21_pad_0, pad_type = key_states_21_pad_type_0, strides = key_states_21_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("key_states_21_cast_fp16")]; tensor value_states_13_strides_0 = const()[name = string("value_states_13_strides_0"), val = tensor([1, 1])]; string value_states_13_pad_type_0 = const()[name = string("value_states_13_pad_type_0"), val = string("valid")]; tensor value_states_13_pad_0 = const()[name = string("value_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_13_dilations_0 = const()[name = string("value_states_13_dilations_0"), val = tensor([1, 1])]; int32 value_states_13_groups_0 = const()[name = string("value_states_13_groups_0"), val = int32(1)]; tensor value_states_13_cast_fp16 = conv(dilations = value_states_13_dilations_0, groups = value_states_13_groups_0, pad = value_states_13_pad_0, pad_type = value_states_13_pad_type_0, strides = value_states_13_strides_0, weight = layers_2_self_attn_v_proj_weight_cast_fp16, x = var_1569_cast_fp16_0)[name = string("value_states_13_cast_fp16")]; tensor concat_24x = const()[name = string("concat_24x"), val = tensor([1, 16, 128, -1])]; tensor x_21_cast_fp16 = reshape(shape = concat_24x, x = query_states_13_cast_fp16)[name = string("x_21_cast_fp16")]; tensor concat_25x = const()[name = string("concat_25x"), val = tensor([1, 2, 128, -1])]; tensor var_1626_cast_fp16 = reshape(shape = concat_25x, x = key_states_21_cast_fp16)[name = string("op_1626_cast_fp16")]; tensor concat_26x = const()[name = string("concat_26x"), val = tensor([1, 2, 128, -1])]; tensor var_1633_cast_fp16 = reshape(shape = concat_26x, x = value_states_13_cast_fp16)[name = string("op_1633_cast_fp16")]; tensor var_1637_cast_fp16 = mul(x = x_21_cast_fp16, y = var_869_cast_fp16)[name = string("op_1637_cast_fp16")]; tensor var_1638_split_sizes_0 = const()[name = string("op_1638_split_sizes_0"), val = tensor([64, 64])]; int32 var_1638_axis_0 = const()[name = string("op_1638_axis_0"), val = int32(-2)]; tensor var_1638_cast_fp16_0, tensor var_1638_cast_fp16_1 = split(axis = var_1638_axis_0, split_sizes = var_1638_split_sizes_0, x = x_21_cast_fp16)[name = string("op_1638_cast_fp16")]; fp16 const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1640_cast_fp16 = mul(x = var_1638_cast_fp16_1, y = const_22_promoted_to_fp16)[name = string("op_1640_cast_fp16")]; int32 var_1642 = const()[name = string("op_1642"), val = int32(-2)]; bool var_1643_interleave_0 = const()[name = string("op_1643_interleave_0"), val = bool(false)]; tensor var_1643_cast_fp16 = concat(axis = var_1642, interleave = var_1643_interleave_0, values = (var_1640_cast_fp16, var_1638_cast_fp16_0))[name = string("op_1643_cast_fp16")]; tensor var_1644_cast_fp16 = mul(x = var_1643_cast_fp16, y = var_878_cast_fp16)[name = string("op_1644_cast_fp16")]; tensor query_states_15_cast_fp16 = add(x = var_1637_cast_fp16, y = var_1644_cast_fp16)[name = string("query_states_15_cast_fp16")]; tensor var_1650_cast_fp16 = mul(x = var_1626_cast_fp16, y = var_869_cast_fp16)[name = string("op_1650_cast_fp16")]; tensor var_1651_split_sizes_0 = const()[name = string("op_1651_split_sizes_0"), val = tensor([64, 64])]; int32 var_1651_axis_0 = const()[name = string("op_1651_axis_0"), val = int32(-2)]; tensor var_1651_cast_fp16_0, tensor var_1651_cast_fp16_1 = split(axis = var_1651_axis_0, split_sizes = var_1651_split_sizes_0, x = var_1626_cast_fp16)[name = string("op_1651_cast_fp16")]; fp16 const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1653_cast_fp16 = mul(x = var_1651_cast_fp16_1, y = const_23_promoted_to_fp16)[name = string("op_1653_cast_fp16")]; int32 var_1655 = const()[name = string("op_1655"), val = int32(-2)]; bool var_1656_interleave_0 = const()[name = string("op_1656_interleave_0"), val = bool(false)]; tensor var_1656_cast_fp16 = concat(axis = var_1655, interleave = var_1656_interleave_0, values = (var_1653_cast_fp16, var_1651_cast_fp16_0))[name = string("op_1656_cast_fp16")]; tensor var_1657_cast_fp16 = mul(x = var_1656_cast_fp16, y = var_878_cast_fp16)[name = string("op_1657_cast_fp16")]; tensor key_states_25_cast_fp16 = add(x = var_1650_cast_fp16, y = var_1657_cast_fp16)[name = string("key_states_25_cast_fp16")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; int32 concat_29_axis_0 = const()[name = string("concat_29_axis_0"), val = int32(0)]; bool concat_29_interleave_0 = const()[name = string("concat_29_interleave_0"), val = bool(false)]; tensor concat_29 = concat(axis = concat_29_axis_0, interleave = concat_29_interleave_0, values = (expand_dims_24, expand_dims_25, position_id, expand_dims_27))[name = string("concat_29")]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; tensor concat_30_values1_0 = const()[name = string("concat_30_values1_0"), val = tensor([0])]; tensor concat_30_values3_0 = const()[name = string("concat_30_values3_0"), val = tensor([0])]; int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_28, concat_30_values1_0, cache_position_end, concat_30_values3_0))[name = string("concat_30")]; tensor key_states_27_perm_0 = const()[name = string("key_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_3_stride_0 = const()[name = string("key_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_27_cast_fp16 = transpose(perm = key_states_27_perm_0, x = key_states_25_cast_fp16)[name = string("transpose_679")]; tensor key_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = key_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = key_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_3_squeeze_mask_0, stride = key_cache_internal_tensor_assign_3_stride_0, update = key_states_27_cast_fp16, x = coreml_update_state_394)[name = string("key_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_3_cast_fp16, input = key_cache)[name = string("coreml_update_state_396_write_state")]; tensor coreml_update_state_396 = read_state(input = key_cache)[name = string("coreml_update_state_396")]; tensor value_states_15_perm_0 = const()[name = string("value_states_15_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_3_stride_0 = const()[name = string("value_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_15_cast_fp16 = transpose(perm = value_states_15_perm_0, x = var_1633_cast_fp16)[name = string("transpose_678")]; tensor value_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = value_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = value_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_3_squeeze_mask_0, stride = value_cache_internal_tensor_assign_3_stride_0, update = value_states_15_cast_fp16, x = coreml_update_state_395)[name = string("value_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_3_cast_fp16, input = value_cache)[name = string("coreml_update_state_397_write_state")]; tensor coreml_update_state_397 = read_state(input = value_cache)[name = string("coreml_update_state_397")]; tensor var_1727_begin_0 = const()[name = string("op_1727_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1727_end_0 = const()[name = string("op_1727_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1727_end_mask_0 = const()[name = string("op_1727_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1727_cast_fp16 = slice_by_index(begin = var_1727_begin_0, end = var_1727_end_0, end_mask = var_1727_end_mask_0, x = coreml_update_state_396)[name = string("op_1727_cast_fp16")]; tensor tile_4 = const()[name = string("tile_4"), val = tensor([1, 1])]; int32 var_1730_axis_0 = const()[name = string("op_1730_axis_0"), val = int32(1)]; tensor var_1730_cast_fp16_0, tensor var_1730_cast_fp16_1 = split(axis = var_1730_axis_0, split_sizes = tile_4, x = var_1727_cast_fp16)[name = string("op_1730_cast_fp16")]; tensor var_1737_begin_0 = const()[name = string("op_1737_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1737_end_0 = const()[name = string("op_1737_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1737_end_mask_0 = const()[name = string("op_1737_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1737_cast_fp16 = slice_by_index(begin = var_1737_begin_0, end = var_1737_end_0, end_mask = var_1737_end_mask_0, x = coreml_update_state_397)[name = string("op_1737_cast_fp16")]; tensor tile_5 = const()[name = string("tile_5"), val = tensor([1, 1])]; int32 var_1740_axis_0 = const()[name = string("op_1740_axis_0"), val = int32(1)]; tensor var_1740_cast_fp16_0, tensor var_1740_cast_fp16_1 = split(axis = var_1740_axis_0, split_sizes = tile_5, x = var_1737_cast_fp16)[name = string("op_1740_cast_fp16")]; tensor var_1743_split_sizes_0 = const()[name = string("op_1743_split_sizes_0"), val = tensor([8, 8])]; int32 var_1743_axis_0 = const()[name = string("op_1743_axis_0"), val = int32(1)]; tensor var_1743_0, tensor var_1743_1 = split(axis = var_1743_axis_0, split_sizes = var_1743_split_sizes_0, x = query_states_15_cast_fp16)[name = string("op_1743")]; bool attn_weights_33_transpose_x_0 = const()[name = string("attn_weights_33_transpose_x_0"), val = bool(false)]; bool attn_weights_33_transpose_y_0 = const()[name = string("attn_weights_33_transpose_y_0"), val = bool(false)]; tensor attn_weights_33_cast_fp16 = matmul(transpose_x = attn_weights_33_transpose_x_0, transpose_y = attn_weights_33_transpose_y_0, x = var_1730_cast_fp16_0, y = var_1743_0)[name = string("attn_weights_33_cast_fp16")]; fp16 var_1746_to_fp16 = const()[name = string("op_1746_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_35_cast_fp16 = mul(x = attn_weights_33_cast_fp16, y = var_1746_to_fp16)[name = string("attn_weights_35_cast_fp16")]; tensor attn_weights_37_cast_fp16 = add(x = attn_weights_35_cast_fp16, y = attn_mask_1)[name = string("attn_weights_37_cast_fp16")]; int32 var_1750 = const()[name = string("op_1750"), val = int32(-2)]; tensor attn_weights_39_cast_fp16 = softmax(axis = var_1750, x = attn_weights_37_cast_fp16)[name = string("attn_weights_39_cast_fp16")]; bool var_1756_transpose_x_1 = const()[name = string("op_1756_transpose_x_1"), val = bool(true)]; bool var_1756_transpose_y_1 = const()[name = string("op_1756_transpose_y_1"), val = bool(false)]; tensor var_1756_cast_fp16 = matmul(transpose_x = var_1756_transpose_x_1, transpose_y = var_1756_transpose_y_1, x = attn_weights_39_cast_fp16, y = var_1740_cast_fp16_0)[name = string("op_1756_cast_fp16")]; bool attn_weights_41_transpose_x_0 = const()[name = string("attn_weights_41_transpose_x_0"), val = bool(false)]; bool attn_weights_41_transpose_y_0 = const()[name = string("attn_weights_41_transpose_y_0"), val = bool(false)]; tensor attn_weights_41_cast_fp16 = matmul(transpose_x = attn_weights_41_transpose_x_0, transpose_y = attn_weights_41_transpose_y_0, x = var_1730_cast_fp16_1, y = var_1743_1)[name = string("attn_weights_41_cast_fp16")]; fp16 var_1758_to_fp16 = const()[name = string("op_1758_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_43_cast_fp16 = mul(x = attn_weights_41_cast_fp16, y = var_1758_to_fp16)[name = string("attn_weights_43_cast_fp16")]; tensor attn_weights_45_cast_fp16 = add(x = attn_weights_43_cast_fp16, y = attn_mask_1)[name = string("attn_weights_45_cast_fp16")]; int32 var_1762 = const()[name = string("op_1762"), val = int32(-2)]; tensor attn_weights_47_cast_fp16 = softmax(axis = var_1762, x = attn_weights_45_cast_fp16)[name = string("attn_weights_47_cast_fp16")]; bool attn_output_17_transpose_x_1 = const()[name = string("attn_output_17_transpose_x_1"), val = bool(true)]; bool attn_output_17_transpose_y_1 = const()[name = string("attn_output_17_transpose_y_1"), val = bool(false)]; tensor attn_output_17_cast_fp16 = matmul(transpose_x = attn_output_17_transpose_x_1, transpose_y = attn_output_17_transpose_y_1, x = attn_weights_47_cast_fp16, y = var_1740_cast_fp16_1)[name = string("attn_output_17_cast_fp16")]; int32 var_1770 = const()[name = string("op_1770"), val = int32(1)]; bool attn_output_19_interleave_0 = const()[name = string("attn_output_19_interleave_0"), val = bool(false)]; tensor attn_output_19_cast_fp16 = concat(axis = var_1770, interleave = attn_output_19_interleave_0, values = (var_1756_cast_fp16, attn_output_17_cast_fp16))[name = string("attn_output_19_cast_fp16")]; tensor var_1774_perm_0 = const()[name = string("op_1774_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_35x = const()[name = string("concat_35x"), val = tensor([1, 2048, 1, -1])]; tensor var_1774_cast_fp16 = transpose(perm = var_1774_perm_0, x = attn_output_19_cast_fp16)[name = string("transpose_677")]; tensor attn_output_23_cast_fp16 = reshape(shape = concat_35x, x = var_1774_cast_fp16)[name = string("attn_output_23_cast_fp16")]; tensor layers_2_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1155169408)))]; tensor hidden_states_23_strides_0 = const()[name = string("hidden_states_23_strides_0"), val = tensor([1, 1])]; string hidden_states_23_pad_type_0 = const()[name = string("hidden_states_23_pad_type_0"), val = string("valid")]; tensor hidden_states_23_pad_0 = const()[name = string("hidden_states_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_23_dilations_0 = const()[name = string("hidden_states_23_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_23_groups_0 = const()[name = string("hidden_states_23_groups_0"), val = int32(1)]; tensor hidden_states_23_cast_fp16 = conv(dilations = hidden_states_23_dilations_0, groups = hidden_states_23_groups_0, pad = hidden_states_23_pad_0, pad_type = hidden_states_23_pad_type_0, strides = hidden_states_23_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16, x = attn_output_23_cast_fp16)[name = string("hidden_states_23_cast_fp16")]; tensor hidden_states_25_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = hidden_states_23_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1807_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_1807_cast_fp16")]; int32 var_1805 = const()[name = string("op_1805"), val = int32(1)]; bool doubled_21_interleave_0 = const()[name = string("doubled_21_interleave_0"), val = bool(false)]; tensor doubled_21_cast_fp16 = concat(axis = var_1805, interleave = doubled_21_interleave_0, values = (hidden_states_25_cast_fp16, var_1807_cast_fp16))[name = string("doubled_21_cast_fp16")]; tensor out_11_axes_0 = const()[name = string("out_11_axes_0"), val = tensor([1])]; tensor out_11_gamma_0_to_fp16 = const()[name = string("out_11_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163558080)))]; fp16 var_1817_to_fp16 = const()[name = string("op_1817_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_11_cast_fp16 = layer_norm(axes = out_11_axes_0, epsilon = var_1817_to_fp16, gamma = out_11_gamma_0_to_fp16, x = doubled_21_cast_fp16)[name = string("out_11_cast_fp16")]; tensor var_1828_split_sizes_0 = const()[name = string("op_1828_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1828_axis_0 = const()[name = string("op_1828_axis_0"), val = int32(1)]; tensor var_1828_cast_fp16_0, tensor var_1828_cast_fp16_1 = split(axis = var_1828_axis_0, split_sizes = var_1828_split_sizes_0, x = out_11_cast_fp16)[name = string("op_1828_cast_fp16")]; tensor layers_2_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163566336)))]; tensor input_5_strides_0 = const()[name = string("input_5_strides_0"), val = tensor([1, 1])]; string input_5_pad_type_0 = const()[name = string("input_5_pad_type_0"), val = string("valid")]; tensor input_5_pad_0 = const()[name = string("input_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_5_dilations_0 = const()[name = string("input_5_dilations_0"), val = tensor([1, 1])]; int32 input_5_groups_0 = const()[name = string("input_5_groups_0"), val = int32(1)]; tensor input_5_cast_fp16 = conv(dilations = input_5_dilations_0, groups = input_5_groups_0, pad = input_5_pad_0, pad_type = input_5_pad_type_0, strides = input_5_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16, x = var_1828_cast_fp16_0)[name = string("input_5_cast_fp16")]; tensor var_1845_cast_fp16 = silu(x = input_5_cast_fp16)[name = string("op_1845_cast_fp16")]; tensor var_1851_strides_0 = const()[name = string("op_1851_strides_0"), val = tensor([1, 1])]; string var_1851_pad_type_0 = const()[name = string("op_1851_pad_type_0"), val = string("valid")]; tensor var_1851_pad_0 = const()[name = string("op_1851_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1851_dilations_0 = const()[name = string("op_1851_dilations_0"), val = tensor([1, 1])]; int32 var_1851_groups_0 = const()[name = string("op_1851_groups_0"), val = int32(1)]; tensor var_1851_cast_fp16 = conv(dilations = var_1851_dilations_0, groups = var_1851_groups_0, pad = var_1851_pad_0, pad_type = var_1851_pad_type_0, strides = var_1851_strides_0, weight = layers_2_mlp_up_proj_weight_cast_fp16, x = var_1828_cast_fp16_0)[name = string("op_1851_cast_fp16")]; tensor x_29_cast_fp16 = mul(x = var_1845_cast_fp16, y = var_1851_cast_fp16)[name = string("x_29_cast_fp16")]; tensor layers_2_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1188732224)))]; tensor hidden_states_27_strides_0 = const()[name = string("hidden_states_27_strides_0"), val = tensor([1, 1])]; string hidden_states_27_pad_type_0 = const()[name = string("hidden_states_27_pad_type_0"), val = string("valid")]; tensor hidden_states_27_pad_0 = const()[name = string("hidden_states_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_27_dilations_0 = const()[name = string("hidden_states_27_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_27_groups_0 = const()[name = string("hidden_states_27_groups_0"), val = int32(1)]; tensor hidden_states_27_cast_fp16 = conv(dilations = hidden_states_27_dilations_0, groups = hidden_states_27_groups_0, pad = hidden_states_27_pad_0, pad_type = hidden_states_27_pad_type_0, strides = hidden_states_27_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16, x = x_29_cast_fp16)[name = string("hidden_states_27_cast_fp16")]; tensor hidden_states_29_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = hidden_states_27_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; fp16 const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1869_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_30_promoted_to_fp16)[name = string("op_1869_cast_fp16")]; int32 var_1867 = const()[name = string("op_1867"), val = int32(1)]; bool doubled_25_interleave_0 = const()[name = string("doubled_25_interleave_0"), val = bool(false)]; tensor doubled_25_cast_fp16 = concat(axis = var_1867, interleave = doubled_25_interleave_0, values = (hidden_states_29_cast_fp16, var_1869_cast_fp16))[name = string("doubled_25_cast_fp16")]; tensor out_13_axes_0 = const()[name = string("out_13_axes_0"), val = tensor([1])]; tensor out_13_gamma_0_to_fp16 = const()[name = string("out_13_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213898112)))]; fp16 var_1879_to_fp16 = const()[name = string("op_1879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_13_cast_fp16 = layer_norm(axes = out_13_axes_0, epsilon = var_1879_to_fp16, gamma = out_13_gamma_0_to_fp16, x = doubled_25_cast_fp16)[name = string("out_13_cast_fp16")]; tensor var_1890_split_sizes_0 = const()[name = string("op_1890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1890_axis_0 = const()[name = string("op_1890_axis_0"), val = int32(1)]; tensor var_1890_cast_fp16_0, tensor var_1890_cast_fp16_1 = split(axis = var_1890_axis_0, split_sizes = var_1890_split_sizes_0, x = out_13_cast_fp16)[name = string("op_1890_cast_fp16")]; tensor layers_3_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213906368)))]; tensor query_states_19_strides_0 = const()[name = string("query_states_19_strides_0"), val = tensor([1, 1])]; string query_states_19_pad_type_0 = const()[name = string("query_states_19_pad_type_0"), val = string("valid")]; tensor query_states_19_pad_0 = const()[name = string("query_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_19_dilations_0 = const()[name = string("query_states_19_dilations_0"), val = tensor([1, 1])]; int32 query_states_19_groups_0 = const()[name = string("query_states_19_groups_0"), val = int32(1)]; tensor query_states_19_cast_fp16 = conv(dilations = query_states_19_dilations_0, groups = query_states_19_groups_0, pad = query_states_19_pad_0, pad_type = query_states_19_pad_type_0, strides = query_states_19_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("query_states_19_cast_fp16")]; tensor layers_3_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1222295040)))]; tensor key_states_31_strides_0 = const()[name = string("key_states_31_strides_0"), val = tensor([1, 1])]; string key_states_31_pad_type_0 = const()[name = string("key_states_31_pad_type_0"), val = string("valid")]; tensor key_states_31_pad_0 = const()[name = string("key_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_31_dilations_0 = const()[name = string("key_states_31_dilations_0"), val = tensor([1, 1])]; int32 key_states_31_groups_0 = const()[name = string("key_states_31_groups_0"), val = int32(1)]; tensor key_states_31_cast_fp16 = conv(dilations = key_states_31_dilations_0, groups = key_states_31_groups_0, pad = key_states_31_pad_0, pad_type = key_states_31_pad_type_0, strides = key_states_31_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("key_states_31_cast_fp16")]; tensor value_states_19_strides_0 = const()[name = string("value_states_19_strides_0"), val = tensor([1, 1])]; string value_states_19_pad_type_0 = const()[name = string("value_states_19_pad_type_0"), val = string("valid")]; tensor value_states_19_pad_0 = const()[name = string("value_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_19_dilations_0 = const()[name = string("value_states_19_dilations_0"), val = tensor([1, 1])]; int32 value_states_19_groups_0 = const()[name = string("value_states_19_groups_0"), val = int32(1)]; tensor value_states_19_cast_fp16 = conv(dilations = value_states_19_dilations_0, groups = value_states_19_groups_0, pad = value_states_19_pad_0, pad_type = value_states_19_pad_type_0, strides = value_states_19_strides_0, weight = layers_3_self_attn_v_proj_weight_cast_fp16, x = var_1890_cast_fp16_0)[name = string("value_states_19_cast_fp16")]; tensor concat_36x = const()[name = string("concat_36x"), val = tensor([1, 16, 128, -1])]; tensor x_31_cast_fp16 = reshape(shape = concat_36x, x = query_states_19_cast_fp16)[name = string("x_31_cast_fp16")]; tensor concat_37x = const()[name = string("concat_37x"), val = tensor([1, 2, 128, -1])]; tensor var_1947_cast_fp16 = reshape(shape = concat_37x, x = key_states_31_cast_fp16)[name = string("op_1947_cast_fp16")]; tensor concat_38x = const()[name = string("concat_38x"), val = tensor([1, 2, 128, -1])]; tensor var_1954_cast_fp16 = reshape(shape = concat_38x, x = value_states_19_cast_fp16)[name = string("op_1954_cast_fp16")]; tensor var_1958_cast_fp16 = mul(x = x_31_cast_fp16, y = var_869_cast_fp16)[name = string("op_1958_cast_fp16")]; tensor var_1959_split_sizes_0 = const()[name = string("op_1959_split_sizes_0"), val = tensor([64, 64])]; int32 var_1959_axis_0 = const()[name = string("op_1959_axis_0"), val = int32(-2)]; tensor var_1959_cast_fp16_0, tensor var_1959_cast_fp16_1 = split(axis = var_1959_axis_0, split_sizes = var_1959_split_sizes_0, x = x_31_cast_fp16)[name = string("op_1959_cast_fp16")]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1961_cast_fp16 = mul(x = var_1959_cast_fp16_1, y = const_32_promoted_to_fp16)[name = string("op_1961_cast_fp16")]; int32 var_1963 = const()[name = string("op_1963"), val = int32(-2)]; bool var_1964_interleave_0 = const()[name = string("op_1964_interleave_0"), val = bool(false)]; tensor var_1964_cast_fp16 = concat(axis = var_1963, interleave = var_1964_interleave_0, values = (var_1961_cast_fp16, var_1959_cast_fp16_0))[name = string("op_1964_cast_fp16")]; tensor var_1965_cast_fp16 = mul(x = var_1964_cast_fp16, y = var_878_cast_fp16)[name = string("op_1965_cast_fp16")]; tensor query_states_21_cast_fp16 = add(x = var_1958_cast_fp16, y = var_1965_cast_fp16)[name = string("query_states_21_cast_fp16")]; tensor var_1971_cast_fp16 = mul(x = var_1947_cast_fp16, y = var_869_cast_fp16)[name = string("op_1971_cast_fp16")]; tensor var_1972_split_sizes_0 = const()[name = string("op_1972_split_sizes_0"), val = tensor([64, 64])]; int32 var_1972_axis_0 = const()[name = string("op_1972_axis_0"), val = int32(-2)]; tensor var_1972_cast_fp16_0, tensor var_1972_cast_fp16_1 = split(axis = var_1972_axis_0, split_sizes = var_1972_split_sizes_0, x = var_1947_cast_fp16)[name = string("op_1972_cast_fp16")]; fp16 const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1974_cast_fp16 = mul(x = var_1972_cast_fp16_1, y = const_33_promoted_to_fp16)[name = string("op_1974_cast_fp16")]; int32 var_1976 = const()[name = string("op_1976"), val = int32(-2)]; bool var_1977_interleave_0 = const()[name = string("op_1977_interleave_0"), val = bool(false)]; tensor var_1977_cast_fp16 = concat(axis = var_1976, interleave = var_1977_interleave_0, values = (var_1974_cast_fp16, var_1972_cast_fp16_0))[name = string("op_1977_cast_fp16")]; tensor var_1978_cast_fp16 = mul(x = var_1977_cast_fp16, y = var_878_cast_fp16)[name = string("op_1978_cast_fp16")]; tensor key_states_35_cast_fp16 = add(x = var_1971_cast_fp16, y = var_1978_cast_fp16)[name = string("key_states_35_cast_fp16")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; int32 concat_41_axis_0 = const()[name = string("concat_41_axis_0"), val = int32(0)]; bool concat_41_interleave_0 = const()[name = string("concat_41_interleave_0"), val = bool(false)]; tensor concat_41 = concat(axis = concat_41_axis_0, interleave = concat_41_interleave_0, values = (expand_dims_36, expand_dims_37, position_id, expand_dims_39))[name = string("concat_41")]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; tensor concat_42_values1_0 = const()[name = string("concat_42_values1_0"), val = tensor([0])]; tensor concat_42_values3_0 = const()[name = string("concat_42_values3_0"), val = tensor([0])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_40, concat_42_values1_0, cache_position_end, concat_42_values3_0))[name = string("concat_42")]; tensor key_states_37_perm_0 = const()[name = string("key_states_37_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_4_stride_0 = const()[name = string("key_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_37_cast_fp16 = transpose(perm = key_states_37_perm_0, x = key_states_35_cast_fp16)[name = string("transpose_676")]; tensor key_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = key_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = key_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_4_squeeze_mask_0, stride = key_cache_internal_tensor_assign_4_stride_0, update = key_states_37_cast_fp16, x = coreml_update_state_396)[name = string("key_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_4_cast_fp16, input = key_cache)[name = string("coreml_update_state_398_write_state")]; tensor coreml_update_state_398 = read_state(input = key_cache)[name = string("coreml_update_state_398")]; tensor value_states_21_perm_0 = const()[name = string("value_states_21_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_4_stride_0 = const()[name = string("value_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_21_cast_fp16 = transpose(perm = value_states_21_perm_0, x = var_1954_cast_fp16)[name = string("transpose_675")]; tensor value_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = value_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = value_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_4_squeeze_mask_0, stride = value_cache_internal_tensor_assign_4_stride_0, update = value_states_21_cast_fp16, x = coreml_update_state_397)[name = string("value_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_4_cast_fp16, input = value_cache)[name = string("coreml_update_state_399_write_state")]; tensor coreml_update_state_399 = read_state(input = value_cache)[name = string("coreml_update_state_399")]; tensor var_2048_begin_0 = const()[name = string("op_2048_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2048_end_0 = const()[name = string("op_2048_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2048_end_mask_0 = const()[name = string("op_2048_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2048_cast_fp16 = slice_by_index(begin = var_2048_begin_0, end = var_2048_end_0, end_mask = var_2048_end_mask_0, x = coreml_update_state_398)[name = string("op_2048_cast_fp16")]; tensor tile_6 = const()[name = string("tile_6"), val = tensor([1, 1])]; int32 var_2051_axis_0 = const()[name = string("op_2051_axis_0"), val = int32(1)]; tensor var_2051_cast_fp16_0, tensor var_2051_cast_fp16_1 = split(axis = var_2051_axis_0, split_sizes = tile_6, x = var_2048_cast_fp16)[name = string("op_2051_cast_fp16")]; tensor var_2058_begin_0 = const()[name = string("op_2058_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2058_end_0 = const()[name = string("op_2058_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2058_end_mask_0 = const()[name = string("op_2058_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2058_cast_fp16 = slice_by_index(begin = var_2058_begin_0, end = var_2058_end_0, end_mask = var_2058_end_mask_0, x = coreml_update_state_399)[name = string("op_2058_cast_fp16")]; tensor tile_7 = const()[name = string("tile_7"), val = tensor([1, 1])]; int32 var_2061_axis_0 = const()[name = string("op_2061_axis_0"), val = int32(1)]; tensor var_2061_cast_fp16_0, tensor var_2061_cast_fp16_1 = split(axis = var_2061_axis_0, split_sizes = tile_7, x = var_2058_cast_fp16)[name = string("op_2061_cast_fp16")]; tensor var_2064_split_sizes_0 = const()[name = string("op_2064_split_sizes_0"), val = tensor([8, 8])]; int32 var_2064_axis_0 = const()[name = string("op_2064_axis_0"), val = int32(1)]; tensor var_2064_0, tensor var_2064_1 = split(axis = var_2064_axis_0, split_sizes = var_2064_split_sizes_0, x = query_states_21_cast_fp16)[name = string("op_2064")]; bool attn_weights_49_transpose_x_0 = const()[name = string("attn_weights_49_transpose_x_0"), val = bool(false)]; bool attn_weights_49_transpose_y_0 = const()[name = string("attn_weights_49_transpose_y_0"), val = bool(false)]; tensor attn_weights_49_cast_fp16 = matmul(transpose_x = attn_weights_49_transpose_x_0, transpose_y = attn_weights_49_transpose_y_0, x = var_2051_cast_fp16_0, y = var_2064_0)[name = string("attn_weights_49_cast_fp16")]; fp16 var_2067_to_fp16 = const()[name = string("op_2067_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_51_cast_fp16 = mul(x = attn_weights_49_cast_fp16, y = var_2067_to_fp16)[name = string("attn_weights_51_cast_fp16")]; tensor attn_weights_53_cast_fp16 = add(x = attn_weights_51_cast_fp16, y = attn_mask_1)[name = string("attn_weights_53_cast_fp16")]; int32 var_2071 = const()[name = string("op_2071"), val = int32(-2)]; tensor attn_weights_55_cast_fp16 = softmax(axis = var_2071, x = attn_weights_53_cast_fp16)[name = string("attn_weights_55_cast_fp16")]; bool var_2077_transpose_x_1 = const()[name = string("op_2077_transpose_x_1"), val = bool(true)]; bool var_2077_transpose_y_1 = const()[name = string("op_2077_transpose_y_1"), val = bool(false)]; tensor var_2077_cast_fp16 = matmul(transpose_x = var_2077_transpose_x_1, transpose_y = var_2077_transpose_y_1, x = attn_weights_55_cast_fp16, y = var_2061_cast_fp16_0)[name = string("op_2077_cast_fp16")]; bool attn_weights_57_transpose_x_0 = const()[name = string("attn_weights_57_transpose_x_0"), val = bool(false)]; bool attn_weights_57_transpose_y_0 = const()[name = string("attn_weights_57_transpose_y_0"), val = bool(false)]; tensor attn_weights_57_cast_fp16 = matmul(transpose_x = attn_weights_57_transpose_x_0, transpose_y = attn_weights_57_transpose_y_0, x = var_2051_cast_fp16_1, y = var_2064_1)[name = string("attn_weights_57_cast_fp16")]; fp16 var_2079_to_fp16 = const()[name = string("op_2079_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_59_cast_fp16 = mul(x = attn_weights_57_cast_fp16, y = var_2079_to_fp16)[name = string("attn_weights_59_cast_fp16")]; tensor attn_weights_61_cast_fp16 = add(x = attn_weights_59_cast_fp16, y = attn_mask_1)[name = string("attn_weights_61_cast_fp16")]; int32 var_2083 = const()[name = string("op_2083"), val = int32(-2)]; tensor attn_weights_63_cast_fp16 = softmax(axis = var_2083, x = attn_weights_61_cast_fp16)[name = string("attn_weights_63_cast_fp16")]; bool attn_output_25_transpose_x_1 = const()[name = string("attn_output_25_transpose_x_1"), val = bool(true)]; bool attn_output_25_transpose_y_1 = const()[name = string("attn_output_25_transpose_y_1"), val = bool(false)]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_1, transpose_y = attn_output_25_transpose_y_1, x = attn_weights_63_cast_fp16, y = var_2061_cast_fp16_1)[name = string("attn_output_25_cast_fp16")]; int32 var_2091 = const()[name = string("op_2091"), val = int32(1)]; bool attn_output_27_interleave_0 = const()[name = string("attn_output_27_interleave_0"), val = bool(false)]; tensor attn_output_27_cast_fp16 = concat(axis = var_2091, interleave = attn_output_27_interleave_0, values = (var_2077_cast_fp16, attn_output_25_cast_fp16))[name = string("attn_output_27_cast_fp16")]; tensor var_2095_perm_0 = const()[name = string("op_2095_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_47x = const()[name = string("concat_47x"), val = tensor([1, 2048, 1, -1])]; tensor var_2095_cast_fp16 = transpose(perm = var_2095_perm_0, x = attn_output_27_cast_fp16)[name = string("transpose_674")]; tensor attn_output_31_cast_fp16 = reshape(shape = concat_47x, x = var_2095_cast_fp16)[name = string("attn_output_31_cast_fp16")]; tensor hidden_states_33_strides_0 = const()[name = string("hidden_states_33_strides_0"), val = tensor([1, 1])]; string hidden_states_33_pad_type_0 = const()[name = string("hidden_states_33_pad_type_0"), val = string("valid")]; tensor hidden_states_33_pad_0 = const()[name = string("hidden_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_33_dilations_0 = const()[name = string("hidden_states_33_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_33_groups_0 = const()[name = string("hidden_states_33_groups_0"), val = int32(1)]; tensor hidden_states_33_cast_fp16 = conv(dilations = hidden_states_33_dilations_0, groups = hidden_states_33_groups_0, pad = hidden_states_33_pad_0, pad_type = hidden_states_33_pad_type_0, strides = hidden_states_33_strides_0, weight = layers_3_self_attn_o_proj_weight_cast_fp16, x = attn_output_31_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; tensor hidden_states_35_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = hidden_states_33_cast_fp16)[name = string("hidden_states_35_cast_fp16")]; fp16 const_38_promoted_to_fp16 = const()[name = string("const_38_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2128_cast_fp16 = mul(x = hidden_states_35_cast_fp16, y = const_38_promoted_to_fp16)[name = string("op_2128_cast_fp16")]; int32 var_2126 = const()[name = string("op_2126"), val = int32(1)]; bool doubled_29_interleave_0 = const()[name = string("doubled_29_interleave_0"), val = bool(false)]; tensor doubled_29_cast_fp16 = concat(axis = var_2126, interleave = doubled_29_interleave_0, values = (hidden_states_35_cast_fp16, var_2128_cast_fp16))[name = string("doubled_29_cast_fp16")]; tensor out_15_axes_0 = const()[name = string("out_15_axes_0"), val = tensor([1])]; tensor out_15_gamma_0_to_fp16 = const()[name = string("out_15_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223343680)))]; fp16 var_2138_to_fp16 = const()[name = string("op_2138_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_15_cast_fp16 = layer_norm(axes = out_15_axes_0, epsilon = var_2138_to_fp16, gamma = out_15_gamma_0_to_fp16, x = doubled_29_cast_fp16)[name = string("out_15_cast_fp16")]; tensor var_2149_split_sizes_0 = const()[name = string("op_2149_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2149_axis_0 = const()[name = string("op_2149_axis_0"), val = int32(1)]; tensor var_2149_cast_fp16_0, tensor var_2149_cast_fp16_1 = split(axis = var_2149_axis_0, split_sizes = var_2149_split_sizes_0, x = out_15_cast_fp16)[name = string("op_2149_cast_fp16")]; tensor layers_3_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223351936)))]; tensor input_7_strides_0 = const()[name = string("input_7_strides_0"), val = tensor([1, 1])]; string input_7_pad_type_0 = const()[name = string("input_7_pad_type_0"), val = string("valid")]; tensor input_7_pad_0 = const()[name = string("input_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_7_dilations_0 = const()[name = string("input_7_dilations_0"), val = tensor([1, 1])]; int32 input_7_groups_0 = const()[name = string("input_7_groups_0"), val = int32(1)]; tensor input_7_cast_fp16 = conv(dilations = input_7_dilations_0, groups = input_7_groups_0, pad = input_7_pad_0, pad_type = input_7_pad_type_0, strides = input_7_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("input_7_cast_fp16")]; tensor var_2166_cast_fp16 = silu(x = input_7_cast_fp16)[name = string("op_2166_cast_fp16")]; tensor layers_3_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1248517824)))]; tensor var_2172_strides_0 = const()[name = string("op_2172_strides_0"), val = tensor([1, 1])]; string var_2172_pad_type_0 = const()[name = string("op_2172_pad_type_0"), val = string("valid")]; tensor var_2172_pad_0 = const()[name = string("op_2172_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2172_dilations_0 = const()[name = string("op_2172_dilations_0"), val = tensor([1, 1])]; int32 var_2172_groups_0 = const()[name = string("op_2172_groups_0"), val = int32(1)]; tensor var_2172_cast_fp16 = conv(dilations = var_2172_dilations_0, groups = var_2172_groups_0, pad = var_2172_pad_0, pad_type = var_2172_pad_type_0, strides = var_2172_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("op_2172_cast_fp16")]; tensor x_39_cast_fp16 = mul(x = var_2166_cast_fp16, y = var_2172_cast_fp16)[name = string("x_39_cast_fp16")]; tensor hidden_states_37_strides_0 = const()[name = string("hidden_states_37_strides_0"), val = tensor([1, 1])]; string hidden_states_37_pad_type_0 = const()[name = string("hidden_states_37_pad_type_0"), val = string("valid")]; tensor hidden_states_37_pad_0 = const()[name = string("hidden_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_37_dilations_0 = const()[name = string("hidden_states_37_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_37_groups_0 = const()[name = string("hidden_states_37_groups_0"), val = int32(1)]; tensor hidden_states_37_cast_fp16 = conv(dilations = hidden_states_37_dilations_0, groups = hidden_states_37_groups_0, pad = hidden_states_37_pad_0, pad_type = hidden_states_37_pad_type_0, strides = hidden_states_37_strides_0, weight = layers_3_mlp_down_proj_weight_cast_fp16, x = x_39_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; tensor hidden_states_39_cast_fp16 = add(x = hidden_states_35_cast_fp16, y = hidden_states_37_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; fp16 const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2190_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_40_promoted_to_fp16)[name = string("op_2190_cast_fp16")]; int32 var_2188 = const()[name = string("op_2188"), val = int32(1)]; bool doubled_33_interleave_0 = const()[name = string("doubled_33_interleave_0"), val = bool(false)]; tensor doubled_33_cast_fp16 = concat(axis = var_2188, interleave = doubled_33_interleave_0, values = (hidden_states_39_cast_fp16, var_2190_cast_fp16))[name = string("doubled_33_cast_fp16")]; tensor out_17_axes_0 = const()[name = string("out_17_axes_0"), val = tensor([1])]; tensor out_17_gamma_0_to_fp16 = const()[name = string("out_17_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273683712)))]; fp16 var_2200_to_fp16 = const()[name = string("op_2200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_17_cast_fp16 = layer_norm(axes = out_17_axes_0, epsilon = var_2200_to_fp16, gamma = out_17_gamma_0_to_fp16, x = doubled_33_cast_fp16)[name = string("out_17_cast_fp16")]; tensor var_2211_split_sizes_0 = const()[name = string("op_2211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2211_axis_0 = const()[name = string("op_2211_axis_0"), val = int32(1)]; tensor var_2211_cast_fp16_0, tensor var_2211_cast_fp16_1 = split(axis = var_2211_axis_0, split_sizes = var_2211_split_sizes_0, x = out_17_cast_fp16)[name = string("op_2211_cast_fp16")]; tensor layers_4_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273691968)))]; tensor query_states_25_strides_0 = const()[name = string("query_states_25_strides_0"), val = tensor([1, 1])]; string query_states_25_pad_type_0 = const()[name = string("query_states_25_pad_type_0"), val = string("valid")]; tensor query_states_25_pad_0 = const()[name = string("query_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_25_dilations_0 = const()[name = string("query_states_25_dilations_0"), val = tensor([1, 1])]; int32 query_states_25_groups_0 = const()[name = string("query_states_25_groups_0"), val = int32(1)]; tensor query_states_25_cast_fp16 = conv(dilations = query_states_25_dilations_0, groups = query_states_25_groups_0, pad = query_states_25_pad_0, pad_type = query_states_25_pad_type_0, strides = query_states_25_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("query_states_25_cast_fp16")]; tensor layers_4_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1282080640)))]; tensor key_states_41_strides_0 = const()[name = string("key_states_41_strides_0"), val = tensor([1, 1])]; string key_states_41_pad_type_0 = const()[name = string("key_states_41_pad_type_0"), val = string("valid")]; tensor key_states_41_pad_0 = const()[name = string("key_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_41_dilations_0 = const()[name = string("key_states_41_dilations_0"), val = tensor([1, 1])]; int32 key_states_41_groups_0 = const()[name = string("key_states_41_groups_0"), val = int32(1)]; tensor key_states_41_cast_fp16 = conv(dilations = key_states_41_dilations_0, groups = key_states_41_groups_0, pad = key_states_41_pad_0, pad_type = key_states_41_pad_type_0, strides = key_states_41_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("key_states_41_cast_fp16")]; tensor value_states_25_strides_0 = const()[name = string("value_states_25_strides_0"), val = tensor([1, 1])]; string value_states_25_pad_type_0 = const()[name = string("value_states_25_pad_type_0"), val = string("valid")]; tensor value_states_25_pad_0 = const()[name = string("value_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_25_dilations_0 = const()[name = string("value_states_25_dilations_0"), val = tensor([1, 1])]; int32 value_states_25_groups_0 = const()[name = string("value_states_25_groups_0"), val = int32(1)]; tensor value_states_25_cast_fp16 = conv(dilations = value_states_25_dilations_0, groups = value_states_25_groups_0, pad = value_states_25_pad_0, pad_type = value_states_25_pad_type_0, strides = value_states_25_strides_0, weight = layers_4_self_attn_v_proj_weight_cast_fp16, x = var_2211_cast_fp16_0)[name = string("value_states_25_cast_fp16")]; tensor concat_48x = const()[name = string("concat_48x"), val = tensor([1, 16, 128, -1])]; tensor x_41_cast_fp16 = reshape(shape = concat_48x, x = query_states_25_cast_fp16)[name = string("x_41_cast_fp16")]; tensor concat_49x = const()[name = string("concat_49x"), val = tensor([1, 2, 128, -1])]; tensor var_2268_cast_fp16 = reshape(shape = concat_49x, x = key_states_41_cast_fp16)[name = string("op_2268_cast_fp16")]; tensor concat_50x = const()[name = string("concat_50x"), val = tensor([1, 2, 128, -1])]; tensor var_2275_cast_fp16 = reshape(shape = concat_50x, x = value_states_25_cast_fp16)[name = string("op_2275_cast_fp16")]; tensor var_2279_cast_fp16 = mul(x = x_41_cast_fp16, y = var_869_cast_fp16)[name = string("op_2279_cast_fp16")]; tensor var_2280_split_sizes_0 = const()[name = string("op_2280_split_sizes_0"), val = tensor([64, 64])]; int32 var_2280_axis_0 = const()[name = string("op_2280_axis_0"), val = int32(-2)]; tensor var_2280_cast_fp16_0, tensor var_2280_cast_fp16_1 = split(axis = var_2280_axis_0, split_sizes = var_2280_split_sizes_0, x = x_41_cast_fp16)[name = string("op_2280_cast_fp16")]; fp16 const_42_promoted_to_fp16 = const()[name = string("const_42_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2282_cast_fp16 = mul(x = var_2280_cast_fp16_1, y = const_42_promoted_to_fp16)[name = string("op_2282_cast_fp16")]; int32 var_2284 = const()[name = string("op_2284"), val = int32(-2)]; bool var_2285_interleave_0 = const()[name = string("op_2285_interleave_0"), val = bool(false)]; tensor var_2285_cast_fp16 = concat(axis = var_2284, interleave = var_2285_interleave_0, values = (var_2282_cast_fp16, var_2280_cast_fp16_0))[name = string("op_2285_cast_fp16")]; tensor var_2286_cast_fp16 = mul(x = var_2285_cast_fp16, y = var_878_cast_fp16)[name = string("op_2286_cast_fp16")]; tensor query_states_27_cast_fp16 = add(x = var_2279_cast_fp16, y = var_2286_cast_fp16)[name = string("query_states_27_cast_fp16")]; tensor var_2292_cast_fp16 = mul(x = var_2268_cast_fp16, y = var_869_cast_fp16)[name = string("op_2292_cast_fp16")]; tensor var_2293_split_sizes_0 = const()[name = string("op_2293_split_sizes_0"), val = tensor([64, 64])]; int32 var_2293_axis_0 = const()[name = string("op_2293_axis_0"), val = int32(-2)]; tensor var_2293_cast_fp16_0, tensor var_2293_cast_fp16_1 = split(axis = var_2293_axis_0, split_sizes = var_2293_split_sizes_0, x = var_2268_cast_fp16)[name = string("op_2293_cast_fp16")]; fp16 const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2295_cast_fp16 = mul(x = var_2293_cast_fp16_1, y = const_43_promoted_to_fp16)[name = string("op_2295_cast_fp16")]; int32 var_2297 = const()[name = string("op_2297"), val = int32(-2)]; bool var_2298_interleave_0 = const()[name = string("op_2298_interleave_0"), val = bool(false)]; tensor var_2298_cast_fp16 = concat(axis = var_2297, interleave = var_2298_interleave_0, values = (var_2295_cast_fp16, var_2293_cast_fp16_0))[name = string("op_2298_cast_fp16")]; tensor var_2299_cast_fp16 = mul(x = var_2298_cast_fp16, y = var_878_cast_fp16)[name = string("op_2299_cast_fp16")]; tensor key_states_45_cast_fp16 = add(x = var_2292_cast_fp16, y = var_2299_cast_fp16)[name = string("key_states_45_cast_fp16")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; int32 concat_53_axis_0 = const()[name = string("concat_53_axis_0"), val = int32(0)]; bool concat_53_interleave_0 = const()[name = string("concat_53_interleave_0"), val = bool(false)]; tensor concat_53 = concat(axis = concat_53_axis_0, interleave = concat_53_interleave_0, values = (expand_dims_48, expand_dims_49, position_id, expand_dims_51))[name = string("concat_53")]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; tensor concat_54_values1_0 = const()[name = string("concat_54_values1_0"), val = tensor([0])]; tensor concat_54_values3_0 = const()[name = string("concat_54_values3_0"), val = tensor([0])]; int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_52, concat_54_values1_0, cache_position_end, concat_54_values3_0))[name = string("concat_54")]; tensor key_states_47_perm_0 = const()[name = string("key_states_47_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_5_stride_0 = const()[name = string("key_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_47_cast_fp16 = transpose(perm = key_states_47_perm_0, x = key_states_45_cast_fp16)[name = string("transpose_673")]; tensor key_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = key_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = key_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_5_squeeze_mask_0, stride = key_cache_internal_tensor_assign_5_stride_0, update = key_states_47_cast_fp16, x = coreml_update_state_398)[name = string("key_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_5_cast_fp16, input = key_cache)[name = string("coreml_update_state_400_write_state")]; tensor coreml_update_state_400 = read_state(input = key_cache)[name = string("coreml_update_state_400")]; tensor value_states_27_perm_0 = const()[name = string("value_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_5_stride_0 = const()[name = string("value_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_27_cast_fp16 = transpose(perm = value_states_27_perm_0, x = var_2275_cast_fp16)[name = string("transpose_672")]; tensor value_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = value_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = value_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_5_squeeze_mask_0, stride = value_cache_internal_tensor_assign_5_stride_0, update = value_states_27_cast_fp16, x = coreml_update_state_399)[name = string("value_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_5_cast_fp16, input = value_cache)[name = string("coreml_update_state_401_write_state")]; tensor coreml_update_state_401 = read_state(input = value_cache)[name = string("coreml_update_state_401")]; tensor var_2369_begin_0 = const()[name = string("op_2369_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2369_end_0 = const()[name = string("op_2369_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2369_end_mask_0 = const()[name = string("op_2369_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2369_cast_fp16 = slice_by_index(begin = var_2369_begin_0, end = var_2369_end_0, end_mask = var_2369_end_mask_0, x = coreml_update_state_400)[name = string("op_2369_cast_fp16")]; tensor tile_8 = const()[name = string("tile_8"), val = tensor([1, 1])]; int32 var_2372_axis_0 = const()[name = string("op_2372_axis_0"), val = int32(1)]; tensor var_2372_cast_fp16_0, tensor var_2372_cast_fp16_1 = split(axis = var_2372_axis_0, split_sizes = tile_8, x = var_2369_cast_fp16)[name = string("op_2372_cast_fp16")]; tensor var_2379_begin_0 = const()[name = string("op_2379_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2379_end_0 = const()[name = string("op_2379_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2379_end_mask_0 = const()[name = string("op_2379_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2379_cast_fp16 = slice_by_index(begin = var_2379_begin_0, end = var_2379_end_0, end_mask = var_2379_end_mask_0, x = coreml_update_state_401)[name = string("op_2379_cast_fp16")]; tensor tile_9 = const()[name = string("tile_9"), val = tensor([1, 1])]; int32 var_2382_axis_0 = const()[name = string("op_2382_axis_0"), val = int32(1)]; tensor var_2382_cast_fp16_0, tensor var_2382_cast_fp16_1 = split(axis = var_2382_axis_0, split_sizes = tile_9, x = var_2379_cast_fp16)[name = string("op_2382_cast_fp16")]; tensor var_2385_split_sizes_0 = const()[name = string("op_2385_split_sizes_0"), val = tensor([8, 8])]; int32 var_2385_axis_0 = const()[name = string("op_2385_axis_0"), val = int32(1)]; tensor var_2385_0, tensor var_2385_1 = split(axis = var_2385_axis_0, split_sizes = var_2385_split_sizes_0, x = query_states_27_cast_fp16)[name = string("op_2385")]; bool attn_weights_65_transpose_x_0 = const()[name = string("attn_weights_65_transpose_x_0"), val = bool(false)]; bool attn_weights_65_transpose_y_0 = const()[name = string("attn_weights_65_transpose_y_0"), val = bool(false)]; tensor attn_weights_65_cast_fp16 = matmul(transpose_x = attn_weights_65_transpose_x_0, transpose_y = attn_weights_65_transpose_y_0, x = var_2372_cast_fp16_0, y = var_2385_0)[name = string("attn_weights_65_cast_fp16")]; fp16 var_2388_to_fp16 = const()[name = string("op_2388_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_67_cast_fp16 = mul(x = attn_weights_65_cast_fp16, y = var_2388_to_fp16)[name = string("attn_weights_67_cast_fp16")]; tensor attn_weights_69_cast_fp16 = add(x = attn_weights_67_cast_fp16, y = attn_mask_1)[name = string("attn_weights_69_cast_fp16")]; int32 var_2392 = const()[name = string("op_2392"), val = int32(-2)]; tensor attn_weights_71_cast_fp16 = softmax(axis = var_2392, x = attn_weights_69_cast_fp16)[name = string("attn_weights_71_cast_fp16")]; bool var_2398_transpose_x_1 = const()[name = string("op_2398_transpose_x_1"), val = bool(true)]; bool var_2398_transpose_y_1 = const()[name = string("op_2398_transpose_y_1"), val = bool(false)]; tensor var_2398_cast_fp16 = matmul(transpose_x = var_2398_transpose_x_1, transpose_y = var_2398_transpose_y_1, x = attn_weights_71_cast_fp16, y = var_2382_cast_fp16_0)[name = string("op_2398_cast_fp16")]; bool attn_weights_73_transpose_x_0 = const()[name = string("attn_weights_73_transpose_x_0"), val = bool(false)]; bool attn_weights_73_transpose_y_0 = const()[name = string("attn_weights_73_transpose_y_0"), val = bool(false)]; tensor attn_weights_73_cast_fp16 = matmul(transpose_x = attn_weights_73_transpose_x_0, transpose_y = attn_weights_73_transpose_y_0, x = var_2372_cast_fp16_1, y = var_2385_1)[name = string("attn_weights_73_cast_fp16")]; fp16 var_2400_to_fp16 = const()[name = string("op_2400_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_75_cast_fp16 = mul(x = attn_weights_73_cast_fp16, y = var_2400_to_fp16)[name = string("attn_weights_75_cast_fp16")]; tensor attn_weights_77_cast_fp16 = add(x = attn_weights_75_cast_fp16, y = attn_mask_1)[name = string("attn_weights_77_cast_fp16")]; int32 var_2404 = const()[name = string("op_2404"), val = int32(-2)]; tensor attn_weights_79_cast_fp16 = softmax(axis = var_2404, x = attn_weights_77_cast_fp16)[name = string("attn_weights_79_cast_fp16")]; bool attn_output_33_transpose_x_1 = const()[name = string("attn_output_33_transpose_x_1"), val = bool(true)]; bool attn_output_33_transpose_y_1 = const()[name = string("attn_output_33_transpose_y_1"), val = bool(false)]; tensor attn_output_33_cast_fp16 = matmul(transpose_x = attn_output_33_transpose_x_1, transpose_y = attn_output_33_transpose_y_1, x = attn_weights_79_cast_fp16, y = var_2382_cast_fp16_1)[name = string("attn_output_33_cast_fp16")]; int32 var_2412 = const()[name = string("op_2412"), val = int32(1)]; bool attn_output_35_interleave_0 = const()[name = string("attn_output_35_interleave_0"), val = bool(false)]; tensor attn_output_35_cast_fp16 = concat(axis = var_2412, interleave = attn_output_35_interleave_0, values = (var_2398_cast_fp16, attn_output_33_cast_fp16))[name = string("attn_output_35_cast_fp16")]; tensor var_2416_perm_0 = const()[name = string("op_2416_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_59x = const()[name = string("concat_59x"), val = tensor([1, 2048, 1, -1])]; tensor var_2416_cast_fp16 = transpose(perm = var_2416_perm_0, x = attn_output_35_cast_fp16)[name = string("transpose_671")]; tensor attn_output_39_cast_fp16 = reshape(shape = concat_59x, x = var_2416_cast_fp16)[name = string("attn_output_39_cast_fp16")]; tensor hidden_states_43_strides_0 = const()[name = string("hidden_states_43_strides_0"), val = tensor([1, 1])]; string hidden_states_43_pad_type_0 = const()[name = string("hidden_states_43_pad_type_0"), val = string("valid")]; tensor hidden_states_43_pad_0 = const()[name = string("hidden_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_43_dilations_0 = const()[name = string("hidden_states_43_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_43_groups_0 = const()[name = string("hidden_states_43_groups_0"), val = int32(1)]; tensor hidden_states_43_cast_fp16 = conv(dilations = hidden_states_43_dilations_0, groups = hidden_states_43_groups_0, pad = hidden_states_43_pad_0, pad_type = hidden_states_43_pad_type_0, strides = hidden_states_43_strides_0, weight = layers_4_self_attn_o_proj_weight_cast_fp16, x = attn_output_39_cast_fp16)[name = string("hidden_states_43_cast_fp16")]; tensor hidden_states_45_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = hidden_states_43_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; fp16 const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2449_cast_fp16 = mul(x = hidden_states_45_cast_fp16, y = const_48_promoted_to_fp16)[name = string("op_2449_cast_fp16")]; int32 var_2447 = const()[name = string("op_2447"), val = int32(1)]; bool doubled_37_interleave_0 = const()[name = string("doubled_37_interleave_0"), val = bool(false)]; tensor doubled_37_cast_fp16 = concat(axis = var_2447, interleave = doubled_37_interleave_0, values = (hidden_states_45_cast_fp16, var_2449_cast_fp16))[name = string("doubled_37_cast_fp16")]; tensor out_19_axes_0 = const()[name = string("out_19_axes_0"), val = tensor([1])]; tensor out_19_gamma_0_to_fp16 = const()[name = string("out_19_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283129280)))]; fp16 var_2459_to_fp16 = const()[name = string("op_2459_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_19_cast_fp16 = layer_norm(axes = out_19_axes_0, epsilon = var_2459_to_fp16, gamma = out_19_gamma_0_to_fp16, x = doubled_37_cast_fp16)[name = string("out_19_cast_fp16")]; tensor var_2470_split_sizes_0 = const()[name = string("op_2470_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2470_axis_0 = const()[name = string("op_2470_axis_0"), val = int32(1)]; tensor var_2470_cast_fp16_0, tensor var_2470_cast_fp16_1 = split(axis = var_2470_axis_0, split_sizes = var_2470_split_sizes_0, x = out_19_cast_fp16)[name = string("op_2470_cast_fp16")]; tensor input_9_strides_0 = const()[name = string("input_9_strides_0"), val = tensor([1, 1])]; string input_9_pad_type_0 = const()[name = string("input_9_pad_type_0"), val = string("valid")]; tensor input_9_pad_0 = const()[name = string("input_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_9_dilations_0 = const()[name = string("input_9_dilations_0"), val = tensor([1, 1])]; int32 input_9_groups_0 = const()[name = string("input_9_groups_0"), val = int32(1)]; tensor input_9_cast_fp16 = conv(dilations = input_9_dilations_0, groups = input_9_groups_0, pad = input_9_pad_0, pad_type = input_9_pad_type_0, strides = input_9_strides_0, weight = layers_4_mlp_gate_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("input_9_cast_fp16")]; tensor var_2487_cast_fp16 = silu(x = input_9_cast_fp16)[name = string("op_2487_cast_fp16")]; tensor var_2493_strides_0 = const()[name = string("op_2493_strides_0"), val = tensor([1, 1])]; string var_2493_pad_type_0 = const()[name = string("op_2493_pad_type_0"), val = string("valid")]; tensor var_2493_pad_0 = const()[name = string("op_2493_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2493_dilations_0 = const()[name = string("op_2493_dilations_0"), val = tensor([1, 1])]; int32 var_2493_groups_0 = const()[name = string("op_2493_groups_0"), val = int32(1)]; tensor var_2493_cast_fp16 = conv(dilations = var_2493_dilations_0, groups = var_2493_groups_0, pad = var_2493_pad_0, pad_type = var_2493_pad_type_0, strides = var_2493_strides_0, weight = layers_4_mlp_up_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("op_2493_cast_fp16")]; tensor x_49_cast_fp16 = mul(x = var_2487_cast_fp16, y = var_2493_cast_fp16)[name = string("x_49_cast_fp16")]; tensor hidden_states_47_strides_0 = const()[name = string("hidden_states_47_strides_0"), val = tensor([1, 1])]; string hidden_states_47_pad_type_0 = const()[name = string("hidden_states_47_pad_type_0"), val = string("valid")]; tensor hidden_states_47_pad_0 = const()[name = string("hidden_states_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_47_dilations_0 = const()[name = string("hidden_states_47_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_47_groups_0 = const()[name = string("hidden_states_47_groups_0"), val = int32(1)]; tensor hidden_states_47_cast_fp16 = conv(dilations = hidden_states_47_dilations_0, groups = hidden_states_47_groups_0, pad = hidden_states_47_pad_0, pad_type = hidden_states_47_pad_type_0, strides = hidden_states_47_strides_0, weight = layers_4_mlp_down_proj_weight_cast_fp16, x = x_49_cast_fp16)[name = string("hidden_states_47_cast_fp16")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = hidden_states_47_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; fp16 const_50_promoted_to_fp16 = const()[name = string("const_50_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2511_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_50_promoted_to_fp16)[name = string("op_2511_cast_fp16")]; int32 var_2509 = const()[name = string("op_2509"), val = int32(1)]; bool doubled_41_interleave_0 = const()[name = string("doubled_41_interleave_0"), val = bool(false)]; tensor doubled_41_cast_fp16 = concat(axis = var_2509, interleave = doubled_41_interleave_0, values = (hidden_states_49_cast_fp16, var_2511_cast_fp16))[name = string("doubled_41_cast_fp16")]; tensor out_21_axes_0 = const()[name = string("out_21_axes_0"), val = tensor([1])]; tensor out_21_gamma_0_to_fp16 = const()[name = string("out_21_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283137536)))]; fp16 var_2521_to_fp16 = const()[name = string("op_2521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_21_cast_fp16 = layer_norm(axes = out_21_axes_0, epsilon = var_2521_to_fp16, gamma = out_21_gamma_0_to_fp16, x = doubled_41_cast_fp16)[name = string("out_21_cast_fp16")]; tensor var_2532_split_sizes_0 = const()[name = string("op_2532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2532_axis_0 = const()[name = string("op_2532_axis_0"), val = int32(1)]; tensor var_2532_cast_fp16_0, tensor var_2532_cast_fp16_1 = split(axis = var_2532_axis_0, split_sizes = var_2532_split_sizes_0, x = out_21_cast_fp16)[name = string("op_2532_cast_fp16")]; tensor layers_5_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283145792)))]; tensor query_states_31_strides_0 = const()[name = string("query_states_31_strides_0"), val = tensor([1, 1])]; string query_states_31_pad_type_0 = const()[name = string("query_states_31_pad_type_0"), val = string("valid")]; tensor query_states_31_pad_0 = const()[name = string("query_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_31_dilations_0 = const()[name = string("query_states_31_dilations_0"), val = tensor([1, 1])]; int32 query_states_31_groups_0 = const()[name = string("query_states_31_groups_0"), val = int32(1)]; tensor query_states_31_cast_fp16 = conv(dilations = query_states_31_dilations_0, groups = query_states_31_groups_0, pad = query_states_31_pad_0, pad_type = query_states_31_pad_type_0, strides = query_states_31_strides_0, weight = layers_5_self_attn_q_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("query_states_31_cast_fp16")]; tensor layers_5_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1291534464)))]; tensor key_states_51_strides_0 = const()[name = string("key_states_51_strides_0"), val = tensor([1, 1])]; string key_states_51_pad_type_0 = const()[name = string("key_states_51_pad_type_0"), val = string("valid")]; tensor key_states_51_pad_0 = const()[name = string("key_states_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_51_dilations_0 = const()[name = string("key_states_51_dilations_0"), val = tensor([1, 1])]; int32 key_states_51_groups_0 = const()[name = string("key_states_51_groups_0"), val = int32(1)]; tensor key_states_51_cast_fp16 = conv(dilations = key_states_51_dilations_0, groups = key_states_51_groups_0, pad = key_states_51_pad_0, pad_type = key_states_51_pad_type_0, strides = key_states_51_strides_0, weight = layers_5_self_attn_k_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("key_states_51_cast_fp16")]; tensor value_states_31_strides_0 = const()[name = string("value_states_31_strides_0"), val = tensor([1, 1])]; string value_states_31_pad_type_0 = const()[name = string("value_states_31_pad_type_0"), val = string("valid")]; tensor value_states_31_pad_0 = const()[name = string("value_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_31_dilations_0 = const()[name = string("value_states_31_dilations_0"), val = tensor([1, 1])]; int32 value_states_31_groups_0 = const()[name = string("value_states_31_groups_0"), val = int32(1)]; tensor value_states_31_cast_fp16 = conv(dilations = value_states_31_dilations_0, groups = value_states_31_groups_0, pad = value_states_31_pad_0, pad_type = value_states_31_pad_type_0, strides = value_states_31_strides_0, weight = layers_5_self_attn_v_proj_weight_cast_fp16, x = var_2532_cast_fp16_0)[name = string("value_states_31_cast_fp16")]; tensor concat_60x = const()[name = string("concat_60x"), val = tensor([1, 16, 128, -1])]; tensor x_51_cast_fp16 = reshape(shape = concat_60x, x = query_states_31_cast_fp16)[name = string("x_51_cast_fp16")]; tensor concat_61x = const()[name = string("concat_61x"), val = tensor([1, 2, 128, -1])]; tensor var_2589_cast_fp16 = reshape(shape = concat_61x, x = key_states_51_cast_fp16)[name = string("op_2589_cast_fp16")]; tensor concat_62x = const()[name = string("concat_62x"), val = tensor([1, 2, 128, -1])]; tensor var_2596_cast_fp16 = reshape(shape = concat_62x, x = value_states_31_cast_fp16)[name = string("op_2596_cast_fp16")]; tensor var_2600_cast_fp16 = mul(x = x_51_cast_fp16, y = var_869_cast_fp16)[name = string("op_2600_cast_fp16")]; tensor var_2601_split_sizes_0 = const()[name = string("op_2601_split_sizes_0"), val = tensor([64, 64])]; int32 var_2601_axis_0 = const()[name = string("op_2601_axis_0"), val = int32(-2)]; tensor var_2601_cast_fp16_0, tensor var_2601_cast_fp16_1 = split(axis = var_2601_axis_0, split_sizes = var_2601_split_sizes_0, x = x_51_cast_fp16)[name = string("op_2601_cast_fp16")]; fp16 const_52_promoted_to_fp16 = const()[name = string("const_52_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2603_cast_fp16 = mul(x = var_2601_cast_fp16_1, y = const_52_promoted_to_fp16)[name = string("op_2603_cast_fp16")]; int32 var_2605 = const()[name = string("op_2605"), val = int32(-2)]; bool var_2606_interleave_0 = const()[name = string("op_2606_interleave_0"), val = bool(false)]; tensor var_2606_cast_fp16 = concat(axis = var_2605, interleave = var_2606_interleave_0, values = (var_2603_cast_fp16, var_2601_cast_fp16_0))[name = string("op_2606_cast_fp16")]; tensor var_2607_cast_fp16 = mul(x = var_2606_cast_fp16, y = var_878_cast_fp16)[name = string("op_2607_cast_fp16")]; tensor query_states_33_cast_fp16 = add(x = var_2600_cast_fp16, y = var_2607_cast_fp16)[name = string("query_states_33_cast_fp16")]; tensor var_2613_cast_fp16 = mul(x = var_2589_cast_fp16, y = var_869_cast_fp16)[name = string("op_2613_cast_fp16")]; tensor var_2614_split_sizes_0 = const()[name = string("op_2614_split_sizes_0"), val = tensor([64, 64])]; int32 var_2614_axis_0 = const()[name = string("op_2614_axis_0"), val = int32(-2)]; tensor var_2614_cast_fp16_0, tensor var_2614_cast_fp16_1 = split(axis = var_2614_axis_0, split_sizes = var_2614_split_sizes_0, x = var_2589_cast_fp16)[name = string("op_2614_cast_fp16")]; fp16 const_53_promoted_to_fp16 = const()[name = string("const_53_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2616_cast_fp16 = mul(x = var_2614_cast_fp16_1, y = const_53_promoted_to_fp16)[name = string("op_2616_cast_fp16")]; int32 var_2618 = const()[name = string("op_2618"), val = int32(-2)]; bool var_2619_interleave_0 = const()[name = string("op_2619_interleave_0"), val = bool(false)]; tensor var_2619_cast_fp16 = concat(axis = var_2618, interleave = var_2619_interleave_0, values = (var_2616_cast_fp16, var_2614_cast_fp16_0))[name = string("op_2619_cast_fp16")]; tensor var_2620_cast_fp16 = mul(x = var_2619_cast_fp16, y = var_878_cast_fp16)[name = string("op_2620_cast_fp16")]; tensor key_states_55_cast_fp16 = add(x = var_2613_cast_fp16, y = var_2620_cast_fp16)[name = string("key_states_55_cast_fp16")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; int32 concat_65_axis_0 = const()[name = string("concat_65_axis_0"), val = int32(0)]; bool concat_65_interleave_0 = const()[name = string("concat_65_interleave_0"), val = bool(false)]; tensor concat_65 = concat(axis = concat_65_axis_0, interleave = concat_65_interleave_0, values = (expand_dims_60, expand_dims_61, position_id, expand_dims_63))[name = string("concat_65")]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; tensor concat_66_values1_0 = const()[name = string("concat_66_values1_0"), val = tensor([0])]; tensor concat_66_values3_0 = const()[name = string("concat_66_values3_0"), val = tensor([0])]; int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_64, concat_66_values1_0, cache_position_end, concat_66_values3_0))[name = string("concat_66")]; tensor key_states_57_perm_0 = const()[name = string("key_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_6_stride_0 = const()[name = string("key_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_57_cast_fp16 = transpose(perm = key_states_57_perm_0, x = key_states_55_cast_fp16)[name = string("transpose_670")]; tensor key_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = key_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = key_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_6_squeeze_mask_0, stride = key_cache_internal_tensor_assign_6_stride_0, update = key_states_57_cast_fp16, x = coreml_update_state_400)[name = string("key_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_6_cast_fp16, input = key_cache)[name = string("coreml_update_state_402_write_state")]; tensor coreml_update_state_402 = read_state(input = key_cache)[name = string("coreml_update_state_402")]; tensor value_states_33_perm_0 = const()[name = string("value_states_33_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_6_stride_0 = const()[name = string("value_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_33_cast_fp16 = transpose(perm = value_states_33_perm_0, x = var_2596_cast_fp16)[name = string("transpose_669")]; tensor value_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = value_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = value_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_6_squeeze_mask_0, stride = value_cache_internal_tensor_assign_6_stride_0, update = value_states_33_cast_fp16, x = coreml_update_state_401)[name = string("value_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_6_cast_fp16, input = value_cache)[name = string("coreml_update_state_403_write_state")]; tensor coreml_update_state_403 = read_state(input = value_cache)[name = string("coreml_update_state_403")]; tensor var_2690_begin_0 = const()[name = string("op_2690_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2690_end_0 = const()[name = string("op_2690_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2690_end_mask_0 = const()[name = string("op_2690_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2690_cast_fp16 = slice_by_index(begin = var_2690_begin_0, end = var_2690_end_0, end_mask = var_2690_end_mask_0, x = coreml_update_state_402)[name = string("op_2690_cast_fp16")]; tensor tile_10 = const()[name = string("tile_10"), val = tensor([1, 1])]; int32 var_2693_axis_0 = const()[name = string("op_2693_axis_0"), val = int32(1)]; tensor var_2693_cast_fp16_0, tensor var_2693_cast_fp16_1 = split(axis = var_2693_axis_0, split_sizes = tile_10, x = var_2690_cast_fp16)[name = string("op_2693_cast_fp16")]; tensor var_2700_begin_0 = const()[name = string("op_2700_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2700_end_0 = const()[name = string("op_2700_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2700_end_mask_0 = const()[name = string("op_2700_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2700_cast_fp16 = slice_by_index(begin = var_2700_begin_0, end = var_2700_end_0, end_mask = var_2700_end_mask_0, x = coreml_update_state_403)[name = string("op_2700_cast_fp16")]; tensor tile_11 = const()[name = string("tile_11"), val = tensor([1, 1])]; int32 var_2703_axis_0 = const()[name = string("op_2703_axis_0"), val = int32(1)]; tensor var_2703_cast_fp16_0, tensor var_2703_cast_fp16_1 = split(axis = var_2703_axis_0, split_sizes = tile_11, x = var_2700_cast_fp16)[name = string("op_2703_cast_fp16")]; tensor var_2706_split_sizes_0 = const()[name = string("op_2706_split_sizes_0"), val = tensor([8, 8])]; int32 var_2706_axis_0 = const()[name = string("op_2706_axis_0"), val = int32(1)]; tensor var_2706_0, tensor var_2706_1 = split(axis = var_2706_axis_0, split_sizes = var_2706_split_sizes_0, x = query_states_33_cast_fp16)[name = string("op_2706")]; bool attn_weights_81_transpose_x_0 = const()[name = string("attn_weights_81_transpose_x_0"), val = bool(false)]; bool attn_weights_81_transpose_y_0 = const()[name = string("attn_weights_81_transpose_y_0"), val = bool(false)]; tensor attn_weights_81_cast_fp16 = matmul(transpose_x = attn_weights_81_transpose_x_0, transpose_y = attn_weights_81_transpose_y_0, x = var_2693_cast_fp16_0, y = var_2706_0)[name = string("attn_weights_81_cast_fp16")]; fp16 var_2709_to_fp16 = const()[name = string("op_2709_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_83_cast_fp16 = mul(x = attn_weights_81_cast_fp16, y = var_2709_to_fp16)[name = string("attn_weights_83_cast_fp16")]; tensor attn_weights_85_cast_fp16 = add(x = attn_weights_83_cast_fp16, y = attn_mask_1)[name = string("attn_weights_85_cast_fp16")]; int32 var_2713 = const()[name = string("op_2713"), val = int32(-2)]; tensor attn_weights_87_cast_fp16 = softmax(axis = var_2713, x = attn_weights_85_cast_fp16)[name = string("attn_weights_87_cast_fp16")]; bool var_2719_transpose_x_1 = const()[name = string("op_2719_transpose_x_1"), val = bool(true)]; bool var_2719_transpose_y_1 = const()[name = string("op_2719_transpose_y_1"), val = bool(false)]; tensor var_2719_cast_fp16 = matmul(transpose_x = var_2719_transpose_x_1, transpose_y = var_2719_transpose_y_1, x = attn_weights_87_cast_fp16, y = var_2703_cast_fp16_0)[name = string("op_2719_cast_fp16")]; bool attn_weights_89_transpose_x_0 = const()[name = string("attn_weights_89_transpose_x_0"), val = bool(false)]; bool attn_weights_89_transpose_y_0 = const()[name = string("attn_weights_89_transpose_y_0"), val = bool(false)]; tensor attn_weights_89_cast_fp16 = matmul(transpose_x = attn_weights_89_transpose_x_0, transpose_y = attn_weights_89_transpose_y_0, x = var_2693_cast_fp16_1, y = var_2706_1)[name = string("attn_weights_89_cast_fp16")]; fp16 var_2721_to_fp16 = const()[name = string("op_2721_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_91_cast_fp16 = mul(x = attn_weights_89_cast_fp16, y = var_2721_to_fp16)[name = string("attn_weights_91_cast_fp16")]; tensor attn_weights_93_cast_fp16 = add(x = attn_weights_91_cast_fp16, y = attn_mask_1)[name = string("attn_weights_93_cast_fp16")]; int32 var_2725 = const()[name = string("op_2725"), val = int32(-2)]; tensor attn_weights_95_cast_fp16 = softmax(axis = var_2725, x = attn_weights_93_cast_fp16)[name = string("attn_weights_95_cast_fp16")]; bool attn_output_41_transpose_x_1 = const()[name = string("attn_output_41_transpose_x_1"), val = bool(true)]; bool attn_output_41_transpose_y_1 = const()[name = string("attn_output_41_transpose_y_1"), val = bool(false)]; tensor attn_output_41_cast_fp16 = matmul(transpose_x = attn_output_41_transpose_x_1, transpose_y = attn_output_41_transpose_y_1, x = attn_weights_95_cast_fp16, y = var_2703_cast_fp16_1)[name = string("attn_output_41_cast_fp16")]; int32 var_2733 = const()[name = string("op_2733"), val = int32(1)]; bool attn_output_43_interleave_0 = const()[name = string("attn_output_43_interleave_0"), val = bool(false)]; tensor attn_output_43_cast_fp16 = concat(axis = var_2733, interleave = attn_output_43_interleave_0, values = (var_2719_cast_fp16, attn_output_41_cast_fp16))[name = string("attn_output_43_cast_fp16")]; tensor var_2737_perm_0 = const()[name = string("op_2737_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_71x = const()[name = string("concat_71x"), val = tensor([1, 2048, 1, -1])]; tensor var_2737_cast_fp16 = transpose(perm = var_2737_perm_0, x = attn_output_43_cast_fp16)[name = string("transpose_668")]; tensor attn_output_47_cast_fp16 = reshape(shape = concat_71x, x = var_2737_cast_fp16)[name = string("attn_output_47_cast_fp16")]; tensor hidden_states_53_strides_0 = const()[name = string("hidden_states_53_strides_0"), val = tensor([1, 1])]; string hidden_states_53_pad_type_0 = const()[name = string("hidden_states_53_pad_type_0"), val = string("valid")]; tensor hidden_states_53_pad_0 = const()[name = string("hidden_states_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_53_dilations_0 = const()[name = string("hidden_states_53_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_53_groups_0 = const()[name = string("hidden_states_53_groups_0"), val = int32(1)]; tensor hidden_states_53_cast_fp16 = conv(dilations = hidden_states_53_dilations_0, groups = hidden_states_53_groups_0, pad = hidden_states_53_pad_0, pad_type = hidden_states_53_pad_type_0, strides = hidden_states_53_strides_0, weight = layers_5_self_attn_o_proj_weight_cast_fp16, x = attn_output_47_cast_fp16)[name = string("hidden_states_53_cast_fp16")]; tensor hidden_states_55_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = hidden_states_53_cast_fp16)[name = string("hidden_states_55_cast_fp16")]; fp16 const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2770_cast_fp16 = mul(x = hidden_states_55_cast_fp16, y = const_58_promoted_to_fp16)[name = string("op_2770_cast_fp16")]; int32 var_2768 = const()[name = string("op_2768"), val = int32(1)]; bool doubled_45_interleave_0 = const()[name = string("doubled_45_interleave_0"), val = bool(false)]; tensor doubled_45_cast_fp16 = concat(axis = var_2768, interleave = doubled_45_interleave_0, values = (hidden_states_55_cast_fp16, var_2770_cast_fp16))[name = string("doubled_45_cast_fp16")]; tensor out_23_axes_0 = const()[name = string("out_23_axes_0"), val = tensor([1])]; tensor out_23_gamma_0_to_fp16 = const()[name = string("out_23_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292583104)))]; fp16 var_2780_to_fp16 = const()[name = string("op_2780_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_23_cast_fp16 = layer_norm(axes = out_23_axes_0, epsilon = var_2780_to_fp16, gamma = out_23_gamma_0_to_fp16, x = doubled_45_cast_fp16)[name = string("out_23_cast_fp16")]; tensor var_2791_split_sizes_0 = const()[name = string("op_2791_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2791_axis_0 = const()[name = string("op_2791_axis_0"), val = int32(1)]; tensor var_2791_cast_fp16_0, tensor var_2791_cast_fp16_1 = split(axis = var_2791_axis_0, split_sizes = var_2791_split_sizes_0, x = out_23_cast_fp16)[name = string("op_2791_cast_fp16")]; tensor layers_5_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_5_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292591360)))]; tensor input_11_strides_0 = const()[name = string("input_11_strides_0"), val = tensor([1, 1])]; string input_11_pad_type_0 = const()[name = string("input_11_pad_type_0"), val = string("valid")]; tensor input_11_pad_0 = const()[name = string("input_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_11_dilations_0 = const()[name = string("input_11_dilations_0"), val = tensor([1, 1])]; int32 input_11_groups_0 = const()[name = string("input_11_groups_0"), val = int32(1)]; tensor input_11_cast_fp16 = conv(dilations = input_11_dilations_0, groups = input_11_groups_0, pad = input_11_pad_0, pad_type = input_11_pad_type_0, strides = input_11_strides_0, weight = layers_5_mlp_gate_proj_weight_to_fp16, x = var_2791_cast_fp16_0)[name = string("input_11_cast_fp16")]; tensor var_2808_cast_fp16 = silu(x = input_11_cast_fp16)[name = string("op_2808_cast_fp16")]; tensor var_2814_strides_0 = const()[name = string("op_2814_strides_0"), val = tensor([1, 1])]; string var_2814_pad_type_0 = const()[name = string("op_2814_pad_type_0"), val = string("valid")]; tensor var_2814_pad_0 = const()[name = string("op_2814_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2814_dilations_0 = const()[name = string("op_2814_dilations_0"), val = tensor([1, 1])]; int32 var_2814_groups_0 = const()[name = string("op_2814_groups_0"), val = int32(1)]; tensor var_2814_cast_fp16 = conv(dilations = var_2814_dilations_0, groups = var_2814_groups_0, pad = var_2814_pad_0, pad_type = var_2814_pad_type_0, strides = var_2814_strides_0, weight = layers_5_mlp_up_proj_weight_cast_fp16, x = var_2791_cast_fp16_0)[name = string("op_2814_cast_fp16")]; tensor x_59_cast_fp16 = mul(x = var_2808_cast_fp16, y = var_2814_cast_fp16)[name = string("x_59_cast_fp16")]; tensor hidden_states_57_strides_0 = const()[name = string("hidden_states_57_strides_0"), val = tensor([1, 1])]; string hidden_states_57_pad_type_0 = const()[name = string("hidden_states_57_pad_type_0"), val = string("valid")]; tensor hidden_states_57_pad_0 = const()[name = string("hidden_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_57_dilations_0 = const()[name = string("hidden_states_57_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_57_groups_0 = const()[name = string("hidden_states_57_groups_0"), val = int32(1)]; tensor hidden_states_57_cast_fp16 = conv(dilations = hidden_states_57_dilations_0, groups = hidden_states_57_groups_0, pad = hidden_states_57_pad_0, pad_type = hidden_states_57_pad_type_0, strides = hidden_states_57_strides_0, weight = layers_5_mlp_down_proj_weight_cast_fp16, x = x_59_cast_fp16)[name = string("hidden_states_57_cast_fp16")]; tensor hidden_states_59_cast_fp16 = add(x = hidden_states_55_cast_fp16, y = hidden_states_57_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2832_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_2832_cast_fp16")]; int32 var_2830 = const()[name = string("op_2830"), val = int32(1)]; bool doubled_49_interleave_0 = const()[name = string("doubled_49_interleave_0"), val = bool(false)]; tensor doubled_49_cast_fp16 = concat(axis = var_2830, interleave = doubled_49_interleave_0, values = (hidden_states_59_cast_fp16, var_2832_cast_fp16))[name = string("doubled_49_cast_fp16")]; tensor out_25_axes_0 = const()[name = string("out_25_axes_0"), val = tensor([1])]; tensor out_25_gamma_0_to_fp16 = const()[name = string("out_25_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317757248)))]; fp16 var_2842_to_fp16 = const()[name = string("op_2842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_25_cast_fp16 = layer_norm(axes = out_25_axes_0, epsilon = var_2842_to_fp16, gamma = out_25_gamma_0_to_fp16, x = doubled_49_cast_fp16)[name = string("out_25_cast_fp16")]; tensor var_2853_split_sizes_0 = const()[name = string("op_2853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2853_axis_0 = const()[name = string("op_2853_axis_0"), val = int32(1)]; tensor var_2853_cast_fp16_0, tensor var_2853_cast_fp16_1 = split(axis = var_2853_axis_0, split_sizes = var_2853_split_sizes_0, x = out_25_cast_fp16)[name = string("op_2853_cast_fp16")]; tensor layers_6_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317765504)))]; tensor query_states_37_strides_0 = const()[name = string("query_states_37_strides_0"), val = tensor([1, 1])]; string query_states_37_pad_type_0 = const()[name = string("query_states_37_pad_type_0"), val = string("valid")]; tensor query_states_37_pad_0 = const()[name = string("query_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_37_dilations_0 = const()[name = string("query_states_37_dilations_0"), val = tensor([1, 1])]; int32 query_states_37_groups_0 = const()[name = string("query_states_37_groups_0"), val = int32(1)]; tensor query_states_37_cast_fp16 = conv(dilations = query_states_37_dilations_0, groups = query_states_37_groups_0, pad = query_states_37_pad_0, pad_type = query_states_37_pad_type_0, strides = query_states_37_strides_0, weight = layers_6_self_attn_q_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("query_states_37_cast_fp16")]; tensor layers_6_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1326154176)))]; tensor key_states_61_strides_0 = const()[name = string("key_states_61_strides_0"), val = tensor([1, 1])]; string key_states_61_pad_type_0 = const()[name = string("key_states_61_pad_type_0"), val = string("valid")]; tensor key_states_61_pad_0 = const()[name = string("key_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_61_dilations_0 = const()[name = string("key_states_61_dilations_0"), val = tensor([1, 1])]; int32 key_states_61_groups_0 = const()[name = string("key_states_61_groups_0"), val = int32(1)]; tensor key_states_61_cast_fp16 = conv(dilations = key_states_61_dilations_0, groups = key_states_61_groups_0, pad = key_states_61_pad_0, pad_type = key_states_61_pad_type_0, strides = key_states_61_strides_0, weight = layers_6_self_attn_k_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("key_states_61_cast_fp16")]; tensor value_states_37_strides_0 = const()[name = string("value_states_37_strides_0"), val = tensor([1, 1])]; string value_states_37_pad_type_0 = const()[name = string("value_states_37_pad_type_0"), val = string("valid")]; tensor value_states_37_pad_0 = const()[name = string("value_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_37_dilations_0 = const()[name = string("value_states_37_dilations_0"), val = tensor([1, 1])]; int32 value_states_37_groups_0 = const()[name = string("value_states_37_groups_0"), val = int32(1)]; tensor value_states_37_cast_fp16 = conv(dilations = value_states_37_dilations_0, groups = value_states_37_groups_0, pad = value_states_37_pad_0, pad_type = value_states_37_pad_type_0, strides = value_states_37_strides_0, weight = layers_6_self_attn_v_proj_weight_cast_fp16, x = var_2853_cast_fp16_0)[name = string("value_states_37_cast_fp16")]; tensor concat_72x = const()[name = string("concat_72x"), val = tensor([1, 16, 128, -1])]; tensor x_61_cast_fp16 = reshape(shape = concat_72x, x = query_states_37_cast_fp16)[name = string("x_61_cast_fp16")]; tensor concat_73x = const()[name = string("concat_73x"), val = tensor([1, 2, 128, -1])]; tensor var_2910_cast_fp16 = reshape(shape = concat_73x, x = key_states_61_cast_fp16)[name = string("op_2910_cast_fp16")]; tensor concat_74x = const()[name = string("concat_74x"), val = tensor([1, 2, 128, -1])]; tensor var_2917_cast_fp16 = reshape(shape = concat_74x, x = value_states_37_cast_fp16)[name = string("op_2917_cast_fp16")]; tensor var_2921_cast_fp16 = mul(x = x_61_cast_fp16, y = var_869_cast_fp16)[name = string("op_2921_cast_fp16")]; tensor var_2922_split_sizes_0 = const()[name = string("op_2922_split_sizes_0"), val = tensor([64, 64])]; int32 var_2922_axis_0 = const()[name = string("op_2922_axis_0"), val = int32(-2)]; tensor var_2922_cast_fp16_0, tensor var_2922_cast_fp16_1 = split(axis = var_2922_axis_0, split_sizes = var_2922_split_sizes_0, x = x_61_cast_fp16)[name = string("op_2922_cast_fp16")]; fp16 const_62_promoted_to_fp16 = const()[name = string("const_62_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2924_cast_fp16 = mul(x = var_2922_cast_fp16_1, y = const_62_promoted_to_fp16)[name = string("op_2924_cast_fp16")]; int32 var_2926 = const()[name = string("op_2926"), val = int32(-2)]; bool var_2927_interleave_0 = const()[name = string("op_2927_interleave_0"), val = bool(false)]; tensor var_2927_cast_fp16 = concat(axis = var_2926, interleave = var_2927_interleave_0, values = (var_2924_cast_fp16, var_2922_cast_fp16_0))[name = string("op_2927_cast_fp16")]; tensor var_2928_cast_fp16 = mul(x = var_2927_cast_fp16, y = var_878_cast_fp16)[name = string("op_2928_cast_fp16")]; tensor query_states_39_cast_fp16 = add(x = var_2921_cast_fp16, y = var_2928_cast_fp16)[name = string("query_states_39_cast_fp16")]; tensor var_2934_cast_fp16 = mul(x = var_2910_cast_fp16, y = var_869_cast_fp16)[name = string("op_2934_cast_fp16")]; tensor var_2935_split_sizes_0 = const()[name = string("op_2935_split_sizes_0"), val = tensor([64, 64])]; int32 var_2935_axis_0 = const()[name = string("op_2935_axis_0"), val = int32(-2)]; tensor var_2935_cast_fp16_0, tensor var_2935_cast_fp16_1 = split(axis = var_2935_axis_0, split_sizes = var_2935_split_sizes_0, x = var_2910_cast_fp16)[name = string("op_2935_cast_fp16")]; fp16 const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2937_cast_fp16 = mul(x = var_2935_cast_fp16_1, y = const_63_promoted_to_fp16)[name = string("op_2937_cast_fp16")]; int32 var_2939 = const()[name = string("op_2939"), val = int32(-2)]; bool var_2940_interleave_0 = const()[name = string("op_2940_interleave_0"), val = bool(false)]; tensor var_2940_cast_fp16 = concat(axis = var_2939, interleave = var_2940_interleave_0, values = (var_2937_cast_fp16, var_2935_cast_fp16_0))[name = string("op_2940_cast_fp16")]; tensor var_2941_cast_fp16 = mul(x = var_2940_cast_fp16, y = var_878_cast_fp16)[name = string("op_2941_cast_fp16")]; tensor key_states_65_cast_fp16 = add(x = var_2934_cast_fp16, y = var_2941_cast_fp16)[name = string("key_states_65_cast_fp16")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; int32 concat_77_axis_0 = const()[name = string("concat_77_axis_0"), val = int32(0)]; bool concat_77_interleave_0 = const()[name = string("concat_77_interleave_0"), val = bool(false)]; tensor concat_77 = concat(axis = concat_77_axis_0, interleave = concat_77_interleave_0, values = (expand_dims_72, expand_dims_73, position_id, expand_dims_75))[name = string("concat_77")]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; tensor concat_78_values1_0 = const()[name = string("concat_78_values1_0"), val = tensor([0])]; tensor concat_78_values3_0 = const()[name = string("concat_78_values3_0"), val = tensor([0])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_76, concat_78_values1_0, cache_position_end, concat_78_values3_0))[name = string("concat_78")]; tensor key_states_67_perm_0 = const()[name = string("key_states_67_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_7_stride_0 = const()[name = string("key_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_67_cast_fp16 = transpose(perm = key_states_67_perm_0, x = key_states_65_cast_fp16)[name = string("transpose_667")]; tensor key_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = key_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = key_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_7_squeeze_mask_0, stride = key_cache_internal_tensor_assign_7_stride_0, update = key_states_67_cast_fp16, x = coreml_update_state_402)[name = string("key_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_7_cast_fp16, input = key_cache)[name = string("coreml_update_state_404_write_state")]; tensor coreml_update_state_404 = read_state(input = key_cache)[name = string("coreml_update_state_404")]; tensor value_states_39_perm_0 = const()[name = string("value_states_39_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_7_stride_0 = const()[name = string("value_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_39_cast_fp16 = transpose(perm = value_states_39_perm_0, x = var_2917_cast_fp16)[name = string("transpose_666")]; tensor value_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = value_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = value_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_7_squeeze_mask_0, stride = value_cache_internal_tensor_assign_7_stride_0, update = value_states_39_cast_fp16, x = coreml_update_state_403)[name = string("value_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_7_cast_fp16, input = value_cache)[name = string("coreml_update_state_405_write_state")]; tensor coreml_update_state_405 = read_state(input = value_cache)[name = string("coreml_update_state_405")]; tensor var_3011_begin_0 = const()[name = string("op_3011_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3011_end_0 = const()[name = string("op_3011_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3011_end_mask_0 = const()[name = string("op_3011_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3011_cast_fp16 = slice_by_index(begin = var_3011_begin_0, end = var_3011_end_0, end_mask = var_3011_end_mask_0, x = coreml_update_state_404)[name = string("op_3011_cast_fp16")]; tensor tile_12 = const()[name = string("tile_12"), val = tensor([1, 1])]; int32 var_3014_axis_0 = const()[name = string("op_3014_axis_0"), val = int32(1)]; tensor var_3014_cast_fp16_0, tensor var_3014_cast_fp16_1 = split(axis = var_3014_axis_0, split_sizes = tile_12, x = var_3011_cast_fp16)[name = string("op_3014_cast_fp16")]; tensor var_3021_begin_0 = const()[name = string("op_3021_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3021_end_0 = const()[name = string("op_3021_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3021_end_mask_0 = const()[name = string("op_3021_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3021_cast_fp16 = slice_by_index(begin = var_3021_begin_0, end = var_3021_end_0, end_mask = var_3021_end_mask_0, x = coreml_update_state_405)[name = string("op_3021_cast_fp16")]; tensor tile_13 = const()[name = string("tile_13"), val = tensor([1, 1])]; int32 var_3024_axis_0 = const()[name = string("op_3024_axis_0"), val = int32(1)]; tensor var_3024_cast_fp16_0, tensor var_3024_cast_fp16_1 = split(axis = var_3024_axis_0, split_sizes = tile_13, x = var_3021_cast_fp16)[name = string("op_3024_cast_fp16")]; tensor var_3027_split_sizes_0 = const()[name = string("op_3027_split_sizes_0"), val = tensor([8, 8])]; int32 var_3027_axis_0 = const()[name = string("op_3027_axis_0"), val = int32(1)]; tensor var_3027_0, tensor var_3027_1 = split(axis = var_3027_axis_0, split_sizes = var_3027_split_sizes_0, x = query_states_39_cast_fp16)[name = string("op_3027")]; bool attn_weights_97_transpose_x_0 = const()[name = string("attn_weights_97_transpose_x_0"), val = bool(false)]; bool attn_weights_97_transpose_y_0 = const()[name = string("attn_weights_97_transpose_y_0"), val = bool(false)]; tensor attn_weights_97_cast_fp16 = matmul(transpose_x = attn_weights_97_transpose_x_0, transpose_y = attn_weights_97_transpose_y_0, x = var_3014_cast_fp16_0, y = var_3027_0)[name = string("attn_weights_97_cast_fp16")]; fp16 var_3030_to_fp16 = const()[name = string("op_3030_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_99_cast_fp16 = mul(x = attn_weights_97_cast_fp16, y = var_3030_to_fp16)[name = string("attn_weights_99_cast_fp16")]; tensor attn_weights_101_cast_fp16 = add(x = attn_weights_99_cast_fp16, y = attn_mask_1)[name = string("attn_weights_101_cast_fp16")]; int32 var_3034 = const()[name = string("op_3034"), val = int32(-2)]; tensor attn_weights_103_cast_fp16 = softmax(axis = var_3034, x = attn_weights_101_cast_fp16)[name = string("attn_weights_103_cast_fp16")]; bool var_3040_transpose_x_1 = const()[name = string("op_3040_transpose_x_1"), val = bool(true)]; bool var_3040_transpose_y_1 = const()[name = string("op_3040_transpose_y_1"), val = bool(false)]; tensor var_3040_cast_fp16 = matmul(transpose_x = var_3040_transpose_x_1, transpose_y = var_3040_transpose_y_1, x = attn_weights_103_cast_fp16, y = var_3024_cast_fp16_0)[name = string("op_3040_cast_fp16")]; bool attn_weights_105_transpose_x_0 = const()[name = string("attn_weights_105_transpose_x_0"), val = bool(false)]; bool attn_weights_105_transpose_y_0 = const()[name = string("attn_weights_105_transpose_y_0"), val = bool(false)]; tensor attn_weights_105_cast_fp16 = matmul(transpose_x = attn_weights_105_transpose_x_0, transpose_y = attn_weights_105_transpose_y_0, x = var_3014_cast_fp16_1, y = var_3027_1)[name = string("attn_weights_105_cast_fp16")]; fp16 var_3042_to_fp16 = const()[name = string("op_3042_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_107_cast_fp16 = mul(x = attn_weights_105_cast_fp16, y = var_3042_to_fp16)[name = string("attn_weights_107_cast_fp16")]; tensor attn_weights_109_cast_fp16 = add(x = attn_weights_107_cast_fp16, y = attn_mask_1)[name = string("attn_weights_109_cast_fp16")]; int32 var_3046 = const()[name = string("op_3046"), val = int32(-2)]; tensor attn_weights_111_cast_fp16 = softmax(axis = var_3046, x = attn_weights_109_cast_fp16)[name = string("attn_weights_111_cast_fp16")]; bool attn_output_49_transpose_x_1 = const()[name = string("attn_output_49_transpose_x_1"), val = bool(true)]; bool attn_output_49_transpose_y_1 = const()[name = string("attn_output_49_transpose_y_1"), val = bool(false)]; tensor attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_1, transpose_y = attn_output_49_transpose_y_1, x = attn_weights_111_cast_fp16, y = var_3024_cast_fp16_1)[name = string("attn_output_49_cast_fp16")]; int32 var_3054 = const()[name = string("op_3054"), val = int32(1)]; bool attn_output_51_interleave_0 = const()[name = string("attn_output_51_interleave_0"), val = bool(false)]; tensor attn_output_51_cast_fp16 = concat(axis = var_3054, interleave = attn_output_51_interleave_0, values = (var_3040_cast_fp16, attn_output_49_cast_fp16))[name = string("attn_output_51_cast_fp16")]; tensor var_3058_perm_0 = const()[name = string("op_3058_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_83x = const()[name = string("concat_83x"), val = tensor([1, 2048, 1, -1])]; tensor var_3058_cast_fp16 = transpose(perm = var_3058_perm_0, x = attn_output_51_cast_fp16)[name = string("transpose_665")]; tensor attn_output_55_cast_fp16 = reshape(shape = concat_83x, x = var_3058_cast_fp16)[name = string("attn_output_55_cast_fp16")]; tensor hidden_states_63_strides_0 = const()[name = string("hidden_states_63_strides_0"), val = tensor([1, 1])]; string hidden_states_63_pad_type_0 = const()[name = string("hidden_states_63_pad_type_0"), val = string("valid")]; tensor hidden_states_63_pad_0 = const()[name = string("hidden_states_63_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_63_dilations_0 = const()[name = string("hidden_states_63_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_63_groups_0 = const()[name = string("hidden_states_63_groups_0"), val = int32(1)]; tensor hidden_states_63_cast_fp16 = conv(dilations = hidden_states_63_dilations_0, groups = hidden_states_63_groups_0, pad = hidden_states_63_pad_0, pad_type = hidden_states_63_pad_type_0, strides = hidden_states_63_strides_0, weight = layers_6_self_attn_o_proj_weight_cast_fp16, x = attn_output_55_cast_fp16)[name = string("hidden_states_63_cast_fp16")]; tensor hidden_states_65_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = hidden_states_63_cast_fp16)[name = string("hidden_states_65_cast_fp16")]; fp16 const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3091_cast_fp16 = mul(x = hidden_states_65_cast_fp16, y = const_68_promoted_to_fp16)[name = string("op_3091_cast_fp16")]; int32 var_3089 = const()[name = string("op_3089"), val = int32(1)]; bool doubled_53_interleave_0 = const()[name = string("doubled_53_interleave_0"), val = bool(false)]; tensor doubled_53_cast_fp16 = concat(axis = var_3089, interleave = doubled_53_interleave_0, values = (hidden_states_65_cast_fp16, var_3091_cast_fp16))[name = string("doubled_53_cast_fp16")]; tensor out_27_axes_0 = const()[name = string("out_27_axes_0"), val = tensor([1])]; tensor out_27_gamma_0_to_fp16 = const()[name = string("out_27_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327202816)))]; fp16 var_3101_to_fp16 = const()[name = string("op_3101_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_27_cast_fp16 = layer_norm(axes = out_27_axes_0, epsilon = var_3101_to_fp16, gamma = out_27_gamma_0_to_fp16, x = doubled_53_cast_fp16)[name = string("out_27_cast_fp16")]; tensor var_3112_split_sizes_0 = const()[name = string("op_3112_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3112_axis_0 = const()[name = string("op_3112_axis_0"), val = int32(1)]; tensor var_3112_cast_fp16_0, tensor var_3112_cast_fp16_1 = split(axis = var_3112_axis_0, split_sizes = var_3112_split_sizes_0, x = out_27_cast_fp16)[name = string("op_3112_cast_fp16")]; tensor input_13_strides_0 = const()[name = string("input_13_strides_0"), val = tensor([1, 1])]; string input_13_pad_type_0 = const()[name = string("input_13_pad_type_0"), val = string("valid")]; tensor input_13_pad_0 = const()[name = string("input_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_13_dilations_0 = const()[name = string("input_13_dilations_0"), val = tensor([1, 1])]; int32 input_13_groups_0 = const()[name = string("input_13_groups_0"), val = int32(1)]; tensor input_13_cast_fp16 = conv(dilations = input_13_dilations_0, groups = input_13_groups_0, pad = input_13_pad_0, pad_type = input_13_pad_type_0, strides = input_13_strides_0, weight = layers_6_mlp_gate_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("input_13_cast_fp16")]; tensor var_3129_cast_fp16 = silu(x = input_13_cast_fp16)[name = string("op_3129_cast_fp16")]; tensor var_3135_strides_0 = const()[name = string("op_3135_strides_0"), val = tensor([1, 1])]; string var_3135_pad_type_0 = const()[name = string("op_3135_pad_type_0"), val = string("valid")]; tensor var_3135_pad_0 = const()[name = string("op_3135_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3135_dilations_0 = const()[name = string("op_3135_dilations_0"), val = tensor([1, 1])]; int32 var_3135_groups_0 = const()[name = string("op_3135_groups_0"), val = int32(1)]; tensor var_3135_cast_fp16 = conv(dilations = var_3135_dilations_0, groups = var_3135_groups_0, pad = var_3135_pad_0, pad_type = var_3135_pad_type_0, strides = var_3135_strides_0, weight = layers_6_mlp_up_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("op_3135_cast_fp16")]; tensor x_69_cast_fp16 = mul(x = var_3129_cast_fp16, y = var_3135_cast_fp16)[name = string("x_69_cast_fp16")]; tensor hidden_states_67_strides_0 = const()[name = string("hidden_states_67_strides_0"), val = tensor([1, 1])]; string hidden_states_67_pad_type_0 = const()[name = string("hidden_states_67_pad_type_0"), val = string("valid")]; tensor hidden_states_67_pad_0 = const()[name = string("hidden_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_67_dilations_0 = const()[name = string("hidden_states_67_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_67_groups_0 = const()[name = string("hidden_states_67_groups_0"), val = int32(1)]; tensor hidden_states_67_cast_fp16 = conv(dilations = hidden_states_67_dilations_0, groups = hidden_states_67_groups_0, pad = hidden_states_67_pad_0, pad_type = hidden_states_67_pad_type_0, strides = hidden_states_67_strides_0, weight = layers_6_mlp_down_proj_weight_cast_fp16, x = x_69_cast_fp16)[name = string("hidden_states_67_cast_fp16")]; tensor hidden_states_69_cast_fp16 = add(x = hidden_states_65_cast_fp16, y = hidden_states_67_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3153_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_70_promoted_to_fp16)[name = string("op_3153_cast_fp16")]; int32 var_3151 = const()[name = string("op_3151"), val = int32(1)]; bool doubled_57_interleave_0 = const()[name = string("doubled_57_interleave_0"), val = bool(false)]; tensor doubled_57_cast_fp16 = concat(axis = var_3151, interleave = doubled_57_interleave_0, values = (hidden_states_69_cast_fp16, var_3153_cast_fp16))[name = string("doubled_57_cast_fp16")]; tensor out_29_axes_0 = const()[name = string("out_29_axes_0"), val = tensor([1])]; tensor out_29_gamma_0_to_fp16 = const()[name = string("out_29_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327211072)))]; fp16 var_3163_to_fp16 = const()[name = string("op_3163_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_29_cast_fp16 = layer_norm(axes = out_29_axes_0, epsilon = var_3163_to_fp16, gamma = out_29_gamma_0_to_fp16, x = doubled_57_cast_fp16)[name = string("out_29_cast_fp16")]; tensor var_3174_split_sizes_0 = const()[name = string("op_3174_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3174_axis_0 = const()[name = string("op_3174_axis_0"), val = int32(1)]; tensor var_3174_cast_fp16_0, tensor var_3174_cast_fp16_1 = split(axis = var_3174_axis_0, split_sizes = var_3174_split_sizes_0, x = out_29_cast_fp16)[name = string("op_3174_cast_fp16")]; tensor layers_7_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327219328)))]; tensor query_states_43_strides_0 = const()[name = string("query_states_43_strides_0"), val = tensor([1, 1])]; string query_states_43_pad_type_0 = const()[name = string("query_states_43_pad_type_0"), val = string("valid")]; tensor query_states_43_pad_0 = const()[name = string("query_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_43_dilations_0 = const()[name = string("query_states_43_dilations_0"), val = tensor([1, 1])]; int32 query_states_43_groups_0 = const()[name = string("query_states_43_groups_0"), val = int32(1)]; tensor query_states_43_cast_fp16 = conv(dilations = query_states_43_dilations_0, groups = query_states_43_groups_0, pad = query_states_43_pad_0, pad_type = query_states_43_pad_type_0, strides = query_states_43_strides_0, weight = layers_7_self_attn_q_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("query_states_43_cast_fp16")]; tensor layers_7_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1335608000)))]; tensor key_states_71_strides_0 = const()[name = string("key_states_71_strides_0"), val = tensor([1, 1])]; string key_states_71_pad_type_0 = const()[name = string("key_states_71_pad_type_0"), val = string("valid")]; tensor key_states_71_pad_0 = const()[name = string("key_states_71_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_71_dilations_0 = const()[name = string("key_states_71_dilations_0"), val = tensor([1, 1])]; int32 key_states_71_groups_0 = const()[name = string("key_states_71_groups_0"), val = int32(1)]; tensor key_states_71_cast_fp16 = conv(dilations = key_states_71_dilations_0, groups = key_states_71_groups_0, pad = key_states_71_pad_0, pad_type = key_states_71_pad_type_0, strides = key_states_71_strides_0, weight = layers_7_self_attn_k_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("key_states_71_cast_fp16")]; tensor value_states_43_strides_0 = const()[name = string("value_states_43_strides_0"), val = tensor([1, 1])]; string value_states_43_pad_type_0 = const()[name = string("value_states_43_pad_type_0"), val = string("valid")]; tensor value_states_43_pad_0 = const()[name = string("value_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_43_dilations_0 = const()[name = string("value_states_43_dilations_0"), val = tensor([1, 1])]; int32 value_states_43_groups_0 = const()[name = string("value_states_43_groups_0"), val = int32(1)]; tensor value_states_43_cast_fp16 = conv(dilations = value_states_43_dilations_0, groups = value_states_43_groups_0, pad = value_states_43_pad_0, pad_type = value_states_43_pad_type_0, strides = value_states_43_strides_0, weight = layers_7_self_attn_v_proj_weight_cast_fp16, x = var_3174_cast_fp16_0)[name = string("value_states_43_cast_fp16")]; tensor concat_84x = const()[name = string("concat_84x"), val = tensor([1, 16, 128, -1])]; tensor x_71_cast_fp16 = reshape(shape = concat_84x, x = query_states_43_cast_fp16)[name = string("x_71_cast_fp16")]; tensor concat_85x = const()[name = string("concat_85x"), val = tensor([1, 2, 128, -1])]; tensor var_3231_cast_fp16 = reshape(shape = concat_85x, x = key_states_71_cast_fp16)[name = string("op_3231_cast_fp16")]; tensor concat_86x = const()[name = string("concat_86x"), val = tensor([1, 2, 128, -1])]; tensor var_3238_cast_fp16 = reshape(shape = concat_86x, x = value_states_43_cast_fp16)[name = string("op_3238_cast_fp16")]; tensor var_3242_cast_fp16 = mul(x = x_71_cast_fp16, y = var_869_cast_fp16)[name = string("op_3242_cast_fp16")]; tensor var_3243_split_sizes_0 = const()[name = string("op_3243_split_sizes_0"), val = tensor([64, 64])]; int32 var_3243_axis_0 = const()[name = string("op_3243_axis_0"), val = int32(-2)]; tensor var_3243_cast_fp16_0, tensor var_3243_cast_fp16_1 = split(axis = var_3243_axis_0, split_sizes = var_3243_split_sizes_0, x = x_71_cast_fp16)[name = string("op_3243_cast_fp16")]; fp16 const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3245_cast_fp16 = mul(x = var_3243_cast_fp16_1, y = const_72_promoted_to_fp16)[name = string("op_3245_cast_fp16")]; int32 var_3247 = const()[name = string("op_3247"), val = int32(-2)]; bool var_3248_interleave_0 = const()[name = string("op_3248_interleave_0"), val = bool(false)]; tensor var_3248_cast_fp16 = concat(axis = var_3247, interleave = var_3248_interleave_0, values = (var_3245_cast_fp16, var_3243_cast_fp16_0))[name = string("op_3248_cast_fp16")]; tensor var_3249_cast_fp16 = mul(x = var_3248_cast_fp16, y = var_878_cast_fp16)[name = string("op_3249_cast_fp16")]; tensor query_states_45_cast_fp16 = add(x = var_3242_cast_fp16, y = var_3249_cast_fp16)[name = string("query_states_45_cast_fp16")]; tensor var_3255_cast_fp16 = mul(x = var_3231_cast_fp16, y = var_869_cast_fp16)[name = string("op_3255_cast_fp16")]; tensor var_3256_split_sizes_0 = const()[name = string("op_3256_split_sizes_0"), val = tensor([64, 64])]; int32 var_3256_axis_0 = const()[name = string("op_3256_axis_0"), val = int32(-2)]; tensor var_3256_cast_fp16_0, tensor var_3256_cast_fp16_1 = split(axis = var_3256_axis_0, split_sizes = var_3256_split_sizes_0, x = var_3231_cast_fp16)[name = string("op_3256_cast_fp16")]; fp16 const_73_promoted_to_fp16 = const()[name = string("const_73_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3258_cast_fp16 = mul(x = var_3256_cast_fp16_1, y = const_73_promoted_to_fp16)[name = string("op_3258_cast_fp16")]; int32 var_3260 = const()[name = string("op_3260"), val = int32(-2)]; bool var_3261_interleave_0 = const()[name = string("op_3261_interleave_0"), val = bool(false)]; tensor var_3261_cast_fp16 = concat(axis = var_3260, interleave = var_3261_interleave_0, values = (var_3258_cast_fp16, var_3256_cast_fp16_0))[name = string("op_3261_cast_fp16")]; tensor var_3262_cast_fp16 = mul(x = var_3261_cast_fp16, y = var_878_cast_fp16)[name = string("op_3262_cast_fp16")]; tensor key_states_75_cast_fp16 = add(x = var_3255_cast_fp16, y = var_3262_cast_fp16)[name = string("key_states_75_cast_fp16")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; int32 concat_89_axis_0 = const()[name = string("concat_89_axis_0"), val = int32(0)]; bool concat_89_interleave_0 = const()[name = string("concat_89_interleave_0"), val = bool(false)]; tensor concat_89 = concat(axis = concat_89_axis_0, interleave = concat_89_interleave_0, values = (expand_dims_84, expand_dims_85, position_id, expand_dims_87))[name = string("concat_89")]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; tensor concat_90_values1_0 = const()[name = string("concat_90_values1_0"), val = tensor([0])]; tensor concat_90_values3_0 = const()[name = string("concat_90_values3_0"), val = tensor([0])]; int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_88, concat_90_values1_0, cache_position_end, concat_90_values3_0))[name = string("concat_90")]; tensor key_states_77_perm_0 = const()[name = string("key_states_77_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_8_stride_0 = const()[name = string("key_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_77_cast_fp16 = transpose(perm = key_states_77_perm_0, x = key_states_75_cast_fp16)[name = string("transpose_664")]; tensor key_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = key_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = key_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_8_squeeze_mask_0, stride = key_cache_internal_tensor_assign_8_stride_0, update = key_states_77_cast_fp16, x = coreml_update_state_404)[name = string("key_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_8_cast_fp16, input = key_cache)[name = string("coreml_update_state_406_write_state")]; tensor coreml_update_state_406 = read_state(input = key_cache)[name = string("coreml_update_state_406")]; tensor value_states_45_perm_0 = const()[name = string("value_states_45_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_8_stride_0 = const()[name = string("value_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_45_cast_fp16 = transpose(perm = value_states_45_perm_0, x = var_3238_cast_fp16)[name = string("transpose_663")]; tensor value_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = value_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = value_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_8_squeeze_mask_0, stride = value_cache_internal_tensor_assign_8_stride_0, update = value_states_45_cast_fp16, x = coreml_update_state_405)[name = string("value_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_8_cast_fp16, input = value_cache)[name = string("coreml_update_state_407_write_state")]; tensor coreml_update_state_407 = read_state(input = value_cache)[name = string("coreml_update_state_407")]; tensor var_3332_begin_0 = const()[name = string("op_3332_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3332_end_0 = const()[name = string("op_3332_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3332_end_mask_0 = const()[name = string("op_3332_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3332_cast_fp16 = slice_by_index(begin = var_3332_begin_0, end = var_3332_end_0, end_mask = var_3332_end_mask_0, x = coreml_update_state_406)[name = string("op_3332_cast_fp16")]; tensor tile_14 = const()[name = string("tile_14"), val = tensor([1, 1])]; int32 var_3335_axis_0 = const()[name = string("op_3335_axis_0"), val = int32(1)]; tensor var_3335_cast_fp16_0, tensor var_3335_cast_fp16_1 = split(axis = var_3335_axis_0, split_sizes = tile_14, x = var_3332_cast_fp16)[name = string("op_3335_cast_fp16")]; tensor var_3342_begin_0 = const()[name = string("op_3342_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3342_end_0 = const()[name = string("op_3342_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3342_end_mask_0 = const()[name = string("op_3342_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3342_cast_fp16 = slice_by_index(begin = var_3342_begin_0, end = var_3342_end_0, end_mask = var_3342_end_mask_0, x = coreml_update_state_407)[name = string("op_3342_cast_fp16")]; tensor tile_15 = const()[name = string("tile_15"), val = tensor([1, 1])]; int32 var_3345_axis_0 = const()[name = string("op_3345_axis_0"), val = int32(1)]; tensor var_3345_cast_fp16_0, tensor var_3345_cast_fp16_1 = split(axis = var_3345_axis_0, split_sizes = tile_15, x = var_3342_cast_fp16)[name = string("op_3345_cast_fp16")]; tensor var_3348_split_sizes_0 = const()[name = string("op_3348_split_sizes_0"), val = tensor([8, 8])]; int32 var_3348_axis_0 = const()[name = string("op_3348_axis_0"), val = int32(1)]; tensor var_3348_0, tensor var_3348_1 = split(axis = var_3348_axis_0, split_sizes = var_3348_split_sizes_0, x = query_states_45_cast_fp16)[name = string("op_3348")]; bool attn_weights_113_transpose_x_0 = const()[name = string("attn_weights_113_transpose_x_0"), val = bool(false)]; bool attn_weights_113_transpose_y_0 = const()[name = string("attn_weights_113_transpose_y_0"), val = bool(false)]; tensor attn_weights_113_cast_fp16 = matmul(transpose_x = attn_weights_113_transpose_x_0, transpose_y = attn_weights_113_transpose_y_0, x = var_3335_cast_fp16_0, y = var_3348_0)[name = string("attn_weights_113_cast_fp16")]; fp16 var_3351_to_fp16 = const()[name = string("op_3351_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_115_cast_fp16 = mul(x = attn_weights_113_cast_fp16, y = var_3351_to_fp16)[name = string("attn_weights_115_cast_fp16")]; tensor attn_weights_117_cast_fp16 = add(x = attn_weights_115_cast_fp16, y = attn_mask_1)[name = string("attn_weights_117_cast_fp16")]; int32 var_3355 = const()[name = string("op_3355"), val = int32(-2)]; tensor attn_weights_119_cast_fp16 = softmax(axis = var_3355, x = attn_weights_117_cast_fp16)[name = string("attn_weights_119_cast_fp16")]; bool var_3361_transpose_x_1 = const()[name = string("op_3361_transpose_x_1"), val = bool(true)]; bool var_3361_transpose_y_1 = const()[name = string("op_3361_transpose_y_1"), val = bool(false)]; tensor var_3361_cast_fp16 = matmul(transpose_x = var_3361_transpose_x_1, transpose_y = var_3361_transpose_y_1, x = attn_weights_119_cast_fp16, y = var_3345_cast_fp16_0)[name = string("op_3361_cast_fp16")]; bool attn_weights_121_transpose_x_0 = const()[name = string("attn_weights_121_transpose_x_0"), val = bool(false)]; bool attn_weights_121_transpose_y_0 = const()[name = string("attn_weights_121_transpose_y_0"), val = bool(false)]; tensor attn_weights_121_cast_fp16 = matmul(transpose_x = attn_weights_121_transpose_x_0, transpose_y = attn_weights_121_transpose_y_0, x = var_3335_cast_fp16_1, y = var_3348_1)[name = string("attn_weights_121_cast_fp16")]; fp16 var_3363_to_fp16 = const()[name = string("op_3363_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_123_cast_fp16 = mul(x = attn_weights_121_cast_fp16, y = var_3363_to_fp16)[name = string("attn_weights_123_cast_fp16")]; tensor attn_weights_125_cast_fp16 = add(x = attn_weights_123_cast_fp16, y = attn_mask_1)[name = string("attn_weights_125_cast_fp16")]; int32 var_3367 = const()[name = string("op_3367"), val = int32(-2)]; tensor attn_weights_127_cast_fp16 = softmax(axis = var_3367, x = attn_weights_125_cast_fp16)[name = string("attn_weights_127_cast_fp16")]; bool attn_output_57_transpose_x_1 = const()[name = string("attn_output_57_transpose_x_1"), val = bool(true)]; bool attn_output_57_transpose_y_1 = const()[name = string("attn_output_57_transpose_y_1"), val = bool(false)]; tensor attn_output_57_cast_fp16 = matmul(transpose_x = attn_output_57_transpose_x_1, transpose_y = attn_output_57_transpose_y_1, x = attn_weights_127_cast_fp16, y = var_3345_cast_fp16_1)[name = string("attn_output_57_cast_fp16")]; int32 var_3375 = const()[name = string("op_3375"), val = int32(1)]; bool attn_output_59_interleave_0 = const()[name = string("attn_output_59_interleave_0"), val = bool(false)]; tensor attn_output_59_cast_fp16 = concat(axis = var_3375, interleave = attn_output_59_interleave_0, values = (var_3361_cast_fp16, attn_output_57_cast_fp16))[name = string("attn_output_59_cast_fp16")]; tensor var_3379_perm_0 = const()[name = string("op_3379_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_95x = const()[name = string("concat_95x"), val = tensor([1, 2048, 1, -1])]; tensor var_3379_cast_fp16 = transpose(perm = var_3379_perm_0, x = attn_output_59_cast_fp16)[name = string("transpose_662")]; tensor attn_output_63_cast_fp16 = reshape(shape = concat_95x, x = var_3379_cast_fp16)[name = string("attn_output_63_cast_fp16")]; tensor hidden_states_73_strides_0 = const()[name = string("hidden_states_73_strides_0"), val = tensor([1, 1])]; string hidden_states_73_pad_type_0 = const()[name = string("hidden_states_73_pad_type_0"), val = string("valid")]; tensor hidden_states_73_pad_0 = const()[name = string("hidden_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_73_dilations_0 = const()[name = string("hidden_states_73_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_73_groups_0 = const()[name = string("hidden_states_73_groups_0"), val = int32(1)]; tensor hidden_states_73_cast_fp16 = conv(dilations = hidden_states_73_dilations_0, groups = hidden_states_73_groups_0, pad = hidden_states_73_pad_0, pad_type = hidden_states_73_pad_type_0, strides = hidden_states_73_strides_0, weight = layers_7_self_attn_o_proj_weight_cast_fp16, x = attn_output_63_cast_fp16)[name = string("hidden_states_73_cast_fp16")]; tensor hidden_states_75_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = hidden_states_73_cast_fp16)[name = string("hidden_states_75_cast_fp16")]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3412_cast_fp16 = mul(x = hidden_states_75_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_3412_cast_fp16")]; int32 var_3410 = const()[name = string("op_3410"), val = int32(1)]; bool doubled_61_interleave_0 = const()[name = string("doubled_61_interleave_0"), val = bool(false)]; tensor doubled_61_cast_fp16 = concat(axis = var_3410, interleave = doubled_61_interleave_0, values = (hidden_states_75_cast_fp16, var_3412_cast_fp16))[name = string("doubled_61_cast_fp16")]; tensor out_31_axes_0 = const()[name = string("out_31_axes_0"), val = tensor([1])]; tensor out_31_gamma_0_to_fp16 = const()[name = string("out_31_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336656640)))]; fp16 var_3422_to_fp16 = const()[name = string("op_3422_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_31_cast_fp16 = layer_norm(axes = out_31_axes_0, epsilon = var_3422_to_fp16, gamma = out_31_gamma_0_to_fp16, x = doubled_61_cast_fp16)[name = string("out_31_cast_fp16")]; tensor var_3433_split_sizes_0 = const()[name = string("op_3433_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3433_axis_0 = const()[name = string("op_3433_axis_0"), val = int32(1)]; tensor var_3433_cast_fp16_0, tensor var_3433_cast_fp16_1 = split(axis = var_3433_axis_0, split_sizes = var_3433_split_sizes_0, x = out_31_cast_fp16)[name = string("op_3433_cast_fp16")]; tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; tensor input_15_cast_fp16 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = layers_7_mlp_gate_proj_weight_cast_fp16, x = var_3433_cast_fp16_0)[name = string("input_15_cast_fp16")]; tensor var_3450_cast_fp16 = silu(x = input_15_cast_fp16)[name = string("op_3450_cast_fp16")]; tensor layers_7_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336664896)))]; tensor var_3456_strides_0 = const()[name = string("op_3456_strides_0"), val = tensor([1, 1])]; string var_3456_pad_type_0 = const()[name = string("op_3456_pad_type_0"), val = string("valid")]; tensor var_3456_pad_0 = const()[name = string("op_3456_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3456_dilations_0 = const()[name = string("op_3456_dilations_0"), val = tensor([1, 1])]; int32 var_3456_groups_0 = const()[name = string("op_3456_groups_0"), val = int32(1)]; tensor var_3456_cast_fp16 = conv(dilations = var_3456_dilations_0, groups = var_3456_groups_0, pad = var_3456_pad_0, pad_type = var_3456_pad_type_0, strides = var_3456_strides_0, weight = layers_7_mlp_up_proj_weight_to_fp16, x = var_3433_cast_fp16_0)[name = string("op_3456_cast_fp16")]; tensor x_79_cast_fp16 = mul(x = var_3450_cast_fp16, y = var_3456_cast_fp16)[name = string("x_79_cast_fp16")]; tensor layers_7_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1361830784)))]; tensor hidden_states_77_strides_0 = const()[name = string("hidden_states_77_strides_0"), val = tensor([1, 1])]; string hidden_states_77_pad_type_0 = const()[name = string("hidden_states_77_pad_type_0"), val = string("valid")]; tensor hidden_states_77_pad_0 = const()[name = string("hidden_states_77_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_77_dilations_0 = const()[name = string("hidden_states_77_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_77_groups_0 = const()[name = string("hidden_states_77_groups_0"), val = int32(1)]; tensor hidden_states_77_cast_fp16 = conv(dilations = hidden_states_77_dilations_0, groups = hidden_states_77_groups_0, pad = hidden_states_77_pad_0, pad_type = hidden_states_77_pad_type_0, strides = hidden_states_77_strides_0, weight = layers_7_mlp_down_proj_weight_to_fp16, x = x_79_cast_fp16)[name = string("hidden_states_77_cast_fp16")]; tensor hidden_states_79_cast_fp16 = add(x = hidden_states_75_cast_fp16, y = hidden_states_77_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3474_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_80_promoted_to_fp16)[name = string("op_3474_cast_fp16")]; int32 var_3472 = const()[name = string("op_3472"), val = int32(1)]; bool doubled_65_interleave_0 = const()[name = string("doubled_65_interleave_0"), val = bool(false)]; tensor doubled_65_cast_fp16 = concat(axis = var_3472, interleave = doubled_65_interleave_0, values = (hidden_states_79_cast_fp16, var_3474_cast_fp16))[name = string("doubled_65_cast_fp16")]; tensor out_33_axes_0 = const()[name = string("out_33_axes_0"), val = tensor([1])]; tensor out_33_gamma_0_to_fp16 = const()[name = string("out_33_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1386996672)))]; fp16 var_3484_to_fp16 = const()[name = string("op_3484_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_33_cast_fp16 = layer_norm(axes = out_33_axes_0, epsilon = var_3484_to_fp16, gamma = out_33_gamma_0_to_fp16, x = doubled_65_cast_fp16)[name = string("out_33_cast_fp16")]; tensor var_3495_split_sizes_0 = const()[name = string("op_3495_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3495_axis_0 = const()[name = string("op_3495_axis_0"), val = int32(1)]; tensor var_3495_cast_fp16_0, tensor var_3495_cast_fp16_1 = split(axis = var_3495_axis_0, split_sizes = var_3495_split_sizes_0, x = out_33_cast_fp16)[name = string("op_3495_cast_fp16")]; tensor layers_8_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1387004928)))]; tensor query_states_49_strides_0 = const()[name = string("query_states_49_strides_0"), val = tensor([1, 1])]; string query_states_49_pad_type_0 = const()[name = string("query_states_49_pad_type_0"), val = string("valid")]; tensor query_states_49_pad_0 = const()[name = string("query_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_49_dilations_0 = const()[name = string("query_states_49_dilations_0"), val = tensor([1, 1])]; int32 query_states_49_groups_0 = const()[name = string("query_states_49_groups_0"), val = int32(1)]; tensor query_states_49_cast_fp16 = conv(dilations = query_states_49_dilations_0, groups = query_states_49_groups_0, pad = query_states_49_pad_0, pad_type = query_states_49_pad_type_0, strides = query_states_49_strides_0, weight = layers_8_self_attn_q_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("query_states_49_cast_fp16")]; tensor layers_8_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1395393600)))]; tensor key_states_81_strides_0 = const()[name = string("key_states_81_strides_0"), val = tensor([1, 1])]; string key_states_81_pad_type_0 = const()[name = string("key_states_81_pad_type_0"), val = string("valid")]; tensor key_states_81_pad_0 = const()[name = string("key_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_81_dilations_0 = const()[name = string("key_states_81_dilations_0"), val = tensor([1, 1])]; int32 key_states_81_groups_0 = const()[name = string("key_states_81_groups_0"), val = int32(1)]; tensor key_states_81_cast_fp16 = conv(dilations = key_states_81_dilations_0, groups = key_states_81_groups_0, pad = key_states_81_pad_0, pad_type = key_states_81_pad_type_0, strides = key_states_81_strides_0, weight = layers_8_self_attn_k_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("key_states_81_cast_fp16")]; tensor value_states_49_strides_0 = const()[name = string("value_states_49_strides_0"), val = tensor([1, 1])]; string value_states_49_pad_type_0 = const()[name = string("value_states_49_pad_type_0"), val = string("valid")]; tensor value_states_49_pad_0 = const()[name = string("value_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_49_dilations_0 = const()[name = string("value_states_49_dilations_0"), val = tensor([1, 1])]; int32 value_states_49_groups_0 = const()[name = string("value_states_49_groups_0"), val = int32(1)]; tensor value_states_49_cast_fp16 = conv(dilations = value_states_49_dilations_0, groups = value_states_49_groups_0, pad = value_states_49_pad_0, pad_type = value_states_49_pad_type_0, strides = value_states_49_strides_0, weight = layers_8_self_attn_v_proj_weight_cast_fp16, x = var_3495_cast_fp16_0)[name = string("value_states_49_cast_fp16")]; tensor concat_96x = const()[name = string("concat_96x"), val = tensor([1, 16, 128, -1])]; tensor x_81_cast_fp16 = reshape(shape = concat_96x, x = query_states_49_cast_fp16)[name = string("x_81_cast_fp16")]; tensor concat_97x = const()[name = string("concat_97x"), val = tensor([1, 2, 128, -1])]; tensor var_3552_cast_fp16 = reshape(shape = concat_97x, x = key_states_81_cast_fp16)[name = string("op_3552_cast_fp16")]; tensor concat_98x = const()[name = string("concat_98x"), val = tensor([1, 2, 128, -1])]; tensor var_3559_cast_fp16 = reshape(shape = concat_98x, x = value_states_49_cast_fp16)[name = string("op_3559_cast_fp16")]; tensor var_3563_cast_fp16 = mul(x = x_81_cast_fp16, y = var_869_cast_fp16)[name = string("op_3563_cast_fp16")]; tensor var_3564_split_sizes_0 = const()[name = string("op_3564_split_sizes_0"), val = tensor([64, 64])]; int32 var_3564_axis_0 = const()[name = string("op_3564_axis_0"), val = int32(-2)]; tensor var_3564_cast_fp16_0, tensor var_3564_cast_fp16_1 = split(axis = var_3564_axis_0, split_sizes = var_3564_split_sizes_0, x = x_81_cast_fp16)[name = string("op_3564_cast_fp16")]; fp16 const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3566_cast_fp16 = mul(x = var_3564_cast_fp16_1, y = const_82_promoted_to_fp16)[name = string("op_3566_cast_fp16")]; int32 var_3568 = const()[name = string("op_3568"), val = int32(-2)]; bool var_3569_interleave_0 = const()[name = string("op_3569_interleave_0"), val = bool(false)]; tensor var_3569_cast_fp16 = concat(axis = var_3568, interleave = var_3569_interleave_0, values = (var_3566_cast_fp16, var_3564_cast_fp16_0))[name = string("op_3569_cast_fp16")]; tensor var_3570_cast_fp16 = mul(x = var_3569_cast_fp16, y = var_878_cast_fp16)[name = string("op_3570_cast_fp16")]; tensor query_states_51_cast_fp16 = add(x = var_3563_cast_fp16, y = var_3570_cast_fp16)[name = string("query_states_51_cast_fp16")]; tensor var_3576_cast_fp16 = mul(x = var_3552_cast_fp16, y = var_869_cast_fp16)[name = string("op_3576_cast_fp16")]; tensor var_3577_split_sizes_0 = const()[name = string("op_3577_split_sizes_0"), val = tensor([64, 64])]; int32 var_3577_axis_0 = const()[name = string("op_3577_axis_0"), val = int32(-2)]; tensor var_3577_cast_fp16_0, tensor var_3577_cast_fp16_1 = split(axis = var_3577_axis_0, split_sizes = var_3577_split_sizes_0, x = var_3552_cast_fp16)[name = string("op_3577_cast_fp16")]; fp16 const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3579_cast_fp16 = mul(x = var_3577_cast_fp16_1, y = const_83_promoted_to_fp16)[name = string("op_3579_cast_fp16")]; int32 var_3581 = const()[name = string("op_3581"), val = int32(-2)]; bool var_3582_interleave_0 = const()[name = string("op_3582_interleave_0"), val = bool(false)]; tensor var_3582_cast_fp16 = concat(axis = var_3581, interleave = var_3582_interleave_0, values = (var_3579_cast_fp16, var_3577_cast_fp16_0))[name = string("op_3582_cast_fp16")]; tensor var_3583_cast_fp16 = mul(x = var_3582_cast_fp16, y = var_878_cast_fp16)[name = string("op_3583_cast_fp16")]; tensor key_states_85_cast_fp16 = add(x = var_3576_cast_fp16, y = var_3583_cast_fp16)[name = string("key_states_85_cast_fp16")]; tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; int32 concat_101_axis_0 = const()[name = string("concat_101_axis_0"), val = int32(0)]; bool concat_101_interleave_0 = const()[name = string("concat_101_interleave_0"), val = bool(false)]; tensor concat_101 = concat(axis = concat_101_axis_0, interleave = concat_101_interleave_0, values = (expand_dims_96, expand_dims_97, position_id, expand_dims_99))[name = string("concat_101")]; tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; tensor concat_102_values1_0 = const()[name = string("concat_102_values1_0"), val = tensor([0])]; tensor concat_102_values3_0 = const()[name = string("concat_102_values3_0"), val = tensor([0])]; int32 concat_102_axis_0 = const()[name = string("concat_102_axis_0"), val = int32(0)]; bool concat_102_interleave_0 = const()[name = string("concat_102_interleave_0"), val = bool(false)]; tensor concat_102 = concat(axis = concat_102_axis_0, interleave = concat_102_interleave_0, values = (expand_dims_100, concat_102_values1_0, cache_position_end, concat_102_values3_0))[name = string("concat_102")]; tensor key_states_87_perm_0 = const()[name = string("key_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_9_stride_0 = const()[name = string("key_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_87_cast_fp16 = transpose(perm = key_states_87_perm_0, x = key_states_85_cast_fp16)[name = string("transpose_661")]; tensor key_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = key_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = key_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_9_squeeze_mask_0, stride = key_cache_internal_tensor_assign_9_stride_0, update = key_states_87_cast_fp16, x = coreml_update_state_406)[name = string("key_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_9_cast_fp16, input = key_cache)[name = string("coreml_update_state_408_write_state")]; tensor coreml_update_state_408 = read_state(input = key_cache)[name = string("coreml_update_state_408")]; tensor value_states_51_perm_0 = const()[name = string("value_states_51_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_9_stride_0 = const()[name = string("value_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_51_cast_fp16 = transpose(perm = value_states_51_perm_0, x = var_3559_cast_fp16)[name = string("transpose_660")]; tensor value_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = value_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = value_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_9_squeeze_mask_0, stride = value_cache_internal_tensor_assign_9_stride_0, update = value_states_51_cast_fp16, x = coreml_update_state_407)[name = string("value_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_9_cast_fp16, input = value_cache)[name = string("coreml_update_state_409_write_state")]; tensor coreml_update_state_409 = read_state(input = value_cache)[name = string("coreml_update_state_409")]; tensor var_3653_begin_0 = const()[name = string("op_3653_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3653_end_0 = const()[name = string("op_3653_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3653_end_mask_0 = const()[name = string("op_3653_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3653_cast_fp16 = slice_by_index(begin = var_3653_begin_0, end = var_3653_end_0, end_mask = var_3653_end_mask_0, x = coreml_update_state_408)[name = string("op_3653_cast_fp16")]; tensor tile_16 = const()[name = string("tile_16"), val = tensor([1, 1])]; int32 var_3656_axis_0 = const()[name = string("op_3656_axis_0"), val = int32(1)]; tensor var_3656_cast_fp16_0, tensor var_3656_cast_fp16_1 = split(axis = var_3656_axis_0, split_sizes = tile_16, x = var_3653_cast_fp16)[name = string("op_3656_cast_fp16")]; tensor var_3663_begin_0 = const()[name = string("op_3663_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3663_end_0 = const()[name = string("op_3663_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3663_end_mask_0 = const()[name = string("op_3663_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3663_cast_fp16 = slice_by_index(begin = var_3663_begin_0, end = var_3663_end_0, end_mask = var_3663_end_mask_0, x = coreml_update_state_409)[name = string("op_3663_cast_fp16")]; tensor tile_17 = const()[name = string("tile_17"), val = tensor([1, 1])]; int32 var_3666_axis_0 = const()[name = string("op_3666_axis_0"), val = int32(1)]; tensor var_3666_cast_fp16_0, tensor var_3666_cast_fp16_1 = split(axis = var_3666_axis_0, split_sizes = tile_17, x = var_3663_cast_fp16)[name = string("op_3666_cast_fp16")]; tensor var_3669_split_sizes_0 = const()[name = string("op_3669_split_sizes_0"), val = tensor([8, 8])]; int32 var_3669_axis_0 = const()[name = string("op_3669_axis_0"), val = int32(1)]; tensor var_3669_0, tensor var_3669_1 = split(axis = var_3669_axis_0, split_sizes = var_3669_split_sizes_0, x = query_states_51_cast_fp16)[name = string("op_3669")]; bool attn_weights_129_transpose_x_0 = const()[name = string("attn_weights_129_transpose_x_0"), val = bool(false)]; bool attn_weights_129_transpose_y_0 = const()[name = string("attn_weights_129_transpose_y_0"), val = bool(false)]; tensor attn_weights_129_cast_fp16 = matmul(transpose_x = attn_weights_129_transpose_x_0, transpose_y = attn_weights_129_transpose_y_0, x = var_3656_cast_fp16_0, y = var_3669_0)[name = string("attn_weights_129_cast_fp16")]; fp16 var_3672_to_fp16 = const()[name = string("op_3672_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_131_cast_fp16 = mul(x = attn_weights_129_cast_fp16, y = var_3672_to_fp16)[name = string("attn_weights_131_cast_fp16")]; tensor attn_weights_133_cast_fp16 = add(x = attn_weights_131_cast_fp16, y = attn_mask_1)[name = string("attn_weights_133_cast_fp16")]; int32 var_3676 = const()[name = string("op_3676"), val = int32(-2)]; tensor attn_weights_135_cast_fp16 = softmax(axis = var_3676, x = attn_weights_133_cast_fp16)[name = string("attn_weights_135_cast_fp16")]; bool var_3682_transpose_x_1 = const()[name = string("op_3682_transpose_x_1"), val = bool(true)]; bool var_3682_transpose_y_1 = const()[name = string("op_3682_transpose_y_1"), val = bool(false)]; tensor var_3682_cast_fp16 = matmul(transpose_x = var_3682_transpose_x_1, transpose_y = var_3682_transpose_y_1, x = attn_weights_135_cast_fp16, y = var_3666_cast_fp16_0)[name = string("op_3682_cast_fp16")]; bool attn_weights_137_transpose_x_0 = const()[name = string("attn_weights_137_transpose_x_0"), val = bool(false)]; bool attn_weights_137_transpose_y_0 = const()[name = string("attn_weights_137_transpose_y_0"), val = bool(false)]; tensor attn_weights_137_cast_fp16 = matmul(transpose_x = attn_weights_137_transpose_x_0, transpose_y = attn_weights_137_transpose_y_0, x = var_3656_cast_fp16_1, y = var_3669_1)[name = string("attn_weights_137_cast_fp16")]; fp16 var_3684_to_fp16 = const()[name = string("op_3684_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_139_cast_fp16 = mul(x = attn_weights_137_cast_fp16, y = var_3684_to_fp16)[name = string("attn_weights_139_cast_fp16")]; tensor attn_weights_141_cast_fp16 = add(x = attn_weights_139_cast_fp16, y = attn_mask_1)[name = string("attn_weights_141_cast_fp16")]; int32 var_3688 = const()[name = string("op_3688"), val = int32(-2)]; tensor attn_weights_143_cast_fp16 = softmax(axis = var_3688, x = attn_weights_141_cast_fp16)[name = string("attn_weights_143_cast_fp16")]; bool attn_output_65_transpose_x_1 = const()[name = string("attn_output_65_transpose_x_1"), val = bool(true)]; bool attn_output_65_transpose_y_1 = const()[name = string("attn_output_65_transpose_y_1"), val = bool(false)]; tensor attn_output_65_cast_fp16 = matmul(transpose_x = attn_output_65_transpose_x_1, transpose_y = attn_output_65_transpose_y_1, x = attn_weights_143_cast_fp16, y = var_3666_cast_fp16_1)[name = string("attn_output_65_cast_fp16")]; int32 var_3696 = const()[name = string("op_3696"), val = int32(1)]; bool attn_output_67_interleave_0 = const()[name = string("attn_output_67_interleave_0"), val = bool(false)]; tensor attn_output_67_cast_fp16 = concat(axis = var_3696, interleave = attn_output_67_interleave_0, values = (var_3682_cast_fp16, attn_output_65_cast_fp16))[name = string("attn_output_67_cast_fp16")]; tensor var_3700_perm_0 = const()[name = string("op_3700_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_107x = const()[name = string("concat_107x"), val = tensor([1, 2048, 1, -1])]; tensor var_3700_cast_fp16 = transpose(perm = var_3700_perm_0, x = attn_output_67_cast_fp16)[name = string("transpose_659")]; tensor attn_output_71_cast_fp16 = reshape(shape = concat_107x, x = var_3700_cast_fp16)[name = string("attn_output_71_cast_fp16")]; tensor hidden_states_83_strides_0 = const()[name = string("hidden_states_83_strides_0"), val = tensor([1, 1])]; string hidden_states_83_pad_type_0 = const()[name = string("hidden_states_83_pad_type_0"), val = string("valid")]; tensor hidden_states_83_pad_0 = const()[name = string("hidden_states_83_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_83_dilations_0 = const()[name = string("hidden_states_83_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_83_groups_0 = const()[name = string("hidden_states_83_groups_0"), val = int32(1)]; tensor hidden_states_83_cast_fp16 = conv(dilations = hidden_states_83_dilations_0, groups = hidden_states_83_groups_0, pad = hidden_states_83_pad_0, pad_type = hidden_states_83_pad_type_0, strides = hidden_states_83_strides_0, weight = layers_8_self_attn_o_proj_weight_cast_fp16, x = attn_output_71_cast_fp16)[name = string("hidden_states_83_cast_fp16")]; tensor hidden_states_85_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = hidden_states_83_cast_fp16)[name = string("hidden_states_85_cast_fp16")]; fp16 const_88_promoted_to_fp16 = const()[name = string("const_88_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3733_cast_fp16 = mul(x = hidden_states_85_cast_fp16, y = const_88_promoted_to_fp16)[name = string("op_3733_cast_fp16")]; int32 var_3731 = const()[name = string("op_3731"), val = int32(1)]; bool doubled_69_interleave_0 = const()[name = string("doubled_69_interleave_0"), val = bool(false)]; tensor doubled_69_cast_fp16 = concat(axis = var_3731, interleave = doubled_69_interleave_0, values = (hidden_states_85_cast_fp16, var_3733_cast_fp16))[name = string("doubled_69_cast_fp16")]; tensor out_35_axes_0 = const()[name = string("out_35_axes_0"), val = tensor([1])]; tensor out_35_gamma_0_to_fp16 = const()[name = string("out_35_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396442240)))]; fp16 var_3743_to_fp16 = const()[name = string("op_3743_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_35_cast_fp16 = layer_norm(axes = out_35_axes_0, epsilon = var_3743_to_fp16, gamma = out_35_gamma_0_to_fp16, x = doubled_69_cast_fp16)[name = string("out_35_cast_fp16")]; tensor var_3754_split_sizes_0 = const()[name = string("op_3754_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3754_axis_0 = const()[name = string("op_3754_axis_0"), val = int32(1)]; tensor var_3754_cast_fp16_0, tensor var_3754_cast_fp16_1 = split(axis = var_3754_axis_0, split_sizes = var_3754_split_sizes_0, x = out_35_cast_fp16)[name = string("op_3754_cast_fp16")]; tensor input_17_strides_0 = const()[name = string("input_17_strides_0"), val = tensor([1, 1])]; string input_17_pad_type_0 = const()[name = string("input_17_pad_type_0"), val = string("valid")]; tensor input_17_pad_0 = const()[name = string("input_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_17_dilations_0 = const()[name = string("input_17_dilations_0"), val = tensor([1, 1])]; int32 input_17_groups_0 = const()[name = string("input_17_groups_0"), val = int32(1)]; tensor input_17_cast_fp16 = conv(dilations = input_17_dilations_0, groups = input_17_groups_0, pad = input_17_pad_0, pad_type = input_17_pad_type_0, strides = input_17_strides_0, weight = layers_8_mlp_gate_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("input_17_cast_fp16")]; tensor var_3771_cast_fp16 = silu(x = input_17_cast_fp16)[name = string("op_3771_cast_fp16")]; tensor var_3777_strides_0 = const()[name = string("op_3777_strides_0"), val = tensor([1, 1])]; string var_3777_pad_type_0 = const()[name = string("op_3777_pad_type_0"), val = string("valid")]; tensor var_3777_pad_0 = const()[name = string("op_3777_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3777_dilations_0 = const()[name = string("op_3777_dilations_0"), val = tensor([1, 1])]; int32 var_3777_groups_0 = const()[name = string("op_3777_groups_0"), val = int32(1)]; tensor var_3777_cast_fp16 = conv(dilations = var_3777_dilations_0, groups = var_3777_groups_0, pad = var_3777_pad_0, pad_type = var_3777_pad_type_0, strides = var_3777_strides_0, weight = layers_8_mlp_up_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("op_3777_cast_fp16")]; tensor x_89_cast_fp16 = mul(x = var_3771_cast_fp16, y = var_3777_cast_fp16)[name = string("x_89_cast_fp16")]; tensor hidden_states_87_strides_0 = const()[name = string("hidden_states_87_strides_0"), val = tensor([1, 1])]; string hidden_states_87_pad_type_0 = const()[name = string("hidden_states_87_pad_type_0"), val = string("valid")]; tensor hidden_states_87_pad_0 = const()[name = string("hidden_states_87_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_87_dilations_0 = const()[name = string("hidden_states_87_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_87_groups_0 = const()[name = string("hidden_states_87_groups_0"), val = int32(1)]; tensor hidden_states_87_cast_fp16 = conv(dilations = hidden_states_87_dilations_0, groups = hidden_states_87_groups_0, pad = hidden_states_87_pad_0, pad_type = hidden_states_87_pad_type_0, strides = hidden_states_87_strides_0, weight = layers_8_mlp_down_proj_weight_cast_fp16, x = x_89_cast_fp16)[name = string("hidden_states_87_cast_fp16")]; tensor hidden_states_89_cast_fp16 = add(x = hidden_states_85_cast_fp16, y = hidden_states_87_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3795_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_3795_cast_fp16")]; int32 var_3793 = const()[name = string("op_3793"), val = int32(1)]; bool doubled_73_interleave_0 = const()[name = string("doubled_73_interleave_0"), val = bool(false)]; tensor doubled_73_cast_fp16 = concat(axis = var_3793, interleave = doubled_73_interleave_0, values = (hidden_states_89_cast_fp16, var_3795_cast_fp16))[name = string("doubled_73_cast_fp16")]; tensor out_37_axes_0 = const()[name = string("out_37_axes_0"), val = tensor([1])]; tensor out_37_gamma_0_to_fp16 = const()[name = string("out_37_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396450496)))]; fp16 var_3805_to_fp16 = const()[name = string("op_3805_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_37_cast_fp16 = layer_norm(axes = out_37_axes_0, epsilon = var_3805_to_fp16, gamma = out_37_gamma_0_to_fp16, x = doubled_73_cast_fp16)[name = string("out_37_cast_fp16")]; tensor var_3816_split_sizes_0 = const()[name = string("op_3816_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3816_axis_0 = const()[name = string("op_3816_axis_0"), val = int32(1)]; tensor var_3816_cast_fp16_0, tensor var_3816_cast_fp16_1 = split(axis = var_3816_axis_0, split_sizes = var_3816_split_sizes_0, x = out_37_cast_fp16)[name = string("op_3816_cast_fp16")]; tensor layers_9_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396458752)))]; tensor query_states_55_strides_0 = const()[name = string("query_states_55_strides_0"), val = tensor([1, 1])]; string query_states_55_pad_type_0 = const()[name = string("query_states_55_pad_type_0"), val = string("valid")]; tensor query_states_55_pad_0 = const()[name = string("query_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_55_dilations_0 = const()[name = string("query_states_55_dilations_0"), val = tensor([1, 1])]; int32 query_states_55_groups_0 = const()[name = string("query_states_55_groups_0"), val = int32(1)]; tensor query_states_55_cast_fp16 = conv(dilations = query_states_55_dilations_0, groups = query_states_55_groups_0, pad = query_states_55_pad_0, pad_type = query_states_55_pad_type_0, strides = query_states_55_strides_0, weight = layers_9_self_attn_q_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("query_states_55_cast_fp16")]; tensor layers_9_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1404847424)))]; tensor key_states_91_strides_0 = const()[name = string("key_states_91_strides_0"), val = tensor([1, 1])]; string key_states_91_pad_type_0 = const()[name = string("key_states_91_pad_type_0"), val = string("valid")]; tensor key_states_91_pad_0 = const()[name = string("key_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_91_dilations_0 = const()[name = string("key_states_91_dilations_0"), val = tensor([1, 1])]; int32 key_states_91_groups_0 = const()[name = string("key_states_91_groups_0"), val = int32(1)]; tensor key_states_91_cast_fp16 = conv(dilations = key_states_91_dilations_0, groups = key_states_91_groups_0, pad = key_states_91_pad_0, pad_type = key_states_91_pad_type_0, strides = key_states_91_strides_0, weight = layers_9_self_attn_k_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("key_states_91_cast_fp16")]; tensor value_states_55_strides_0 = const()[name = string("value_states_55_strides_0"), val = tensor([1, 1])]; string value_states_55_pad_type_0 = const()[name = string("value_states_55_pad_type_0"), val = string("valid")]; tensor value_states_55_pad_0 = const()[name = string("value_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_55_dilations_0 = const()[name = string("value_states_55_dilations_0"), val = tensor([1, 1])]; int32 value_states_55_groups_0 = const()[name = string("value_states_55_groups_0"), val = int32(1)]; tensor value_states_55_cast_fp16 = conv(dilations = value_states_55_dilations_0, groups = value_states_55_groups_0, pad = value_states_55_pad_0, pad_type = value_states_55_pad_type_0, strides = value_states_55_strides_0, weight = layers_9_self_attn_v_proj_weight_cast_fp16, x = var_3816_cast_fp16_0)[name = string("value_states_55_cast_fp16")]; tensor concat_108x = const()[name = string("concat_108x"), val = tensor([1, 16, 128, -1])]; tensor x_91_cast_fp16 = reshape(shape = concat_108x, x = query_states_55_cast_fp16)[name = string("x_91_cast_fp16")]; tensor concat_109x = const()[name = string("concat_109x"), val = tensor([1, 2, 128, -1])]; tensor var_3873_cast_fp16 = reshape(shape = concat_109x, x = key_states_91_cast_fp16)[name = string("op_3873_cast_fp16")]; tensor concat_110x = const()[name = string("concat_110x"), val = tensor([1, 2, 128, -1])]; tensor var_3880_cast_fp16 = reshape(shape = concat_110x, x = value_states_55_cast_fp16)[name = string("op_3880_cast_fp16")]; tensor var_3884_cast_fp16 = mul(x = x_91_cast_fp16, y = var_869_cast_fp16)[name = string("op_3884_cast_fp16")]; tensor var_3885_split_sizes_0 = const()[name = string("op_3885_split_sizes_0"), val = tensor([64, 64])]; int32 var_3885_axis_0 = const()[name = string("op_3885_axis_0"), val = int32(-2)]; tensor var_3885_cast_fp16_0, tensor var_3885_cast_fp16_1 = split(axis = var_3885_axis_0, split_sizes = var_3885_split_sizes_0, x = x_91_cast_fp16)[name = string("op_3885_cast_fp16")]; fp16 const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3887_cast_fp16 = mul(x = var_3885_cast_fp16_1, y = const_92_promoted_to_fp16)[name = string("op_3887_cast_fp16")]; int32 var_3889 = const()[name = string("op_3889"), val = int32(-2)]; bool var_3890_interleave_0 = const()[name = string("op_3890_interleave_0"), val = bool(false)]; tensor var_3890_cast_fp16 = concat(axis = var_3889, interleave = var_3890_interleave_0, values = (var_3887_cast_fp16, var_3885_cast_fp16_0))[name = string("op_3890_cast_fp16")]; tensor var_3891_cast_fp16 = mul(x = var_3890_cast_fp16, y = var_878_cast_fp16)[name = string("op_3891_cast_fp16")]; tensor query_states_57_cast_fp16 = add(x = var_3884_cast_fp16, y = var_3891_cast_fp16)[name = string("query_states_57_cast_fp16")]; tensor var_3897_cast_fp16 = mul(x = var_3873_cast_fp16, y = var_869_cast_fp16)[name = string("op_3897_cast_fp16")]; tensor var_3898_split_sizes_0 = const()[name = string("op_3898_split_sizes_0"), val = tensor([64, 64])]; int32 var_3898_axis_0 = const()[name = string("op_3898_axis_0"), val = int32(-2)]; tensor var_3898_cast_fp16_0, tensor var_3898_cast_fp16_1 = split(axis = var_3898_axis_0, split_sizes = var_3898_split_sizes_0, x = var_3873_cast_fp16)[name = string("op_3898_cast_fp16")]; fp16 const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3900_cast_fp16 = mul(x = var_3898_cast_fp16_1, y = const_93_promoted_to_fp16)[name = string("op_3900_cast_fp16")]; int32 var_3902 = const()[name = string("op_3902"), val = int32(-2)]; bool var_3903_interleave_0 = const()[name = string("op_3903_interleave_0"), val = bool(false)]; tensor var_3903_cast_fp16 = concat(axis = var_3902, interleave = var_3903_interleave_0, values = (var_3900_cast_fp16, var_3898_cast_fp16_0))[name = string("op_3903_cast_fp16")]; tensor var_3904_cast_fp16 = mul(x = var_3903_cast_fp16, y = var_878_cast_fp16)[name = string("op_3904_cast_fp16")]; tensor key_states_95_cast_fp16 = add(x = var_3897_cast_fp16, y = var_3904_cast_fp16)[name = string("key_states_95_cast_fp16")]; tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; int32 concat_113_axis_0 = const()[name = string("concat_113_axis_0"), val = int32(0)]; bool concat_113_interleave_0 = const()[name = string("concat_113_interleave_0"), val = bool(false)]; tensor concat_113 = concat(axis = concat_113_axis_0, interleave = concat_113_interleave_0, values = (expand_dims_108, expand_dims_109, position_id, expand_dims_111))[name = string("concat_113")]; tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; tensor concat_114_values1_0 = const()[name = string("concat_114_values1_0"), val = tensor([0])]; tensor concat_114_values3_0 = const()[name = string("concat_114_values3_0"), val = tensor([0])]; int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_112, concat_114_values1_0, cache_position_end, concat_114_values3_0))[name = string("concat_114")]; tensor key_states_97_perm_0 = const()[name = string("key_states_97_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_10_stride_0 = const()[name = string("key_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_97_cast_fp16 = transpose(perm = key_states_97_perm_0, x = key_states_95_cast_fp16)[name = string("transpose_658")]; tensor key_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = key_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = key_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_10_squeeze_mask_0, stride = key_cache_internal_tensor_assign_10_stride_0, update = key_states_97_cast_fp16, x = coreml_update_state_408)[name = string("key_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_10_cast_fp16, input = key_cache)[name = string("coreml_update_state_410_write_state")]; tensor coreml_update_state_410 = read_state(input = key_cache)[name = string("coreml_update_state_410")]; tensor value_states_57_perm_0 = const()[name = string("value_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_10_stride_0 = const()[name = string("value_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_57_cast_fp16 = transpose(perm = value_states_57_perm_0, x = var_3880_cast_fp16)[name = string("transpose_657")]; tensor value_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = value_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = value_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_10_squeeze_mask_0, stride = value_cache_internal_tensor_assign_10_stride_0, update = value_states_57_cast_fp16, x = coreml_update_state_409)[name = string("value_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_10_cast_fp16, input = value_cache)[name = string("coreml_update_state_411_write_state")]; tensor coreml_update_state_411 = read_state(input = value_cache)[name = string("coreml_update_state_411")]; tensor var_3974_begin_0 = const()[name = string("op_3974_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3974_end_0 = const()[name = string("op_3974_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3974_end_mask_0 = const()[name = string("op_3974_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3974_cast_fp16 = slice_by_index(begin = var_3974_begin_0, end = var_3974_end_0, end_mask = var_3974_end_mask_0, x = coreml_update_state_410)[name = string("op_3974_cast_fp16")]; tensor tile_18 = const()[name = string("tile_18"), val = tensor([1, 1])]; int32 var_3977_axis_0 = const()[name = string("op_3977_axis_0"), val = int32(1)]; tensor var_3977_cast_fp16_0, tensor var_3977_cast_fp16_1 = split(axis = var_3977_axis_0, split_sizes = tile_18, x = var_3974_cast_fp16)[name = string("op_3977_cast_fp16")]; tensor var_3984_begin_0 = const()[name = string("op_3984_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3984_end_0 = const()[name = string("op_3984_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3984_end_mask_0 = const()[name = string("op_3984_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3984_cast_fp16 = slice_by_index(begin = var_3984_begin_0, end = var_3984_end_0, end_mask = var_3984_end_mask_0, x = coreml_update_state_411)[name = string("op_3984_cast_fp16")]; tensor tile_19 = const()[name = string("tile_19"), val = tensor([1, 1])]; int32 var_3987_axis_0 = const()[name = string("op_3987_axis_0"), val = int32(1)]; tensor var_3987_cast_fp16_0, tensor var_3987_cast_fp16_1 = split(axis = var_3987_axis_0, split_sizes = tile_19, x = var_3984_cast_fp16)[name = string("op_3987_cast_fp16")]; tensor var_3990_split_sizes_0 = const()[name = string("op_3990_split_sizes_0"), val = tensor([8, 8])]; int32 var_3990_axis_0 = const()[name = string("op_3990_axis_0"), val = int32(1)]; tensor var_3990_0, tensor var_3990_1 = split(axis = var_3990_axis_0, split_sizes = var_3990_split_sizes_0, x = query_states_57_cast_fp16)[name = string("op_3990")]; bool attn_weights_145_transpose_x_0 = const()[name = string("attn_weights_145_transpose_x_0"), val = bool(false)]; bool attn_weights_145_transpose_y_0 = const()[name = string("attn_weights_145_transpose_y_0"), val = bool(false)]; tensor attn_weights_145_cast_fp16 = matmul(transpose_x = attn_weights_145_transpose_x_0, transpose_y = attn_weights_145_transpose_y_0, x = var_3977_cast_fp16_0, y = var_3990_0)[name = string("attn_weights_145_cast_fp16")]; fp16 var_3993_to_fp16 = const()[name = string("op_3993_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_147_cast_fp16 = mul(x = attn_weights_145_cast_fp16, y = var_3993_to_fp16)[name = string("attn_weights_147_cast_fp16")]; tensor attn_weights_149_cast_fp16 = add(x = attn_weights_147_cast_fp16, y = attn_mask_1)[name = string("attn_weights_149_cast_fp16")]; int32 var_3997 = const()[name = string("op_3997"), val = int32(-2)]; tensor attn_weights_151_cast_fp16 = softmax(axis = var_3997, x = attn_weights_149_cast_fp16)[name = string("attn_weights_151_cast_fp16")]; bool var_4003_transpose_x_1 = const()[name = string("op_4003_transpose_x_1"), val = bool(true)]; bool var_4003_transpose_y_1 = const()[name = string("op_4003_transpose_y_1"), val = bool(false)]; tensor var_4003_cast_fp16 = matmul(transpose_x = var_4003_transpose_x_1, transpose_y = var_4003_transpose_y_1, x = attn_weights_151_cast_fp16, y = var_3987_cast_fp16_0)[name = string("op_4003_cast_fp16")]; bool attn_weights_153_transpose_x_0 = const()[name = string("attn_weights_153_transpose_x_0"), val = bool(false)]; bool attn_weights_153_transpose_y_0 = const()[name = string("attn_weights_153_transpose_y_0"), val = bool(false)]; tensor attn_weights_153_cast_fp16 = matmul(transpose_x = attn_weights_153_transpose_x_0, transpose_y = attn_weights_153_transpose_y_0, x = var_3977_cast_fp16_1, y = var_3990_1)[name = string("attn_weights_153_cast_fp16")]; fp16 var_4005_to_fp16 = const()[name = string("op_4005_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_155_cast_fp16 = mul(x = attn_weights_153_cast_fp16, y = var_4005_to_fp16)[name = string("attn_weights_155_cast_fp16")]; tensor attn_weights_157_cast_fp16 = add(x = attn_weights_155_cast_fp16, y = attn_mask_1)[name = string("attn_weights_157_cast_fp16")]; int32 var_4009 = const()[name = string("op_4009"), val = int32(-2)]; tensor attn_weights_159_cast_fp16 = softmax(axis = var_4009, x = attn_weights_157_cast_fp16)[name = string("attn_weights_159_cast_fp16")]; bool attn_output_73_transpose_x_1 = const()[name = string("attn_output_73_transpose_x_1"), val = bool(true)]; bool attn_output_73_transpose_y_1 = const()[name = string("attn_output_73_transpose_y_1"), val = bool(false)]; tensor attn_output_73_cast_fp16 = matmul(transpose_x = attn_output_73_transpose_x_1, transpose_y = attn_output_73_transpose_y_1, x = attn_weights_159_cast_fp16, y = var_3987_cast_fp16_1)[name = string("attn_output_73_cast_fp16")]; int32 var_4017 = const()[name = string("op_4017"), val = int32(1)]; bool attn_output_75_interleave_0 = const()[name = string("attn_output_75_interleave_0"), val = bool(false)]; tensor attn_output_75_cast_fp16 = concat(axis = var_4017, interleave = attn_output_75_interleave_0, values = (var_4003_cast_fp16, attn_output_73_cast_fp16))[name = string("attn_output_75_cast_fp16")]; tensor var_4021_perm_0 = const()[name = string("op_4021_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_119x = const()[name = string("concat_119x"), val = tensor([1, 2048, 1, -1])]; tensor var_4021_cast_fp16 = transpose(perm = var_4021_perm_0, x = attn_output_75_cast_fp16)[name = string("transpose_656")]; tensor attn_output_79_cast_fp16 = reshape(shape = concat_119x, x = var_4021_cast_fp16)[name = string("attn_output_79_cast_fp16")]; tensor hidden_states_93_strides_0 = const()[name = string("hidden_states_93_strides_0"), val = tensor([1, 1])]; string hidden_states_93_pad_type_0 = const()[name = string("hidden_states_93_pad_type_0"), val = string("valid")]; tensor hidden_states_93_pad_0 = const()[name = string("hidden_states_93_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_93_dilations_0 = const()[name = string("hidden_states_93_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_93_groups_0 = const()[name = string("hidden_states_93_groups_0"), val = int32(1)]; tensor hidden_states_93_cast_fp16 = conv(dilations = hidden_states_93_dilations_0, groups = hidden_states_93_groups_0, pad = hidden_states_93_pad_0, pad_type = hidden_states_93_pad_type_0, strides = hidden_states_93_strides_0, weight = layers_9_self_attn_o_proj_weight_cast_fp16, x = attn_output_79_cast_fp16)[name = string("hidden_states_93_cast_fp16")]; tensor hidden_states_95_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = hidden_states_93_cast_fp16)[name = string("hidden_states_95_cast_fp16")]; fp16 const_98_promoted_to_fp16 = const()[name = string("const_98_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4054_cast_fp16 = mul(x = hidden_states_95_cast_fp16, y = const_98_promoted_to_fp16)[name = string("op_4054_cast_fp16")]; int32 var_4052 = const()[name = string("op_4052"), val = int32(1)]; bool doubled_77_interleave_0 = const()[name = string("doubled_77_interleave_0"), val = bool(false)]; tensor doubled_77_cast_fp16 = concat(axis = var_4052, interleave = doubled_77_interleave_0, values = (hidden_states_95_cast_fp16, var_4054_cast_fp16))[name = string("doubled_77_cast_fp16")]; tensor out_39_axes_0 = const()[name = string("out_39_axes_0"), val = tensor([1])]; tensor out_39_gamma_0_to_fp16 = const()[name = string("out_39_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405896064)))]; fp16 var_4064_to_fp16 = const()[name = string("op_4064_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_39_cast_fp16 = layer_norm(axes = out_39_axes_0, epsilon = var_4064_to_fp16, gamma = out_39_gamma_0_to_fp16, x = doubled_77_cast_fp16)[name = string("out_39_cast_fp16")]; tensor var_4075_split_sizes_0 = const()[name = string("op_4075_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4075_axis_0 = const()[name = string("op_4075_axis_0"), val = int32(1)]; tensor var_4075_cast_fp16_0, tensor var_4075_cast_fp16_1 = split(axis = var_4075_axis_0, split_sizes = var_4075_split_sizes_0, x = out_39_cast_fp16)[name = string("op_4075_cast_fp16")]; tensor input_19_strides_0 = const()[name = string("input_19_strides_0"), val = tensor([1, 1])]; string input_19_pad_type_0 = const()[name = string("input_19_pad_type_0"), val = string("valid")]; tensor input_19_pad_0 = const()[name = string("input_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_19_dilations_0 = const()[name = string("input_19_dilations_0"), val = tensor([1, 1])]; int32 input_19_groups_0 = const()[name = string("input_19_groups_0"), val = int32(1)]; tensor input_19_cast_fp16 = conv(dilations = input_19_dilations_0, groups = input_19_groups_0, pad = input_19_pad_0, pad_type = input_19_pad_type_0, strides = input_19_strides_0, weight = layers_9_mlp_gate_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("input_19_cast_fp16")]; tensor var_4092_cast_fp16 = silu(x = input_19_cast_fp16)[name = string("op_4092_cast_fp16")]; tensor var_4098_strides_0 = const()[name = string("op_4098_strides_0"), val = tensor([1, 1])]; string var_4098_pad_type_0 = const()[name = string("op_4098_pad_type_0"), val = string("valid")]; tensor var_4098_pad_0 = const()[name = string("op_4098_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4098_dilations_0 = const()[name = string("op_4098_dilations_0"), val = tensor([1, 1])]; int32 var_4098_groups_0 = const()[name = string("op_4098_groups_0"), val = int32(1)]; tensor var_4098_cast_fp16 = conv(dilations = var_4098_dilations_0, groups = var_4098_groups_0, pad = var_4098_pad_0, pad_type = var_4098_pad_type_0, strides = var_4098_strides_0, weight = layers_9_mlp_up_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("op_4098_cast_fp16")]; tensor x_99_cast_fp16 = mul(x = var_4092_cast_fp16, y = var_4098_cast_fp16)[name = string("x_99_cast_fp16")]; tensor hidden_states_97_strides_0 = const()[name = string("hidden_states_97_strides_0"), val = tensor([1, 1])]; string hidden_states_97_pad_type_0 = const()[name = string("hidden_states_97_pad_type_0"), val = string("valid")]; tensor hidden_states_97_pad_0 = const()[name = string("hidden_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_97_dilations_0 = const()[name = string("hidden_states_97_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_97_groups_0 = const()[name = string("hidden_states_97_groups_0"), val = int32(1)]; tensor hidden_states_97_cast_fp16 = conv(dilations = hidden_states_97_dilations_0, groups = hidden_states_97_groups_0, pad = hidden_states_97_pad_0, pad_type = hidden_states_97_pad_type_0, strides = hidden_states_97_strides_0, weight = layers_9_mlp_down_proj_weight_cast_fp16, x = x_99_cast_fp16)[name = string("hidden_states_97_cast_fp16")]; tensor hidden_states_99_cast_fp16 = add(x = hidden_states_95_cast_fp16, y = hidden_states_97_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; fp16 const_100_promoted_to_fp16 = const()[name = string("const_100_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4116_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_100_promoted_to_fp16)[name = string("op_4116_cast_fp16")]; int32 var_4114 = const()[name = string("op_4114"), val = int32(1)]; bool doubled_81_interleave_0 = const()[name = string("doubled_81_interleave_0"), val = bool(false)]; tensor doubled_81_cast_fp16 = concat(axis = var_4114, interleave = doubled_81_interleave_0, values = (hidden_states_99_cast_fp16, var_4116_cast_fp16))[name = string("doubled_81_cast_fp16")]; tensor out_41_axes_0 = const()[name = string("out_41_axes_0"), val = tensor([1])]; tensor out_41_gamma_0_to_fp16 = const()[name = string("out_41_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405904320)))]; fp16 var_4126_to_fp16 = const()[name = string("op_4126_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_41_cast_fp16 = layer_norm(axes = out_41_axes_0, epsilon = var_4126_to_fp16, gamma = out_41_gamma_0_to_fp16, x = doubled_81_cast_fp16)[name = string("out_41_cast_fp16")]; tensor var_4137_split_sizes_0 = const()[name = string("op_4137_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4137_axis_0 = const()[name = string("op_4137_axis_0"), val = int32(1)]; tensor var_4137_cast_fp16_0, tensor var_4137_cast_fp16_1 = split(axis = var_4137_axis_0, split_sizes = var_4137_split_sizes_0, x = out_41_cast_fp16)[name = string("op_4137_cast_fp16")]; tensor layers_10_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405912576)))]; tensor query_states_61_strides_0 = const()[name = string("query_states_61_strides_0"), val = tensor([1, 1])]; string query_states_61_pad_type_0 = const()[name = string("query_states_61_pad_type_0"), val = string("valid")]; tensor query_states_61_pad_0 = const()[name = string("query_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_61_dilations_0 = const()[name = string("query_states_61_dilations_0"), val = tensor([1, 1])]; int32 query_states_61_groups_0 = const()[name = string("query_states_61_groups_0"), val = int32(1)]; tensor query_states_61_cast_fp16 = conv(dilations = query_states_61_dilations_0, groups = query_states_61_groups_0, pad = query_states_61_pad_0, pad_type = query_states_61_pad_type_0, strides = query_states_61_strides_0, weight = layers_10_self_attn_q_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("query_states_61_cast_fp16")]; tensor layers_10_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1414301248)))]; tensor key_states_101_strides_0 = const()[name = string("key_states_101_strides_0"), val = tensor([1, 1])]; string key_states_101_pad_type_0 = const()[name = string("key_states_101_pad_type_0"), val = string("valid")]; tensor key_states_101_pad_0 = const()[name = string("key_states_101_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_101_dilations_0 = const()[name = string("key_states_101_dilations_0"), val = tensor([1, 1])]; int32 key_states_101_groups_0 = const()[name = string("key_states_101_groups_0"), val = int32(1)]; tensor key_states_101_cast_fp16 = conv(dilations = key_states_101_dilations_0, groups = key_states_101_groups_0, pad = key_states_101_pad_0, pad_type = key_states_101_pad_type_0, strides = key_states_101_strides_0, weight = layers_10_self_attn_k_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("key_states_101_cast_fp16")]; tensor value_states_61_strides_0 = const()[name = string("value_states_61_strides_0"), val = tensor([1, 1])]; string value_states_61_pad_type_0 = const()[name = string("value_states_61_pad_type_0"), val = string("valid")]; tensor value_states_61_pad_0 = const()[name = string("value_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_61_dilations_0 = const()[name = string("value_states_61_dilations_0"), val = tensor([1, 1])]; int32 value_states_61_groups_0 = const()[name = string("value_states_61_groups_0"), val = int32(1)]; tensor value_states_61_cast_fp16 = conv(dilations = value_states_61_dilations_0, groups = value_states_61_groups_0, pad = value_states_61_pad_0, pad_type = value_states_61_pad_type_0, strides = value_states_61_strides_0, weight = layers_10_self_attn_v_proj_weight_cast_fp16, x = var_4137_cast_fp16_0)[name = string("value_states_61_cast_fp16")]; tensor concat_120x = const()[name = string("concat_120x"), val = tensor([1, 16, 128, -1])]; tensor x_101_cast_fp16 = reshape(shape = concat_120x, x = query_states_61_cast_fp16)[name = string("x_101_cast_fp16")]; tensor concat_121x = const()[name = string("concat_121x"), val = tensor([1, 2, 128, -1])]; tensor var_4194_cast_fp16 = reshape(shape = concat_121x, x = key_states_101_cast_fp16)[name = string("op_4194_cast_fp16")]; tensor concat_122x = const()[name = string("concat_122x"), val = tensor([1, 2, 128, -1])]; tensor var_4201_cast_fp16 = reshape(shape = concat_122x, x = value_states_61_cast_fp16)[name = string("op_4201_cast_fp16")]; tensor var_4205_cast_fp16 = mul(x = x_101_cast_fp16, y = var_869_cast_fp16)[name = string("op_4205_cast_fp16")]; tensor var_4206_split_sizes_0 = const()[name = string("op_4206_split_sizes_0"), val = tensor([64, 64])]; int32 var_4206_axis_0 = const()[name = string("op_4206_axis_0"), val = int32(-2)]; tensor var_4206_cast_fp16_0, tensor var_4206_cast_fp16_1 = split(axis = var_4206_axis_0, split_sizes = var_4206_split_sizes_0, x = x_101_cast_fp16)[name = string("op_4206_cast_fp16")]; fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4208_cast_fp16 = mul(x = var_4206_cast_fp16_1, y = const_102_promoted_to_fp16)[name = string("op_4208_cast_fp16")]; int32 var_4210 = const()[name = string("op_4210"), val = int32(-2)]; bool var_4211_interleave_0 = const()[name = string("op_4211_interleave_0"), val = bool(false)]; tensor var_4211_cast_fp16 = concat(axis = var_4210, interleave = var_4211_interleave_0, values = (var_4208_cast_fp16, var_4206_cast_fp16_0))[name = string("op_4211_cast_fp16")]; tensor var_4212_cast_fp16 = mul(x = var_4211_cast_fp16, y = var_878_cast_fp16)[name = string("op_4212_cast_fp16")]; tensor query_states_63_cast_fp16 = add(x = var_4205_cast_fp16, y = var_4212_cast_fp16)[name = string("query_states_63_cast_fp16")]; tensor var_4218_cast_fp16 = mul(x = var_4194_cast_fp16, y = var_869_cast_fp16)[name = string("op_4218_cast_fp16")]; tensor var_4219_split_sizes_0 = const()[name = string("op_4219_split_sizes_0"), val = tensor([64, 64])]; int32 var_4219_axis_0 = const()[name = string("op_4219_axis_0"), val = int32(-2)]; tensor var_4219_cast_fp16_0, tensor var_4219_cast_fp16_1 = split(axis = var_4219_axis_0, split_sizes = var_4219_split_sizes_0, x = var_4194_cast_fp16)[name = string("op_4219_cast_fp16")]; fp16 const_103_promoted_to_fp16 = const()[name = string("const_103_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4221_cast_fp16 = mul(x = var_4219_cast_fp16_1, y = const_103_promoted_to_fp16)[name = string("op_4221_cast_fp16")]; int32 var_4223 = const()[name = string("op_4223"), val = int32(-2)]; bool var_4224_interleave_0 = const()[name = string("op_4224_interleave_0"), val = bool(false)]; tensor var_4224_cast_fp16 = concat(axis = var_4223, interleave = var_4224_interleave_0, values = (var_4221_cast_fp16, var_4219_cast_fp16_0))[name = string("op_4224_cast_fp16")]; tensor var_4225_cast_fp16 = mul(x = var_4224_cast_fp16, y = var_878_cast_fp16)[name = string("op_4225_cast_fp16")]; tensor key_states_105_cast_fp16 = add(x = var_4218_cast_fp16, y = var_4225_cast_fp16)[name = string("key_states_105_cast_fp16")]; tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; int32 concat_125_axis_0 = const()[name = string("concat_125_axis_0"), val = int32(0)]; bool concat_125_interleave_0 = const()[name = string("concat_125_interleave_0"), val = bool(false)]; tensor concat_125 = concat(axis = concat_125_axis_0, interleave = concat_125_interleave_0, values = (expand_dims_120, expand_dims_121, position_id, expand_dims_123))[name = string("concat_125")]; tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; tensor concat_126_values1_0 = const()[name = string("concat_126_values1_0"), val = tensor([0])]; tensor concat_126_values3_0 = const()[name = string("concat_126_values3_0"), val = tensor([0])]; int32 concat_126_axis_0 = const()[name = string("concat_126_axis_0"), val = int32(0)]; bool concat_126_interleave_0 = const()[name = string("concat_126_interleave_0"), val = bool(false)]; tensor concat_126 = concat(axis = concat_126_axis_0, interleave = concat_126_interleave_0, values = (expand_dims_124, concat_126_values1_0, cache_position_end, concat_126_values3_0))[name = string("concat_126")]; tensor key_states_107_perm_0 = const()[name = string("key_states_107_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_11_stride_0 = const()[name = string("key_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_107_cast_fp16 = transpose(perm = key_states_107_perm_0, x = key_states_105_cast_fp16)[name = string("transpose_655")]; tensor key_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = key_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = key_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_11_squeeze_mask_0, stride = key_cache_internal_tensor_assign_11_stride_0, update = key_states_107_cast_fp16, x = coreml_update_state_410)[name = string("key_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_11_cast_fp16, input = key_cache)[name = string("coreml_update_state_412_write_state")]; tensor coreml_update_state_412 = read_state(input = key_cache)[name = string("coreml_update_state_412")]; tensor value_states_63_perm_0 = const()[name = string("value_states_63_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_11_stride_0 = const()[name = string("value_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_63_cast_fp16 = transpose(perm = value_states_63_perm_0, x = var_4201_cast_fp16)[name = string("transpose_654")]; tensor value_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = value_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = value_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_11_squeeze_mask_0, stride = value_cache_internal_tensor_assign_11_stride_0, update = value_states_63_cast_fp16, x = coreml_update_state_411)[name = string("value_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_11_cast_fp16, input = value_cache)[name = string("coreml_update_state_413_write_state")]; tensor coreml_update_state_413 = read_state(input = value_cache)[name = string("coreml_update_state_413")]; tensor var_4295_begin_0 = const()[name = string("op_4295_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4295_end_0 = const()[name = string("op_4295_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4295_end_mask_0 = const()[name = string("op_4295_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4295_cast_fp16 = slice_by_index(begin = var_4295_begin_0, end = var_4295_end_0, end_mask = var_4295_end_mask_0, x = coreml_update_state_412)[name = string("op_4295_cast_fp16")]; tensor tile_20 = const()[name = string("tile_20"), val = tensor([1, 1])]; int32 var_4298_axis_0 = const()[name = string("op_4298_axis_0"), val = int32(1)]; tensor var_4298_cast_fp16_0, tensor var_4298_cast_fp16_1 = split(axis = var_4298_axis_0, split_sizes = tile_20, x = var_4295_cast_fp16)[name = string("op_4298_cast_fp16")]; tensor var_4305_begin_0 = const()[name = string("op_4305_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4305_end_0 = const()[name = string("op_4305_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4305_end_mask_0 = const()[name = string("op_4305_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4305_cast_fp16 = slice_by_index(begin = var_4305_begin_0, end = var_4305_end_0, end_mask = var_4305_end_mask_0, x = coreml_update_state_413)[name = string("op_4305_cast_fp16")]; tensor tile_21 = const()[name = string("tile_21"), val = tensor([1, 1])]; int32 var_4308_axis_0 = const()[name = string("op_4308_axis_0"), val = int32(1)]; tensor var_4308_cast_fp16_0, tensor var_4308_cast_fp16_1 = split(axis = var_4308_axis_0, split_sizes = tile_21, x = var_4305_cast_fp16)[name = string("op_4308_cast_fp16")]; tensor var_4311_split_sizes_0 = const()[name = string("op_4311_split_sizes_0"), val = tensor([8, 8])]; int32 var_4311_axis_0 = const()[name = string("op_4311_axis_0"), val = int32(1)]; tensor var_4311_0, tensor var_4311_1 = split(axis = var_4311_axis_0, split_sizes = var_4311_split_sizes_0, x = query_states_63_cast_fp16)[name = string("op_4311")]; bool attn_weights_161_transpose_x_0 = const()[name = string("attn_weights_161_transpose_x_0"), val = bool(false)]; bool attn_weights_161_transpose_y_0 = const()[name = string("attn_weights_161_transpose_y_0"), val = bool(false)]; tensor attn_weights_161_cast_fp16 = matmul(transpose_x = attn_weights_161_transpose_x_0, transpose_y = attn_weights_161_transpose_y_0, x = var_4298_cast_fp16_0, y = var_4311_0)[name = string("attn_weights_161_cast_fp16")]; fp16 var_4314_to_fp16 = const()[name = string("op_4314_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_163_cast_fp16 = mul(x = attn_weights_161_cast_fp16, y = var_4314_to_fp16)[name = string("attn_weights_163_cast_fp16")]; tensor attn_weights_165_cast_fp16 = add(x = attn_weights_163_cast_fp16, y = attn_mask_1)[name = string("attn_weights_165_cast_fp16")]; int32 var_4318 = const()[name = string("op_4318"), val = int32(-2)]; tensor attn_weights_167_cast_fp16 = softmax(axis = var_4318, x = attn_weights_165_cast_fp16)[name = string("attn_weights_167_cast_fp16")]; bool var_4324_transpose_x_1 = const()[name = string("op_4324_transpose_x_1"), val = bool(true)]; bool var_4324_transpose_y_1 = const()[name = string("op_4324_transpose_y_1"), val = bool(false)]; tensor var_4324_cast_fp16 = matmul(transpose_x = var_4324_transpose_x_1, transpose_y = var_4324_transpose_y_1, x = attn_weights_167_cast_fp16, y = var_4308_cast_fp16_0)[name = string("op_4324_cast_fp16")]; bool attn_weights_169_transpose_x_0 = const()[name = string("attn_weights_169_transpose_x_0"), val = bool(false)]; bool attn_weights_169_transpose_y_0 = const()[name = string("attn_weights_169_transpose_y_0"), val = bool(false)]; tensor attn_weights_169_cast_fp16 = matmul(transpose_x = attn_weights_169_transpose_x_0, transpose_y = attn_weights_169_transpose_y_0, x = var_4298_cast_fp16_1, y = var_4311_1)[name = string("attn_weights_169_cast_fp16")]; fp16 var_4326_to_fp16 = const()[name = string("op_4326_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_171_cast_fp16 = mul(x = attn_weights_169_cast_fp16, y = var_4326_to_fp16)[name = string("attn_weights_171_cast_fp16")]; tensor attn_weights_173_cast_fp16 = add(x = attn_weights_171_cast_fp16, y = attn_mask_1)[name = string("attn_weights_173_cast_fp16")]; int32 var_4330 = const()[name = string("op_4330"), val = int32(-2)]; tensor attn_weights_175_cast_fp16 = softmax(axis = var_4330, x = attn_weights_173_cast_fp16)[name = string("attn_weights_175_cast_fp16")]; bool attn_output_81_transpose_x_1 = const()[name = string("attn_output_81_transpose_x_1"), val = bool(true)]; bool attn_output_81_transpose_y_1 = const()[name = string("attn_output_81_transpose_y_1"), val = bool(false)]; tensor attn_output_81_cast_fp16 = matmul(transpose_x = attn_output_81_transpose_x_1, transpose_y = attn_output_81_transpose_y_1, x = attn_weights_175_cast_fp16, y = var_4308_cast_fp16_1)[name = string("attn_output_81_cast_fp16")]; int32 var_4338 = const()[name = string("op_4338"), val = int32(1)]; bool attn_output_83_interleave_0 = const()[name = string("attn_output_83_interleave_0"), val = bool(false)]; tensor attn_output_83_cast_fp16 = concat(axis = var_4338, interleave = attn_output_83_interleave_0, values = (var_4324_cast_fp16, attn_output_81_cast_fp16))[name = string("attn_output_83_cast_fp16")]; tensor var_4342_perm_0 = const()[name = string("op_4342_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_131x = const()[name = string("concat_131x"), val = tensor([1, 2048, 1, -1])]; tensor var_4342_cast_fp16 = transpose(perm = var_4342_perm_0, x = attn_output_83_cast_fp16)[name = string("transpose_653")]; tensor attn_output_87_cast_fp16 = reshape(shape = concat_131x, x = var_4342_cast_fp16)[name = string("attn_output_87_cast_fp16")]; tensor hidden_states_103_strides_0 = const()[name = string("hidden_states_103_strides_0"), val = tensor([1, 1])]; string hidden_states_103_pad_type_0 = const()[name = string("hidden_states_103_pad_type_0"), val = string("valid")]; tensor hidden_states_103_pad_0 = const()[name = string("hidden_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_103_dilations_0 = const()[name = string("hidden_states_103_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_103_groups_0 = const()[name = string("hidden_states_103_groups_0"), val = int32(1)]; tensor hidden_states_103_cast_fp16 = conv(dilations = hidden_states_103_dilations_0, groups = hidden_states_103_groups_0, pad = hidden_states_103_pad_0, pad_type = hidden_states_103_pad_type_0, strides = hidden_states_103_strides_0, weight = layers_10_self_attn_o_proj_weight_cast_fp16, x = attn_output_87_cast_fp16)[name = string("hidden_states_103_cast_fp16")]; tensor hidden_states_105_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = hidden_states_103_cast_fp16)[name = string("hidden_states_105_cast_fp16")]; fp16 const_108_promoted_to_fp16 = const()[name = string("const_108_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4375_cast_fp16 = mul(x = hidden_states_105_cast_fp16, y = const_108_promoted_to_fp16)[name = string("op_4375_cast_fp16")]; int32 var_4373 = const()[name = string("op_4373"), val = int32(1)]; bool doubled_85_interleave_0 = const()[name = string("doubled_85_interleave_0"), val = bool(false)]; tensor doubled_85_cast_fp16 = concat(axis = var_4373, interleave = doubled_85_interleave_0, values = (hidden_states_105_cast_fp16, var_4375_cast_fp16))[name = string("doubled_85_cast_fp16")]; tensor out_43_axes_0 = const()[name = string("out_43_axes_0"), val = tensor([1])]; tensor out_43_gamma_0_to_fp16 = const()[name = string("out_43_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415349888)))]; fp16 var_4385_to_fp16 = const()[name = string("op_4385_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_43_cast_fp16 = layer_norm(axes = out_43_axes_0, epsilon = var_4385_to_fp16, gamma = out_43_gamma_0_to_fp16, x = doubled_85_cast_fp16)[name = string("out_43_cast_fp16")]; tensor var_4396_split_sizes_0 = const()[name = string("op_4396_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4396_axis_0 = const()[name = string("op_4396_axis_0"), val = int32(1)]; tensor var_4396_cast_fp16_0, tensor var_4396_cast_fp16_1 = split(axis = var_4396_axis_0, split_sizes = var_4396_split_sizes_0, x = out_43_cast_fp16)[name = string("op_4396_cast_fp16")]; tensor input_21_strides_0 = const()[name = string("input_21_strides_0"), val = tensor([1, 1])]; string input_21_pad_type_0 = const()[name = string("input_21_pad_type_0"), val = string("valid")]; tensor input_21_pad_0 = const()[name = string("input_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_21_dilations_0 = const()[name = string("input_21_dilations_0"), val = tensor([1, 1])]; int32 input_21_groups_0 = const()[name = string("input_21_groups_0"), val = int32(1)]; tensor input_21_cast_fp16 = conv(dilations = input_21_dilations_0, groups = input_21_groups_0, pad = input_21_pad_0, pad_type = input_21_pad_type_0, strides = input_21_strides_0, weight = layers_10_mlp_gate_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("input_21_cast_fp16")]; tensor var_4413_cast_fp16 = silu(x = input_21_cast_fp16)[name = string("op_4413_cast_fp16")]; tensor var_4419_strides_0 = const()[name = string("op_4419_strides_0"), val = tensor([1, 1])]; string var_4419_pad_type_0 = const()[name = string("op_4419_pad_type_0"), val = string("valid")]; tensor var_4419_pad_0 = const()[name = string("op_4419_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4419_dilations_0 = const()[name = string("op_4419_dilations_0"), val = tensor([1, 1])]; int32 var_4419_groups_0 = const()[name = string("op_4419_groups_0"), val = int32(1)]; tensor var_4419_cast_fp16 = conv(dilations = var_4419_dilations_0, groups = var_4419_groups_0, pad = var_4419_pad_0, pad_type = var_4419_pad_type_0, strides = var_4419_strides_0, weight = layers_10_mlp_up_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("op_4419_cast_fp16")]; tensor x_109_cast_fp16 = mul(x = var_4413_cast_fp16, y = var_4419_cast_fp16)[name = string("x_109_cast_fp16")]; tensor hidden_states_107_strides_0 = const()[name = string("hidden_states_107_strides_0"), val = tensor([1, 1])]; string hidden_states_107_pad_type_0 = const()[name = string("hidden_states_107_pad_type_0"), val = string("valid")]; tensor hidden_states_107_pad_0 = const()[name = string("hidden_states_107_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_107_dilations_0 = const()[name = string("hidden_states_107_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_107_groups_0 = const()[name = string("hidden_states_107_groups_0"), val = int32(1)]; tensor hidden_states_107_cast_fp16 = conv(dilations = hidden_states_107_dilations_0, groups = hidden_states_107_groups_0, pad = hidden_states_107_pad_0, pad_type = hidden_states_107_pad_type_0, strides = hidden_states_107_strides_0, weight = layers_10_mlp_down_proj_weight_cast_fp16, x = x_109_cast_fp16)[name = string("hidden_states_107_cast_fp16")]; tensor hidden_states_109_cast_fp16 = add(x = hidden_states_105_cast_fp16, y = hidden_states_107_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; fp16 const_110_promoted_to_fp16 = const()[name = string("const_110_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4437_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_110_promoted_to_fp16)[name = string("op_4437_cast_fp16")]; int32 var_4435 = const()[name = string("op_4435"), val = int32(1)]; bool doubled_89_interleave_0 = const()[name = string("doubled_89_interleave_0"), val = bool(false)]; tensor doubled_89_cast_fp16 = concat(axis = var_4435, interleave = doubled_89_interleave_0, values = (hidden_states_109_cast_fp16, var_4437_cast_fp16))[name = string("doubled_89_cast_fp16")]; tensor out_45_axes_0 = const()[name = string("out_45_axes_0"), val = tensor([1])]; tensor out_45_gamma_0_to_fp16 = const()[name = string("out_45_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415358144)))]; fp16 var_4447_to_fp16 = const()[name = string("op_4447_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_45_cast_fp16 = layer_norm(axes = out_45_axes_0, epsilon = var_4447_to_fp16, gamma = out_45_gamma_0_to_fp16, x = doubled_89_cast_fp16)[name = string("out_45_cast_fp16")]; tensor var_4458_split_sizes_0 = const()[name = string("op_4458_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4458_axis_0 = const()[name = string("op_4458_axis_0"), val = int32(1)]; tensor var_4458_cast_fp16_0, tensor var_4458_cast_fp16_1 = split(axis = var_4458_axis_0, split_sizes = var_4458_split_sizes_0, x = out_45_cast_fp16)[name = string("op_4458_cast_fp16")]; tensor query_states_67_strides_0 = const()[name = string("query_states_67_strides_0"), val = tensor([1, 1])]; string query_states_67_pad_type_0 = const()[name = string("query_states_67_pad_type_0"), val = string("valid")]; tensor query_states_67_pad_0 = const()[name = string("query_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_67_dilations_0 = const()[name = string("query_states_67_dilations_0"), val = tensor([1, 1])]; int32 query_states_67_groups_0 = const()[name = string("query_states_67_groups_0"), val = int32(1)]; tensor query_states_67_cast_fp16 = conv(dilations = query_states_67_dilations_0, groups = query_states_67_groups_0, pad = query_states_67_pad_0, pad_type = query_states_67_pad_type_0, strides = query_states_67_strides_0, weight = layers_11_self_attn_q_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("query_states_67_cast_fp16")]; tensor key_states_111_strides_0 = const()[name = string("key_states_111_strides_0"), val = tensor([1, 1])]; string key_states_111_pad_type_0 = const()[name = string("key_states_111_pad_type_0"), val = string("valid")]; tensor key_states_111_pad_0 = const()[name = string("key_states_111_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_111_dilations_0 = const()[name = string("key_states_111_dilations_0"), val = tensor([1, 1])]; int32 key_states_111_groups_0 = const()[name = string("key_states_111_groups_0"), val = int32(1)]; tensor key_states_111_cast_fp16 = conv(dilations = key_states_111_dilations_0, groups = key_states_111_groups_0, pad = key_states_111_pad_0, pad_type = key_states_111_pad_type_0, strides = key_states_111_strides_0, weight = layers_11_self_attn_k_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("key_states_111_cast_fp16")]; tensor value_states_67_strides_0 = const()[name = string("value_states_67_strides_0"), val = tensor([1, 1])]; string value_states_67_pad_type_0 = const()[name = string("value_states_67_pad_type_0"), val = string("valid")]; tensor value_states_67_pad_0 = const()[name = string("value_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_67_dilations_0 = const()[name = string("value_states_67_dilations_0"), val = tensor([1, 1])]; int32 value_states_67_groups_0 = const()[name = string("value_states_67_groups_0"), val = int32(1)]; tensor value_states_67_cast_fp16 = conv(dilations = value_states_67_dilations_0, groups = value_states_67_groups_0, pad = value_states_67_pad_0, pad_type = value_states_67_pad_type_0, strides = value_states_67_strides_0, weight = layers_11_self_attn_v_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("value_states_67_cast_fp16")]; tensor concat_132x = const()[name = string("concat_132x"), val = tensor([1, 16, 128, -1])]; tensor x_111_cast_fp16 = reshape(shape = concat_132x, x = query_states_67_cast_fp16)[name = string("x_111_cast_fp16")]; tensor concat_133x = const()[name = string("concat_133x"), val = tensor([1, 2, 128, -1])]; tensor var_4515_cast_fp16 = reshape(shape = concat_133x, x = key_states_111_cast_fp16)[name = string("op_4515_cast_fp16")]; tensor concat_134x = const()[name = string("concat_134x"), val = tensor([1, 2, 128, -1])]; tensor var_4522_cast_fp16 = reshape(shape = concat_134x, x = value_states_67_cast_fp16)[name = string("op_4522_cast_fp16")]; tensor var_4526_cast_fp16 = mul(x = x_111_cast_fp16, y = var_869_cast_fp16)[name = string("op_4526_cast_fp16")]; tensor var_4527_split_sizes_0 = const()[name = string("op_4527_split_sizes_0"), val = tensor([64, 64])]; int32 var_4527_axis_0 = const()[name = string("op_4527_axis_0"), val = int32(-2)]; tensor var_4527_cast_fp16_0, tensor var_4527_cast_fp16_1 = split(axis = var_4527_axis_0, split_sizes = var_4527_split_sizes_0, x = x_111_cast_fp16)[name = string("op_4527_cast_fp16")]; fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4529_cast_fp16 = mul(x = var_4527_cast_fp16_1, y = const_112_promoted_to_fp16)[name = string("op_4529_cast_fp16")]; int32 var_4531 = const()[name = string("op_4531"), val = int32(-2)]; bool var_4532_interleave_0 = const()[name = string("op_4532_interleave_0"), val = bool(false)]; tensor var_4532_cast_fp16 = concat(axis = var_4531, interleave = var_4532_interleave_0, values = (var_4529_cast_fp16, var_4527_cast_fp16_0))[name = string("op_4532_cast_fp16")]; tensor var_4533_cast_fp16 = mul(x = var_4532_cast_fp16, y = var_878_cast_fp16)[name = string("op_4533_cast_fp16")]; tensor query_states_69_cast_fp16 = add(x = var_4526_cast_fp16, y = var_4533_cast_fp16)[name = string("query_states_69_cast_fp16")]; tensor var_4539_cast_fp16 = mul(x = var_4515_cast_fp16, y = var_869_cast_fp16)[name = string("op_4539_cast_fp16")]; tensor var_4540_split_sizes_0 = const()[name = string("op_4540_split_sizes_0"), val = tensor([64, 64])]; int32 var_4540_axis_0 = const()[name = string("op_4540_axis_0"), val = int32(-2)]; tensor var_4540_cast_fp16_0, tensor var_4540_cast_fp16_1 = split(axis = var_4540_axis_0, split_sizes = var_4540_split_sizes_0, x = var_4515_cast_fp16)[name = string("op_4540_cast_fp16")]; fp16 const_113_promoted_to_fp16 = const()[name = string("const_113_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4542_cast_fp16 = mul(x = var_4540_cast_fp16_1, y = const_113_promoted_to_fp16)[name = string("op_4542_cast_fp16")]; int32 var_4544 = const()[name = string("op_4544"), val = int32(-2)]; bool var_4545_interleave_0 = const()[name = string("op_4545_interleave_0"), val = bool(false)]; tensor var_4545_cast_fp16 = concat(axis = var_4544, interleave = var_4545_interleave_0, values = (var_4542_cast_fp16, var_4540_cast_fp16_0))[name = string("op_4545_cast_fp16")]; tensor var_4546_cast_fp16 = mul(x = var_4545_cast_fp16, y = var_878_cast_fp16)[name = string("op_4546_cast_fp16")]; tensor key_states_115_cast_fp16 = add(x = var_4539_cast_fp16, y = var_4546_cast_fp16)[name = string("key_states_115_cast_fp16")]; tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; int32 concat_137_axis_0 = const()[name = string("concat_137_axis_0"), val = int32(0)]; bool concat_137_interleave_0 = const()[name = string("concat_137_interleave_0"), val = bool(false)]; tensor concat_137 = concat(axis = concat_137_axis_0, interleave = concat_137_interleave_0, values = (expand_dims_132, expand_dims_133, position_id, expand_dims_135))[name = string("concat_137")]; tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; tensor concat_138_values1_0 = const()[name = string("concat_138_values1_0"), val = tensor([0])]; tensor concat_138_values3_0 = const()[name = string("concat_138_values3_0"), val = tensor([0])]; int32 concat_138_axis_0 = const()[name = string("concat_138_axis_0"), val = int32(0)]; bool concat_138_interleave_0 = const()[name = string("concat_138_interleave_0"), val = bool(false)]; tensor concat_138 = concat(axis = concat_138_axis_0, interleave = concat_138_interleave_0, values = (expand_dims_136, concat_138_values1_0, cache_position_end, concat_138_values3_0))[name = string("concat_138")]; tensor key_states_117_perm_0 = const()[name = string("key_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_12_stride_0 = const()[name = string("key_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_117_cast_fp16 = transpose(perm = key_states_117_perm_0, x = key_states_115_cast_fp16)[name = string("transpose_652")]; tensor key_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = key_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = key_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_12_squeeze_mask_0, stride = key_cache_internal_tensor_assign_12_stride_0, update = key_states_117_cast_fp16, x = coreml_update_state_412)[name = string("key_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_12_cast_fp16, input = key_cache)[name = string("coreml_update_state_414_write_state")]; tensor coreml_update_state_414 = read_state(input = key_cache)[name = string("coreml_update_state_414")]; tensor value_states_69_perm_0 = const()[name = string("value_states_69_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_12_stride_0 = const()[name = string("value_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_69_cast_fp16 = transpose(perm = value_states_69_perm_0, x = var_4522_cast_fp16)[name = string("transpose_651")]; tensor value_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = value_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = value_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_12_squeeze_mask_0, stride = value_cache_internal_tensor_assign_12_stride_0, update = value_states_69_cast_fp16, x = coreml_update_state_413)[name = string("value_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_12_cast_fp16, input = value_cache)[name = string("coreml_update_state_415_write_state")]; tensor coreml_update_state_415 = read_state(input = value_cache)[name = string("coreml_update_state_415")]; tensor var_4616_begin_0 = const()[name = string("op_4616_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4616_end_0 = const()[name = string("op_4616_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4616_end_mask_0 = const()[name = string("op_4616_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4616_cast_fp16 = slice_by_index(begin = var_4616_begin_0, end = var_4616_end_0, end_mask = var_4616_end_mask_0, x = coreml_update_state_414)[name = string("op_4616_cast_fp16")]; tensor tile_22 = const()[name = string("tile_22"), val = tensor([1, 1])]; int32 var_4619_axis_0 = const()[name = string("op_4619_axis_0"), val = int32(1)]; tensor var_4619_cast_fp16_0, tensor var_4619_cast_fp16_1 = split(axis = var_4619_axis_0, split_sizes = tile_22, x = var_4616_cast_fp16)[name = string("op_4619_cast_fp16")]; tensor var_4626_begin_0 = const()[name = string("op_4626_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4626_end_0 = const()[name = string("op_4626_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4626_end_mask_0 = const()[name = string("op_4626_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4626_cast_fp16 = slice_by_index(begin = var_4626_begin_0, end = var_4626_end_0, end_mask = var_4626_end_mask_0, x = coreml_update_state_415)[name = string("op_4626_cast_fp16")]; tensor tile_23 = const()[name = string("tile_23"), val = tensor([1, 1])]; int32 var_4629_axis_0 = const()[name = string("op_4629_axis_0"), val = int32(1)]; tensor var_4629_cast_fp16_0, tensor var_4629_cast_fp16_1 = split(axis = var_4629_axis_0, split_sizes = tile_23, x = var_4626_cast_fp16)[name = string("op_4629_cast_fp16")]; tensor var_4632_split_sizes_0 = const()[name = string("op_4632_split_sizes_0"), val = tensor([8, 8])]; int32 var_4632_axis_0 = const()[name = string("op_4632_axis_0"), val = int32(1)]; tensor var_4632_0, tensor var_4632_1 = split(axis = var_4632_axis_0, split_sizes = var_4632_split_sizes_0, x = query_states_69_cast_fp16)[name = string("op_4632")]; bool attn_weights_177_transpose_x_0 = const()[name = string("attn_weights_177_transpose_x_0"), val = bool(false)]; bool attn_weights_177_transpose_y_0 = const()[name = string("attn_weights_177_transpose_y_0"), val = bool(false)]; tensor attn_weights_177_cast_fp16 = matmul(transpose_x = attn_weights_177_transpose_x_0, transpose_y = attn_weights_177_transpose_y_0, x = var_4619_cast_fp16_0, y = var_4632_0)[name = string("attn_weights_177_cast_fp16")]; fp16 var_4635_to_fp16 = const()[name = string("op_4635_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_179_cast_fp16 = mul(x = attn_weights_177_cast_fp16, y = var_4635_to_fp16)[name = string("attn_weights_179_cast_fp16")]; tensor attn_weights_181_cast_fp16 = add(x = attn_weights_179_cast_fp16, y = attn_mask_1)[name = string("attn_weights_181_cast_fp16")]; int32 var_4639 = const()[name = string("op_4639"), val = int32(-2)]; tensor attn_weights_183_cast_fp16 = softmax(axis = var_4639, x = attn_weights_181_cast_fp16)[name = string("attn_weights_183_cast_fp16")]; bool var_4645_transpose_x_1 = const()[name = string("op_4645_transpose_x_1"), val = bool(true)]; bool var_4645_transpose_y_1 = const()[name = string("op_4645_transpose_y_1"), val = bool(false)]; tensor var_4645_cast_fp16 = matmul(transpose_x = var_4645_transpose_x_1, transpose_y = var_4645_transpose_y_1, x = attn_weights_183_cast_fp16, y = var_4629_cast_fp16_0)[name = string("op_4645_cast_fp16")]; bool attn_weights_185_transpose_x_0 = const()[name = string("attn_weights_185_transpose_x_0"), val = bool(false)]; bool attn_weights_185_transpose_y_0 = const()[name = string("attn_weights_185_transpose_y_0"), val = bool(false)]; tensor attn_weights_185_cast_fp16 = matmul(transpose_x = attn_weights_185_transpose_x_0, transpose_y = attn_weights_185_transpose_y_0, x = var_4619_cast_fp16_1, y = var_4632_1)[name = string("attn_weights_185_cast_fp16")]; fp16 var_4647_to_fp16 = const()[name = string("op_4647_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_187_cast_fp16 = mul(x = attn_weights_185_cast_fp16, y = var_4647_to_fp16)[name = string("attn_weights_187_cast_fp16")]; tensor attn_weights_189_cast_fp16 = add(x = attn_weights_187_cast_fp16, y = attn_mask_1)[name = string("attn_weights_189_cast_fp16")]; int32 var_4651 = const()[name = string("op_4651"), val = int32(-2)]; tensor attn_weights_191_cast_fp16 = softmax(axis = var_4651, x = attn_weights_189_cast_fp16)[name = string("attn_weights_191_cast_fp16")]; bool attn_output_89_transpose_x_1 = const()[name = string("attn_output_89_transpose_x_1"), val = bool(true)]; bool attn_output_89_transpose_y_1 = const()[name = string("attn_output_89_transpose_y_1"), val = bool(false)]; tensor attn_output_89_cast_fp16 = matmul(transpose_x = attn_output_89_transpose_x_1, transpose_y = attn_output_89_transpose_y_1, x = attn_weights_191_cast_fp16, y = var_4629_cast_fp16_1)[name = string("attn_output_89_cast_fp16")]; int32 var_4659 = const()[name = string("op_4659"), val = int32(1)]; bool attn_output_91_interleave_0 = const()[name = string("attn_output_91_interleave_0"), val = bool(false)]; tensor attn_output_91_cast_fp16 = concat(axis = var_4659, interleave = attn_output_91_interleave_0, values = (var_4645_cast_fp16, attn_output_89_cast_fp16))[name = string("attn_output_91_cast_fp16")]; tensor var_4663_perm_0 = const()[name = string("op_4663_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_143x = const()[name = string("concat_143x"), val = tensor([1, 2048, 1, -1])]; tensor var_4663_cast_fp16 = transpose(perm = var_4663_perm_0, x = attn_output_91_cast_fp16)[name = string("transpose_650")]; tensor attn_output_95_cast_fp16 = reshape(shape = concat_143x, x = var_4663_cast_fp16)[name = string("attn_output_95_cast_fp16")]; tensor hidden_states_113_strides_0 = const()[name = string("hidden_states_113_strides_0"), val = tensor([1, 1])]; string hidden_states_113_pad_type_0 = const()[name = string("hidden_states_113_pad_type_0"), val = string("valid")]; tensor hidden_states_113_pad_0 = const()[name = string("hidden_states_113_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_113_dilations_0 = const()[name = string("hidden_states_113_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_113_groups_0 = const()[name = string("hidden_states_113_groups_0"), val = int32(1)]; tensor hidden_states_113_cast_fp16 = conv(dilations = hidden_states_113_dilations_0, groups = hidden_states_113_groups_0, pad = hidden_states_113_pad_0, pad_type = hidden_states_113_pad_type_0, strides = hidden_states_113_strides_0, weight = layers_11_self_attn_o_proj_weight_cast_fp16, x = attn_output_95_cast_fp16)[name = string("hidden_states_113_cast_fp16")]; tensor hidden_states_115_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = hidden_states_113_cast_fp16)[name = string("hidden_states_115_cast_fp16")]; fp16 const_118_promoted_to_fp16 = const()[name = string("const_118_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4696_cast_fp16 = mul(x = hidden_states_115_cast_fp16, y = const_118_promoted_to_fp16)[name = string("op_4696_cast_fp16")]; int32 var_4694 = const()[name = string("op_4694"), val = int32(1)]; bool doubled_93_interleave_0 = const()[name = string("doubled_93_interleave_0"), val = bool(false)]; tensor doubled_93_cast_fp16 = concat(axis = var_4694, interleave = doubled_93_interleave_0, values = (hidden_states_115_cast_fp16, var_4696_cast_fp16))[name = string("doubled_93_cast_fp16")]; tensor out_47_axes_0 = const()[name = string("out_47_axes_0"), val = tensor([1])]; tensor out_47_gamma_0_to_fp16 = const()[name = string("out_47_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415366400)))]; fp16 var_4706_to_fp16 = const()[name = string("op_4706_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_47_cast_fp16 = layer_norm(axes = out_47_axes_0, epsilon = var_4706_to_fp16, gamma = out_47_gamma_0_to_fp16, x = doubled_93_cast_fp16)[name = string("out_47_cast_fp16")]; tensor var_4717_split_sizes_0 = const()[name = string("op_4717_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4717_axis_0 = const()[name = string("op_4717_axis_0"), val = int32(1)]; tensor var_4717_cast_fp16_0, tensor var_4717_cast_fp16_1 = split(axis = var_4717_axis_0, split_sizes = var_4717_split_sizes_0, x = out_47_cast_fp16)[name = string("op_4717_cast_fp16")]; tensor input_23_strides_0 = const()[name = string("input_23_strides_0"), val = tensor([1, 1])]; string input_23_pad_type_0 = const()[name = string("input_23_pad_type_0"), val = string("valid")]; tensor input_23_pad_0 = const()[name = string("input_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_23_dilations_0 = const()[name = string("input_23_dilations_0"), val = tensor([1, 1])]; int32 input_23_groups_0 = const()[name = string("input_23_groups_0"), val = int32(1)]; tensor input_23_cast_fp16 = conv(dilations = input_23_dilations_0, groups = input_23_groups_0, pad = input_23_pad_0, pad_type = input_23_pad_type_0, strides = input_23_strides_0, weight = layers_11_mlp_gate_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("input_23_cast_fp16")]; tensor var_4734_cast_fp16 = silu(x = input_23_cast_fp16)[name = string("op_4734_cast_fp16")]; tensor var_4740_strides_0 = const()[name = string("op_4740_strides_0"), val = tensor([1, 1])]; string var_4740_pad_type_0 = const()[name = string("op_4740_pad_type_0"), val = string("valid")]; tensor var_4740_pad_0 = const()[name = string("op_4740_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4740_dilations_0 = const()[name = string("op_4740_dilations_0"), val = tensor([1, 1])]; int32 var_4740_groups_0 = const()[name = string("op_4740_groups_0"), val = int32(1)]; tensor var_4740_cast_fp16 = conv(dilations = var_4740_dilations_0, groups = var_4740_groups_0, pad = var_4740_pad_0, pad_type = var_4740_pad_type_0, strides = var_4740_strides_0, weight = layers_11_mlp_up_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("op_4740_cast_fp16")]; tensor x_119_cast_fp16 = mul(x = var_4734_cast_fp16, y = var_4740_cast_fp16)[name = string("x_119_cast_fp16")]; tensor hidden_states_117_strides_0 = const()[name = string("hidden_states_117_strides_0"), val = tensor([1, 1])]; string hidden_states_117_pad_type_0 = const()[name = string("hidden_states_117_pad_type_0"), val = string("valid")]; tensor hidden_states_117_pad_0 = const()[name = string("hidden_states_117_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_117_dilations_0 = const()[name = string("hidden_states_117_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_117_groups_0 = const()[name = string("hidden_states_117_groups_0"), val = int32(1)]; tensor hidden_states_117_cast_fp16 = conv(dilations = hidden_states_117_dilations_0, groups = hidden_states_117_groups_0, pad = hidden_states_117_pad_0, pad_type = hidden_states_117_pad_type_0, strides = hidden_states_117_strides_0, weight = layers_11_mlp_down_proj_weight_cast_fp16, x = x_119_cast_fp16)[name = string("hidden_states_117_cast_fp16")]; tensor hidden_states_119_cast_fp16 = add(x = hidden_states_115_cast_fp16, y = hidden_states_117_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; fp16 const_120_promoted_to_fp16 = const()[name = string("const_120_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4758_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_120_promoted_to_fp16)[name = string("op_4758_cast_fp16")]; int32 var_4756 = const()[name = string("op_4756"), val = int32(1)]; bool doubled_97_interleave_0 = const()[name = string("doubled_97_interleave_0"), val = bool(false)]; tensor doubled_97_cast_fp16 = concat(axis = var_4756, interleave = doubled_97_interleave_0, values = (hidden_states_119_cast_fp16, var_4758_cast_fp16))[name = string("doubled_97_cast_fp16")]; tensor out_49_axes_0 = const()[name = string("out_49_axes_0"), val = tensor([1])]; tensor out_49_gamma_0_to_fp16 = const()[name = string("out_49_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415374656)))]; fp16 var_4768_to_fp16 = const()[name = string("op_4768_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_49_cast_fp16 = layer_norm(axes = out_49_axes_0, epsilon = var_4768_to_fp16, gamma = out_49_gamma_0_to_fp16, x = doubled_97_cast_fp16)[name = string("out_49_cast_fp16")]; tensor var_4779_split_sizes_0 = const()[name = string("op_4779_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4779_axis_0 = const()[name = string("op_4779_axis_0"), val = int32(1)]; tensor var_4779_cast_fp16_0, tensor var_4779_cast_fp16_1 = split(axis = var_4779_axis_0, split_sizes = var_4779_split_sizes_0, x = out_49_cast_fp16)[name = string("op_4779_cast_fp16")]; tensor query_states_73_strides_0 = const()[name = string("query_states_73_strides_0"), val = tensor([1, 1])]; string query_states_73_pad_type_0 = const()[name = string("query_states_73_pad_type_0"), val = string("valid")]; tensor query_states_73_pad_0 = const()[name = string("query_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_73_dilations_0 = const()[name = string("query_states_73_dilations_0"), val = tensor([1, 1])]; int32 query_states_73_groups_0 = const()[name = string("query_states_73_groups_0"), val = int32(1)]; tensor query_states_73_cast_fp16 = conv(dilations = query_states_73_dilations_0, groups = query_states_73_groups_0, pad = query_states_73_pad_0, pad_type = query_states_73_pad_type_0, strides = query_states_73_strides_0, weight = layers_12_self_attn_q_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("query_states_73_cast_fp16")]; tensor key_states_121_strides_0 = const()[name = string("key_states_121_strides_0"), val = tensor([1, 1])]; string key_states_121_pad_type_0 = const()[name = string("key_states_121_pad_type_0"), val = string("valid")]; tensor key_states_121_pad_0 = const()[name = string("key_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_121_dilations_0 = const()[name = string("key_states_121_dilations_0"), val = tensor([1, 1])]; int32 key_states_121_groups_0 = const()[name = string("key_states_121_groups_0"), val = int32(1)]; tensor key_states_121_cast_fp16 = conv(dilations = key_states_121_dilations_0, groups = key_states_121_groups_0, pad = key_states_121_pad_0, pad_type = key_states_121_pad_type_0, strides = key_states_121_strides_0, weight = layers_12_self_attn_k_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("key_states_121_cast_fp16")]; tensor value_states_73_strides_0 = const()[name = string("value_states_73_strides_0"), val = tensor([1, 1])]; string value_states_73_pad_type_0 = const()[name = string("value_states_73_pad_type_0"), val = string("valid")]; tensor value_states_73_pad_0 = const()[name = string("value_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_73_dilations_0 = const()[name = string("value_states_73_dilations_0"), val = tensor([1, 1])]; int32 value_states_73_groups_0 = const()[name = string("value_states_73_groups_0"), val = int32(1)]; tensor value_states_73_cast_fp16 = conv(dilations = value_states_73_dilations_0, groups = value_states_73_groups_0, pad = value_states_73_pad_0, pad_type = value_states_73_pad_type_0, strides = value_states_73_strides_0, weight = layers_12_self_attn_v_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("value_states_73_cast_fp16")]; tensor concat_144x = const()[name = string("concat_144x"), val = tensor([1, 16, 128, -1])]; tensor x_121_cast_fp16 = reshape(shape = concat_144x, x = query_states_73_cast_fp16)[name = string("x_121_cast_fp16")]; tensor concat_145x = const()[name = string("concat_145x"), val = tensor([1, 2, 128, -1])]; tensor var_4836_cast_fp16 = reshape(shape = concat_145x, x = key_states_121_cast_fp16)[name = string("op_4836_cast_fp16")]; tensor concat_146x = const()[name = string("concat_146x"), val = tensor([1, 2, 128, -1])]; tensor var_4843_cast_fp16 = reshape(shape = concat_146x, x = value_states_73_cast_fp16)[name = string("op_4843_cast_fp16")]; tensor var_4847_cast_fp16 = mul(x = x_121_cast_fp16, y = var_869_cast_fp16)[name = string("op_4847_cast_fp16")]; tensor var_4848_split_sizes_0 = const()[name = string("op_4848_split_sizes_0"), val = tensor([64, 64])]; int32 var_4848_axis_0 = const()[name = string("op_4848_axis_0"), val = int32(-2)]; tensor var_4848_cast_fp16_0, tensor var_4848_cast_fp16_1 = split(axis = var_4848_axis_0, split_sizes = var_4848_split_sizes_0, x = x_121_cast_fp16)[name = string("op_4848_cast_fp16")]; fp16 const_122_promoted_to_fp16 = const()[name = string("const_122_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4850_cast_fp16 = mul(x = var_4848_cast_fp16_1, y = const_122_promoted_to_fp16)[name = string("op_4850_cast_fp16")]; int32 var_4852 = const()[name = string("op_4852"), val = int32(-2)]; bool var_4853_interleave_0 = const()[name = string("op_4853_interleave_0"), val = bool(false)]; tensor var_4853_cast_fp16 = concat(axis = var_4852, interleave = var_4853_interleave_0, values = (var_4850_cast_fp16, var_4848_cast_fp16_0))[name = string("op_4853_cast_fp16")]; tensor var_4854_cast_fp16 = mul(x = var_4853_cast_fp16, y = var_878_cast_fp16)[name = string("op_4854_cast_fp16")]; tensor query_states_75_cast_fp16 = add(x = var_4847_cast_fp16, y = var_4854_cast_fp16)[name = string("query_states_75_cast_fp16")]; tensor var_4860_cast_fp16 = mul(x = var_4836_cast_fp16, y = var_869_cast_fp16)[name = string("op_4860_cast_fp16")]; tensor var_4861_split_sizes_0 = const()[name = string("op_4861_split_sizes_0"), val = tensor([64, 64])]; int32 var_4861_axis_0 = const()[name = string("op_4861_axis_0"), val = int32(-2)]; tensor var_4861_cast_fp16_0, tensor var_4861_cast_fp16_1 = split(axis = var_4861_axis_0, split_sizes = var_4861_split_sizes_0, x = var_4836_cast_fp16)[name = string("op_4861_cast_fp16")]; fp16 const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4863_cast_fp16 = mul(x = var_4861_cast_fp16_1, y = const_123_promoted_to_fp16)[name = string("op_4863_cast_fp16")]; int32 var_4865 = const()[name = string("op_4865"), val = int32(-2)]; bool var_4866_interleave_0 = const()[name = string("op_4866_interleave_0"), val = bool(false)]; tensor var_4866_cast_fp16 = concat(axis = var_4865, interleave = var_4866_interleave_0, values = (var_4863_cast_fp16, var_4861_cast_fp16_0))[name = string("op_4866_cast_fp16")]; tensor var_4867_cast_fp16 = mul(x = var_4866_cast_fp16, y = var_878_cast_fp16)[name = string("op_4867_cast_fp16")]; tensor key_states_125_cast_fp16 = add(x = var_4860_cast_fp16, y = var_4867_cast_fp16)[name = string("key_states_125_cast_fp16")]; tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; int32 concat_149_axis_0 = const()[name = string("concat_149_axis_0"), val = int32(0)]; bool concat_149_interleave_0 = const()[name = string("concat_149_interleave_0"), val = bool(false)]; tensor concat_149 = concat(axis = concat_149_axis_0, interleave = concat_149_interleave_0, values = (expand_dims_144, expand_dims_145, position_id, expand_dims_147))[name = string("concat_149")]; tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; tensor concat_150_values1_0 = const()[name = string("concat_150_values1_0"), val = tensor([0])]; tensor concat_150_values3_0 = const()[name = string("concat_150_values3_0"), val = tensor([0])]; int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_148, concat_150_values1_0, cache_position_end, concat_150_values3_0))[name = string("concat_150")]; tensor key_states_127_perm_0 = const()[name = string("key_states_127_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_13_stride_0 = const()[name = string("key_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_127_cast_fp16 = transpose(perm = key_states_127_perm_0, x = key_states_125_cast_fp16)[name = string("transpose_649")]; tensor key_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = key_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = key_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_13_squeeze_mask_0, stride = key_cache_internal_tensor_assign_13_stride_0, update = key_states_127_cast_fp16, x = coreml_update_state_414)[name = string("key_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_13_cast_fp16, input = key_cache)[name = string("coreml_update_state_416_write_state")]; tensor coreml_update_state_416 = read_state(input = key_cache)[name = string("coreml_update_state_416")]; tensor value_states_75_perm_0 = const()[name = string("value_states_75_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_13_stride_0 = const()[name = string("value_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_75_cast_fp16 = transpose(perm = value_states_75_perm_0, x = var_4843_cast_fp16)[name = string("transpose_648")]; tensor value_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = value_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = value_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_13_squeeze_mask_0, stride = value_cache_internal_tensor_assign_13_stride_0, update = value_states_75_cast_fp16, x = coreml_update_state_415)[name = string("value_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_13_cast_fp16, input = value_cache)[name = string("coreml_update_state_417_write_state")]; tensor coreml_update_state_417 = read_state(input = value_cache)[name = string("coreml_update_state_417")]; tensor var_4937_begin_0 = const()[name = string("op_4937_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4937_end_0 = const()[name = string("op_4937_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4937_end_mask_0 = const()[name = string("op_4937_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4937_cast_fp16 = slice_by_index(begin = var_4937_begin_0, end = var_4937_end_0, end_mask = var_4937_end_mask_0, x = coreml_update_state_416)[name = string("op_4937_cast_fp16")]; tensor tile_24 = const()[name = string("tile_24"), val = tensor([1, 1])]; int32 var_4940_axis_0 = const()[name = string("op_4940_axis_0"), val = int32(1)]; tensor var_4940_cast_fp16_0, tensor var_4940_cast_fp16_1 = split(axis = var_4940_axis_0, split_sizes = tile_24, x = var_4937_cast_fp16)[name = string("op_4940_cast_fp16")]; tensor var_4947_begin_0 = const()[name = string("op_4947_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4947_end_0 = const()[name = string("op_4947_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4947_end_mask_0 = const()[name = string("op_4947_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4947_cast_fp16 = slice_by_index(begin = var_4947_begin_0, end = var_4947_end_0, end_mask = var_4947_end_mask_0, x = coreml_update_state_417)[name = string("op_4947_cast_fp16")]; tensor tile_25 = const()[name = string("tile_25"), val = tensor([1, 1])]; int32 var_4950_axis_0 = const()[name = string("op_4950_axis_0"), val = int32(1)]; tensor var_4950_cast_fp16_0, tensor var_4950_cast_fp16_1 = split(axis = var_4950_axis_0, split_sizes = tile_25, x = var_4947_cast_fp16)[name = string("op_4950_cast_fp16")]; tensor var_4953_split_sizes_0 = const()[name = string("op_4953_split_sizes_0"), val = tensor([8, 8])]; int32 var_4953_axis_0 = const()[name = string("op_4953_axis_0"), val = int32(1)]; tensor var_4953_0, tensor var_4953_1 = split(axis = var_4953_axis_0, split_sizes = var_4953_split_sizes_0, x = query_states_75_cast_fp16)[name = string("op_4953")]; bool attn_weights_193_transpose_x_0 = const()[name = string("attn_weights_193_transpose_x_0"), val = bool(false)]; bool attn_weights_193_transpose_y_0 = const()[name = string("attn_weights_193_transpose_y_0"), val = bool(false)]; tensor attn_weights_193_cast_fp16 = matmul(transpose_x = attn_weights_193_transpose_x_0, transpose_y = attn_weights_193_transpose_y_0, x = var_4940_cast_fp16_0, y = var_4953_0)[name = string("attn_weights_193_cast_fp16")]; fp16 var_4956_to_fp16 = const()[name = string("op_4956_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_195_cast_fp16 = mul(x = attn_weights_193_cast_fp16, y = var_4956_to_fp16)[name = string("attn_weights_195_cast_fp16")]; tensor attn_weights_197_cast_fp16 = add(x = attn_weights_195_cast_fp16, y = attn_mask_1)[name = string("attn_weights_197_cast_fp16")]; int32 var_4960 = const()[name = string("op_4960"), val = int32(-2)]; tensor attn_weights_199_cast_fp16 = softmax(axis = var_4960, x = attn_weights_197_cast_fp16)[name = string("attn_weights_199_cast_fp16")]; bool var_4966_transpose_x_1 = const()[name = string("op_4966_transpose_x_1"), val = bool(true)]; bool var_4966_transpose_y_1 = const()[name = string("op_4966_transpose_y_1"), val = bool(false)]; tensor var_4966_cast_fp16 = matmul(transpose_x = var_4966_transpose_x_1, transpose_y = var_4966_transpose_y_1, x = attn_weights_199_cast_fp16, y = var_4950_cast_fp16_0)[name = string("op_4966_cast_fp16")]; bool attn_weights_201_transpose_x_0 = const()[name = string("attn_weights_201_transpose_x_0"), val = bool(false)]; bool attn_weights_201_transpose_y_0 = const()[name = string("attn_weights_201_transpose_y_0"), val = bool(false)]; tensor attn_weights_201_cast_fp16 = matmul(transpose_x = attn_weights_201_transpose_x_0, transpose_y = attn_weights_201_transpose_y_0, x = var_4940_cast_fp16_1, y = var_4953_1)[name = string("attn_weights_201_cast_fp16")]; fp16 var_4968_to_fp16 = const()[name = string("op_4968_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_203_cast_fp16 = mul(x = attn_weights_201_cast_fp16, y = var_4968_to_fp16)[name = string("attn_weights_203_cast_fp16")]; tensor attn_weights_205_cast_fp16 = add(x = attn_weights_203_cast_fp16, y = attn_mask_1)[name = string("attn_weights_205_cast_fp16")]; int32 var_4972 = const()[name = string("op_4972"), val = int32(-2)]; tensor attn_weights_207_cast_fp16 = softmax(axis = var_4972, x = attn_weights_205_cast_fp16)[name = string("attn_weights_207_cast_fp16")]; bool attn_output_97_transpose_x_1 = const()[name = string("attn_output_97_transpose_x_1"), val = bool(true)]; bool attn_output_97_transpose_y_1 = const()[name = string("attn_output_97_transpose_y_1"), val = bool(false)]; tensor attn_output_97_cast_fp16 = matmul(transpose_x = attn_output_97_transpose_x_1, transpose_y = attn_output_97_transpose_y_1, x = attn_weights_207_cast_fp16, y = var_4950_cast_fp16_1)[name = string("attn_output_97_cast_fp16")]; int32 var_4980 = const()[name = string("op_4980"), val = int32(1)]; bool attn_output_99_interleave_0 = const()[name = string("attn_output_99_interleave_0"), val = bool(false)]; tensor attn_output_99_cast_fp16 = concat(axis = var_4980, interleave = attn_output_99_interleave_0, values = (var_4966_cast_fp16, attn_output_97_cast_fp16))[name = string("attn_output_99_cast_fp16")]; tensor var_4984_perm_0 = const()[name = string("op_4984_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_155x = const()[name = string("concat_155x"), val = tensor([1, 2048, 1, -1])]; tensor var_4984_cast_fp16 = transpose(perm = var_4984_perm_0, x = attn_output_99_cast_fp16)[name = string("transpose_647")]; tensor attn_output_103_cast_fp16 = reshape(shape = concat_155x, x = var_4984_cast_fp16)[name = string("attn_output_103_cast_fp16")]; tensor hidden_states_123_strides_0 = const()[name = string("hidden_states_123_strides_0"), val = tensor([1, 1])]; string hidden_states_123_pad_type_0 = const()[name = string("hidden_states_123_pad_type_0"), val = string("valid")]; tensor hidden_states_123_pad_0 = const()[name = string("hidden_states_123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_123_dilations_0 = const()[name = string("hidden_states_123_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_123_groups_0 = const()[name = string("hidden_states_123_groups_0"), val = int32(1)]; tensor hidden_states_123_cast_fp16 = conv(dilations = hidden_states_123_dilations_0, groups = hidden_states_123_groups_0, pad = hidden_states_123_pad_0, pad_type = hidden_states_123_pad_type_0, strides = hidden_states_123_strides_0, weight = layers_12_self_attn_o_proj_weight_cast_fp16, x = attn_output_103_cast_fp16)[name = string("hidden_states_123_cast_fp16")]; tensor hidden_states_125_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = hidden_states_123_cast_fp16)[name = string("hidden_states_125_cast_fp16")]; fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5017_cast_fp16 = mul(x = hidden_states_125_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_5017_cast_fp16")]; int32 var_5015 = const()[name = string("op_5015"), val = int32(1)]; bool doubled_101_interleave_0 = const()[name = string("doubled_101_interleave_0"), val = bool(false)]; tensor doubled_101_cast_fp16 = concat(axis = var_5015, interleave = doubled_101_interleave_0, values = (hidden_states_125_cast_fp16, var_5017_cast_fp16))[name = string("doubled_101_cast_fp16")]; tensor out_51_axes_0 = const()[name = string("out_51_axes_0"), val = tensor([1])]; tensor out_51_gamma_0_to_fp16 = const()[name = string("out_51_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415382912)))]; fp16 var_5027_to_fp16 = const()[name = string("op_5027_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_51_cast_fp16 = layer_norm(axes = out_51_axes_0, epsilon = var_5027_to_fp16, gamma = out_51_gamma_0_to_fp16, x = doubled_101_cast_fp16)[name = string("out_51_cast_fp16")]; tensor var_5038_split_sizes_0 = const()[name = string("op_5038_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5038_axis_0 = const()[name = string("op_5038_axis_0"), val = int32(1)]; tensor var_5038_cast_fp16_0, tensor var_5038_cast_fp16_1 = split(axis = var_5038_axis_0, split_sizes = var_5038_split_sizes_0, x = out_51_cast_fp16)[name = string("op_5038_cast_fp16")]; tensor input_25_strides_0 = const()[name = string("input_25_strides_0"), val = tensor([1, 1])]; string input_25_pad_type_0 = const()[name = string("input_25_pad_type_0"), val = string("valid")]; tensor input_25_pad_0 = const()[name = string("input_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_25_dilations_0 = const()[name = string("input_25_dilations_0"), val = tensor([1, 1])]; int32 input_25_groups_0 = const()[name = string("input_25_groups_0"), val = int32(1)]; tensor input_25_cast_fp16 = conv(dilations = input_25_dilations_0, groups = input_25_groups_0, pad = input_25_pad_0, pad_type = input_25_pad_type_0, strides = input_25_strides_0, weight = layers_12_mlp_gate_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("input_25_cast_fp16")]; tensor var_5055_cast_fp16 = silu(x = input_25_cast_fp16)[name = string("op_5055_cast_fp16")]; tensor var_5061_strides_0 = const()[name = string("op_5061_strides_0"), val = tensor([1, 1])]; string var_5061_pad_type_0 = const()[name = string("op_5061_pad_type_0"), val = string("valid")]; tensor var_5061_pad_0 = const()[name = string("op_5061_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5061_dilations_0 = const()[name = string("op_5061_dilations_0"), val = tensor([1, 1])]; int32 var_5061_groups_0 = const()[name = string("op_5061_groups_0"), val = int32(1)]; tensor var_5061_cast_fp16 = conv(dilations = var_5061_dilations_0, groups = var_5061_groups_0, pad = var_5061_pad_0, pad_type = var_5061_pad_type_0, strides = var_5061_strides_0, weight = layers_12_mlp_up_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("op_5061_cast_fp16")]; tensor x_129_cast_fp16 = mul(x = var_5055_cast_fp16, y = var_5061_cast_fp16)[name = string("x_129_cast_fp16")]; tensor hidden_states_127_strides_0 = const()[name = string("hidden_states_127_strides_0"), val = tensor([1, 1])]; string hidden_states_127_pad_type_0 = const()[name = string("hidden_states_127_pad_type_0"), val = string("valid")]; tensor hidden_states_127_pad_0 = const()[name = string("hidden_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_127_dilations_0 = const()[name = string("hidden_states_127_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_127_groups_0 = const()[name = string("hidden_states_127_groups_0"), val = int32(1)]; tensor hidden_states_127_cast_fp16 = conv(dilations = hidden_states_127_dilations_0, groups = hidden_states_127_groups_0, pad = hidden_states_127_pad_0, pad_type = hidden_states_127_pad_type_0, strides = hidden_states_127_strides_0, weight = layers_12_mlp_down_proj_weight_cast_fp16, x = x_129_cast_fp16)[name = string("hidden_states_127_cast_fp16")]; tensor hidden_states_129_cast_fp16 = add(x = hidden_states_125_cast_fp16, y = hidden_states_127_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5079_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_130_promoted_to_fp16)[name = string("op_5079_cast_fp16")]; int32 var_5077 = const()[name = string("op_5077"), val = int32(1)]; bool doubled_105_interleave_0 = const()[name = string("doubled_105_interleave_0"), val = bool(false)]; tensor doubled_105_cast_fp16 = concat(axis = var_5077, interleave = doubled_105_interleave_0, values = (hidden_states_129_cast_fp16, var_5079_cast_fp16))[name = string("doubled_105_cast_fp16")]; tensor out_53_axes_0 = const()[name = string("out_53_axes_0"), val = tensor([1])]; tensor out_53_gamma_0_to_fp16 = const()[name = string("out_53_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415391168)))]; fp16 var_5089_to_fp16 = const()[name = string("op_5089_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_53_cast_fp16 = layer_norm(axes = out_53_axes_0, epsilon = var_5089_to_fp16, gamma = out_53_gamma_0_to_fp16, x = doubled_105_cast_fp16)[name = string("out_53_cast_fp16")]; tensor var_5100_split_sizes_0 = const()[name = string("op_5100_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5100_axis_0 = const()[name = string("op_5100_axis_0"), val = int32(1)]; tensor var_5100_cast_fp16_0, tensor var_5100_cast_fp16_1 = split(axis = var_5100_axis_0, split_sizes = var_5100_split_sizes_0, x = out_53_cast_fp16)[name = string("op_5100_cast_fp16")]; tensor query_states_79_strides_0 = const()[name = string("query_states_79_strides_0"), val = tensor([1, 1])]; string query_states_79_pad_type_0 = const()[name = string("query_states_79_pad_type_0"), val = string("valid")]; tensor query_states_79_pad_0 = const()[name = string("query_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_79_dilations_0 = const()[name = string("query_states_79_dilations_0"), val = tensor([1, 1])]; int32 query_states_79_groups_0 = const()[name = string("query_states_79_groups_0"), val = int32(1)]; tensor query_states_79_cast_fp16 = conv(dilations = query_states_79_dilations_0, groups = query_states_79_groups_0, pad = query_states_79_pad_0, pad_type = query_states_79_pad_type_0, strides = query_states_79_strides_0, weight = layers_13_self_attn_q_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("query_states_79_cast_fp16")]; tensor key_states_131_strides_0 = const()[name = string("key_states_131_strides_0"), val = tensor([1, 1])]; string key_states_131_pad_type_0 = const()[name = string("key_states_131_pad_type_0"), val = string("valid")]; tensor key_states_131_pad_0 = const()[name = string("key_states_131_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_131_dilations_0 = const()[name = string("key_states_131_dilations_0"), val = tensor([1, 1])]; int32 key_states_131_groups_0 = const()[name = string("key_states_131_groups_0"), val = int32(1)]; tensor key_states_131_cast_fp16 = conv(dilations = key_states_131_dilations_0, groups = key_states_131_groups_0, pad = key_states_131_pad_0, pad_type = key_states_131_pad_type_0, strides = key_states_131_strides_0, weight = layers_13_self_attn_k_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("key_states_131_cast_fp16")]; tensor value_states_79_strides_0 = const()[name = string("value_states_79_strides_0"), val = tensor([1, 1])]; string value_states_79_pad_type_0 = const()[name = string("value_states_79_pad_type_0"), val = string("valid")]; tensor value_states_79_pad_0 = const()[name = string("value_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_79_dilations_0 = const()[name = string("value_states_79_dilations_0"), val = tensor([1, 1])]; int32 value_states_79_groups_0 = const()[name = string("value_states_79_groups_0"), val = int32(1)]; tensor value_states_79_cast_fp16 = conv(dilations = value_states_79_dilations_0, groups = value_states_79_groups_0, pad = value_states_79_pad_0, pad_type = value_states_79_pad_type_0, strides = value_states_79_strides_0, weight = layers_13_self_attn_v_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("value_states_79_cast_fp16")]; tensor concat_156x = const()[name = string("concat_156x"), val = tensor([1, 16, 128, -1])]; tensor x_131_cast_fp16 = reshape(shape = concat_156x, x = query_states_79_cast_fp16)[name = string("x_131_cast_fp16")]; tensor concat_157x = const()[name = string("concat_157x"), val = tensor([1, 2, 128, -1])]; tensor var_5157_cast_fp16 = reshape(shape = concat_157x, x = key_states_131_cast_fp16)[name = string("op_5157_cast_fp16")]; tensor concat_158x = const()[name = string("concat_158x"), val = tensor([1, 2, 128, -1])]; tensor var_5164_cast_fp16 = reshape(shape = concat_158x, x = value_states_79_cast_fp16)[name = string("op_5164_cast_fp16")]; tensor var_5168_cast_fp16 = mul(x = x_131_cast_fp16, y = var_869_cast_fp16)[name = string("op_5168_cast_fp16")]; tensor var_5169_split_sizes_0 = const()[name = string("op_5169_split_sizes_0"), val = tensor([64, 64])]; int32 var_5169_axis_0 = const()[name = string("op_5169_axis_0"), val = int32(-2)]; tensor var_5169_cast_fp16_0, tensor var_5169_cast_fp16_1 = split(axis = var_5169_axis_0, split_sizes = var_5169_split_sizes_0, x = x_131_cast_fp16)[name = string("op_5169_cast_fp16")]; fp16 const_132_promoted_to_fp16 = const()[name = string("const_132_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5171_cast_fp16 = mul(x = var_5169_cast_fp16_1, y = const_132_promoted_to_fp16)[name = string("op_5171_cast_fp16")]; int32 var_5173 = const()[name = string("op_5173"), val = int32(-2)]; bool var_5174_interleave_0 = const()[name = string("op_5174_interleave_0"), val = bool(false)]; tensor var_5174_cast_fp16 = concat(axis = var_5173, interleave = var_5174_interleave_0, values = (var_5171_cast_fp16, var_5169_cast_fp16_0))[name = string("op_5174_cast_fp16")]; tensor var_5175_cast_fp16 = mul(x = var_5174_cast_fp16, y = var_878_cast_fp16)[name = string("op_5175_cast_fp16")]; tensor query_states_81_cast_fp16 = add(x = var_5168_cast_fp16, y = var_5175_cast_fp16)[name = string("query_states_81_cast_fp16")]; tensor var_5181_cast_fp16 = mul(x = var_5157_cast_fp16, y = var_869_cast_fp16)[name = string("op_5181_cast_fp16")]; tensor var_5182_split_sizes_0 = const()[name = string("op_5182_split_sizes_0"), val = tensor([64, 64])]; int32 var_5182_axis_0 = const()[name = string("op_5182_axis_0"), val = int32(-2)]; tensor var_5182_cast_fp16_0, tensor var_5182_cast_fp16_1 = split(axis = var_5182_axis_0, split_sizes = var_5182_split_sizes_0, x = var_5157_cast_fp16)[name = string("op_5182_cast_fp16")]; fp16 const_133_promoted_to_fp16 = const()[name = string("const_133_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5184_cast_fp16 = mul(x = var_5182_cast_fp16_1, y = const_133_promoted_to_fp16)[name = string("op_5184_cast_fp16")]; int32 var_5186 = const()[name = string("op_5186"), val = int32(-2)]; bool var_5187_interleave_0 = const()[name = string("op_5187_interleave_0"), val = bool(false)]; tensor var_5187_cast_fp16 = concat(axis = var_5186, interleave = var_5187_interleave_0, values = (var_5184_cast_fp16, var_5182_cast_fp16_0))[name = string("op_5187_cast_fp16")]; tensor var_5188_cast_fp16 = mul(x = var_5187_cast_fp16, y = var_878_cast_fp16)[name = string("op_5188_cast_fp16")]; tensor key_states_135_cast_fp16 = add(x = var_5181_cast_fp16, y = var_5188_cast_fp16)[name = string("key_states_135_cast_fp16")]; tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; int32 concat_161_axis_0 = const()[name = string("concat_161_axis_0"), val = int32(0)]; bool concat_161_interleave_0 = const()[name = string("concat_161_interleave_0"), val = bool(false)]; tensor concat_161 = concat(axis = concat_161_axis_0, interleave = concat_161_interleave_0, values = (expand_dims_156, expand_dims_157, position_id, expand_dims_159))[name = string("concat_161")]; tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; tensor concat_162_values1_0 = const()[name = string("concat_162_values1_0"), val = tensor([0])]; tensor concat_162_values3_0 = const()[name = string("concat_162_values3_0"), val = tensor([0])]; int32 concat_162_axis_0 = const()[name = string("concat_162_axis_0"), val = int32(0)]; bool concat_162_interleave_0 = const()[name = string("concat_162_interleave_0"), val = bool(false)]; tensor concat_162 = concat(axis = concat_162_axis_0, interleave = concat_162_interleave_0, values = (expand_dims_160, concat_162_values1_0, cache_position_end, concat_162_values3_0))[name = string("concat_162")]; tensor key_states_137_perm_0 = const()[name = string("key_states_137_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_14_stride_0 = const()[name = string("key_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_137_cast_fp16 = transpose(perm = key_states_137_perm_0, x = key_states_135_cast_fp16)[name = string("transpose_646")]; tensor key_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = key_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = key_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_14_squeeze_mask_0, stride = key_cache_internal_tensor_assign_14_stride_0, update = key_states_137_cast_fp16, x = coreml_update_state_416)[name = string("key_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_14_cast_fp16, input = key_cache)[name = string("coreml_update_state_418_write_state")]; tensor coreml_update_state_418 = read_state(input = key_cache)[name = string("coreml_update_state_418")]; tensor value_states_81_perm_0 = const()[name = string("value_states_81_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_14_stride_0 = const()[name = string("value_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_81_cast_fp16 = transpose(perm = value_states_81_perm_0, x = var_5164_cast_fp16)[name = string("transpose_645")]; tensor value_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = value_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = value_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_14_squeeze_mask_0, stride = value_cache_internal_tensor_assign_14_stride_0, update = value_states_81_cast_fp16, x = coreml_update_state_417)[name = string("value_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_14_cast_fp16, input = value_cache)[name = string("coreml_update_state_419_write_state")]; tensor coreml_update_state_419 = read_state(input = value_cache)[name = string("coreml_update_state_419")]; tensor var_5258_begin_0 = const()[name = string("op_5258_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5258_end_0 = const()[name = string("op_5258_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5258_end_mask_0 = const()[name = string("op_5258_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5258_cast_fp16 = slice_by_index(begin = var_5258_begin_0, end = var_5258_end_0, end_mask = var_5258_end_mask_0, x = coreml_update_state_418)[name = string("op_5258_cast_fp16")]; tensor tile_26 = const()[name = string("tile_26"), val = tensor([1, 1])]; int32 var_5261_axis_0 = const()[name = string("op_5261_axis_0"), val = int32(1)]; tensor var_5261_cast_fp16_0, tensor var_5261_cast_fp16_1 = split(axis = var_5261_axis_0, split_sizes = tile_26, x = var_5258_cast_fp16)[name = string("op_5261_cast_fp16")]; tensor var_5268_begin_0 = const()[name = string("op_5268_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5268_end_0 = const()[name = string("op_5268_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5268_end_mask_0 = const()[name = string("op_5268_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5268_cast_fp16 = slice_by_index(begin = var_5268_begin_0, end = var_5268_end_0, end_mask = var_5268_end_mask_0, x = coreml_update_state_419)[name = string("op_5268_cast_fp16")]; tensor tile_27 = const()[name = string("tile_27"), val = tensor([1, 1])]; int32 var_5271_axis_0 = const()[name = string("op_5271_axis_0"), val = int32(1)]; tensor var_5271_cast_fp16_0, tensor var_5271_cast_fp16_1 = split(axis = var_5271_axis_0, split_sizes = tile_27, x = var_5268_cast_fp16)[name = string("op_5271_cast_fp16")]; tensor var_5274_split_sizes_0 = const()[name = string("op_5274_split_sizes_0"), val = tensor([8, 8])]; int32 var_5274_axis_0 = const()[name = string("op_5274_axis_0"), val = int32(1)]; tensor var_5274_0, tensor var_5274_1 = split(axis = var_5274_axis_0, split_sizes = var_5274_split_sizes_0, x = query_states_81_cast_fp16)[name = string("op_5274")]; bool attn_weights_209_transpose_x_0 = const()[name = string("attn_weights_209_transpose_x_0"), val = bool(false)]; bool attn_weights_209_transpose_y_0 = const()[name = string("attn_weights_209_transpose_y_0"), val = bool(false)]; tensor attn_weights_209_cast_fp16 = matmul(transpose_x = attn_weights_209_transpose_x_0, transpose_y = attn_weights_209_transpose_y_0, x = var_5261_cast_fp16_0, y = var_5274_0)[name = string("attn_weights_209_cast_fp16")]; fp16 var_5277_to_fp16 = const()[name = string("op_5277_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_211_cast_fp16 = mul(x = attn_weights_209_cast_fp16, y = var_5277_to_fp16)[name = string("attn_weights_211_cast_fp16")]; tensor attn_weights_213_cast_fp16 = add(x = attn_weights_211_cast_fp16, y = attn_mask_1)[name = string("attn_weights_213_cast_fp16")]; int32 var_5281 = const()[name = string("op_5281"), val = int32(-2)]; tensor attn_weights_215_cast_fp16 = softmax(axis = var_5281, x = attn_weights_213_cast_fp16)[name = string("attn_weights_215_cast_fp16")]; bool var_5287_transpose_x_1 = const()[name = string("op_5287_transpose_x_1"), val = bool(true)]; bool var_5287_transpose_y_1 = const()[name = string("op_5287_transpose_y_1"), val = bool(false)]; tensor var_5287_cast_fp16 = matmul(transpose_x = var_5287_transpose_x_1, transpose_y = var_5287_transpose_y_1, x = attn_weights_215_cast_fp16, y = var_5271_cast_fp16_0)[name = string("op_5287_cast_fp16")]; bool attn_weights_217_transpose_x_0 = const()[name = string("attn_weights_217_transpose_x_0"), val = bool(false)]; bool attn_weights_217_transpose_y_0 = const()[name = string("attn_weights_217_transpose_y_0"), val = bool(false)]; tensor attn_weights_217_cast_fp16 = matmul(transpose_x = attn_weights_217_transpose_x_0, transpose_y = attn_weights_217_transpose_y_0, x = var_5261_cast_fp16_1, y = var_5274_1)[name = string("attn_weights_217_cast_fp16")]; fp16 var_5289_to_fp16 = const()[name = string("op_5289_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_219_cast_fp16 = mul(x = attn_weights_217_cast_fp16, y = var_5289_to_fp16)[name = string("attn_weights_219_cast_fp16")]; tensor attn_weights_221_cast_fp16 = add(x = attn_weights_219_cast_fp16, y = attn_mask_1)[name = string("attn_weights_221_cast_fp16")]; int32 var_5293 = const()[name = string("op_5293"), val = int32(-2)]; tensor attn_weights_223_cast_fp16 = softmax(axis = var_5293, x = attn_weights_221_cast_fp16)[name = string("attn_weights_223_cast_fp16")]; bool attn_output_105_transpose_x_1 = const()[name = string("attn_output_105_transpose_x_1"), val = bool(true)]; bool attn_output_105_transpose_y_1 = const()[name = string("attn_output_105_transpose_y_1"), val = bool(false)]; tensor attn_output_105_cast_fp16 = matmul(transpose_x = attn_output_105_transpose_x_1, transpose_y = attn_output_105_transpose_y_1, x = attn_weights_223_cast_fp16, y = var_5271_cast_fp16_1)[name = string("attn_output_105_cast_fp16")]; int32 var_5301 = const()[name = string("op_5301"), val = int32(1)]; bool attn_output_107_interleave_0 = const()[name = string("attn_output_107_interleave_0"), val = bool(false)]; tensor attn_output_107_cast_fp16 = concat(axis = var_5301, interleave = attn_output_107_interleave_0, values = (var_5287_cast_fp16, attn_output_105_cast_fp16))[name = string("attn_output_107_cast_fp16")]; tensor var_5305_perm_0 = const()[name = string("op_5305_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_167x = const()[name = string("concat_167x"), val = tensor([1, 2048, 1, -1])]; tensor var_5305_cast_fp16 = transpose(perm = var_5305_perm_0, x = attn_output_107_cast_fp16)[name = string("transpose_644")]; tensor attn_output_111_cast_fp16 = reshape(shape = concat_167x, x = var_5305_cast_fp16)[name = string("attn_output_111_cast_fp16")]; tensor hidden_states_133_strides_0 = const()[name = string("hidden_states_133_strides_0"), val = tensor([1, 1])]; string hidden_states_133_pad_type_0 = const()[name = string("hidden_states_133_pad_type_0"), val = string("valid")]; tensor hidden_states_133_pad_0 = const()[name = string("hidden_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_133_dilations_0 = const()[name = string("hidden_states_133_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_133_groups_0 = const()[name = string("hidden_states_133_groups_0"), val = int32(1)]; tensor hidden_states_133_cast_fp16 = conv(dilations = hidden_states_133_dilations_0, groups = hidden_states_133_groups_0, pad = hidden_states_133_pad_0, pad_type = hidden_states_133_pad_type_0, strides = hidden_states_133_strides_0, weight = layers_13_self_attn_o_proj_weight_cast_fp16, x = attn_output_111_cast_fp16)[name = string("hidden_states_133_cast_fp16")]; tensor hidden_states_135_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = hidden_states_133_cast_fp16)[name = string("hidden_states_135_cast_fp16")]; fp16 const_138_promoted_to_fp16 = const()[name = string("const_138_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5338_cast_fp16 = mul(x = hidden_states_135_cast_fp16, y = const_138_promoted_to_fp16)[name = string("op_5338_cast_fp16")]; int32 var_5336 = const()[name = string("op_5336"), val = int32(1)]; bool doubled_109_interleave_0 = const()[name = string("doubled_109_interleave_0"), val = bool(false)]; tensor doubled_109_cast_fp16 = concat(axis = var_5336, interleave = doubled_109_interleave_0, values = (hidden_states_135_cast_fp16, var_5338_cast_fp16))[name = string("doubled_109_cast_fp16")]; tensor out_55_axes_0 = const()[name = string("out_55_axes_0"), val = tensor([1])]; tensor out_55_gamma_0_to_fp16 = const()[name = string("out_55_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415399424)))]; fp16 var_5348_to_fp16 = const()[name = string("op_5348_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_55_cast_fp16 = layer_norm(axes = out_55_axes_0, epsilon = var_5348_to_fp16, gamma = out_55_gamma_0_to_fp16, x = doubled_109_cast_fp16)[name = string("out_55_cast_fp16")]; tensor var_5359_split_sizes_0 = const()[name = string("op_5359_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5359_axis_0 = const()[name = string("op_5359_axis_0"), val = int32(1)]; tensor var_5359_cast_fp16_0, tensor var_5359_cast_fp16_1 = split(axis = var_5359_axis_0, split_sizes = var_5359_split_sizes_0, x = out_55_cast_fp16)[name = string("op_5359_cast_fp16")]; tensor input_27_strides_0 = const()[name = string("input_27_strides_0"), val = tensor([1, 1])]; string input_27_pad_type_0 = const()[name = string("input_27_pad_type_0"), val = string("valid")]; tensor input_27_pad_0 = const()[name = string("input_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_27_dilations_0 = const()[name = string("input_27_dilations_0"), val = tensor([1, 1])]; int32 input_27_groups_0 = const()[name = string("input_27_groups_0"), val = int32(1)]; tensor input_27_cast_fp16 = conv(dilations = input_27_dilations_0, groups = input_27_groups_0, pad = input_27_pad_0, pad_type = input_27_pad_type_0, strides = input_27_strides_0, weight = layers_13_mlp_gate_proj_weight_cast_fp16, x = var_5359_cast_fp16_0)[name = string("input_27_cast_fp16")]; tensor var_5376_cast_fp16 = silu(x = input_27_cast_fp16)[name = string("op_5376_cast_fp16")]; tensor layers_13_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_13_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415407680)))]; tensor var_5382_strides_0 = const()[name = string("op_5382_strides_0"), val = tensor([1, 1])]; string var_5382_pad_type_0 = const()[name = string("op_5382_pad_type_0"), val = string("valid")]; tensor var_5382_pad_0 = const()[name = string("op_5382_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5382_dilations_0 = const()[name = string("op_5382_dilations_0"), val = tensor([1, 1])]; int32 var_5382_groups_0 = const()[name = string("op_5382_groups_0"), val = int32(1)]; tensor var_5382_cast_fp16 = conv(dilations = var_5382_dilations_0, groups = var_5382_groups_0, pad = var_5382_pad_0, pad_type = var_5382_pad_type_0, strides = var_5382_strides_0, weight = layers_13_mlp_up_proj_weight_to_fp16, x = var_5359_cast_fp16_0)[name = string("op_5382_cast_fp16")]; tensor x_139_cast_fp16 = mul(x = var_5376_cast_fp16, y = var_5382_cast_fp16)[name = string("x_139_cast_fp16")]; tensor hidden_states_137_strides_0 = const()[name = string("hidden_states_137_strides_0"), val = tensor([1, 1])]; string hidden_states_137_pad_type_0 = const()[name = string("hidden_states_137_pad_type_0"), val = string("valid")]; tensor hidden_states_137_pad_0 = const()[name = string("hidden_states_137_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_137_dilations_0 = const()[name = string("hidden_states_137_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_137_groups_0 = const()[name = string("hidden_states_137_groups_0"), val = int32(1)]; tensor hidden_states_137_cast_fp16 = conv(dilations = hidden_states_137_dilations_0, groups = hidden_states_137_groups_0, pad = hidden_states_137_pad_0, pad_type = hidden_states_137_pad_type_0, strides = hidden_states_137_strides_0, weight = layers_13_mlp_down_proj_weight_cast_fp16, x = x_139_cast_fp16)[name = string("hidden_states_137_cast_fp16")]; tensor hidden_states_139_cast_fp16 = add(x = hidden_states_135_cast_fp16, y = hidden_states_137_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; fp16 const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5400_cast_fp16 = mul(x = hidden_states_139_cast_fp16, y = const_140_promoted_to_fp16)[name = string("op_5400_cast_fp16")]; int32 var_5398 = const()[name = string("op_5398"), val = int32(1)]; bool doubled_113_interleave_0 = const()[name = string("doubled_113_interleave_0"), val = bool(false)]; tensor doubled_113_cast_fp16 = concat(axis = var_5398, interleave = doubled_113_interleave_0, values = (hidden_states_139_cast_fp16, var_5400_cast_fp16))[name = string("doubled_113_cast_fp16")]; tensor out_57_axes_0 = const()[name = string("out_57_axes_0"), val = tensor([1])]; tensor out_57_gamma_0_to_fp16 = const()[name = string("out_57_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440573568)))]; fp16 var_5410_to_fp16 = const()[name = string("op_5410_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_57_cast_fp16 = layer_norm(axes = out_57_axes_0, epsilon = var_5410_to_fp16, gamma = out_57_gamma_0_to_fp16, x = doubled_113_cast_fp16)[name = string("out_57_cast_fp16")]; tensor var_5421_split_sizes_0 = const()[name = string("op_5421_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5421_axis_0 = const()[name = string("op_5421_axis_0"), val = int32(1)]; tensor var_5421_cast_fp16_0, tensor var_5421_cast_fp16_1 = split(axis = var_5421_axis_0, split_sizes = var_5421_split_sizes_0, x = out_57_cast_fp16)[name = string("op_5421_cast_fp16")]; tensor query_states_85_strides_0 = const()[name = string("query_states_85_strides_0"), val = tensor([1, 1])]; string query_states_85_pad_type_0 = const()[name = string("query_states_85_pad_type_0"), val = string("valid")]; tensor query_states_85_pad_0 = const()[name = string("query_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_85_dilations_0 = const()[name = string("query_states_85_dilations_0"), val = tensor([1, 1])]; int32 query_states_85_groups_0 = const()[name = string("query_states_85_groups_0"), val = int32(1)]; tensor query_states_85_cast_fp16 = conv(dilations = query_states_85_dilations_0, groups = query_states_85_groups_0, pad = query_states_85_pad_0, pad_type = query_states_85_pad_type_0, strides = query_states_85_strides_0, weight = layers_14_self_attn_q_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("query_states_85_cast_fp16")]; tensor layers_14_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_14_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440581824)))]; tensor key_states_141_strides_0 = const()[name = string("key_states_141_strides_0"), val = tensor([1, 1])]; string key_states_141_pad_type_0 = const()[name = string("key_states_141_pad_type_0"), val = string("valid")]; tensor key_states_141_pad_0 = const()[name = string("key_states_141_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_141_dilations_0 = const()[name = string("key_states_141_dilations_0"), val = tensor([1, 1])]; int32 key_states_141_groups_0 = const()[name = string("key_states_141_groups_0"), val = int32(1)]; tensor key_states_141_cast_fp16 = conv(dilations = key_states_141_dilations_0, groups = key_states_141_groups_0, pad = key_states_141_pad_0, pad_type = key_states_141_pad_type_0, strides = key_states_141_strides_0, weight = layers_14_self_attn_k_proj_weight_to_fp16, x = var_5421_cast_fp16_0)[name = string("key_states_141_cast_fp16")]; tensor value_states_85_strides_0 = const()[name = string("value_states_85_strides_0"), val = tensor([1, 1])]; string value_states_85_pad_type_0 = const()[name = string("value_states_85_pad_type_0"), val = string("valid")]; tensor value_states_85_pad_0 = const()[name = string("value_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_85_dilations_0 = const()[name = string("value_states_85_dilations_0"), val = tensor([1, 1])]; int32 value_states_85_groups_0 = const()[name = string("value_states_85_groups_0"), val = int32(1)]; tensor value_states_85_cast_fp16 = conv(dilations = value_states_85_dilations_0, groups = value_states_85_groups_0, pad = value_states_85_pad_0, pad_type = value_states_85_pad_type_0, strides = value_states_85_strides_0, weight = layers_14_self_attn_v_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("value_states_85_cast_fp16")]; tensor concat_168x = const()[name = string("concat_168x"), val = tensor([1, 16, 128, -1])]; tensor x_141_cast_fp16 = reshape(shape = concat_168x, x = query_states_85_cast_fp16)[name = string("x_141_cast_fp16")]; tensor concat_169x = const()[name = string("concat_169x"), val = tensor([1, 2, 128, -1])]; tensor var_5478_cast_fp16 = reshape(shape = concat_169x, x = key_states_141_cast_fp16)[name = string("op_5478_cast_fp16")]; tensor concat_170x = const()[name = string("concat_170x"), val = tensor([1, 2, 128, -1])]; tensor var_5485_cast_fp16 = reshape(shape = concat_170x, x = value_states_85_cast_fp16)[name = string("op_5485_cast_fp16")]; tensor var_5489_cast_fp16 = mul(x = x_141_cast_fp16, y = var_869_cast_fp16)[name = string("op_5489_cast_fp16")]; tensor var_5490_split_sizes_0 = const()[name = string("op_5490_split_sizes_0"), val = tensor([64, 64])]; int32 var_5490_axis_0 = const()[name = string("op_5490_axis_0"), val = int32(-2)]; tensor var_5490_cast_fp16_0, tensor var_5490_cast_fp16_1 = split(axis = var_5490_axis_0, split_sizes = var_5490_split_sizes_0, x = x_141_cast_fp16)[name = string("op_5490_cast_fp16")]; fp16 const_142_promoted_to_fp16 = const()[name = string("const_142_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5492_cast_fp16 = mul(x = var_5490_cast_fp16_1, y = const_142_promoted_to_fp16)[name = string("op_5492_cast_fp16")]; int32 var_5494 = const()[name = string("op_5494"), val = int32(-2)]; bool var_5495_interleave_0 = const()[name = string("op_5495_interleave_0"), val = bool(false)]; tensor var_5495_cast_fp16 = concat(axis = var_5494, interleave = var_5495_interleave_0, values = (var_5492_cast_fp16, var_5490_cast_fp16_0))[name = string("op_5495_cast_fp16")]; tensor var_5496_cast_fp16 = mul(x = var_5495_cast_fp16, y = var_878_cast_fp16)[name = string("op_5496_cast_fp16")]; tensor query_states_87_cast_fp16 = add(x = var_5489_cast_fp16, y = var_5496_cast_fp16)[name = string("query_states_87_cast_fp16")]; tensor var_5502_cast_fp16 = mul(x = var_5478_cast_fp16, y = var_869_cast_fp16)[name = string("op_5502_cast_fp16")]; tensor var_5503_split_sizes_0 = const()[name = string("op_5503_split_sizes_0"), val = tensor([64, 64])]; int32 var_5503_axis_0 = const()[name = string("op_5503_axis_0"), val = int32(-2)]; tensor var_5503_cast_fp16_0, tensor var_5503_cast_fp16_1 = split(axis = var_5503_axis_0, split_sizes = var_5503_split_sizes_0, x = var_5478_cast_fp16)[name = string("op_5503_cast_fp16")]; fp16 const_143_promoted_to_fp16 = const()[name = string("const_143_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5505_cast_fp16 = mul(x = var_5503_cast_fp16_1, y = const_143_promoted_to_fp16)[name = string("op_5505_cast_fp16")]; int32 var_5507 = const()[name = string("op_5507"), val = int32(-2)]; bool var_5508_interleave_0 = const()[name = string("op_5508_interleave_0"), val = bool(false)]; tensor var_5508_cast_fp16 = concat(axis = var_5507, interleave = var_5508_interleave_0, values = (var_5505_cast_fp16, var_5503_cast_fp16_0))[name = string("op_5508_cast_fp16")]; tensor var_5509_cast_fp16 = mul(x = var_5508_cast_fp16, y = var_878_cast_fp16)[name = string("op_5509_cast_fp16")]; tensor key_states_145_cast_fp16 = add(x = var_5502_cast_fp16, y = var_5509_cast_fp16)[name = string("key_states_145_cast_fp16")]; tensor expand_dims_168 = const()[name = string("expand_dims_168"), val = tensor([14])]; tensor expand_dims_169 = const()[name = string("expand_dims_169"), val = tensor([0])]; tensor expand_dims_171 = const()[name = string("expand_dims_171"), val = tensor([0])]; int32 concat_173_axis_0 = const()[name = string("concat_173_axis_0"), val = int32(0)]; bool concat_173_interleave_0 = const()[name = string("concat_173_interleave_0"), val = bool(false)]; tensor concat_173 = concat(axis = concat_173_axis_0, interleave = concat_173_interleave_0, values = (expand_dims_168, expand_dims_169, position_id, expand_dims_171))[name = string("concat_173")]; tensor expand_dims_172 = const()[name = string("expand_dims_172"), val = tensor([15])]; tensor concat_174_values1_0 = const()[name = string("concat_174_values1_0"), val = tensor([0])]; tensor concat_174_values3_0 = const()[name = string("concat_174_values3_0"), val = tensor([0])]; int32 concat_174_axis_0 = const()[name = string("concat_174_axis_0"), val = int32(0)]; bool concat_174_interleave_0 = const()[name = string("concat_174_interleave_0"), val = bool(false)]; tensor concat_174 = concat(axis = concat_174_axis_0, interleave = concat_174_interleave_0, values = (expand_dims_172, concat_174_values1_0, cache_position_end, concat_174_values3_0))[name = string("concat_174")]; tensor key_states_147_perm_0 = const()[name = string("key_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_15_stride_0 = const()[name = string("key_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_147_cast_fp16 = transpose(perm = key_states_147_perm_0, x = key_states_145_cast_fp16)[name = string("transpose_643")]; tensor key_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = key_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = key_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_15_squeeze_mask_0, stride = key_cache_internal_tensor_assign_15_stride_0, update = key_states_147_cast_fp16, x = coreml_update_state_418)[name = string("key_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_15_cast_fp16, input = key_cache)[name = string("coreml_update_state_420_write_state")]; tensor coreml_update_state_420 = read_state(input = key_cache)[name = string("coreml_update_state_420")]; tensor value_states_87_perm_0 = const()[name = string("value_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_15_stride_0 = const()[name = string("value_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_87_cast_fp16 = transpose(perm = value_states_87_perm_0, x = var_5485_cast_fp16)[name = string("transpose_642")]; tensor value_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = value_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = value_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_15_squeeze_mask_0, stride = value_cache_internal_tensor_assign_15_stride_0, update = value_states_87_cast_fp16, x = coreml_update_state_419)[name = string("value_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_15_cast_fp16, input = value_cache)[name = string("coreml_update_state_421_write_state")]; tensor coreml_update_state_421 = read_state(input = value_cache)[name = string("coreml_update_state_421")]; tensor var_5579_begin_0 = const()[name = string("op_5579_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5579_end_0 = const()[name = string("op_5579_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5579_end_mask_0 = const()[name = string("op_5579_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5579_cast_fp16 = slice_by_index(begin = var_5579_begin_0, end = var_5579_end_0, end_mask = var_5579_end_mask_0, x = coreml_update_state_420)[name = string("op_5579_cast_fp16")]; tensor tile_28 = const()[name = string("tile_28"), val = tensor([1, 1])]; int32 var_5582_axis_0 = const()[name = string("op_5582_axis_0"), val = int32(1)]; tensor var_5582_cast_fp16_0, tensor var_5582_cast_fp16_1 = split(axis = var_5582_axis_0, split_sizes = tile_28, x = var_5579_cast_fp16)[name = string("op_5582_cast_fp16")]; tensor var_5589_begin_0 = const()[name = string("op_5589_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5589_end_0 = const()[name = string("op_5589_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5589_end_mask_0 = const()[name = string("op_5589_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5589_cast_fp16 = slice_by_index(begin = var_5589_begin_0, end = var_5589_end_0, end_mask = var_5589_end_mask_0, x = coreml_update_state_421)[name = string("op_5589_cast_fp16")]; tensor tile_29 = const()[name = string("tile_29"), val = tensor([1, 1])]; int32 var_5592_axis_0 = const()[name = string("op_5592_axis_0"), val = int32(1)]; tensor var_5592_cast_fp16_0, tensor var_5592_cast_fp16_1 = split(axis = var_5592_axis_0, split_sizes = tile_29, x = var_5589_cast_fp16)[name = string("op_5592_cast_fp16")]; tensor var_5595_split_sizes_0 = const()[name = string("op_5595_split_sizes_0"), val = tensor([8, 8])]; int32 var_5595_axis_0 = const()[name = string("op_5595_axis_0"), val = int32(1)]; tensor var_5595_0, tensor var_5595_1 = split(axis = var_5595_axis_0, split_sizes = var_5595_split_sizes_0, x = query_states_87_cast_fp16)[name = string("op_5595")]; bool attn_weights_225_transpose_x_0 = const()[name = string("attn_weights_225_transpose_x_0"), val = bool(false)]; bool attn_weights_225_transpose_y_0 = const()[name = string("attn_weights_225_transpose_y_0"), val = bool(false)]; tensor attn_weights_225_cast_fp16 = matmul(transpose_x = attn_weights_225_transpose_x_0, transpose_y = attn_weights_225_transpose_y_0, x = var_5582_cast_fp16_0, y = var_5595_0)[name = string("attn_weights_225_cast_fp16")]; fp16 var_5598_to_fp16 = const()[name = string("op_5598_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_227_cast_fp16 = mul(x = attn_weights_225_cast_fp16, y = var_5598_to_fp16)[name = string("attn_weights_227_cast_fp16")]; tensor attn_weights_229_cast_fp16 = add(x = attn_weights_227_cast_fp16, y = attn_mask_1)[name = string("attn_weights_229_cast_fp16")]; int32 var_5602 = const()[name = string("op_5602"), val = int32(-2)]; tensor attn_weights_231_cast_fp16 = softmax(axis = var_5602, x = attn_weights_229_cast_fp16)[name = string("attn_weights_231_cast_fp16")]; bool var_5608_transpose_x_1 = const()[name = string("op_5608_transpose_x_1"), val = bool(true)]; bool var_5608_transpose_y_1 = const()[name = string("op_5608_transpose_y_1"), val = bool(false)]; tensor var_5608_cast_fp16 = matmul(transpose_x = var_5608_transpose_x_1, transpose_y = var_5608_transpose_y_1, x = attn_weights_231_cast_fp16, y = var_5592_cast_fp16_0)[name = string("op_5608_cast_fp16")]; bool attn_weights_233_transpose_x_0 = const()[name = string("attn_weights_233_transpose_x_0"), val = bool(false)]; bool attn_weights_233_transpose_y_0 = const()[name = string("attn_weights_233_transpose_y_0"), val = bool(false)]; tensor attn_weights_233_cast_fp16 = matmul(transpose_x = attn_weights_233_transpose_x_0, transpose_y = attn_weights_233_transpose_y_0, x = var_5582_cast_fp16_1, y = var_5595_1)[name = string("attn_weights_233_cast_fp16")]; fp16 var_5610_to_fp16 = const()[name = string("op_5610_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_235_cast_fp16 = mul(x = attn_weights_233_cast_fp16, y = var_5610_to_fp16)[name = string("attn_weights_235_cast_fp16")]; tensor attn_weights_237_cast_fp16 = add(x = attn_weights_235_cast_fp16, y = attn_mask_1)[name = string("attn_weights_237_cast_fp16")]; int32 var_5614 = const()[name = string("op_5614"), val = int32(-2)]; tensor attn_weights_239_cast_fp16 = softmax(axis = var_5614, x = attn_weights_237_cast_fp16)[name = string("attn_weights_239_cast_fp16")]; bool attn_output_113_transpose_x_1 = const()[name = string("attn_output_113_transpose_x_1"), val = bool(true)]; bool attn_output_113_transpose_y_1 = const()[name = string("attn_output_113_transpose_y_1"), val = bool(false)]; tensor attn_output_113_cast_fp16 = matmul(transpose_x = attn_output_113_transpose_x_1, transpose_y = attn_output_113_transpose_y_1, x = attn_weights_239_cast_fp16, y = var_5592_cast_fp16_1)[name = string("attn_output_113_cast_fp16")]; int32 var_5622 = const()[name = string("op_5622"), val = int32(1)]; bool attn_output_115_interleave_0 = const()[name = string("attn_output_115_interleave_0"), val = bool(false)]; tensor attn_output_115_cast_fp16 = concat(axis = var_5622, interleave = attn_output_115_interleave_0, values = (var_5608_cast_fp16, attn_output_113_cast_fp16))[name = string("attn_output_115_cast_fp16")]; tensor var_5626_perm_0 = const()[name = string("op_5626_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_179x = const()[name = string("concat_179x"), val = tensor([1, 2048, 1, -1])]; tensor var_5626_cast_fp16 = transpose(perm = var_5626_perm_0, x = attn_output_115_cast_fp16)[name = string("transpose_641")]; tensor attn_output_119_cast_fp16 = reshape(shape = concat_179x, x = var_5626_cast_fp16)[name = string("attn_output_119_cast_fp16")]; tensor hidden_states_143_strides_0 = const()[name = string("hidden_states_143_strides_0"), val = tensor([1, 1])]; string hidden_states_143_pad_type_0 = const()[name = string("hidden_states_143_pad_type_0"), val = string("valid")]; tensor hidden_states_143_pad_0 = const()[name = string("hidden_states_143_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_143_dilations_0 = const()[name = string("hidden_states_143_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_143_groups_0 = const()[name = string("hidden_states_143_groups_0"), val = int32(1)]; tensor hidden_states_143_cast_fp16 = conv(dilations = hidden_states_143_dilations_0, groups = hidden_states_143_groups_0, pad = hidden_states_143_pad_0, pad_type = hidden_states_143_pad_type_0, strides = hidden_states_143_strides_0, weight = layers_14_self_attn_o_proj_weight_cast_fp16, x = attn_output_119_cast_fp16)[name = string("hidden_states_143_cast_fp16")]; tensor hidden_states_145_cast_fp16 = add(x = hidden_states_139_cast_fp16, y = hidden_states_143_cast_fp16)[name = string("hidden_states_145_cast_fp16")]; fp16 const_148_promoted_to_fp16 = const()[name = string("const_148_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5659_cast_fp16 = mul(x = hidden_states_145_cast_fp16, y = const_148_promoted_to_fp16)[name = string("op_5659_cast_fp16")]; int32 var_5657 = const()[name = string("op_5657"), val = int32(1)]; bool doubled_117_interleave_0 = const()[name = string("doubled_117_interleave_0"), val = bool(false)]; tensor doubled_117_cast_fp16 = concat(axis = var_5657, interleave = doubled_117_interleave_0, values = (hidden_states_145_cast_fp16, var_5659_cast_fp16))[name = string("doubled_117_cast_fp16")]; tensor out_59_axes_0 = const()[name = string("out_59_axes_0"), val = tensor([1])]; tensor out_59_gamma_0_to_fp16 = const()[name = string("out_59_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441630464)))]; fp16 var_5669_to_fp16 = const()[name = string("op_5669_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_59_cast_fp16 = layer_norm(axes = out_59_axes_0, epsilon = var_5669_to_fp16, gamma = out_59_gamma_0_to_fp16, x = doubled_117_cast_fp16)[name = string("out_59_cast_fp16")]; tensor var_5680_split_sizes_0 = const()[name = string("op_5680_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5680_axis_0 = const()[name = string("op_5680_axis_0"), val = int32(1)]; tensor var_5680_cast_fp16_0, tensor var_5680_cast_fp16_1 = split(axis = var_5680_axis_0, split_sizes = var_5680_split_sizes_0, x = out_59_cast_fp16)[name = string("op_5680_cast_fp16")]; tensor input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor([1, 1])]; string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; tensor input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor([1, 1])]; int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; tensor input_29_cast_fp16 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_14_mlp_gate_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("input_29_cast_fp16")]; tensor var_5697_cast_fp16 = silu(x = input_29_cast_fp16)[name = string("op_5697_cast_fp16")]; tensor var_5703_strides_0 = const()[name = string("op_5703_strides_0"), val = tensor([1, 1])]; string var_5703_pad_type_0 = const()[name = string("op_5703_pad_type_0"), val = string("valid")]; tensor var_5703_pad_0 = const()[name = string("op_5703_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5703_dilations_0 = const()[name = string("op_5703_dilations_0"), val = tensor([1, 1])]; int32 var_5703_groups_0 = const()[name = string("op_5703_groups_0"), val = int32(1)]; tensor var_5703_cast_fp16 = conv(dilations = var_5703_dilations_0, groups = var_5703_groups_0, pad = var_5703_pad_0, pad_type = var_5703_pad_type_0, strides = var_5703_strides_0, weight = layers_14_mlp_up_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("op_5703_cast_fp16")]; tensor x_149_cast_fp16 = mul(x = var_5697_cast_fp16, y = var_5703_cast_fp16)[name = string("x_149_cast_fp16")]; tensor hidden_states_147_strides_0 = const()[name = string("hidden_states_147_strides_0"), val = tensor([1, 1])]; string hidden_states_147_pad_type_0 = const()[name = string("hidden_states_147_pad_type_0"), val = string("valid")]; tensor hidden_states_147_pad_0 = const()[name = string("hidden_states_147_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_147_dilations_0 = const()[name = string("hidden_states_147_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_147_groups_0 = const()[name = string("hidden_states_147_groups_0"), val = int32(1)]; tensor hidden_states_147_cast_fp16 = conv(dilations = hidden_states_147_dilations_0, groups = hidden_states_147_groups_0, pad = hidden_states_147_pad_0, pad_type = hidden_states_147_pad_type_0, strides = hidden_states_147_strides_0, weight = layers_14_mlp_down_proj_weight_cast_fp16, x = x_149_cast_fp16)[name = string("hidden_states_147_cast_fp16")]; tensor hidden_states_149_cast_fp16 = add(x = hidden_states_145_cast_fp16, y = hidden_states_147_cast_fp16)[name = string("hidden_states_149_cast_fp16")]; fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5721_cast_fp16 = mul(x = hidden_states_149_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_5721_cast_fp16")]; int32 var_5719 = const()[name = string("op_5719"), val = int32(1)]; bool doubled_121_interleave_0 = const()[name = string("doubled_121_interleave_0"), val = bool(false)]; tensor doubled_121_cast_fp16 = concat(axis = var_5719, interleave = doubled_121_interleave_0, values = (hidden_states_149_cast_fp16, var_5721_cast_fp16))[name = string("doubled_121_cast_fp16")]; tensor out_61_axes_0 = const()[name = string("out_61_axes_0"), val = tensor([1])]; tensor out_61_gamma_0_to_fp16 = const()[name = string("out_61_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441638720)))]; fp16 var_5731_to_fp16 = const()[name = string("op_5731_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_61_cast_fp16 = layer_norm(axes = out_61_axes_0, epsilon = var_5731_to_fp16, gamma = out_61_gamma_0_to_fp16, x = doubled_121_cast_fp16)[name = string("out_61_cast_fp16")]; tensor var_5742_split_sizes_0 = const()[name = string("op_5742_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5742_axis_0 = const()[name = string("op_5742_axis_0"), val = int32(1)]; tensor var_5742_cast_fp16_0, tensor var_5742_cast_fp16_1 = split(axis = var_5742_axis_0, split_sizes = var_5742_split_sizes_0, x = out_61_cast_fp16)[name = string("op_5742_cast_fp16")]; tensor query_states_91_strides_0 = const()[name = string("query_states_91_strides_0"), val = tensor([1, 1])]; string query_states_91_pad_type_0 = const()[name = string("query_states_91_pad_type_0"), val = string("valid")]; tensor query_states_91_pad_0 = const()[name = string("query_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_91_dilations_0 = const()[name = string("query_states_91_dilations_0"), val = tensor([1, 1])]; int32 query_states_91_groups_0 = const()[name = string("query_states_91_groups_0"), val = int32(1)]; tensor query_states_91_cast_fp16 = conv(dilations = query_states_91_dilations_0, groups = query_states_91_groups_0, pad = query_states_91_pad_0, pad_type = query_states_91_pad_type_0, strides = query_states_91_strides_0, weight = layers_15_self_attn_q_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("query_states_91_cast_fp16")]; tensor key_states_151_strides_0 = const()[name = string("key_states_151_strides_0"), val = tensor([1, 1])]; string key_states_151_pad_type_0 = const()[name = string("key_states_151_pad_type_0"), val = string("valid")]; tensor key_states_151_pad_0 = const()[name = string("key_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_151_dilations_0 = const()[name = string("key_states_151_dilations_0"), val = tensor([1, 1])]; int32 key_states_151_groups_0 = const()[name = string("key_states_151_groups_0"), val = int32(1)]; tensor key_states_151_cast_fp16 = conv(dilations = key_states_151_dilations_0, groups = key_states_151_groups_0, pad = key_states_151_pad_0, pad_type = key_states_151_pad_type_0, strides = key_states_151_strides_0, weight = layers_15_self_attn_k_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("key_states_151_cast_fp16")]; tensor value_states_91_strides_0 = const()[name = string("value_states_91_strides_0"), val = tensor([1, 1])]; string value_states_91_pad_type_0 = const()[name = string("value_states_91_pad_type_0"), val = string("valid")]; tensor value_states_91_pad_0 = const()[name = string("value_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_91_dilations_0 = const()[name = string("value_states_91_dilations_0"), val = tensor([1, 1])]; int32 value_states_91_groups_0 = const()[name = string("value_states_91_groups_0"), val = int32(1)]; tensor value_states_91_cast_fp16 = conv(dilations = value_states_91_dilations_0, groups = value_states_91_groups_0, pad = value_states_91_pad_0, pad_type = value_states_91_pad_type_0, strides = value_states_91_strides_0, weight = layers_15_self_attn_v_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("value_states_91_cast_fp16")]; tensor concat_180x = const()[name = string("concat_180x"), val = tensor([1, 16, 128, -1])]; tensor x_151_cast_fp16 = reshape(shape = concat_180x, x = query_states_91_cast_fp16)[name = string("x_151_cast_fp16")]; tensor concat_181x = const()[name = string("concat_181x"), val = tensor([1, 2, 128, -1])]; tensor var_5799_cast_fp16 = reshape(shape = concat_181x, x = key_states_151_cast_fp16)[name = string("op_5799_cast_fp16")]; tensor concat_182x = const()[name = string("concat_182x"), val = tensor([1, 2, 128, -1])]; tensor var_5806_cast_fp16 = reshape(shape = concat_182x, x = value_states_91_cast_fp16)[name = string("op_5806_cast_fp16")]; tensor var_5810_cast_fp16 = mul(x = x_151_cast_fp16, y = var_869_cast_fp16)[name = string("op_5810_cast_fp16")]; tensor var_5811_split_sizes_0 = const()[name = string("op_5811_split_sizes_0"), val = tensor([64, 64])]; int32 var_5811_axis_0 = const()[name = string("op_5811_axis_0"), val = int32(-2)]; tensor var_5811_cast_fp16_0, tensor var_5811_cast_fp16_1 = split(axis = var_5811_axis_0, split_sizes = var_5811_split_sizes_0, x = x_151_cast_fp16)[name = string("op_5811_cast_fp16")]; fp16 const_152_promoted_to_fp16 = const()[name = string("const_152_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5813_cast_fp16 = mul(x = var_5811_cast_fp16_1, y = const_152_promoted_to_fp16)[name = string("op_5813_cast_fp16")]; int32 var_5815 = const()[name = string("op_5815"), val = int32(-2)]; bool var_5816_interleave_0 = const()[name = string("op_5816_interleave_0"), val = bool(false)]; tensor var_5816_cast_fp16 = concat(axis = var_5815, interleave = var_5816_interleave_0, values = (var_5813_cast_fp16, var_5811_cast_fp16_0))[name = string("op_5816_cast_fp16")]; tensor var_5817_cast_fp16 = mul(x = var_5816_cast_fp16, y = var_878_cast_fp16)[name = string("op_5817_cast_fp16")]; tensor query_states_93_cast_fp16 = add(x = var_5810_cast_fp16, y = var_5817_cast_fp16)[name = string("query_states_93_cast_fp16")]; tensor var_5823_cast_fp16 = mul(x = var_5799_cast_fp16, y = var_869_cast_fp16)[name = string("op_5823_cast_fp16")]; tensor var_5824_split_sizes_0 = const()[name = string("op_5824_split_sizes_0"), val = tensor([64, 64])]; int32 var_5824_axis_0 = const()[name = string("op_5824_axis_0"), val = int32(-2)]; tensor var_5824_cast_fp16_0, tensor var_5824_cast_fp16_1 = split(axis = var_5824_axis_0, split_sizes = var_5824_split_sizes_0, x = var_5799_cast_fp16)[name = string("op_5824_cast_fp16")]; fp16 const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5826_cast_fp16 = mul(x = var_5824_cast_fp16_1, y = const_153_promoted_to_fp16)[name = string("op_5826_cast_fp16")]; int32 var_5828 = const()[name = string("op_5828"), val = int32(-2)]; bool var_5829_interleave_0 = const()[name = string("op_5829_interleave_0"), val = bool(false)]; tensor var_5829_cast_fp16 = concat(axis = var_5828, interleave = var_5829_interleave_0, values = (var_5826_cast_fp16, var_5824_cast_fp16_0))[name = string("op_5829_cast_fp16")]; tensor var_5830_cast_fp16 = mul(x = var_5829_cast_fp16, y = var_878_cast_fp16)[name = string("op_5830_cast_fp16")]; tensor key_states_155_cast_fp16 = add(x = var_5823_cast_fp16, y = var_5830_cast_fp16)[name = string("key_states_155_cast_fp16")]; tensor expand_dims_180 = const()[name = string("expand_dims_180"), val = tensor([15])]; tensor expand_dims_181 = const()[name = string("expand_dims_181"), val = tensor([0])]; tensor expand_dims_183 = const()[name = string("expand_dims_183"), val = tensor([0])]; int32 concat_185_axis_0 = const()[name = string("concat_185_axis_0"), val = int32(0)]; bool concat_185_interleave_0 = const()[name = string("concat_185_interleave_0"), val = bool(false)]; tensor concat_185 = concat(axis = concat_185_axis_0, interleave = concat_185_interleave_0, values = (expand_dims_180, expand_dims_181, position_id, expand_dims_183))[name = string("concat_185")]; tensor expand_dims_184 = const()[name = string("expand_dims_184"), val = tensor([16])]; tensor concat_186_values1_0 = const()[name = string("concat_186_values1_0"), val = tensor([0])]; tensor concat_186_values3_0 = const()[name = string("concat_186_values3_0"), val = tensor([0])]; int32 concat_186_axis_0 = const()[name = string("concat_186_axis_0"), val = int32(0)]; bool concat_186_interleave_0 = const()[name = string("concat_186_interleave_0"), val = bool(false)]; tensor concat_186 = concat(axis = concat_186_axis_0, interleave = concat_186_interleave_0, values = (expand_dims_184, concat_186_values1_0, cache_position_end, concat_186_values3_0))[name = string("concat_186")]; tensor key_states_157_perm_0 = const()[name = string("key_states_157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_16_stride_0 = const()[name = string("key_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_157_cast_fp16 = transpose(perm = key_states_157_perm_0, x = key_states_155_cast_fp16)[name = string("transpose_640")]; tensor key_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = key_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = key_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_16_squeeze_mask_0, stride = key_cache_internal_tensor_assign_16_stride_0, update = key_states_157_cast_fp16, x = coreml_update_state_420)[name = string("key_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_16_cast_fp16, input = key_cache)[name = string("coreml_update_state_422_write_state")]; tensor coreml_update_state_422 = read_state(input = key_cache)[name = string("coreml_update_state_422")]; tensor value_states_93_perm_0 = const()[name = string("value_states_93_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_16_stride_0 = const()[name = string("value_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_93_cast_fp16 = transpose(perm = value_states_93_perm_0, x = var_5806_cast_fp16)[name = string("transpose_639")]; tensor value_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = value_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = value_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_16_squeeze_mask_0, stride = value_cache_internal_tensor_assign_16_stride_0, update = value_states_93_cast_fp16, x = coreml_update_state_421)[name = string("value_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_16_cast_fp16, input = value_cache)[name = string("coreml_update_state_423_write_state")]; tensor coreml_update_state_423 = read_state(input = value_cache)[name = string("coreml_update_state_423")]; tensor var_5900_begin_0 = const()[name = string("op_5900_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5900_end_0 = const()[name = string("op_5900_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5900_end_mask_0 = const()[name = string("op_5900_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5900_cast_fp16 = slice_by_index(begin = var_5900_begin_0, end = var_5900_end_0, end_mask = var_5900_end_mask_0, x = coreml_update_state_422)[name = string("op_5900_cast_fp16")]; tensor tile_30 = const()[name = string("tile_30"), val = tensor([1, 1])]; int32 var_5903_axis_0 = const()[name = string("op_5903_axis_0"), val = int32(1)]; tensor var_5903_cast_fp16_0, tensor var_5903_cast_fp16_1 = split(axis = var_5903_axis_0, split_sizes = tile_30, x = var_5900_cast_fp16)[name = string("op_5903_cast_fp16")]; tensor var_5910_begin_0 = const()[name = string("op_5910_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5910_end_0 = const()[name = string("op_5910_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5910_end_mask_0 = const()[name = string("op_5910_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5910_cast_fp16 = slice_by_index(begin = var_5910_begin_0, end = var_5910_end_0, end_mask = var_5910_end_mask_0, x = coreml_update_state_423)[name = string("op_5910_cast_fp16")]; tensor tile_31 = const()[name = string("tile_31"), val = tensor([1, 1])]; int32 var_5913_axis_0 = const()[name = string("op_5913_axis_0"), val = int32(1)]; tensor var_5913_cast_fp16_0, tensor var_5913_cast_fp16_1 = split(axis = var_5913_axis_0, split_sizes = tile_31, x = var_5910_cast_fp16)[name = string("op_5913_cast_fp16")]; tensor var_5916_split_sizes_0 = const()[name = string("op_5916_split_sizes_0"), val = tensor([8, 8])]; int32 var_5916_axis_0 = const()[name = string("op_5916_axis_0"), val = int32(1)]; tensor var_5916_0, tensor var_5916_1 = split(axis = var_5916_axis_0, split_sizes = var_5916_split_sizes_0, x = query_states_93_cast_fp16)[name = string("op_5916")]; bool attn_weights_241_transpose_x_0 = const()[name = string("attn_weights_241_transpose_x_0"), val = bool(false)]; bool attn_weights_241_transpose_y_0 = const()[name = string("attn_weights_241_transpose_y_0"), val = bool(false)]; tensor attn_weights_241_cast_fp16 = matmul(transpose_x = attn_weights_241_transpose_x_0, transpose_y = attn_weights_241_transpose_y_0, x = var_5903_cast_fp16_0, y = var_5916_0)[name = string("attn_weights_241_cast_fp16")]; fp16 var_5919_to_fp16 = const()[name = string("op_5919_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_243_cast_fp16 = mul(x = attn_weights_241_cast_fp16, y = var_5919_to_fp16)[name = string("attn_weights_243_cast_fp16")]; tensor attn_weights_245_cast_fp16 = add(x = attn_weights_243_cast_fp16, y = attn_mask_1)[name = string("attn_weights_245_cast_fp16")]; int32 var_5923 = const()[name = string("op_5923"), val = int32(-2)]; tensor attn_weights_247_cast_fp16 = softmax(axis = var_5923, x = attn_weights_245_cast_fp16)[name = string("attn_weights_247_cast_fp16")]; bool var_5929_transpose_x_1 = const()[name = string("op_5929_transpose_x_1"), val = bool(true)]; bool var_5929_transpose_y_1 = const()[name = string("op_5929_transpose_y_1"), val = bool(false)]; tensor var_5929_cast_fp16 = matmul(transpose_x = var_5929_transpose_x_1, transpose_y = var_5929_transpose_y_1, x = attn_weights_247_cast_fp16, y = var_5913_cast_fp16_0)[name = string("op_5929_cast_fp16")]; bool attn_weights_249_transpose_x_0 = const()[name = string("attn_weights_249_transpose_x_0"), val = bool(false)]; bool attn_weights_249_transpose_y_0 = const()[name = string("attn_weights_249_transpose_y_0"), val = bool(false)]; tensor attn_weights_249_cast_fp16 = matmul(transpose_x = attn_weights_249_transpose_x_0, transpose_y = attn_weights_249_transpose_y_0, x = var_5903_cast_fp16_1, y = var_5916_1)[name = string("attn_weights_249_cast_fp16")]; fp16 var_5931_to_fp16 = const()[name = string("op_5931_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_251_cast_fp16 = mul(x = attn_weights_249_cast_fp16, y = var_5931_to_fp16)[name = string("attn_weights_251_cast_fp16")]; tensor attn_weights_253_cast_fp16 = add(x = attn_weights_251_cast_fp16, y = attn_mask_1)[name = string("attn_weights_253_cast_fp16")]; int32 var_5935 = const()[name = string("op_5935"), val = int32(-2)]; tensor attn_weights_255_cast_fp16 = softmax(axis = var_5935, x = attn_weights_253_cast_fp16)[name = string("attn_weights_255_cast_fp16")]; bool attn_output_121_transpose_x_1 = const()[name = string("attn_output_121_transpose_x_1"), val = bool(true)]; bool attn_output_121_transpose_y_1 = const()[name = string("attn_output_121_transpose_y_1"), val = bool(false)]; tensor attn_output_121_cast_fp16 = matmul(transpose_x = attn_output_121_transpose_x_1, transpose_y = attn_output_121_transpose_y_1, x = attn_weights_255_cast_fp16, y = var_5913_cast_fp16_1)[name = string("attn_output_121_cast_fp16")]; int32 var_5943 = const()[name = string("op_5943"), val = int32(1)]; bool attn_output_123_interleave_0 = const()[name = string("attn_output_123_interleave_0"), val = bool(false)]; tensor attn_output_123_cast_fp16 = concat(axis = var_5943, interleave = attn_output_123_interleave_0, values = (var_5929_cast_fp16, attn_output_121_cast_fp16))[name = string("attn_output_123_cast_fp16")]; tensor var_5947_perm_0 = const()[name = string("op_5947_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_191x = const()[name = string("concat_191x"), val = tensor([1, 2048, 1, -1])]; tensor var_5947_cast_fp16 = transpose(perm = var_5947_perm_0, x = attn_output_123_cast_fp16)[name = string("transpose_638")]; tensor attn_output_127_cast_fp16 = reshape(shape = concat_191x, x = var_5947_cast_fp16)[name = string("attn_output_127_cast_fp16")]; tensor hidden_states_153_strides_0 = const()[name = string("hidden_states_153_strides_0"), val = tensor([1, 1])]; string hidden_states_153_pad_type_0 = const()[name = string("hidden_states_153_pad_type_0"), val = string("valid")]; tensor hidden_states_153_pad_0 = const()[name = string("hidden_states_153_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_153_dilations_0 = const()[name = string("hidden_states_153_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_153_groups_0 = const()[name = string("hidden_states_153_groups_0"), val = int32(1)]; tensor hidden_states_153_cast_fp16 = conv(dilations = hidden_states_153_dilations_0, groups = hidden_states_153_groups_0, pad = hidden_states_153_pad_0, pad_type = hidden_states_153_pad_type_0, strides = hidden_states_153_strides_0, weight = layers_15_self_attn_o_proj_weight_cast_fp16, x = attn_output_127_cast_fp16)[name = string("hidden_states_153_cast_fp16")]; tensor hidden_states_155_cast_fp16 = add(x = hidden_states_149_cast_fp16, y = hidden_states_153_cast_fp16)[name = string("hidden_states_155_cast_fp16")]; fp16 const_158_promoted_to_fp16 = const()[name = string("const_158_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5980_cast_fp16 = mul(x = hidden_states_155_cast_fp16, y = const_158_promoted_to_fp16)[name = string("op_5980_cast_fp16")]; int32 var_5978 = const()[name = string("op_5978"), val = int32(1)]; bool doubled_125_interleave_0 = const()[name = string("doubled_125_interleave_0"), val = bool(false)]; tensor doubled_125_cast_fp16 = concat(axis = var_5978, interleave = doubled_125_interleave_0, values = (hidden_states_155_cast_fp16, var_5980_cast_fp16))[name = string("doubled_125_cast_fp16")]; tensor out_63_axes_0 = const()[name = string("out_63_axes_0"), val = tensor([1])]; tensor out_63_gamma_0_to_fp16 = const()[name = string("out_63_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441646976)))]; fp16 var_5990_to_fp16 = const()[name = string("op_5990_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_63_cast_fp16 = layer_norm(axes = out_63_axes_0, epsilon = var_5990_to_fp16, gamma = out_63_gamma_0_to_fp16, x = doubled_125_cast_fp16)[name = string("out_63_cast_fp16")]; tensor var_6001_split_sizes_0 = const()[name = string("op_6001_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6001_axis_0 = const()[name = string("op_6001_axis_0"), val = int32(1)]; tensor var_6001_cast_fp16_0, tensor var_6001_cast_fp16_1 = split(axis = var_6001_axis_0, split_sizes = var_6001_split_sizes_0, x = out_63_cast_fp16)[name = string("op_6001_cast_fp16")]; tensor input_31_strides_0 = const()[name = string("input_31_strides_0"), val = tensor([1, 1])]; string input_31_pad_type_0 = const()[name = string("input_31_pad_type_0"), val = string("valid")]; tensor input_31_pad_0 = const()[name = string("input_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_31_dilations_0 = const()[name = string("input_31_dilations_0"), val = tensor([1, 1])]; int32 input_31_groups_0 = const()[name = string("input_31_groups_0"), val = int32(1)]; tensor input_31_cast_fp16 = conv(dilations = input_31_dilations_0, groups = input_31_groups_0, pad = input_31_pad_0, pad_type = input_31_pad_type_0, strides = input_31_strides_0, weight = layers_15_mlp_gate_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("input_31_cast_fp16")]; tensor var_6018_cast_fp16 = silu(x = input_31_cast_fp16)[name = string("op_6018_cast_fp16")]; tensor var_6024_strides_0 = const()[name = string("op_6024_strides_0"), val = tensor([1, 1])]; string var_6024_pad_type_0 = const()[name = string("op_6024_pad_type_0"), val = string("valid")]; tensor var_6024_pad_0 = const()[name = string("op_6024_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6024_dilations_0 = const()[name = string("op_6024_dilations_0"), val = tensor([1, 1])]; int32 var_6024_groups_0 = const()[name = string("op_6024_groups_0"), val = int32(1)]; tensor var_6024_cast_fp16 = conv(dilations = var_6024_dilations_0, groups = var_6024_groups_0, pad = var_6024_pad_0, pad_type = var_6024_pad_type_0, strides = var_6024_strides_0, weight = layers_15_mlp_up_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("op_6024_cast_fp16")]; tensor x_159_cast_fp16 = mul(x = var_6018_cast_fp16, y = var_6024_cast_fp16)[name = string("x_159_cast_fp16")]; tensor hidden_states_157_strides_0 = const()[name = string("hidden_states_157_strides_0"), val = tensor([1, 1])]; string hidden_states_157_pad_type_0 = const()[name = string("hidden_states_157_pad_type_0"), val = string("valid")]; tensor hidden_states_157_pad_0 = const()[name = string("hidden_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_157_dilations_0 = const()[name = string("hidden_states_157_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_157_groups_0 = const()[name = string("hidden_states_157_groups_0"), val = int32(1)]; tensor hidden_states_157_cast_fp16 = conv(dilations = hidden_states_157_dilations_0, groups = hidden_states_157_groups_0, pad = hidden_states_157_pad_0, pad_type = hidden_states_157_pad_type_0, strides = hidden_states_157_strides_0, weight = layers_15_mlp_down_proj_weight_cast_fp16, x = x_159_cast_fp16)[name = string("hidden_states_157_cast_fp16")]; tensor hidden_states_159_cast_fp16 = add(x = hidden_states_155_cast_fp16, y = hidden_states_157_cast_fp16)[name = string("hidden_states_159_cast_fp16")]; fp16 const_160_promoted_to_fp16 = const()[name = string("const_160_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6042_cast_fp16 = mul(x = hidden_states_159_cast_fp16, y = const_160_promoted_to_fp16)[name = string("op_6042_cast_fp16")]; int32 var_6040 = const()[name = string("op_6040"), val = int32(1)]; bool doubled_129_interleave_0 = const()[name = string("doubled_129_interleave_0"), val = bool(false)]; tensor doubled_129_cast_fp16 = concat(axis = var_6040, interleave = doubled_129_interleave_0, values = (hidden_states_159_cast_fp16, var_6042_cast_fp16))[name = string("doubled_129_cast_fp16")]; tensor out_65_axes_0 = const()[name = string("out_65_axes_0"), val = tensor([1])]; tensor out_65_gamma_0_to_fp16 = const()[name = string("out_65_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441655232)))]; fp16 var_6052_to_fp16 = const()[name = string("op_6052_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_65_cast_fp16 = layer_norm(axes = out_65_axes_0, epsilon = var_6052_to_fp16, gamma = out_65_gamma_0_to_fp16, x = doubled_129_cast_fp16)[name = string("out_65_cast_fp16")]; tensor var_6063_split_sizes_0 = const()[name = string("op_6063_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6063_axis_0 = const()[name = string("op_6063_axis_0"), val = int32(1)]; tensor var_6063_cast_fp16_0, tensor var_6063_cast_fp16_1 = split(axis = var_6063_axis_0, split_sizes = var_6063_split_sizes_0, x = out_65_cast_fp16)[name = string("op_6063_cast_fp16")]; tensor query_states_97_strides_0 = const()[name = string("query_states_97_strides_0"), val = tensor([1, 1])]; string query_states_97_pad_type_0 = const()[name = string("query_states_97_pad_type_0"), val = string("valid")]; tensor query_states_97_pad_0 = const()[name = string("query_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_97_dilations_0 = const()[name = string("query_states_97_dilations_0"), val = tensor([1, 1])]; int32 query_states_97_groups_0 = const()[name = string("query_states_97_groups_0"), val = int32(1)]; tensor query_states_97_cast_fp16 = conv(dilations = query_states_97_dilations_0, groups = query_states_97_groups_0, pad = query_states_97_pad_0, pad_type = query_states_97_pad_type_0, strides = query_states_97_strides_0, weight = layers_16_self_attn_q_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("query_states_97_cast_fp16")]; tensor key_states_161_strides_0 = const()[name = string("key_states_161_strides_0"), val = tensor([1, 1])]; string key_states_161_pad_type_0 = const()[name = string("key_states_161_pad_type_0"), val = string("valid")]; tensor key_states_161_pad_0 = const()[name = string("key_states_161_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_161_dilations_0 = const()[name = string("key_states_161_dilations_0"), val = tensor([1, 1])]; int32 key_states_161_groups_0 = const()[name = string("key_states_161_groups_0"), val = int32(1)]; tensor key_states_161_cast_fp16 = conv(dilations = key_states_161_dilations_0, groups = key_states_161_groups_0, pad = key_states_161_pad_0, pad_type = key_states_161_pad_type_0, strides = key_states_161_strides_0, weight = layers_16_self_attn_k_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("key_states_161_cast_fp16")]; tensor value_states_97_strides_0 = const()[name = string("value_states_97_strides_0"), val = tensor([1, 1])]; string value_states_97_pad_type_0 = const()[name = string("value_states_97_pad_type_0"), val = string("valid")]; tensor value_states_97_pad_0 = const()[name = string("value_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_97_dilations_0 = const()[name = string("value_states_97_dilations_0"), val = tensor([1, 1])]; int32 value_states_97_groups_0 = const()[name = string("value_states_97_groups_0"), val = int32(1)]; tensor value_states_97_cast_fp16 = conv(dilations = value_states_97_dilations_0, groups = value_states_97_groups_0, pad = value_states_97_pad_0, pad_type = value_states_97_pad_type_0, strides = value_states_97_strides_0, weight = layers_16_self_attn_v_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("value_states_97_cast_fp16")]; tensor concat_192x = const()[name = string("concat_192x"), val = tensor([1, 16, 128, -1])]; tensor x_161_cast_fp16 = reshape(shape = concat_192x, x = query_states_97_cast_fp16)[name = string("x_161_cast_fp16")]; tensor concat_193x = const()[name = string("concat_193x"), val = tensor([1, 2, 128, -1])]; tensor var_6120_cast_fp16 = reshape(shape = concat_193x, x = key_states_161_cast_fp16)[name = string("op_6120_cast_fp16")]; tensor concat_194x = const()[name = string("concat_194x"), val = tensor([1, 2, 128, -1])]; tensor var_6127_cast_fp16 = reshape(shape = concat_194x, x = value_states_97_cast_fp16)[name = string("op_6127_cast_fp16")]; tensor var_6131_cast_fp16 = mul(x = x_161_cast_fp16, y = var_869_cast_fp16)[name = string("op_6131_cast_fp16")]; tensor var_6132_split_sizes_0 = const()[name = string("op_6132_split_sizes_0"), val = tensor([64, 64])]; int32 var_6132_axis_0 = const()[name = string("op_6132_axis_0"), val = int32(-2)]; tensor var_6132_cast_fp16_0, tensor var_6132_cast_fp16_1 = split(axis = var_6132_axis_0, split_sizes = var_6132_split_sizes_0, x = x_161_cast_fp16)[name = string("op_6132_cast_fp16")]; fp16 const_162_promoted_to_fp16 = const()[name = string("const_162_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6134_cast_fp16 = mul(x = var_6132_cast_fp16_1, y = const_162_promoted_to_fp16)[name = string("op_6134_cast_fp16")]; int32 var_6136 = const()[name = string("op_6136"), val = int32(-2)]; bool var_6137_interleave_0 = const()[name = string("op_6137_interleave_0"), val = bool(false)]; tensor var_6137_cast_fp16 = concat(axis = var_6136, interleave = var_6137_interleave_0, values = (var_6134_cast_fp16, var_6132_cast_fp16_0))[name = string("op_6137_cast_fp16")]; tensor var_6138_cast_fp16 = mul(x = var_6137_cast_fp16, y = var_878_cast_fp16)[name = string("op_6138_cast_fp16")]; tensor query_states_99_cast_fp16 = add(x = var_6131_cast_fp16, y = var_6138_cast_fp16)[name = string("query_states_99_cast_fp16")]; tensor var_6144_cast_fp16 = mul(x = var_6120_cast_fp16, y = var_869_cast_fp16)[name = string("op_6144_cast_fp16")]; tensor var_6145_split_sizes_0 = const()[name = string("op_6145_split_sizes_0"), val = tensor([64, 64])]; int32 var_6145_axis_0 = const()[name = string("op_6145_axis_0"), val = int32(-2)]; tensor var_6145_cast_fp16_0, tensor var_6145_cast_fp16_1 = split(axis = var_6145_axis_0, split_sizes = var_6145_split_sizes_0, x = var_6120_cast_fp16)[name = string("op_6145_cast_fp16")]; fp16 const_163_promoted_to_fp16 = const()[name = string("const_163_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6147_cast_fp16 = mul(x = var_6145_cast_fp16_1, y = const_163_promoted_to_fp16)[name = string("op_6147_cast_fp16")]; int32 var_6149 = const()[name = string("op_6149"), val = int32(-2)]; bool var_6150_interleave_0 = const()[name = string("op_6150_interleave_0"), val = bool(false)]; tensor var_6150_cast_fp16 = concat(axis = var_6149, interleave = var_6150_interleave_0, values = (var_6147_cast_fp16, var_6145_cast_fp16_0))[name = string("op_6150_cast_fp16")]; tensor var_6151_cast_fp16 = mul(x = var_6150_cast_fp16, y = var_878_cast_fp16)[name = string("op_6151_cast_fp16")]; tensor key_states_165_cast_fp16 = add(x = var_6144_cast_fp16, y = var_6151_cast_fp16)[name = string("key_states_165_cast_fp16")]; tensor expand_dims_192 = const()[name = string("expand_dims_192"), val = tensor([16])]; tensor expand_dims_193 = const()[name = string("expand_dims_193"), val = tensor([0])]; tensor expand_dims_195 = const()[name = string("expand_dims_195"), val = tensor([0])]; int32 concat_197_axis_0 = const()[name = string("concat_197_axis_0"), val = int32(0)]; bool concat_197_interleave_0 = const()[name = string("concat_197_interleave_0"), val = bool(false)]; tensor concat_197 = concat(axis = concat_197_axis_0, interleave = concat_197_interleave_0, values = (expand_dims_192, expand_dims_193, position_id, expand_dims_195))[name = string("concat_197")]; tensor expand_dims_196 = const()[name = string("expand_dims_196"), val = tensor([17])]; tensor concat_198_values1_0 = const()[name = string("concat_198_values1_0"), val = tensor([0])]; tensor concat_198_values3_0 = const()[name = string("concat_198_values3_0"), val = tensor([0])]; int32 concat_198_axis_0 = const()[name = string("concat_198_axis_0"), val = int32(0)]; bool concat_198_interleave_0 = const()[name = string("concat_198_interleave_0"), val = bool(false)]; tensor concat_198 = concat(axis = concat_198_axis_0, interleave = concat_198_interleave_0, values = (expand_dims_196, concat_198_values1_0, cache_position_end, concat_198_values3_0))[name = string("concat_198")]; tensor key_states_167_perm_0 = const()[name = string("key_states_167_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_17_stride_0 = const()[name = string("key_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_167_cast_fp16 = transpose(perm = key_states_167_perm_0, x = key_states_165_cast_fp16)[name = string("transpose_637")]; tensor key_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = key_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = key_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_17_squeeze_mask_0, stride = key_cache_internal_tensor_assign_17_stride_0, update = key_states_167_cast_fp16, x = coreml_update_state_422)[name = string("key_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_17_cast_fp16, input = key_cache)[name = string("coreml_update_state_424_write_state")]; tensor coreml_update_state_424 = read_state(input = key_cache)[name = string("coreml_update_state_424")]; tensor value_states_99_perm_0 = const()[name = string("value_states_99_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_17_stride_0 = const()[name = string("value_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_99_cast_fp16 = transpose(perm = value_states_99_perm_0, x = var_6127_cast_fp16)[name = string("transpose_636")]; tensor value_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = value_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = value_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_17_squeeze_mask_0, stride = value_cache_internal_tensor_assign_17_stride_0, update = value_states_99_cast_fp16, x = coreml_update_state_423)[name = string("value_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_17_cast_fp16, input = value_cache)[name = string("coreml_update_state_425_write_state")]; tensor coreml_update_state_425 = read_state(input = value_cache)[name = string("coreml_update_state_425")]; tensor var_6221_begin_0 = const()[name = string("op_6221_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6221_end_0 = const()[name = string("op_6221_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6221_end_mask_0 = const()[name = string("op_6221_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6221_cast_fp16 = slice_by_index(begin = var_6221_begin_0, end = var_6221_end_0, end_mask = var_6221_end_mask_0, x = coreml_update_state_424)[name = string("op_6221_cast_fp16")]; tensor tile_32 = const()[name = string("tile_32"), val = tensor([1, 1])]; int32 var_6224_axis_0 = const()[name = string("op_6224_axis_0"), val = int32(1)]; tensor var_6224_cast_fp16_0, tensor var_6224_cast_fp16_1 = split(axis = var_6224_axis_0, split_sizes = tile_32, x = var_6221_cast_fp16)[name = string("op_6224_cast_fp16")]; tensor var_6231_begin_0 = const()[name = string("op_6231_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6231_end_0 = const()[name = string("op_6231_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6231_end_mask_0 = const()[name = string("op_6231_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6231_cast_fp16 = slice_by_index(begin = var_6231_begin_0, end = var_6231_end_0, end_mask = var_6231_end_mask_0, x = coreml_update_state_425)[name = string("op_6231_cast_fp16")]; tensor tile_33 = const()[name = string("tile_33"), val = tensor([1, 1])]; int32 var_6234_axis_0 = const()[name = string("op_6234_axis_0"), val = int32(1)]; tensor var_6234_cast_fp16_0, tensor var_6234_cast_fp16_1 = split(axis = var_6234_axis_0, split_sizes = tile_33, x = var_6231_cast_fp16)[name = string("op_6234_cast_fp16")]; tensor var_6237_split_sizes_0 = const()[name = string("op_6237_split_sizes_0"), val = tensor([8, 8])]; int32 var_6237_axis_0 = const()[name = string("op_6237_axis_0"), val = int32(1)]; tensor var_6237_0, tensor var_6237_1 = split(axis = var_6237_axis_0, split_sizes = var_6237_split_sizes_0, x = query_states_99_cast_fp16)[name = string("op_6237")]; bool attn_weights_257_transpose_x_0 = const()[name = string("attn_weights_257_transpose_x_0"), val = bool(false)]; bool attn_weights_257_transpose_y_0 = const()[name = string("attn_weights_257_transpose_y_0"), val = bool(false)]; tensor attn_weights_257_cast_fp16 = matmul(transpose_x = attn_weights_257_transpose_x_0, transpose_y = attn_weights_257_transpose_y_0, x = var_6224_cast_fp16_0, y = var_6237_0)[name = string("attn_weights_257_cast_fp16")]; fp16 var_6240_to_fp16 = const()[name = string("op_6240_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_259_cast_fp16 = mul(x = attn_weights_257_cast_fp16, y = var_6240_to_fp16)[name = string("attn_weights_259_cast_fp16")]; tensor attn_weights_261_cast_fp16 = add(x = attn_weights_259_cast_fp16, y = attn_mask_1)[name = string("attn_weights_261_cast_fp16")]; int32 var_6244 = const()[name = string("op_6244"), val = int32(-2)]; tensor attn_weights_263_cast_fp16 = softmax(axis = var_6244, x = attn_weights_261_cast_fp16)[name = string("attn_weights_263_cast_fp16")]; bool var_6250_transpose_x_1 = const()[name = string("op_6250_transpose_x_1"), val = bool(true)]; bool var_6250_transpose_y_1 = const()[name = string("op_6250_transpose_y_1"), val = bool(false)]; tensor var_6250_cast_fp16 = matmul(transpose_x = var_6250_transpose_x_1, transpose_y = var_6250_transpose_y_1, x = attn_weights_263_cast_fp16, y = var_6234_cast_fp16_0)[name = string("op_6250_cast_fp16")]; bool attn_weights_265_transpose_x_0 = const()[name = string("attn_weights_265_transpose_x_0"), val = bool(false)]; bool attn_weights_265_transpose_y_0 = const()[name = string("attn_weights_265_transpose_y_0"), val = bool(false)]; tensor attn_weights_265_cast_fp16 = matmul(transpose_x = attn_weights_265_transpose_x_0, transpose_y = attn_weights_265_transpose_y_0, x = var_6224_cast_fp16_1, y = var_6237_1)[name = string("attn_weights_265_cast_fp16")]; fp16 var_6252_to_fp16 = const()[name = string("op_6252_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_267_cast_fp16 = mul(x = attn_weights_265_cast_fp16, y = var_6252_to_fp16)[name = string("attn_weights_267_cast_fp16")]; tensor attn_weights_269_cast_fp16 = add(x = attn_weights_267_cast_fp16, y = attn_mask_1)[name = string("attn_weights_269_cast_fp16")]; int32 var_6256 = const()[name = string("op_6256"), val = int32(-2)]; tensor attn_weights_271_cast_fp16 = softmax(axis = var_6256, x = attn_weights_269_cast_fp16)[name = string("attn_weights_271_cast_fp16")]; bool attn_output_129_transpose_x_1 = const()[name = string("attn_output_129_transpose_x_1"), val = bool(true)]; bool attn_output_129_transpose_y_1 = const()[name = string("attn_output_129_transpose_y_1"), val = bool(false)]; tensor attn_output_129_cast_fp16 = matmul(transpose_x = attn_output_129_transpose_x_1, transpose_y = attn_output_129_transpose_y_1, x = attn_weights_271_cast_fp16, y = var_6234_cast_fp16_1)[name = string("attn_output_129_cast_fp16")]; int32 var_6264 = const()[name = string("op_6264"), val = int32(1)]; bool attn_output_131_interleave_0 = const()[name = string("attn_output_131_interleave_0"), val = bool(false)]; tensor attn_output_131_cast_fp16 = concat(axis = var_6264, interleave = attn_output_131_interleave_0, values = (var_6250_cast_fp16, attn_output_129_cast_fp16))[name = string("attn_output_131_cast_fp16")]; tensor var_6268_perm_0 = const()[name = string("op_6268_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_203x = const()[name = string("concat_203x"), val = tensor([1, 2048, 1, -1])]; tensor var_6268_cast_fp16 = transpose(perm = var_6268_perm_0, x = attn_output_131_cast_fp16)[name = string("transpose_635")]; tensor attn_output_135_cast_fp16 = reshape(shape = concat_203x, x = var_6268_cast_fp16)[name = string("attn_output_135_cast_fp16")]; tensor hidden_states_163_strides_0 = const()[name = string("hidden_states_163_strides_0"), val = tensor([1, 1])]; string hidden_states_163_pad_type_0 = const()[name = string("hidden_states_163_pad_type_0"), val = string("valid")]; tensor hidden_states_163_pad_0 = const()[name = string("hidden_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_163_dilations_0 = const()[name = string("hidden_states_163_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_163_groups_0 = const()[name = string("hidden_states_163_groups_0"), val = int32(1)]; tensor hidden_states_163_cast_fp16 = conv(dilations = hidden_states_163_dilations_0, groups = hidden_states_163_groups_0, pad = hidden_states_163_pad_0, pad_type = hidden_states_163_pad_type_0, strides = hidden_states_163_strides_0, weight = layers_16_self_attn_o_proj_weight_cast_fp16, x = attn_output_135_cast_fp16)[name = string("hidden_states_163_cast_fp16")]; tensor hidden_states_165_cast_fp16 = add(x = hidden_states_159_cast_fp16, y = hidden_states_163_cast_fp16)[name = string("hidden_states_165_cast_fp16")]; fp16 const_168_promoted_to_fp16 = const()[name = string("const_168_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6301_cast_fp16 = mul(x = hidden_states_165_cast_fp16, y = const_168_promoted_to_fp16)[name = string("op_6301_cast_fp16")]; int32 var_6299 = const()[name = string("op_6299"), val = int32(1)]; bool doubled_133_interleave_0 = const()[name = string("doubled_133_interleave_0"), val = bool(false)]; tensor doubled_133_cast_fp16 = concat(axis = var_6299, interleave = doubled_133_interleave_0, values = (hidden_states_165_cast_fp16, var_6301_cast_fp16))[name = string("doubled_133_cast_fp16")]; tensor out_67_axes_0 = const()[name = string("out_67_axes_0"), val = tensor([1])]; tensor out_67_gamma_0_to_fp16 = const()[name = string("out_67_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441663488)))]; fp16 var_6311_to_fp16 = const()[name = string("op_6311_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_67_cast_fp16 = layer_norm(axes = out_67_axes_0, epsilon = var_6311_to_fp16, gamma = out_67_gamma_0_to_fp16, x = doubled_133_cast_fp16)[name = string("out_67_cast_fp16")]; tensor var_6322_split_sizes_0 = const()[name = string("op_6322_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6322_axis_0 = const()[name = string("op_6322_axis_0"), val = int32(1)]; tensor var_6322_cast_fp16_0, tensor var_6322_cast_fp16_1 = split(axis = var_6322_axis_0, split_sizes = var_6322_split_sizes_0, x = out_67_cast_fp16)[name = string("op_6322_cast_fp16")]; tensor layers_16_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441671744)))]; tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; tensor input_33_cast_fp16 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = layers_16_mlp_gate_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("input_33_cast_fp16")]; tensor var_6339_cast_fp16 = silu(x = input_33_cast_fp16)[name = string("op_6339_cast_fp16")]; tensor layers_16_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1466837632)))]; tensor var_6345_strides_0 = const()[name = string("op_6345_strides_0"), val = tensor([1, 1])]; string var_6345_pad_type_0 = const()[name = string("op_6345_pad_type_0"), val = string("valid")]; tensor var_6345_pad_0 = const()[name = string("op_6345_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6345_dilations_0 = const()[name = string("op_6345_dilations_0"), val = tensor([1, 1])]; int32 var_6345_groups_0 = const()[name = string("op_6345_groups_0"), val = int32(1)]; tensor var_6345_cast_fp16 = conv(dilations = var_6345_dilations_0, groups = var_6345_groups_0, pad = var_6345_pad_0, pad_type = var_6345_pad_type_0, strides = var_6345_strides_0, weight = layers_16_mlp_up_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("op_6345_cast_fp16")]; tensor x_169_cast_fp16 = mul(x = var_6339_cast_fp16, y = var_6345_cast_fp16)[name = string("x_169_cast_fp16")]; tensor hidden_states_167_strides_0 = const()[name = string("hidden_states_167_strides_0"), val = tensor([1, 1])]; string hidden_states_167_pad_type_0 = const()[name = string("hidden_states_167_pad_type_0"), val = string("valid")]; tensor hidden_states_167_pad_0 = const()[name = string("hidden_states_167_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_167_dilations_0 = const()[name = string("hidden_states_167_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_167_groups_0 = const()[name = string("hidden_states_167_groups_0"), val = int32(1)]; tensor hidden_states_167_cast_fp16 = conv(dilations = hidden_states_167_dilations_0, groups = hidden_states_167_groups_0, pad = hidden_states_167_pad_0, pad_type = hidden_states_167_pad_type_0, strides = hidden_states_167_strides_0, weight = layers_16_mlp_down_proj_weight_cast_fp16, x = x_169_cast_fp16)[name = string("hidden_states_167_cast_fp16")]; tensor hidden_states_169_cast_fp16 = add(x = hidden_states_165_cast_fp16, y = hidden_states_167_cast_fp16)[name = string("hidden_states_169_cast_fp16")]; fp16 const_170_promoted_to_fp16 = const()[name = string("const_170_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6363_cast_fp16 = mul(x = hidden_states_169_cast_fp16, y = const_170_promoted_to_fp16)[name = string("op_6363_cast_fp16")]; int32 var_6361 = const()[name = string("op_6361"), val = int32(1)]; bool doubled_137_interleave_0 = const()[name = string("doubled_137_interleave_0"), val = bool(false)]; tensor doubled_137_cast_fp16 = concat(axis = var_6361, interleave = doubled_137_interleave_0, values = (hidden_states_169_cast_fp16, var_6363_cast_fp16))[name = string("doubled_137_cast_fp16")]; tensor out_69_axes_0 = const()[name = string("out_69_axes_0"), val = tensor([1])]; tensor out_69_gamma_0_to_fp16 = const()[name = string("out_69_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492003520)))]; fp16 var_6373_to_fp16 = const()[name = string("op_6373_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_69_cast_fp16 = layer_norm(axes = out_69_axes_0, epsilon = var_6373_to_fp16, gamma = out_69_gamma_0_to_fp16, x = doubled_137_cast_fp16)[name = string("out_69_cast_fp16")]; tensor var_6384_split_sizes_0 = const()[name = string("op_6384_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6384_axis_0 = const()[name = string("op_6384_axis_0"), val = int32(1)]; tensor var_6384_cast_fp16_0, tensor var_6384_cast_fp16_1 = split(axis = var_6384_axis_0, split_sizes = var_6384_split_sizes_0, x = out_69_cast_fp16)[name = string("op_6384_cast_fp16")]; tensor query_states_103_strides_0 = const()[name = string("query_states_103_strides_0"), val = tensor([1, 1])]; string query_states_103_pad_type_0 = const()[name = string("query_states_103_pad_type_0"), val = string("valid")]; tensor query_states_103_pad_0 = const()[name = string("query_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_103_dilations_0 = const()[name = string("query_states_103_dilations_0"), val = tensor([1, 1])]; int32 query_states_103_groups_0 = const()[name = string("query_states_103_groups_0"), val = int32(1)]; tensor query_states_103_cast_fp16 = conv(dilations = query_states_103_dilations_0, groups = query_states_103_groups_0, pad = query_states_103_pad_0, pad_type = query_states_103_pad_type_0, strides = query_states_103_strides_0, weight = layers_17_self_attn_q_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("query_states_103_cast_fp16")]; tensor key_states_171_strides_0 = const()[name = string("key_states_171_strides_0"), val = tensor([1, 1])]; string key_states_171_pad_type_0 = const()[name = string("key_states_171_pad_type_0"), val = string("valid")]; tensor key_states_171_pad_0 = const()[name = string("key_states_171_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_171_dilations_0 = const()[name = string("key_states_171_dilations_0"), val = tensor([1, 1])]; int32 key_states_171_groups_0 = const()[name = string("key_states_171_groups_0"), val = int32(1)]; tensor key_states_171_cast_fp16 = conv(dilations = key_states_171_dilations_0, groups = key_states_171_groups_0, pad = key_states_171_pad_0, pad_type = key_states_171_pad_type_0, strides = key_states_171_strides_0, weight = layers_17_self_attn_k_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("key_states_171_cast_fp16")]; tensor value_states_103_strides_0 = const()[name = string("value_states_103_strides_0"), val = tensor([1, 1])]; string value_states_103_pad_type_0 = const()[name = string("value_states_103_pad_type_0"), val = string("valid")]; tensor value_states_103_pad_0 = const()[name = string("value_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_103_dilations_0 = const()[name = string("value_states_103_dilations_0"), val = tensor([1, 1])]; int32 value_states_103_groups_0 = const()[name = string("value_states_103_groups_0"), val = int32(1)]; tensor value_states_103_cast_fp16 = conv(dilations = value_states_103_dilations_0, groups = value_states_103_groups_0, pad = value_states_103_pad_0, pad_type = value_states_103_pad_type_0, strides = value_states_103_strides_0, weight = layers_17_self_attn_v_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("value_states_103_cast_fp16")]; tensor concat_204x = const()[name = string("concat_204x"), val = tensor([1, 16, 128, -1])]; tensor x_171_cast_fp16 = reshape(shape = concat_204x, x = query_states_103_cast_fp16)[name = string("x_171_cast_fp16")]; tensor concat_205x = const()[name = string("concat_205x"), val = tensor([1, 2, 128, -1])]; tensor var_6441_cast_fp16 = reshape(shape = concat_205x, x = key_states_171_cast_fp16)[name = string("op_6441_cast_fp16")]; tensor concat_206x = const()[name = string("concat_206x"), val = tensor([1, 2, 128, -1])]; tensor var_6448_cast_fp16 = reshape(shape = concat_206x, x = value_states_103_cast_fp16)[name = string("op_6448_cast_fp16")]; tensor var_6452_cast_fp16 = mul(x = x_171_cast_fp16, y = var_869_cast_fp16)[name = string("op_6452_cast_fp16")]; tensor var_6453_split_sizes_0 = const()[name = string("op_6453_split_sizes_0"), val = tensor([64, 64])]; int32 var_6453_axis_0 = const()[name = string("op_6453_axis_0"), val = int32(-2)]; tensor var_6453_cast_fp16_0, tensor var_6453_cast_fp16_1 = split(axis = var_6453_axis_0, split_sizes = var_6453_split_sizes_0, x = x_171_cast_fp16)[name = string("op_6453_cast_fp16")]; fp16 const_172_promoted_to_fp16 = const()[name = string("const_172_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6455_cast_fp16 = mul(x = var_6453_cast_fp16_1, y = const_172_promoted_to_fp16)[name = string("op_6455_cast_fp16")]; int32 var_6457 = const()[name = string("op_6457"), val = int32(-2)]; bool var_6458_interleave_0 = const()[name = string("op_6458_interleave_0"), val = bool(false)]; tensor var_6458_cast_fp16 = concat(axis = var_6457, interleave = var_6458_interleave_0, values = (var_6455_cast_fp16, var_6453_cast_fp16_0))[name = string("op_6458_cast_fp16")]; tensor var_6459_cast_fp16 = mul(x = var_6458_cast_fp16, y = var_878_cast_fp16)[name = string("op_6459_cast_fp16")]; tensor query_states_105_cast_fp16 = add(x = var_6452_cast_fp16, y = var_6459_cast_fp16)[name = string("query_states_105_cast_fp16")]; tensor var_6465_cast_fp16 = mul(x = var_6441_cast_fp16, y = var_869_cast_fp16)[name = string("op_6465_cast_fp16")]; tensor var_6466_split_sizes_0 = const()[name = string("op_6466_split_sizes_0"), val = tensor([64, 64])]; int32 var_6466_axis_0 = const()[name = string("op_6466_axis_0"), val = int32(-2)]; tensor var_6466_cast_fp16_0, tensor var_6466_cast_fp16_1 = split(axis = var_6466_axis_0, split_sizes = var_6466_split_sizes_0, x = var_6441_cast_fp16)[name = string("op_6466_cast_fp16")]; fp16 const_173_promoted_to_fp16 = const()[name = string("const_173_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6468_cast_fp16 = mul(x = var_6466_cast_fp16_1, y = const_173_promoted_to_fp16)[name = string("op_6468_cast_fp16")]; int32 var_6470 = const()[name = string("op_6470"), val = int32(-2)]; bool var_6471_interleave_0 = const()[name = string("op_6471_interleave_0"), val = bool(false)]; tensor var_6471_cast_fp16 = concat(axis = var_6470, interleave = var_6471_interleave_0, values = (var_6468_cast_fp16, var_6466_cast_fp16_0))[name = string("op_6471_cast_fp16")]; tensor var_6472_cast_fp16 = mul(x = var_6471_cast_fp16, y = var_878_cast_fp16)[name = string("op_6472_cast_fp16")]; tensor key_states_175_cast_fp16 = add(x = var_6465_cast_fp16, y = var_6472_cast_fp16)[name = string("key_states_175_cast_fp16")]; tensor expand_dims_204 = const()[name = string("expand_dims_204"), val = tensor([17])]; tensor expand_dims_205 = const()[name = string("expand_dims_205"), val = tensor([0])]; tensor expand_dims_207 = const()[name = string("expand_dims_207"), val = tensor([0])]; int32 concat_209_axis_0 = const()[name = string("concat_209_axis_0"), val = int32(0)]; bool concat_209_interleave_0 = const()[name = string("concat_209_interleave_0"), val = bool(false)]; tensor concat_209 = concat(axis = concat_209_axis_0, interleave = concat_209_interleave_0, values = (expand_dims_204, expand_dims_205, position_id, expand_dims_207))[name = string("concat_209")]; tensor expand_dims_208 = const()[name = string("expand_dims_208"), val = tensor([18])]; tensor concat_210_values1_0 = const()[name = string("concat_210_values1_0"), val = tensor([0])]; tensor concat_210_values3_0 = const()[name = string("concat_210_values3_0"), val = tensor([0])]; int32 concat_210_axis_0 = const()[name = string("concat_210_axis_0"), val = int32(0)]; bool concat_210_interleave_0 = const()[name = string("concat_210_interleave_0"), val = bool(false)]; tensor concat_210 = concat(axis = concat_210_axis_0, interleave = concat_210_interleave_0, values = (expand_dims_208, concat_210_values1_0, cache_position_end, concat_210_values3_0))[name = string("concat_210")]; tensor key_states_177_perm_0 = const()[name = string("key_states_177_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_18_stride_0 = const()[name = string("key_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_177_cast_fp16 = transpose(perm = key_states_177_perm_0, x = key_states_175_cast_fp16)[name = string("transpose_634")]; tensor key_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = key_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = key_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_18_squeeze_mask_0, stride = key_cache_internal_tensor_assign_18_stride_0, update = key_states_177_cast_fp16, x = coreml_update_state_424)[name = string("key_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_18_cast_fp16, input = key_cache)[name = string("coreml_update_state_426_write_state")]; tensor coreml_update_state_426 = read_state(input = key_cache)[name = string("coreml_update_state_426")]; tensor value_states_105_perm_0 = const()[name = string("value_states_105_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_18_stride_0 = const()[name = string("value_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_105_cast_fp16 = transpose(perm = value_states_105_perm_0, x = var_6448_cast_fp16)[name = string("transpose_633")]; tensor value_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = value_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = value_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_18_squeeze_mask_0, stride = value_cache_internal_tensor_assign_18_stride_0, update = value_states_105_cast_fp16, x = coreml_update_state_425)[name = string("value_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_18_cast_fp16, input = value_cache)[name = string("coreml_update_state_427_write_state")]; tensor coreml_update_state_427 = read_state(input = value_cache)[name = string("coreml_update_state_427")]; tensor var_6542_begin_0 = const()[name = string("op_6542_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6542_end_0 = const()[name = string("op_6542_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6542_end_mask_0 = const()[name = string("op_6542_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6542_cast_fp16 = slice_by_index(begin = var_6542_begin_0, end = var_6542_end_0, end_mask = var_6542_end_mask_0, x = coreml_update_state_426)[name = string("op_6542_cast_fp16")]; tensor tile_34 = const()[name = string("tile_34"), val = tensor([1, 1])]; int32 var_6545_axis_0 = const()[name = string("op_6545_axis_0"), val = int32(1)]; tensor var_6545_cast_fp16_0, tensor var_6545_cast_fp16_1 = split(axis = var_6545_axis_0, split_sizes = tile_34, x = var_6542_cast_fp16)[name = string("op_6545_cast_fp16")]; tensor var_6552_begin_0 = const()[name = string("op_6552_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6552_end_0 = const()[name = string("op_6552_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6552_end_mask_0 = const()[name = string("op_6552_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6552_cast_fp16 = slice_by_index(begin = var_6552_begin_0, end = var_6552_end_0, end_mask = var_6552_end_mask_0, x = coreml_update_state_427)[name = string("op_6552_cast_fp16")]; tensor tile_35 = const()[name = string("tile_35"), val = tensor([1, 1])]; int32 var_6555_axis_0 = const()[name = string("op_6555_axis_0"), val = int32(1)]; tensor var_6555_cast_fp16_0, tensor var_6555_cast_fp16_1 = split(axis = var_6555_axis_0, split_sizes = tile_35, x = var_6552_cast_fp16)[name = string("op_6555_cast_fp16")]; tensor var_6558_split_sizes_0 = const()[name = string("op_6558_split_sizes_0"), val = tensor([8, 8])]; int32 var_6558_axis_0 = const()[name = string("op_6558_axis_0"), val = int32(1)]; tensor var_6558_0, tensor var_6558_1 = split(axis = var_6558_axis_0, split_sizes = var_6558_split_sizes_0, x = query_states_105_cast_fp16)[name = string("op_6558")]; bool attn_weights_273_transpose_x_0 = const()[name = string("attn_weights_273_transpose_x_0"), val = bool(false)]; bool attn_weights_273_transpose_y_0 = const()[name = string("attn_weights_273_transpose_y_0"), val = bool(false)]; tensor attn_weights_273_cast_fp16 = matmul(transpose_x = attn_weights_273_transpose_x_0, transpose_y = attn_weights_273_transpose_y_0, x = var_6545_cast_fp16_0, y = var_6558_0)[name = string("attn_weights_273_cast_fp16")]; fp16 var_6561_to_fp16 = const()[name = string("op_6561_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_275_cast_fp16 = mul(x = attn_weights_273_cast_fp16, y = var_6561_to_fp16)[name = string("attn_weights_275_cast_fp16")]; tensor attn_weights_277_cast_fp16 = add(x = attn_weights_275_cast_fp16, y = attn_mask_1)[name = string("attn_weights_277_cast_fp16")]; int32 var_6565 = const()[name = string("op_6565"), val = int32(-2)]; tensor attn_weights_279_cast_fp16 = softmax(axis = var_6565, x = attn_weights_277_cast_fp16)[name = string("attn_weights_279_cast_fp16")]; bool var_6571_transpose_x_1 = const()[name = string("op_6571_transpose_x_1"), val = bool(true)]; bool var_6571_transpose_y_1 = const()[name = string("op_6571_transpose_y_1"), val = bool(false)]; tensor var_6571_cast_fp16 = matmul(transpose_x = var_6571_transpose_x_1, transpose_y = var_6571_transpose_y_1, x = attn_weights_279_cast_fp16, y = var_6555_cast_fp16_0)[name = string("op_6571_cast_fp16")]; bool attn_weights_281_transpose_x_0 = const()[name = string("attn_weights_281_transpose_x_0"), val = bool(false)]; bool attn_weights_281_transpose_y_0 = const()[name = string("attn_weights_281_transpose_y_0"), val = bool(false)]; tensor attn_weights_281_cast_fp16 = matmul(transpose_x = attn_weights_281_transpose_x_0, transpose_y = attn_weights_281_transpose_y_0, x = var_6545_cast_fp16_1, y = var_6558_1)[name = string("attn_weights_281_cast_fp16")]; fp16 var_6573_to_fp16 = const()[name = string("op_6573_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_283_cast_fp16 = mul(x = attn_weights_281_cast_fp16, y = var_6573_to_fp16)[name = string("attn_weights_283_cast_fp16")]; tensor attn_weights_285_cast_fp16 = add(x = attn_weights_283_cast_fp16, y = attn_mask_1)[name = string("attn_weights_285_cast_fp16")]; int32 var_6577 = const()[name = string("op_6577"), val = int32(-2)]; tensor attn_weights_287_cast_fp16 = softmax(axis = var_6577, x = attn_weights_285_cast_fp16)[name = string("attn_weights_287_cast_fp16")]; bool attn_output_137_transpose_x_1 = const()[name = string("attn_output_137_transpose_x_1"), val = bool(true)]; bool attn_output_137_transpose_y_1 = const()[name = string("attn_output_137_transpose_y_1"), val = bool(false)]; tensor attn_output_137_cast_fp16 = matmul(transpose_x = attn_output_137_transpose_x_1, transpose_y = attn_output_137_transpose_y_1, x = attn_weights_287_cast_fp16, y = var_6555_cast_fp16_1)[name = string("attn_output_137_cast_fp16")]; int32 var_6585 = const()[name = string("op_6585"), val = int32(1)]; bool attn_output_139_interleave_0 = const()[name = string("attn_output_139_interleave_0"), val = bool(false)]; tensor attn_output_139_cast_fp16 = concat(axis = var_6585, interleave = attn_output_139_interleave_0, values = (var_6571_cast_fp16, attn_output_137_cast_fp16))[name = string("attn_output_139_cast_fp16")]; tensor var_6589_perm_0 = const()[name = string("op_6589_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_215x = const()[name = string("concat_215x"), val = tensor([1, 2048, 1, -1])]; tensor var_6589_cast_fp16 = transpose(perm = var_6589_perm_0, x = attn_output_139_cast_fp16)[name = string("transpose_632")]; tensor attn_output_143_cast_fp16 = reshape(shape = concat_215x, x = var_6589_cast_fp16)[name = string("attn_output_143_cast_fp16")]; tensor hidden_states_173_strides_0 = const()[name = string("hidden_states_173_strides_0"), val = tensor([1, 1])]; string hidden_states_173_pad_type_0 = const()[name = string("hidden_states_173_pad_type_0"), val = string("valid")]; tensor hidden_states_173_pad_0 = const()[name = string("hidden_states_173_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_173_dilations_0 = const()[name = string("hidden_states_173_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_173_groups_0 = const()[name = string("hidden_states_173_groups_0"), val = int32(1)]; tensor hidden_states_173_cast_fp16 = conv(dilations = hidden_states_173_dilations_0, groups = hidden_states_173_groups_0, pad = hidden_states_173_pad_0, pad_type = hidden_states_173_pad_type_0, strides = hidden_states_173_strides_0, weight = layers_17_self_attn_o_proj_weight_cast_fp16, x = attn_output_143_cast_fp16)[name = string("hidden_states_173_cast_fp16")]; tensor hidden_states_175_cast_fp16 = add(x = hidden_states_169_cast_fp16, y = hidden_states_173_cast_fp16)[name = string("hidden_states_175_cast_fp16")]; fp16 const_178_promoted_to_fp16 = const()[name = string("const_178_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6622_cast_fp16 = mul(x = hidden_states_175_cast_fp16, y = const_178_promoted_to_fp16)[name = string("op_6622_cast_fp16")]; int32 var_6620 = const()[name = string("op_6620"), val = int32(1)]; bool doubled_141_interleave_0 = const()[name = string("doubled_141_interleave_0"), val = bool(false)]; tensor doubled_141_cast_fp16 = concat(axis = var_6620, interleave = doubled_141_interleave_0, values = (hidden_states_175_cast_fp16, var_6622_cast_fp16))[name = string("doubled_141_cast_fp16")]; tensor out_71_axes_0 = const()[name = string("out_71_axes_0"), val = tensor([1])]; tensor out_71_gamma_0_to_fp16 = const()[name = string("out_71_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492011776)))]; fp16 var_6632_to_fp16 = const()[name = string("op_6632_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_71_cast_fp16 = layer_norm(axes = out_71_axes_0, epsilon = var_6632_to_fp16, gamma = out_71_gamma_0_to_fp16, x = doubled_141_cast_fp16)[name = string("out_71_cast_fp16")]; tensor var_6643_split_sizes_0 = const()[name = string("op_6643_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6643_axis_0 = const()[name = string("op_6643_axis_0"), val = int32(1)]; tensor var_6643_cast_fp16_0, tensor var_6643_cast_fp16_1 = split(axis = var_6643_axis_0, split_sizes = var_6643_split_sizes_0, x = out_71_cast_fp16)[name = string("op_6643_cast_fp16")]; tensor input_35_strides_0 = const()[name = string("input_35_strides_0"), val = tensor([1, 1])]; string input_35_pad_type_0 = const()[name = string("input_35_pad_type_0"), val = string("valid")]; tensor input_35_pad_0 = const()[name = string("input_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_35_dilations_0 = const()[name = string("input_35_dilations_0"), val = tensor([1, 1])]; int32 input_35_groups_0 = const()[name = string("input_35_groups_0"), val = int32(1)]; tensor input_35_cast_fp16 = conv(dilations = input_35_dilations_0, groups = input_35_groups_0, pad = input_35_pad_0, pad_type = input_35_pad_type_0, strides = input_35_strides_0, weight = layers_17_mlp_gate_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("input_35_cast_fp16")]; tensor var_6660_cast_fp16 = silu(x = input_35_cast_fp16)[name = string("op_6660_cast_fp16")]; tensor var_6666_strides_0 = const()[name = string("op_6666_strides_0"), val = tensor([1, 1])]; string var_6666_pad_type_0 = const()[name = string("op_6666_pad_type_0"), val = string("valid")]; tensor var_6666_pad_0 = const()[name = string("op_6666_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6666_dilations_0 = const()[name = string("op_6666_dilations_0"), val = tensor([1, 1])]; int32 var_6666_groups_0 = const()[name = string("op_6666_groups_0"), val = int32(1)]; tensor var_6666_cast_fp16 = conv(dilations = var_6666_dilations_0, groups = var_6666_groups_0, pad = var_6666_pad_0, pad_type = var_6666_pad_type_0, strides = var_6666_strides_0, weight = layers_17_mlp_up_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("op_6666_cast_fp16")]; tensor x_179_cast_fp16 = mul(x = var_6660_cast_fp16, y = var_6666_cast_fp16)[name = string("x_179_cast_fp16")]; tensor hidden_states_177_strides_0 = const()[name = string("hidden_states_177_strides_0"), val = tensor([1, 1])]; string hidden_states_177_pad_type_0 = const()[name = string("hidden_states_177_pad_type_0"), val = string("valid")]; tensor hidden_states_177_pad_0 = const()[name = string("hidden_states_177_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_177_dilations_0 = const()[name = string("hidden_states_177_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_177_groups_0 = const()[name = string("hidden_states_177_groups_0"), val = int32(1)]; tensor hidden_states_177_cast_fp16 = conv(dilations = hidden_states_177_dilations_0, groups = hidden_states_177_groups_0, pad = hidden_states_177_pad_0, pad_type = hidden_states_177_pad_type_0, strides = hidden_states_177_strides_0, weight = layers_17_mlp_down_proj_weight_cast_fp16, x = x_179_cast_fp16)[name = string("hidden_states_177_cast_fp16")]; tensor hidden_states_179_cast_fp16 = add(x = hidden_states_175_cast_fp16, y = hidden_states_177_cast_fp16)[name = string("hidden_states_179_cast_fp16")]; fp16 const_180_promoted_to_fp16 = const()[name = string("const_180_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6684_cast_fp16 = mul(x = hidden_states_179_cast_fp16, y = const_180_promoted_to_fp16)[name = string("op_6684_cast_fp16")]; int32 var_6682 = const()[name = string("op_6682"), val = int32(1)]; bool doubled_145_interleave_0 = const()[name = string("doubled_145_interleave_0"), val = bool(false)]; tensor doubled_145_cast_fp16 = concat(axis = var_6682, interleave = doubled_145_interleave_0, values = (hidden_states_179_cast_fp16, var_6684_cast_fp16))[name = string("doubled_145_cast_fp16")]; tensor out_73_axes_0 = const()[name = string("out_73_axes_0"), val = tensor([1])]; tensor out_73_gamma_0_to_fp16 = const()[name = string("out_73_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492020032)))]; fp16 var_6694_to_fp16 = const()[name = string("op_6694_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_73_cast_fp16 = layer_norm(axes = out_73_axes_0, epsilon = var_6694_to_fp16, gamma = out_73_gamma_0_to_fp16, x = doubled_145_cast_fp16)[name = string("out_73_cast_fp16")]; tensor var_6705_split_sizes_0 = const()[name = string("op_6705_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6705_axis_0 = const()[name = string("op_6705_axis_0"), val = int32(1)]; tensor var_6705_cast_fp16_0, tensor var_6705_cast_fp16_1 = split(axis = var_6705_axis_0, split_sizes = var_6705_split_sizes_0, x = out_73_cast_fp16)[name = string("op_6705_cast_fp16")]; tensor query_states_109_strides_0 = const()[name = string("query_states_109_strides_0"), val = tensor([1, 1])]; string query_states_109_pad_type_0 = const()[name = string("query_states_109_pad_type_0"), val = string("valid")]; tensor query_states_109_pad_0 = const()[name = string("query_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_109_dilations_0 = const()[name = string("query_states_109_dilations_0"), val = tensor([1, 1])]; int32 query_states_109_groups_0 = const()[name = string("query_states_109_groups_0"), val = int32(1)]; tensor query_states_109_cast_fp16 = conv(dilations = query_states_109_dilations_0, groups = query_states_109_groups_0, pad = query_states_109_pad_0, pad_type = query_states_109_pad_type_0, strides = query_states_109_strides_0, weight = layers_18_self_attn_q_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("query_states_109_cast_fp16")]; tensor key_states_181_strides_0 = const()[name = string("key_states_181_strides_0"), val = tensor([1, 1])]; string key_states_181_pad_type_0 = const()[name = string("key_states_181_pad_type_0"), val = string("valid")]; tensor key_states_181_pad_0 = const()[name = string("key_states_181_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_181_dilations_0 = const()[name = string("key_states_181_dilations_0"), val = tensor([1, 1])]; int32 key_states_181_groups_0 = const()[name = string("key_states_181_groups_0"), val = int32(1)]; tensor key_states_181_cast_fp16 = conv(dilations = key_states_181_dilations_0, groups = key_states_181_groups_0, pad = key_states_181_pad_0, pad_type = key_states_181_pad_type_0, strides = key_states_181_strides_0, weight = layers_18_self_attn_k_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("key_states_181_cast_fp16")]; tensor value_states_109_strides_0 = const()[name = string("value_states_109_strides_0"), val = tensor([1, 1])]; string value_states_109_pad_type_0 = const()[name = string("value_states_109_pad_type_0"), val = string("valid")]; tensor value_states_109_pad_0 = const()[name = string("value_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_109_dilations_0 = const()[name = string("value_states_109_dilations_0"), val = tensor([1, 1])]; int32 value_states_109_groups_0 = const()[name = string("value_states_109_groups_0"), val = int32(1)]; tensor value_states_109_cast_fp16 = conv(dilations = value_states_109_dilations_0, groups = value_states_109_groups_0, pad = value_states_109_pad_0, pad_type = value_states_109_pad_type_0, strides = value_states_109_strides_0, weight = layers_18_self_attn_v_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("value_states_109_cast_fp16")]; tensor concat_216x = const()[name = string("concat_216x"), val = tensor([1, 16, 128, -1])]; tensor x_181_cast_fp16 = reshape(shape = concat_216x, x = query_states_109_cast_fp16)[name = string("x_181_cast_fp16")]; tensor concat_217x = const()[name = string("concat_217x"), val = tensor([1, 2, 128, -1])]; tensor var_6762_cast_fp16 = reshape(shape = concat_217x, x = key_states_181_cast_fp16)[name = string("op_6762_cast_fp16")]; tensor concat_218x = const()[name = string("concat_218x"), val = tensor([1, 2, 128, -1])]; tensor var_6769_cast_fp16 = reshape(shape = concat_218x, x = value_states_109_cast_fp16)[name = string("op_6769_cast_fp16")]; tensor var_6773_cast_fp16 = mul(x = x_181_cast_fp16, y = var_869_cast_fp16)[name = string("op_6773_cast_fp16")]; tensor var_6774_split_sizes_0 = const()[name = string("op_6774_split_sizes_0"), val = tensor([64, 64])]; int32 var_6774_axis_0 = const()[name = string("op_6774_axis_0"), val = int32(-2)]; tensor var_6774_cast_fp16_0, tensor var_6774_cast_fp16_1 = split(axis = var_6774_axis_0, split_sizes = var_6774_split_sizes_0, x = x_181_cast_fp16)[name = string("op_6774_cast_fp16")]; fp16 const_182_promoted_to_fp16 = const()[name = string("const_182_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6776_cast_fp16 = mul(x = var_6774_cast_fp16_1, y = const_182_promoted_to_fp16)[name = string("op_6776_cast_fp16")]; int32 var_6778 = const()[name = string("op_6778"), val = int32(-2)]; bool var_6779_interleave_0 = const()[name = string("op_6779_interleave_0"), val = bool(false)]; tensor var_6779_cast_fp16 = concat(axis = var_6778, interleave = var_6779_interleave_0, values = (var_6776_cast_fp16, var_6774_cast_fp16_0))[name = string("op_6779_cast_fp16")]; tensor var_6780_cast_fp16 = mul(x = var_6779_cast_fp16, y = var_878_cast_fp16)[name = string("op_6780_cast_fp16")]; tensor query_states_111_cast_fp16 = add(x = var_6773_cast_fp16, y = var_6780_cast_fp16)[name = string("query_states_111_cast_fp16")]; tensor var_6786_cast_fp16 = mul(x = var_6762_cast_fp16, y = var_869_cast_fp16)[name = string("op_6786_cast_fp16")]; tensor var_6787_split_sizes_0 = const()[name = string("op_6787_split_sizes_0"), val = tensor([64, 64])]; int32 var_6787_axis_0 = const()[name = string("op_6787_axis_0"), val = int32(-2)]; tensor var_6787_cast_fp16_0, tensor var_6787_cast_fp16_1 = split(axis = var_6787_axis_0, split_sizes = var_6787_split_sizes_0, x = var_6762_cast_fp16)[name = string("op_6787_cast_fp16")]; fp16 const_183_promoted_to_fp16 = const()[name = string("const_183_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6789_cast_fp16 = mul(x = var_6787_cast_fp16_1, y = const_183_promoted_to_fp16)[name = string("op_6789_cast_fp16")]; int32 var_6791 = const()[name = string("op_6791"), val = int32(-2)]; bool var_6792_interleave_0 = const()[name = string("op_6792_interleave_0"), val = bool(false)]; tensor var_6792_cast_fp16 = concat(axis = var_6791, interleave = var_6792_interleave_0, values = (var_6789_cast_fp16, var_6787_cast_fp16_0))[name = string("op_6792_cast_fp16")]; tensor var_6793_cast_fp16 = mul(x = var_6792_cast_fp16, y = var_878_cast_fp16)[name = string("op_6793_cast_fp16")]; tensor key_states_185_cast_fp16 = add(x = var_6786_cast_fp16, y = var_6793_cast_fp16)[name = string("key_states_185_cast_fp16")]; tensor expand_dims_216 = const()[name = string("expand_dims_216"), val = tensor([18])]; tensor expand_dims_217 = const()[name = string("expand_dims_217"), val = tensor([0])]; tensor expand_dims_219 = const()[name = string("expand_dims_219"), val = tensor([0])]; int32 concat_221_axis_0 = const()[name = string("concat_221_axis_0"), val = int32(0)]; bool concat_221_interleave_0 = const()[name = string("concat_221_interleave_0"), val = bool(false)]; tensor concat_221 = concat(axis = concat_221_axis_0, interleave = concat_221_interleave_0, values = (expand_dims_216, expand_dims_217, position_id, expand_dims_219))[name = string("concat_221")]; tensor expand_dims_220 = const()[name = string("expand_dims_220"), val = tensor([19])]; tensor concat_222_values1_0 = const()[name = string("concat_222_values1_0"), val = tensor([0])]; tensor concat_222_values3_0 = const()[name = string("concat_222_values3_0"), val = tensor([0])]; int32 concat_222_axis_0 = const()[name = string("concat_222_axis_0"), val = int32(0)]; bool concat_222_interleave_0 = const()[name = string("concat_222_interleave_0"), val = bool(false)]; tensor concat_222 = concat(axis = concat_222_axis_0, interleave = concat_222_interleave_0, values = (expand_dims_220, concat_222_values1_0, cache_position_end, concat_222_values3_0))[name = string("concat_222")]; tensor key_states_187_perm_0 = const()[name = string("key_states_187_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_19_stride_0 = const()[name = string("key_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_187_cast_fp16 = transpose(perm = key_states_187_perm_0, x = key_states_185_cast_fp16)[name = string("transpose_631")]; tensor key_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = key_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = key_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_19_squeeze_mask_0, stride = key_cache_internal_tensor_assign_19_stride_0, update = key_states_187_cast_fp16, x = coreml_update_state_426)[name = string("key_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_19_cast_fp16, input = key_cache)[name = string("coreml_update_state_428_write_state")]; tensor coreml_update_state_428 = read_state(input = key_cache)[name = string("coreml_update_state_428")]; tensor value_states_111_perm_0 = const()[name = string("value_states_111_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_19_stride_0 = const()[name = string("value_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_111_cast_fp16 = transpose(perm = value_states_111_perm_0, x = var_6769_cast_fp16)[name = string("transpose_630")]; tensor value_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = value_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = value_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_19_squeeze_mask_0, stride = value_cache_internal_tensor_assign_19_stride_0, update = value_states_111_cast_fp16, x = coreml_update_state_427)[name = string("value_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_19_cast_fp16, input = value_cache)[name = string("coreml_update_state_429_write_state")]; tensor coreml_update_state_429 = read_state(input = value_cache)[name = string("coreml_update_state_429")]; tensor var_6863_begin_0 = const()[name = string("op_6863_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6863_end_0 = const()[name = string("op_6863_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6863_end_mask_0 = const()[name = string("op_6863_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6863_cast_fp16 = slice_by_index(begin = var_6863_begin_0, end = var_6863_end_0, end_mask = var_6863_end_mask_0, x = coreml_update_state_428)[name = string("op_6863_cast_fp16")]; tensor tile_36 = const()[name = string("tile_36"), val = tensor([1, 1])]; int32 var_6866_axis_0 = const()[name = string("op_6866_axis_0"), val = int32(1)]; tensor var_6866_cast_fp16_0, tensor var_6866_cast_fp16_1 = split(axis = var_6866_axis_0, split_sizes = tile_36, x = var_6863_cast_fp16)[name = string("op_6866_cast_fp16")]; tensor var_6873_begin_0 = const()[name = string("op_6873_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6873_end_0 = const()[name = string("op_6873_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6873_end_mask_0 = const()[name = string("op_6873_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6873_cast_fp16 = slice_by_index(begin = var_6873_begin_0, end = var_6873_end_0, end_mask = var_6873_end_mask_0, x = coreml_update_state_429)[name = string("op_6873_cast_fp16")]; tensor tile_37 = const()[name = string("tile_37"), val = tensor([1, 1])]; int32 var_6876_axis_0 = const()[name = string("op_6876_axis_0"), val = int32(1)]; tensor var_6876_cast_fp16_0, tensor var_6876_cast_fp16_1 = split(axis = var_6876_axis_0, split_sizes = tile_37, x = var_6873_cast_fp16)[name = string("op_6876_cast_fp16")]; tensor var_6879_split_sizes_0 = const()[name = string("op_6879_split_sizes_0"), val = tensor([8, 8])]; int32 var_6879_axis_0 = const()[name = string("op_6879_axis_0"), val = int32(1)]; tensor var_6879_0, tensor var_6879_1 = split(axis = var_6879_axis_0, split_sizes = var_6879_split_sizes_0, x = query_states_111_cast_fp16)[name = string("op_6879")]; bool attn_weights_289_transpose_x_0 = const()[name = string("attn_weights_289_transpose_x_0"), val = bool(false)]; bool attn_weights_289_transpose_y_0 = const()[name = string("attn_weights_289_transpose_y_0"), val = bool(false)]; tensor attn_weights_289_cast_fp16 = matmul(transpose_x = attn_weights_289_transpose_x_0, transpose_y = attn_weights_289_transpose_y_0, x = var_6866_cast_fp16_0, y = var_6879_0)[name = string("attn_weights_289_cast_fp16")]; fp16 var_6882_to_fp16 = const()[name = string("op_6882_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_291_cast_fp16 = mul(x = attn_weights_289_cast_fp16, y = var_6882_to_fp16)[name = string("attn_weights_291_cast_fp16")]; tensor attn_weights_293_cast_fp16 = add(x = attn_weights_291_cast_fp16, y = attn_mask_1)[name = string("attn_weights_293_cast_fp16")]; int32 var_6886 = const()[name = string("op_6886"), val = int32(-2)]; tensor attn_weights_295_cast_fp16 = softmax(axis = var_6886, x = attn_weights_293_cast_fp16)[name = string("attn_weights_295_cast_fp16")]; bool var_6892_transpose_x_1 = const()[name = string("op_6892_transpose_x_1"), val = bool(true)]; bool var_6892_transpose_y_1 = const()[name = string("op_6892_transpose_y_1"), val = bool(false)]; tensor var_6892_cast_fp16 = matmul(transpose_x = var_6892_transpose_x_1, transpose_y = var_6892_transpose_y_1, x = attn_weights_295_cast_fp16, y = var_6876_cast_fp16_0)[name = string("op_6892_cast_fp16")]; bool attn_weights_297_transpose_x_0 = const()[name = string("attn_weights_297_transpose_x_0"), val = bool(false)]; bool attn_weights_297_transpose_y_0 = const()[name = string("attn_weights_297_transpose_y_0"), val = bool(false)]; tensor attn_weights_297_cast_fp16 = matmul(transpose_x = attn_weights_297_transpose_x_0, transpose_y = attn_weights_297_transpose_y_0, x = var_6866_cast_fp16_1, y = var_6879_1)[name = string("attn_weights_297_cast_fp16")]; fp16 var_6894_to_fp16 = const()[name = string("op_6894_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_299_cast_fp16 = mul(x = attn_weights_297_cast_fp16, y = var_6894_to_fp16)[name = string("attn_weights_299_cast_fp16")]; tensor attn_weights_301_cast_fp16 = add(x = attn_weights_299_cast_fp16, y = attn_mask_1)[name = string("attn_weights_301_cast_fp16")]; int32 var_6898 = const()[name = string("op_6898"), val = int32(-2)]; tensor attn_weights_303_cast_fp16 = softmax(axis = var_6898, x = attn_weights_301_cast_fp16)[name = string("attn_weights_303_cast_fp16")]; bool attn_output_145_transpose_x_1 = const()[name = string("attn_output_145_transpose_x_1"), val = bool(true)]; bool attn_output_145_transpose_y_1 = const()[name = string("attn_output_145_transpose_y_1"), val = bool(false)]; tensor attn_output_145_cast_fp16 = matmul(transpose_x = attn_output_145_transpose_x_1, transpose_y = attn_output_145_transpose_y_1, x = attn_weights_303_cast_fp16, y = var_6876_cast_fp16_1)[name = string("attn_output_145_cast_fp16")]; int32 var_6906 = const()[name = string("op_6906"), val = int32(1)]; bool attn_output_147_interleave_0 = const()[name = string("attn_output_147_interleave_0"), val = bool(false)]; tensor attn_output_147_cast_fp16 = concat(axis = var_6906, interleave = attn_output_147_interleave_0, values = (var_6892_cast_fp16, attn_output_145_cast_fp16))[name = string("attn_output_147_cast_fp16")]; tensor var_6910_perm_0 = const()[name = string("op_6910_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_227x = const()[name = string("concat_227x"), val = tensor([1, 2048, 1, -1])]; tensor var_6910_cast_fp16 = transpose(perm = var_6910_perm_0, x = attn_output_147_cast_fp16)[name = string("transpose_629")]; tensor attn_output_151_cast_fp16 = reshape(shape = concat_227x, x = var_6910_cast_fp16)[name = string("attn_output_151_cast_fp16")]; tensor hidden_states_183_strides_0 = const()[name = string("hidden_states_183_strides_0"), val = tensor([1, 1])]; string hidden_states_183_pad_type_0 = const()[name = string("hidden_states_183_pad_type_0"), val = string("valid")]; tensor hidden_states_183_pad_0 = const()[name = string("hidden_states_183_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_183_dilations_0 = const()[name = string("hidden_states_183_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_183_groups_0 = const()[name = string("hidden_states_183_groups_0"), val = int32(1)]; tensor hidden_states_183_cast_fp16 = conv(dilations = hidden_states_183_dilations_0, groups = hidden_states_183_groups_0, pad = hidden_states_183_pad_0, pad_type = hidden_states_183_pad_type_0, strides = hidden_states_183_strides_0, weight = layers_18_self_attn_o_proj_weight_cast_fp16, x = attn_output_151_cast_fp16)[name = string("hidden_states_183_cast_fp16")]; tensor hidden_states_185_cast_fp16 = add(x = hidden_states_179_cast_fp16, y = hidden_states_183_cast_fp16)[name = string("hidden_states_185_cast_fp16")]; fp16 const_188_promoted_to_fp16 = const()[name = string("const_188_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6943_cast_fp16 = mul(x = hidden_states_185_cast_fp16, y = const_188_promoted_to_fp16)[name = string("op_6943_cast_fp16")]; int32 var_6941 = const()[name = string("op_6941"), val = int32(1)]; bool doubled_149_interleave_0 = const()[name = string("doubled_149_interleave_0"), val = bool(false)]; tensor doubled_149_cast_fp16 = concat(axis = var_6941, interleave = doubled_149_interleave_0, values = (hidden_states_185_cast_fp16, var_6943_cast_fp16))[name = string("doubled_149_cast_fp16")]; tensor out_75_axes_0 = const()[name = string("out_75_axes_0"), val = tensor([1])]; tensor out_75_gamma_0_to_fp16 = const()[name = string("out_75_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492028288)))]; fp16 var_6953_to_fp16 = const()[name = string("op_6953_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_75_cast_fp16 = layer_norm(axes = out_75_axes_0, epsilon = var_6953_to_fp16, gamma = out_75_gamma_0_to_fp16, x = doubled_149_cast_fp16)[name = string("out_75_cast_fp16")]; tensor var_6964_split_sizes_0 = const()[name = string("op_6964_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6964_axis_0 = const()[name = string("op_6964_axis_0"), val = int32(1)]; tensor var_6964_cast_fp16_0, tensor var_6964_cast_fp16_1 = split(axis = var_6964_axis_0, split_sizes = var_6964_split_sizes_0, x = out_75_cast_fp16)[name = string("op_6964_cast_fp16")]; tensor input_37_strides_0 = const()[name = string("input_37_strides_0"), val = tensor([1, 1])]; string input_37_pad_type_0 = const()[name = string("input_37_pad_type_0"), val = string("valid")]; tensor input_37_pad_0 = const()[name = string("input_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_37_dilations_0 = const()[name = string("input_37_dilations_0"), val = tensor([1, 1])]; int32 input_37_groups_0 = const()[name = string("input_37_groups_0"), val = int32(1)]; tensor input_37_cast_fp16 = conv(dilations = input_37_dilations_0, groups = input_37_groups_0, pad = input_37_pad_0, pad_type = input_37_pad_type_0, strides = input_37_strides_0, weight = layers_18_mlp_gate_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("input_37_cast_fp16")]; tensor var_6981_cast_fp16 = silu(x = input_37_cast_fp16)[name = string("op_6981_cast_fp16")]; tensor var_6987_strides_0 = const()[name = string("op_6987_strides_0"), val = tensor([1, 1])]; string var_6987_pad_type_0 = const()[name = string("op_6987_pad_type_0"), val = string("valid")]; tensor var_6987_pad_0 = const()[name = string("op_6987_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6987_dilations_0 = const()[name = string("op_6987_dilations_0"), val = tensor([1, 1])]; int32 var_6987_groups_0 = const()[name = string("op_6987_groups_0"), val = int32(1)]; tensor var_6987_cast_fp16 = conv(dilations = var_6987_dilations_0, groups = var_6987_groups_0, pad = var_6987_pad_0, pad_type = var_6987_pad_type_0, strides = var_6987_strides_0, weight = layers_18_mlp_up_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("op_6987_cast_fp16")]; tensor x_189_cast_fp16 = mul(x = var_6981_cast_fp16, y = var_6987_cast_fp16)[name = string("x_189_cast_fp16")]; tensor hidden_states_187_strides_0 = const()[name = string("hidden_states_187_strides_0"), val = tensor([1, 1])]; string hidden_states_187_pad_type_0 = const()[name = string("hidden_states_187_pad_type_0"), val = string("valid")]; tensor hidden_states_187_pad_0 = const()[name = string("hidden_states_187_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_187_dilations_0 = const()[name = string("hidden_states_187_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_187_groups_0 = const()[name = string("hidden_states_187_groups_0"), val = int32(1)]; tensor hidden_states_187_cast_fp16 = conv(dilations = hidden_states_187_dilations_0, groups = hidden_states_187_groups_0, pad = hidden_states_187_pad_0, pad_type = hidden_states_187_pad_type_0, strides = hidden_states_187_strides_0, weight = layers_18_mlp_down_proj_weight_cast_fp16, x = x_189_cast_fp16)[name = string("hidden_states_187_cast_fp16")]; tensor hidden_states_189_cast_fp16 = add(x = hidden_states_185_cast_fp16, y = hidden_states_187_cast_fp16)[name = string("hidden_states_189_cast_fp16")]; fp16 const_190_promoted_to_fp16 = const()[name = string("const_190_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7005_cast_fp16 = mul(x = hidden_states_189_cast_fp16, y = const_190_promoted_to_fp16)[name = string("op_7005_cast_fp16")]; int32 var_7003 = const()[name = string("op_7003"), val = int32(1)]; bool doubled_153_interleave_0 = const()[name = string("doubled_153_interleave_0"), val = bool(false)]; tensor doubled_153_cast_fp16 = concat(axis = var_7003, interleave = doubled_153_interleave_0, values = (hidden_states_189_cast_fp16, var_7005_cast_fp16))[name = string("doubled_153_cast_fp16")]; tensor out_77_axes_0 = const()[name = string("out_77_axes_0"), val = tensor([1])]; tensor out_77_gamma_0_to_fp16 = const()[name = string("out_77_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492036544)))]; fp16 var_7015_to_fp16 = const()[name = string("op_7015_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_77_cast_fp16 = layer_norm(axes = out_77_axes_0, epsilon = var_7015_to_fp16, gamma = out_77_gamma_0_to_fp16, x = doubled_153_cast_fp16)[name = string("out_77_cast_fp16")]; tensor var_7026_split_sizes_0 = const()[name = string("op_7026_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7026_axis_0 = const()[name = string("op_7026_axis_0"), val = int32(1)]; tensor var_7026_cast_fp16_0, tensor var_7026_cast_fp16_1 = split(axis = var_7026_axis_0, split_sizes = var_7026_split_sizes_0, x = out_77_cast_fp16)[name = string("op_7026_cast_fp16")]; tensor query_states_115_strides_0 = const()[name = string("query_states_115_strides_0"), val = tensor([1, 1])]; string query_states_115_pad_type_0 = const()[name = string("query_states_115_pad_type_0"), val = string("valid")]; tensor query_states_115_pad_0 = const()[name = string("query_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_115_dilations_0 = const()[name = string("query_states_115_dilations_0"), val = tensor([1, 1])]; int32 query_states_115_groups_0 = const()[name = string("query_states_115_groups_0"), val = int32(1)]; tensor query_states_115_cast_fp16 = conv(dilations = query_states_115_dilations_0, groups = query_states_115_groups_0, pad = query_states_115_pad_0, pad_type = query_states_115_pad_type_0, strides = query_states_115_strides_0, weight = layers_19_self_attn_q_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("query_states_115_cast_fp16")]; tensor key_states_191_strides_0 = const()[name = string("key_states_191_strides_0"), val = tensor([1, 1])]; string key_states_191_pad_type_0 = const()[name = string("key_states_191_pad_type_0"), val = string("valid")]; tensor key_states_191_pad_0 = const()[name = string("key_states_191_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_191_dilations_0 = const()[name = string("key_states_191_dilations_0"), val = tensor([1, 1])]; int32 key_states_191_groups_0 = const()[name = string("key_states_191_groups_0"), val = int32(1)]; tensor key_states_191_cast_fp16 = conv(dilations = key_states_191_dilations_0, groups = key_states_191_groups_0, pad = key_states_191_pad_0, pad_type = key_states_191_pad_type_0, strides = key_states_191_strides_0, weight = layers_19_self_attn_k_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("key_states_191_cast_fp16")]; tensor layers_19_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492044800)))]; tensor value_states_115_strides_0 = const()[name = string("value_states_115_strides_0"), val = tensor([1, 1])]; string value_states_115_pad_type_0 = const()[name = string("value_states_115_pad_type_0"), val = string("valid")]; tensor value_states_115_pad_0 = const()[name = string("value_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_115_dilations_0 = const()[name = string("value_states_115_dilations_0"), val = tensor([1, 1])]; int32 value_states_115_groups_0 = const()[name = string("value_states_115_groups_0"), val = int32(1)]; tensor value_states_115_cast_fp16 = conv(dilations = value_states_115_dilations_0, groups = value_states_115_groups_0, pad = value_states_115_pad_0, pad_type = value_states_115_pad_type_0, strides = value_states_115_strides_0, weight = layers_19_self_attn_v_proj_weight_to_fp16, x = var_7026_cast_fp16_0)[name = string("value_states_115_cast_fp16")]; tensor concat_228x = const()[name = string("concat_228x"), val = tensor([1, 16, 128, -1])]; tensor x_191_cast_fp16 = reshape(shape = concat_228x, x = query_states_115_cast_fp16)[name = string("x_191_cast_fp16")]; tensor concat_229x = const()[name = string("concat_229x"), val = tensor([1, 2, 128, -1])]; tensor var_7083_cast_fp16 = reshape(shape = concat_229x, x = key_states_191_cast_fp16)[name = string("op_7083_cast_fp16")]; tensor concat_230x = const()[name = string("concat_230x"), val = tensor([1, 2, 128, -1])]; tensor var_7090_cast_fp16 = reshape(shape = concat_230x, x = value_states_115_cast_fp16)[name = string("op_7090_cast_fp16")]; tensor var_7094_cast_fp16 = mul(x = x_191_cast_fp16, y = var_869_cast_fp16)[name = string("op_7094_cast_fp16")]; tensor var_7095_split_sizes_0 = const()[name = string("op_7095_split_sizes_0"), val = tensor([64, 64])]; int32 var_7095_axis_0 = const()[name = string("op_7095_axis_0"), val = int32(-2)]; tensor var_7095_cast_fp16_0, tensor var_7095_cast_fp16_1 = split(axis = var_7095_axis_0, split_sizes = var_7095_split_sizes_0, x = x_191_cast_fp16)[name = string("op_7095_cast_fp16")]; fp16 const_192_promoted_to_fp16 = const()[name = string("const_192_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7097_cast_fp16 = mul(x = var_7095_cast_fp16_1, y = const_192_promoted_to_fp16)[name = string("op_7097_cast_fp16")]; int32 var_7099 = const()[name = string("op_7099"), val = int32(-2)]; bool var_7100_interleave_0 = const()[name = string("op_7100_interleave_0"), val = bool(false)]; tensor var_7100_cast_fp16 = concat(axis = var_7099, interleave = var_7100_interleave_0, values = (var_7097_cast_fp16, var_7095_cast_fp16_0))[name = string("op_7100_cast_fp16")]; tensor var_7101_cast_fp16 = mul(x = var_7100_cast_fp16, y = var_878_cast_fp16)[name = string("op_7101_cast_fp16")]; tensor query_states_117_cast_fp16 = add(x = var_7094_cast_fp16, y = var_7101_cast_fp16)[name = string("query_states_117_cast_fp16")]; tensor var_7107_cast_fp16 = mul(x = var_7083_cast_fp16, y = var_869_cast_fp16)[name = string("op_7107_cast_fp16")]; tensor var_7108_split_sizes_0 = const()[name = string("op_7108_split_sizes_0"), val = tensor([64, 64])]; int32 var_7108_axis_0 = const()[name = string("op_7108_axis_0"), val = int32(-2)]; tensor var_7108_cast_fp16_0, tensor var_7108_cast_fp16_1 = split(axis = var_7108_axis_0, split_sizes = var_7108_split_sizes_0, x = var_7083_cast_fp16)[name = string("op_7108_cast_fp16")]; fp16 const_193_promoted_to_fp16 = const()[name = string("const_193_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7110_cast_fp16 = mul(x = var_7108_cast_fp16_1, y = const_193_promoted_to_fp16)[name = string("op_7110_cast_fp16")]; int32 var_7112 = const()[name = string("op_7112"), val = int32(-2)]; bool var_7113_interleave_0 = const()[name = string("op_7113_interleave_0"), val = bool(false)]; tensor var_7113_cast_fp16 = concat(axis = var_7112, interleave = var_7113_interleave_0, values = (var_7110_cast_fp16, var_7108_cast_fp16_0))[name = string("op_7113_cast_fp16")]; tensor var_7114_cast_fp16 = mul(x = var_7113_cast_fp16, y = var_878_cast_fp16)[name = string("op_7114_cast_fp16")]; tensor key_states_195_cast_fp16 = add(x = var_7107_cast_fp16, y = var_7114_cast_fp16)[name = string("key_states_195_cast_fp16")]; tensor expand_dims_228 = const()[name = string("expand_dims_228"), val = tensor([19])]; tensor expand_dims_229 = const()[name = string("expand_dims_229"), val = tensor([0])]; tensor expand_dims_231 = const()[name = string("expand_dims_231"), val = tensor([0])]; int32 concat_233_axis_0 = const()[name = string("concat_233_axis_0"), val = int32(0)]; bool concat_233_interleave_0 = const()[name = string("concat_233_interleave_0"), val = bool(false)]; tensor concat_233 = concat(axis = concat_233_axis_0, interleave = concat_233_interleave_0, values = (expand_dims_228, expand_dims_229, position_id, expand_dims_231))[name = string("concat_233")]; tensor expand_dims_232 = const()[name = string("expand_dims_232"), val = tensor([20])]; tensor concat_234_values1_0 = const()[name = string("concat_234_values1_0"), val = tensor([0])]; tensor concat_234_values3_0 = const()[name = string("concat_234_values3_0"), val = tensor([0])]; int32 concat_234_axis_0 = const()[name = string("concat_234_axis_0"), val = int32(0)]; bool concat_234_interleave_0 = const()[name = string("concat_234_interleave_0"), val = bool(false)]; tensor concat_234 = concat(axis = concat_234_axis_0, interleave = concat_234_interleave_0, values = (expand_dims_232, concat_234_values1_0, cache_position_end, concat_234_values3_0))[name = string("concat_234")]; tensor key_states_197_perm_0 = const()[name = string("key_states_197_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_20_stride_0 = const()[name = string("key_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_197_cast_fp16 = transpose(perm = key_states_197_perm_0, x = key_states_195_cast_fp16)[name = string("transpose_628")]; tensor key_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = key_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = key_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_20_squeeze_mask_0, stride = key_cache_internal_tensor_assign_20_stride_0, update = key_states_197_cast_fp16, x = coreml_update_state_428)[name = string("key_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_20_cast_fp16, input = key_cache)[name = string("coreml_update_state_430_write_state")]; tensor coreml_update_state_430 = read_state(input = key_cache)[name = string("coreml_update_state_430")]; tensor value_states_117_perm_0 = const()[name = string("value_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_20_stride_0 = const()[name = string("value_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_117_cast_fp16 = transpose(perm = value_states_117_perm_0, x = var_7090_cast_fp16)[name = string("transpose_627")]; tensor value_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = value_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = value_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_20_squeeze_mask_0, stride = value_cache_internal_tensor_assign_20_stride_0, update = value_states_117_cast_fp16, x = coreml_update_state_429)[name = string("value_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_20_cast_fp16, input = value_cache)[name = string("coreml_update_state_431_write_state")]; tensor coreml_update_state_431 = read_state(input = value_cache)[name = string("coreml_update_state_431")]; tensor var_7184_begin_0 = const()[name = string("op_7184_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7184_end_0 = const()[name = string("op_7184_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7184_end_mask_0 = const()[name = string("op_7184_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7184_cast_fp16 = slice_by_index(begin = var_7184_begin_0, end = var_7184_end_0, end_mask = var_7184_end_mask_0, x = coreml_update_state_430)[name = string("op_7184_cast_fp16")]; tensor tile_38 = const()[name = string("tile_38"), val = tensor([1, 1])]; int32 var_7187_axis_0 = const()[name = string("op_7187_axis_0"), val = int32(1)]; tensor var_7187_cast_fp16_0, tensor var_7187_cast_fp16_1 = split(axis = var_7187_axis_0, split_sizes = tile_38, x = var_7184_cast_fp16)[name = string("op_7187_cast_fp16")]; tensor var_7194_begin_0 = const()[name = string("op_7194_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7194_end_0 = const()[name = string("op_7194_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7194_end_mask_0 = const()[name = string("op_7194_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7194_cast_fp16 = slice_by_index(begin = var_7194_begin_0, end = var_7194_end_0, end_mask = var_7194_end_mask_0, x = coreml_update_state_431)[name = string("op_7194_cast_fp16")]; tensor tile_39 = const()[name = string("tile_39"), val = tensor([1, 1])]; int32 var_7197_axis_0 = const()[name = string("op_7197_axis_0"), val = int32(1)]; tensor var_7197_cast_fp16_0, tensor var_7197_cast_fp16_1 = split(axis = var_7197_axis_0, split_sizes = tile_39, x = var_7194_cast_fp16)[name = string("op_7197_cast_fp16")]; tensor var_7200_split_sizes_0 = const()[name = string("op_7200_split_sizes_0"), val = tensor([8, 8])]; int32 var_7200_axis_0 = const()[name = string("op_7200_axis_0"), val = int32(1)]; tensor var_7200_0, tensor var_7200_1 = split(axis = var_7200_axis_0, split_sizes = var_7200_split_sizes_0, x = query_states_117_cast_fp16)[name = string("op_7200")]; bool attn_weights_305_transpose_x_0 = const()[name = string("attn_weights_305_transpose_x_0"), val = bool(false)]; bool attn_weights_305_transpose_y_0 = const()[name = string("attn_weights_305_transpose_y_0"), val = bool(false)]; tensor attn_weights_305_cast_fp16 = matmul(transpose_x = attn_weights_305_transpose_x_0, transpose_y = attn_weights_305_transpose_y_0, x = var_7187_cast_fp16_0, y = var_7200_0)[name = string("attn_weights_305_cast_fp16")]; fp16 var_7203_to_fp16 = const()[name = string("op_7203_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_307_cast_fp16 = mul(x = attn_weights_305_cast_fp16, y = var_7203_to_fp16)[name = string("attn_weights_307_cast_fp16")]; tensor attn_weights_309_cast_fp16 = add(x = attn_weights_307_cast_fp16, y = attn_mask_1)[name = string("attn_weights_309_cast_fp16")]; int32 var_7207 = const()[name = string("op_7207"), val = int32(-2)]; tensor attn_weights_311_cast_fp16 = softmax(axis = var_7207, x = attn_weights_309_cast_fp16)[name = string("attn_weights_311_cast_fp16")]; bool var_7213_transpose_x_1 = const()[name = string("op_7213_transpose_x_1"), val = bool(true)]; bool var_7213_transpose_y_1 = const()[name = string("op_7213_transpose_y_1"), val = bool(false)]; tensor var_7213_cast_fp16 = matmul(transpose_x = var_7213_transpose_x_1, transpose_y = var_7213_transpose_y_1, x = attn_weights_311_cast_fp16, y = var_7197_cast_fp16_0)[name = string("op_7213_cast_fp16")]; bool attn_weights_313_transpose_x_0 = const()[name = string("attn_weights_313_transpose_x_0"), val = bool(false)]; bool attn_weights_313_transpose_y_0 = const()[name = string("attn_weights_313_transpose_y_0"), val = bool(false)]; tensor attn_weights_313_cast_fp16 = matmul(transpose_x = attn_weights_313_transpose_x_0, transpose_y = attn_weights_313_transpose_y_0, x = var_7187_cast_fp16_1, y = var_7200_1)[name = string("attn_weights_313_cast_fp16")]; fp16 var_7215_to_fp16 = const()[name = string("op_7215_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_315_cast_fp16 = mul(x = attn_weights_313_cast_fp16, y = var_7215_to_fp16)[name = string("attn_weights_315_cast_fp16")]; tensor attn_weights_317_cast_fp16 = add(x = attn_weights_315_cast_fp16, y = attn_mask_1)[name = string("attn_weights_317_cast_fp16")]; int32 var_7219 = const()[name = string("op_7219"), val = int32(-2)]; tensor attn_weights_319_cast_fp16 = softmax(axis = var_7219, x = attn_weights_317_cast_fp16)[name = string("attn_weights_319_cast_fp16")]; bool attn_output_153_transpose_x_1 = const()[name = string("attn_output_153_transpose_x_1"), val = bool(true)]; bool attn_output_153_transpose_y_1 = const()[name = string("attn_output_153_transpose_y_1"), val = bool(false)]; tensor attn_output_153_cast_fp16 = matmul(transpose_x = attn_output_153_transpose_x_1, transpose_y = attn_output_153_transpose_y_1, x = attn_weights_319_cast_fp16, y = var_7197_cast_fp16_1)[name = string("attn_output_153_cast_fp16")]; int32 var_7227 = const()[name = string("op_7227"), val = int32(1)]; bool attn_output_155_interleave_0 = const()[name = string("attn_output_155_interleave_0"), val = bool(false)]; tensor attn_output_155_cast_fp16 = concat(axis = var_7227, interleave = attn_output_155_interleave_0, values = (var_7213_cast_fp16, attn_output_153_cast_fp16))[name = string("attn_output_155_cast_fp16")]; tensor var_7231_perm_0 = const()[name = string("op_7231_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_239x = const()[name = string("concat_239x"), val = tensor([1, 2048, 1, -1])]; tensor var_7231_cast_fp16 = transpose(perm = var_7231_perm_0, x = attn_output_155_cast_fp16)[name = string("transpose_626")]; tensor attn_output_159_cast_fp16 = reshape(shape = concat_239x, x = var_7231_cast_fp16)[name = string("attn_output_159_cast_fp16")]; tensor layers_19_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1493093440)))]; tensor hidden_states_193_strides_0 = const()[name = string("hidden_states_193_strides_0"), val = tensor([1, 1])]; string hidden_states_193_pad_type_0 = const()[name = string("hidden_states_193_pad_type_0"), val = string("valid")]; tensor hidden_states_193_pad_0 = const()[name = string("hidden_states_193_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_193_dilations_0 = const()[name = string("hidden_states_193_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_193_groups_0 = const()[name = string("hidden_states_193_groups_0"), val = int32(1)]; tensor hidden_states_193_cast_fp16 = conv(dilations = hidden_states_193_dilations_0, groups = hidden_states_193_groups_0, pad = hidden_states_193_pad_0, pad_type = hidden_states_193_pad_type_0, strides = hidden_states_193_strides_0, weight = layers_19_self_attn_o_proj_weight_to_fp16, x = attn_output_159_cast_fp16)[name = string("hidden_states_193_cast_fp16")]; tensor hidden_states_195_cast_fp16 = add(x = hidden_states_189_cast_fp16, y = hidden_states_193_cast_fp16)[name = string("hidden_states_195_cast_fp16")]; fp16 const_198_promoted_to_fp16 = const()[name = string("const_198_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7264_cast_fp16 = mul(x = hidden_states_195_cast_fp16, y = const_198_promoted_to_fp16)[name = string("op_7264_cast_fp16")]; int32 var_7262 = const()[name = string("op_7262"), val = int32(1)]; bool doubled_157_interleave_0 = const()[name = string("doubled_157_interleave_0"), val = bool(false)]; tensor doubled_157_cast_fp16 = concat(axis = var_7262, interleave = doubled_157_interleave_0, values = (hidden_states_195_cast_fp16, var_7264_cast_fp16))[name = string("doubled_157_cast_fp16")]; tensor out_79_axes_0 = const()[name = string("out_79_axes_0"), val = tensor([1])]; tensor out_79_gamma_0_to_fp16 = const()[name = string("out_79_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501482112)))]; fp16 var_7274_to_fp16 = const()[name = string("op_7274_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_79_cast_fp16 = layer_norm(axes = out_79_axes_0, epsilon = var_7274_to_fp16, gamma = out_79_gamma_0_to_fp16, x = doubled_157_cast_fp16)[name = string("out_79_cast_fp16")]; tensor var_7285_split_sizes_0 = const()[name = string("op_7285_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7285_axis_0 = const()[name = string("op_7285_axis_0"), val = int32(1)]; tensor var_7285_cast_fp16_0, tensor var_7285_cast_fp16_1 = split(axis = var_7285_axis_0, split_sizes = var_7285_split_sizes_0, x = out_79_cast_fp16)[name = string("op_7285_cast_fp16")]; tensor input_39_strides_0 = const()[name = string("input_39_strides_0"), val = tensor([1, 1])]; string input_39_pad_type_0 = const()[name = string("input_39_pad_type_0"), val = string("valid")]; tensor input_39_pad_0 = const()[name = string("input_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_39_dilations_0 = const()[name = string("input_39_dilations_0"), val = tensor([1, 1])]; int32 input_39_groups_0 = const()[name = string("input_39_groups_0"), val = int32(1)]; tensor input_39_cast_fp16 = conv(dilations = input_39_dilations_0, groups = input_39_groups_0, pad = input_39_pad_0, pad_type = input_39_pad_type_0, strides = input_39_strides_0, weight = layers_19_mlp_gate_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("input_39_cast_fp16")]; tensor var_7302_cast_fp16 = silu(x = input_39_cast_fp16)[name = string("op_7302_cast_fp16")]; tensor var_7308_strides_0 = const()[name = string("op_7308_strides_0"), val = tensor([1, 1])]; string var_7308_pad_type_0 = const()[name = string("op_7308_pad_type_0"), val = string("valid")]; tensor var_7308_pad_0 = const()[name = string("op_7308_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7308_dilations_0 = const()[name = string("op_7308_dilations_0"), val = tensor([1, 1])]; int32 var_7308_groups_0 = const()[name = string("op_7308_groups_0"), val = int32(1)]; tensor var_7308_cast_fp16 = conv(dilations = var_7308_dilations_0, groups = var_7308_groups_0, pad = var_7308_pad_0, pad_type = var_7308_pad_type_0, strides = var_7308_strides_0, weight = layers_19_mlp_up_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("op_7308_cast_fp16")]; tensor x_199_cast_fp16 = mul(x = var_7302_cast_fp16, y = var_7308_cast_fp16)[name = string("x_199_cast_fp16")]; tensor hidden_states_197_strides_0 = const()[name = string("hidden_states_197_strides_0"), val = tensor([1, 1])]; string hidden_states_197_pad_type_0 = const()[name = string("hidden_states_197_pad_type_0"), val = string("valid")]; tensor hidden_states_197_pad_0 = const()[name = string("hidden_states_197_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_197_dilations_0 = const()[name = string("hidden_states_197_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_197_groups_0 = const()[name = string("hidden_states_197_groups_0"), val = int32(1)]; tensor hidden_states_197_cast_fp16 = conv(dilations = hidden_states_197_dilations_0, groups = hidden_states_197_groups_0, pad = hidden_states_197_pad_0, pad_type = hidden_states_197_pad_type_0, strides = hidden_states_197_strides_0, weight = layers_19_mlp_down_proj_weight_cast_fp16, x = x_199_cast_fp16)[name = string("hidden_states_197_cast_fp16")]; tensor hidden_states_199_cast_fp16 = add(x = hidden_states_195_cast_fp16, y = hidden_states_197_cast_fp16)[name = string("hidden_states_199_cast_fp16")]; fp16 const_200_promoted_to_fp16 = const()[name = string("const_200_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7326_cast_fp16 = mul(x = hidden_states_199_cast_fp16, y = const_200_promoted_to_fp16)[name = string("op_7326_cast_fp16")]; int32 var_7324 = const()[name = string("op_7324"), val = int32(1)]; bool doubled_161_interleave_0 = const()[name = string("doubled_161_interleave_0"), val = bool(false)]; tensor doubled_161_cast_fp16 = concat(axis = var_7324, interleave = doubled_161_interleave_0, values = (hidden_states_199_cast_fp16, var_7326_cast_fp16))[name = string("doubled_161_cast_fp16")]; tensor out_81_axes_0 = const()[name = string("out_81_axes_0"), val = tensor([1])]; tensor out_81_gamma_0_to_fp16 = const()[name = string("out_81_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501490368)))]; fp16 var_7336_to_fp16 = const()[name = string("op_7336_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_81_cast_fp16 = layer_norm(axes = out_81_axes_0, epsilon = var_7336_to_fp16, gamma = out_81_gamma_0_to_fp16, x = doubled_161_cast_fp16)[name = string("out_81_cast_fp16")]; tensor var_7347_split_sizes_0 = const()[name = string("op_7347_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7347_axis_0 = const()[name = string("op_7347_axis_0"), val = int32(1)]; tensor var_7347_cast_fp16_0, tensor var_7347_cast_fp16_1 = split(axis = var_7347_axis_0, split_sizes = var_7347_split_sizes_0, x = out_81_cast_fp16)[name = string("op_7347_cast_fp16")]; tensor query_states_121_strides_0 = const()[name = string("query_states_121_strides_0"), val = tensor([1, 1])]; string query_states_121_pad_type_0 = const()[name = string("query_states_121_pad_type_0"), val = string("valid")]; tensor query_states_121_pad_0 = const()[name = string("query_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_121_dilations_0 = const()[name = string("query_states_121_dilations_0"), val = tensor([1, 1])]; int32 query_states_121_groups_0 = const()[name = string("query_states_121_groups_0"), val = int32(1)]; tensor query_states_121_cast_fp16 = conv(dilations = query_states_121_dilations_0, groups = query_states_121_groups_0, pad = query_states_121_pad_0, pad_type = query_states_121_pad_type_0, strides = query_states_121_strides_0, weight = layers_20_self_attn_q_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("query_states_121_cast_fp16")]; tensor key_states_201_strides_0 = const()[name = string("key_states_201_strides_0"), val = tensor([1, 1])]; string key_states_201_pad_type_0 = const()[name = string("key_states_201_pad_type_0"), val = string("valid")]; tensor key_states_201_pad_0 = const()[name = string("key_states_201_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_201_dilations_0 = const()[name = string("key_states_201_dilations_0"), val = tensor([1, 1])]; int32 key_states_201_groups_0 = const()[name = string("key_states_201_groups_0"), val = int32(1)]; tensor key_states_201_cast_fp16 = conv(dilations = key_states_201_dilations_0, groups = key_states_201_groups_0, pad = key_states_201_pad_0, pad_type = key_states_201_pad_type_0, strides = key_states_201_strides_0, weight = layers_20_self_attn_k_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("key_states_201_cast_fp16")]; tensor layers_20_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_20_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501498624)))]; tensor value_states_121_strides_0 = const()[name = string("value_states_121_strides_0"), val = tensor([1, 1])]; string value_states_121_pad_type_0 = const()[name = string("value_states_121_pad_type_0"), val = string("valid")]; tensor value_states_121_pad_0 = const()[name = string("value_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_121_dilations_0 = const()[name = string("value_states_121_dilations_0"), val = tensor([1, 1])]; int32 value_states_121_groups_0 = const()[name = string("value_states_121_groups_0"), val = int32(1)]; tensor value_states_121_cast_fp16 = conv(dilations = value_states_121_dilations_0, groups = value_states_121_groups_0, pad = value_states_121_pad_0, pad_type = value_states_121_pad_type_0, strides = value_states_121_strides_0, weight = layers_20_self_attn_v_proj_weight_to_fp16, x = var_7347_cast_fp16_0)[name = string("value_states_121_cast_fp16")]; tensor concat_240x = const()[name = string("concat_240x"), val = tensor([1, 16, 128, -1])]; tensor x_201_cast_fp16 = reshape(shape = concat_240x, x = query_states_121_cast_fp16)[name = string("x_201_cast_fp16")]; tensor concat_241x = const()[name = string("concat_241x"), val = tensor([1, 2, 128, -1])]; tensor var_7404_cast_fp16 = reshape(shape = concat_241x, x = key_states_201_cast_fp16)[name = string("op_7404_cast_fp16")]; tensor concat_242x = const()[name = string("concat_242x"), val = tensor([1, 2, 128, -1])]; tensor var_7411_cast_fp16 = reshape(shape = concat_242x, x = value_states_121_cast_fp16)[name = string("op_7411_cast_fp16")]; tensor var_7415_cast_fp16 = mul(x = x_201_cast_fp16, y = var_869_cast_fp16)[name = string("op_7415_cast_fp16")]; tensor var_7416_split_sizes_0 = const()[name = string("op_7416_split_sizes_0"), val = tensor([64, 64])]; int32 var_7416_axis_0 = const()[name = string("op_7416_axis_0"), val = int32(-2)]; tensor var_7416_cast_fp16_0, tensor var_7416_cast_fp16_1 = split(axis = var_7416_axis_0, split_sizes = var_7416_split_sizes_0, x = x_201_cast_fp16)[name = string("op_7416_cast_fp16")]; fp16 const_202_promoted_to_fp16 = const()[name = string("const_202_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7418_cast_fp16 = mul(x = var_7416_cast_fp16_1, y = const_202_promoted_to_fp16)[name = string("op_7418_cast_fp16")]; int32 var_7420 = const()[name = string("op_7420"), val = int32(-2)]; bool var_7421_interleave_0 = const()[name = string("op_7421_interleave_0"), val = bool(false)]; tensor var_7421_cast_fp16 = concat(axis = var_7420, interleave = var_7421_interleave_0, values = (var_7418_cast_fp16, var_7416_cast_fp16_0))[name = string("op_7421_cast_fp16")]; tensor var_7422_cast_fp16 = mul(x = var_7421_cast_fp16, y = var_878_cast_fp16)[name = string("op_7422_cast_fp16")]; tensor query_states_123_cast_fp16 = add(x = var_7415_cast_fp16, y = var_7422_cast_fp16)[name = string("query_states_123_cast_fp16")]; tensor var_7428_cast_fp16 = mul(x = var_7404_cast_fp16, y = var_869_cast_fp16)[name = string("op_7428_cast_fp16")]; tensor var_7429_split_sizes_0 = const()[name = string("op_7429_split_sizes_0"), val = tensor([64, 64])]; int32 var_7429_axis_0 = const()[name = string("op_7429_axis_0"), val = int32(-2)]; tensor var_7429_cast_fp16_0, tensor var_7429_cast_fp16_1 = split(axis = var_7429_axis_0, split_sizes = var_7429_split_sizes_0, x = var_7404_cast_fp16)[name = string("op_7429_cast_fp16")]; fp16 const_203_promoted_to_fp16 = const()[name = string("const_203_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7431_cast_fp16 = mul(x = var_7429_cast_fp16_1, y = const_203_promoted_to_fp16)[name = string("op_7431_cast_fp16")]; int32 var_7433 = const()[name = string("op_7433"), val = int32(-2)]; bool var_7434_interleave_0 = const()[name = string("op_7434_interleave_0"), val = bool(false)]; tensor var_7434_cast_fp16 = concat(axis = var_7433, interleave = var_7434_interleave_0, values = (var_7431_cast_fp16, var_7429_cast_fp16_0))[name = string("op_7434_cast_fp16")]; tensor var_7435_cast_fp16 = mul(x = var_7434_cast_fp16, y = var_878_cast_fp16)[name = string("op_7435_cast_fp16")]; tensor key_states_205_cast_fp16 = add(x = var_7428_cast_fp16, y = var_7435_cast_fp16)[name = string("key_states_205_cast_fp16")]; tensor expand_dims_240 = const()[name = string("expand_dims_240"), val = tensor([20])]; tensor expand_dims_241 = const()[name = string("expand_dims_241"), val = tensor([0])]; tensor expand_dims_243 = const()[name = string("expand_dims_243"), val = tensor([0])]; int32 concat_245_axis_0 = const()[name = string("concat_245_axis_0"), val = int32(0)]; bool concat_245_interleave_0 = const()[name = string("concat_245_interleave_0"), val = bool(false)]; tensor concat_245 = concat(axis = concat_245_axis_0, interleave = concat_245_interleave_0, values = (expand_dims_240, expand_dims_241, position_id, expand_dims_243))[name = string("concat_245")]; tensor expand_dims_244 = const()[name = string("expand_dims_244"), val = tensor([21])]; tensor concat_246_values1_0 = const()[name = string("concat_246_values1_0"), val = tensor([0])]; tensor concat_246_values3_0 = const()[name = string("concat_246_values3_0"), val = tensor([0])]; int32 concat_246_axis_0 = const()[name = string("concat_246_axis_0"), val = int32(0)]; bool concat_246_interleave_0 = const()[name = string("concat_246_interleave_0"), val = bool(false)]; tensor concat_246 = concat(axis = concat_246_axis_0, interleave = concat_246_interleave_0, values = (expand_dims_244, concat_246_values1_0, cache_position_end, concat_246_values3_0))[name = string("concat_246")]; tensor key_states_207_perm_0 = const()[name = string("key_states_207_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_21_stride_0 = const()[name = string("key_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_207_cast_fp16 = transpose(perm = key_states_207_perm_0, x = key_states_205_cast_fp16)[name = string("transpose_625")]; tensor key_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = key_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = key_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_21_squeeze_mask_0, stride = key_cache_internal_tensor_assign_21_stride_0, update = key_states_207_cast_fp16, x = coreml_update_state_430)[name = string("key_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_21_cast_fp16, input = key_cache)[name = string("coreml_update_state_432_write_state")]; tensor coreml_update_state_432 = read_state(input = key_cache)[name = string("coreml_update_state_432")]; tensor value_states_123_perm_0 = const()[name = string("value_states_123_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_21_stride_0 = const()[name = string("value_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_123_cast_fp16 = transpose(perm = value_states_123_perm_0, x = var_7411_cast_fp16)[name = string("transpose_624")]; tensor value_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = value_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = value_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_21_squeeze_mask_0, stride = value_cache_internal_tensor_assign_21_stride_0, update = value_states_123_cast_fp16, x = coreml_update_state_431)[name = string("value_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_21_cast_fp16, input = value_cache)[name = string("coreml_update_state_433_write_state")]; tensor coreml_update_state_433 = read_state(input = value_cache)[name = string("coreml_update_state_433")]; tensor var_7505_begin_0 = const()[name = string("op_7505_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7505_end_0 = const()[name = string("op_7505_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7505_end_mask_0 = const()[name = string("op_7505_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7505_cast_fp16 = slice_by_index(begin = var_7505_begin_0, end = var_7505_end_0, end_mask = var_7505_end_mask_0, x = coreml_update_state_432)[name = string("op_7505_cast_fp16")]; tensor tile_40 = const()[name = string("tile_40"), val = tensor([1, 1])]; int32 var_7508_axis_0 = const()[name = string("op_7508_axis_0"), val = int32(1)]; tensor var_7508_cast_fp16_0, tensor var_7508_cast_fp16_1 = split(axis = var_7508_axis_0, split_sizes = tile_40, x = var_7505_cast_fp16)[name = string("op_7508_cast_fp16")]; tensor var_7515_begin_0 = const()[name = string("op_7515_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7515_end_0 = const()[name = string("op_7515_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7515_end_mask_0 = const()[name = string("op_7515_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7515_cast_fp16 = slice_by_index(begin = var_7515_begin_0, end = var_7515_end_0, end_mask = var_7515_end_mask_0, x = coreml_update_state_433)[name = string("op_7515_cast_fp16")]; tensor tile_41 = const()[name = string("tile_41"), val = tensor([1, 1])]; int32 var_7518_axis_0 = const()[name = string("op_7518_axis_0"), val = int32(1)]; tensor var_7518_cast_fp16_0, tensor var_7518_cast_fp16_1 = split(axis = var_7518_axis_0, split_sizes = tile_41, x = var_7515_cast_fp16)[name = string("op_7518_cast_fp16")]; tensor var_7521_split_sizes_0 = const()[name = string("op_7521_split_sizes_0"), val = tensor([8, 8])]; int32 var_7521_axis_0 = const()[name = string("op_7521_axis_0"), val = int32(1)]; tensor var_7521_0, tensor var_7521_1 = split(axis = var_7521_axis_0, split_sizes = var_7521_split_sizes_0, x = query_states_123_cast_fp16)[name = string("op_7521")]; bool attn_weights_321_transpose_x_0 = const()[name = string("attn_weights_321_transpose_x_0"), val = bool(false)]; bool attn_weights_321_transpose_y_0 = const()[name = string("attn_weights_321_transpose_y_0"), val = bool(false)]; tensor attn_weights_321_cast_fp16 = matmul(transpose_x = attn_weights_321_transpose_x_0, transpose_y = attn_weights_321_transpose_y_0, x = var_7508_cast_fp16_0, y = var_7521_0)[name = string("attn_weights_321_cast_fp16")]; fp16 var_7524_to_fp16 = const()[name = string("op_7524_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_323_cast_fp16 = mul(x = attn_weights_321_cast_fp16, y = var_7524_to_fp16)[name = string("attn_weights_323_cast_fp16")]; tensor attn_weights_325_cast_fp16 = add(x = attn_weights_323_cast_fp16, y = attn_mask_1)[name = string("attn_weights_325_cast_fp16")]; int32 var_7528 = const()[name = string("op_7528"), val = int32(-2)]; tensor attn_weights_327_cast_fp16 = softmax(axis = var_7528, x = attn_weights_325_cast_fp16)[name = string("attn_weights_327_cast_fp16")]; bool var_7534_transpose_x_1 = const()[name = string("op_7534_transpose_x_1"), val = bool(true)]; bool var_7534_transpose_y_1 = const()[name = string("op_7534_transpose_y_1"), val = bool(false)]; tensor var_7534_cast_fp16 = matmul(transpose_x = var_7534_transpose_x_1, transpose_y = var_7534_transpose_y_1, x = attn_weights_327_cast_fp16, y = var_7518_cast_fp16_0)[name = string("op_7534_cast_fp16")]; bool attn_weights_329_transpose_x_0 = const()[name = string("attn_weights_329_transpose_x_0"), val = bool(false)]; bool attn_weights_329_transpose_y_0 = const()[name = string("attn_weights_329_transpose_y_0"), val = bool(false)]; tensor attn_weights_329_cast_fp16 = matmul(transpose_x = attn_weights_329_transpose_x_0, transpose_y = attn_weights_329_transpose_y_0, x = var_7508_cast_fp16_1, y = var_7521_1)[name = string("attn_weights_329_cast_fp16")]; fp16 var_7536_to_fp16 = const()[name = string("op_7536_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_331_cast_fp16 = mul(x = attn_weights_329_cast_fp16, y = var_7536_to_fp16)[name = string("attn_weights_331_cast_fp16")]; tensor attn_weights_333_cast_fp16 = add(x = attn_weights_331_cast_fp16, y = attn_mask_1)[name = string("attn_weights_333_cast_fp16")]; int32 var_7540 = const()[name = string("op_7540"), val = int32(-2)]; tensor attn_weights_335_cast_fp16 = softmax(axis = var_7540, x = attn_weights_333_cast_fp16)[name = string("attn_weights_335_cast_fp16")]; bool attn_output_161_transpose_x_1 = const()[name = string("attn_output_161_transpose_x_1"), val = bool(true)]; bool attn_output_161_transpose_y_1 = const()[name = string("attn_output_161_transpose_y_1"), val = bool(false)]; tensor attn_output_161_cast_fp16 = matmul(transpose_x = attn_output_161_transpose_x_1, transpose_y = attn_output_161_transpose_y_1, x = attn_weights_335_cast_fp16, y = var_7518_cast_fp16_1)[name = string("attn_output_161_cast_fp16")]; int32 var_7548 = const()[name = string("op_7548"), val = int32(1)]; bool attn_output_163_interleave_0 = const()[name = string("attn_output_163_interleave_0"), val = bool(false)]; tensor attn_output_163_cast_fp16 = concat(axis = var_7548, interleave = attn_output_163_interleave_0, values = (var_7534_cast_fp16, attn_output_161_cast_fp16))[name = string("attn_output_163_cast_fp16")]; tensor var_7552_perm_0 = const()[name = string("op_7552_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_251x = const()[name = string("concat_251x"), val = tensor([1, 2048, 1, -1])]; tensor var_7552_cast_fp16 = transpose(perm = var_7552_perm_0, x = attn_output_163_cast_fp16)[name = string("transpose_623")]; tensor attn_output_167_cast_fp16 = reshape(shape = concat_251x, x = var_7552_cast_fp16)[name = string("attn_output_167_cast_fp16")]; tensor hidden_states_203_strides_0 = const()[name = string("hidden_states_203_strides_0"), val = tensor([1, 1])]; string hidden_states_203_pad_type_0 = const()[name = string("hidden_states_203_pad_type_0"), val = string("valid")]; tensor hidden_states_203_pad_0 = const()[name = string("hidden_states_203_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_203_dilations_0 = const()[name = string("hidden_states_203_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_203_groups_0 = const()[name = string("hidden_states_203_groups_0"), val = int32(1)]; tensor hidden_states_203_cast_fp16 = conv(dilations = hidden_states_203_dilations_0, groups = hidden_states_203_groups_0, pad = hidden_states_203_pad_0, pad_type = hidden_states_203_pad_type_0, strides = hidden_states_203_strides_0, weight = layers_20_self_attn_o_proj_weight_cast_fp16, x = attn_output_167_cast_fp16)[name = string("hidden_states_203_cast_fp16")]; tensor hidden_states_205_cast_fp16 = add(x = hidden_states_199_cast_fp16, y = hidden_states_203_cast_fp16)[name = string("hidden_states_205_cast_fp16")]; fp16 const_208_promoted_to_fp16 = const()[name = string("const_208_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7585_cast_fp16 = mul(x = hidden_states_205_cast_fp16, y = const_208_promoted_to_fp16)[name = string("op_7585_cast_fp16")]; int32 var_7583 = const()[name = string("op_7583"), val = int32(1)]; bool doubled_165_interleave_0 = const()[name = string("doubled_165_interleave_0"), val = bool(false)]; tensor doubled_165_cast_fp16 = concat(axis = var_7583, interleave = doubled_165_interleave_0, values = (hidden_states_205_cast_fp16, var_7585_cast_fp16))[name = string("doubled_165_cast_fp16")]; tensor out_83_axes_0 = const()[name = string("out_83_axes_0"), val = tensor([1])]; tensor out_83_gamma_0_to_fp16 = const()[name = string("out_83_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502547264)))]; fp16 var_7595_to_fp16 = const()[name = string("op_7595_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_83_cast_fp16 = layer_norm(axes = out_83_axes_0, epsilon = var_7595_to_fp16, gamma = out_83_gamma_0_to_fp16, x = doubled_165_cast_fp16)[name = string("out_83_cast_fp16")]; tensor var_7606_split_sizes_0 = const()[name = string("op_7606_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7606_axis_0 = const()[name = string("op_7606_axis_0"), val = int32(1)]; tensor var_7606_cast_fp16_0, tensor var_7606_cast_fp16_1 = split(axis = var_7606_axis_0, split_sizes = var_7606_split_sizes_0, x = out_83_cast_fp16)[name = string("op_7606_cast_fp16")]; tensor input_41_strides_0 = const()[name = string("input_41_strides_0"), val = tensor([1, 1])]; string input_41_pad_type_0 = const()[name = string("input_41_pad_type_0"), val = string("valid")]; tensor input_41_pad_0 = const()[name = string("input_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_41_dilations_0 = const()[name = string("input_41_dilations_0"), val = tensor([1, 1])]; int32 input_41_groups_0 = const()[name = string("input_41_groups_0"), val = int32(1)]; tensor input_41_cast_fp16 = conv(dilations = input_41_dilations_0, groups = input_41_groups_0, pad = input_41_pad_0, pad_type = input_41_pad_type_0, strides = input_41_strides_0, weight = layers_20_mlp_gate_proj_weight_cast_fp16, x = var_7606_cast_fp16_0)[name = string("input_41_cast_fp16")]; tensor var_7623_cast_fp16 = silu(x = input_41_cast_fp16)[name = string("op_7623_cast_fp16")]; tensor layers_20_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_20_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502555520)))]; tensor var_7629_strides_0 = const()[name = string("op_7629_strides_0"), val = tensor([1, 1])]; string var_7629_pad_type_0 = const()[name = string("op_7629_pad_type_0"), val = string("valid")]; tensor var_7629_pad_0 = const()[name = string("op_7629_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7629_dilations_0 = const()[name = string("op_7629_dilations_0"), val = tensor([1, 1])]; int32 var_7629_groups_0 = const()[name = string("op_7629_groups_0"), val = int32(1)]; tensor var_7629_cast_fp16 = conv(dilations = var_7629_dilations_0, groups = var_7629_groups_0, pad = var_7629_pad_0, pad_type = var_7629_pad_type_0, strides = var_7629_strides_0, weight = layers_20_mlp_up_proj_weight_to_fp16, x = var_7606_cast_fp16_0)[name = string("op_7629_cast_fp16")]; tensor x_209_cast_fp16 = mul(x = var_7623_cast_fp16, y = var_7629_cast_fp16)[name = string("x_209_cast_fp16")]; tensor hidden_states_207_strides_0 = const()[name = string("hidden_states_207_strides_0"), val = tensor([1, 1])]; string hidden_states_207_pad_type_0 = const()[name = string("hidden_states_207_pad_type_0"), val = string("valid")]; tensor hidden_states_207_pad_0 = const()[name = string("hidden_states_207_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_207_dilations_0 = const()[name = string("hidden_states_207_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_207_groups_0 = const()[name = string("hidden_states_207_groups_0"), val = int32(1)]; tensor hidden_states_207_cast_fp16 = conv(dilations = hidden_states_207_dilations_0, groups = hidden_states_207_groups_0, pad = hidden_states_207_pad_0, pad_type = hidden_states_207_pad_type_0, strides = hidden_states_207_strides_0, weight = layers_20_mlp_down_proj_weight_cast_fp16, x = x_209_cast_fp16)[name = string("hidden_states_207_cast_fp16")]; tensor hidden_states_209_cast_fp16 = add(x = hidden_states_205_cast_fp16, y = hidden_states_207_cast_fp16)[name = string("hidden_states_209_cast_fp16")]; fp16 const_210_promoted_to_fp16 = const()[name = string("const_210_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7647_cast_fp16 = mul(x = hidden_states_209_cast_fp16, y = const_210_promoted_to_fp16)[name = string("op_7647_cast_fp16")]; int32 var_7645 = const()[name = string("op_7645"), val = int32(1)]; bool doubled_169_interleave_0 = const()[name = string("doubled_169_interleave_0"), val = bool(false)]; tensor doubled_169_cast_fp16 = concat(axis = var_7645, interleave = doubled_169_interleave_0, values = (hidden_states_209_cast_fp16, var_7647_cast_fp16))[name = string("doubled_169_cast_fp16")]; tensor out_85_axes_0 = const()[name = string("out_85_axes_0"), val = tensor([1])]; tensor out_85_gamma_0_to_fp16 = const()[name = string("out_85_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527721408)))]; fp16 var_7657_to_fp16 = const()[name = string("op_7657_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_85_cast_fp16 = layer_norm(axes = out_85_axes_0, epsilon = var_7657_to_fp16, gamma = out_85_gamma_0_to_fp16, x = doubled_169_cast_fp16)[name = string("out_85_cast_fp16")]; tensor var_7668_split_sizes_0 = const()[name = string("op_7668_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7668_axis_0 = const()[name = string("op_7668_axis_0"), val = int32(1)]; tensor var_7668_cast_fp16_0, tensor var_7668_cast_fp16_1 = split(axis = var_7668_axis_0, split_sizes = var_7668_split_sizes_0, x = out_85_cast_fp16)[name = string("op_7668_cast_fp16")]; tensor query_states_127_strides_0 = const()[name = string("query_states_127_strides_0"), val = tensor([1, 1])]; string query_states_127_pad_type_0 = const()[name = string("query_states_127_pad_type_0"), val = string("valid")]; tensor query_states_127_pad_0 = const()[name = string("query_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_127_dilations_0 = const()[name = string("query_states_127_dilations_0"), val = tensor([1, 1])]; int32 query_states_127_groups_0 = const()[name = string("query_states_127_groups_0"), val = int32(1)]; tensor query_states_127_cast_fp16 = conv(dilations = query_states_127_dilations_0, groups = query_states_127_groups_0, pad = query_states_127_pad_0, pad_type = query_states_127_pad_type_0, strides = query_states_127_strides_0, weight = layers_21_self_attn_q_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("query_states_127_cast_fp16")]; tensor key_states_211_strides_0 = const()[name = string("key_states_211_strides_0"), val = tensor([1, 1])]; string key_states_211_pad_type_0 = const()[name = string("key_states_211_pad_type_0"), val = string("valid")]; tensor key_states_211_pad_0 = const()[name = string("key_states_211_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_211_dilations_0 = const()[name = string("key_states_211_dilations_0"), val = tensor([1, 1])]; int32 key_states_211_groups_0 = const()[name = string("key_states_211_groups_0"), val = int32(1)]; tensor key_states_211_cast_fp16 = conv(dilations = key_states_211_dilations_0, groups = key_states_211_groups_0, pad = key_states_211_pad_0, pad_type = key_states_211_pad_type_0, strides = key_states_211_strides_0, weight = layers_21_self_attn_k_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("key_states_211_cast_fp16")]; tensor layers_21_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_21_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527729664)))]; tensor value_states_127_strides_0 = const()[name = string("value_states_127_strides_0"), val = tensor([1, 1])]; string value_states_127_pad_type_0 = const()[name = string("value_states_127_pad_type_0"), val = string("valid")]; tensor value_states_127_pad_0 = const()[name = string("value_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_127_dilations_0 = const()[name = string("value_states_127_dilations_0"), val = tensor([1, 1])]; int32 value_states_127_groups_0 = const()[name = string("value_states_127_groups_0"), val = int32(1)]; tensor value_states_127_cast_fp16 = conv(dilations = value_states_127_dilations_0, groups = value_states_127_groups_0, pad = value_states_127_pad_0, pad_type = value_states_127_pad_type_0, strides = value_states_127_strides_0, weight = layers_21_self_attn_v_proj_weight_to_fp16, x = var_7668_cast_fp16_0)[name = string("value_states_127_cast_fp16")]; tensor concat_252x = const()[name = string("concat_252x"), val = tensor([1, 16, 128, -1])]; tensor x_211_cast_fp16 = reshape(shape = concat_252x, x = query_states_127_cast_fp16)[name = string("x_211_cast_fp16")]; tensor concat_253x = const()[name = string("concat_253x"), val = tensor([1, 2, 128, -1])]; tensor var_7725_cast_fp16 = reshape(shape = concat_253x, x = key_states_211_cast_fp16)[name = string("op_7725_cast_fp16")]; tensor concat_254x = const()[name = string("concat_254x"), val = tensor([1, 2, 128, -1])]; tensor var_7732_cast_fp16 = reshape(shape = concat_254x, x = value_states_127_cast_fp16)[name = string("op_7732_cast_fp16")]; tensor var_7736_cast_fp16 = mul(x = x_211_cast_fp16, y = var_869_cast_fp16)[name = string("op_7736_cast_fp16")]; tensor var_7737_split_sizes_0 = const()[name = string("op_7737_split_sizes_0"), val = tensor([64, 64])]; int32 var_7737_axis_0 = const()[name = string("op_7737_axis_0"), val = int32(-2)]; tensor var_7737_cast_fp16_0, tensor var_7737_cast_fp16_1 = split(axis = var_7737_axis_0, split_sizes = var_7737_split_sizes_0, x = x_211_cast_fp16)[name = string("op_7737_cast_fp16")]; fp16 const_212_promoted_to_fp16 = const()[name = string("const_212_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7739_cast_fp16 = mul(x = var_7737_cast_fp16_1, y = const_212_promoted_to_fp16)[name = string("op_7739_cast_fp16")]; int32 var_7741 = const()[name = string("op_7741"), val = int32(-2)]; bool var_7742_interleave_0 = const()[name = string("op_7742_interleave_0"), val = bool(false)]; tensor var_7742_cast_fp16 = concat(axis = var_7741, interleave = var_7742_interleave_0, values = (var_7739_cast_fp16, var_7737_cast_fp16_0))[name = string("op_7742_cast_fp16")]; tensor var_7743_cast_fp16 = mul(x = var_7742_cast_fp16, y = var_878_cast_fp16)[name = string("op_7743_cast_fp16")]; tensor query_states_129_cast_fp16 = add(x = var_7736_cast_fp16, y = var_7743_cast_fp16)[name = string("query_states_129_cast_fp16")]; tensor var_7749_cast_fp16 = mul(x = var_7725_cast_fp16, y = var_869_cast_fp16)[name = string("op_7749_cast_fp16")]; tensor var_7750_split_sizes_0 = const()[name = string("op_7750_split_sizes_0"), val = tensor([64, 64])]; int32 var_7750_axis_0 = const()[name = string("op_7750_axis_0"), val = int32(-2)]; tensor var_7750_cast_fp16_0, tensor var_7750_cast_fp16_1 = split(axis = var_7750_axis_0, split_sizes = var_7750_split_sizes_0, x = var_7725_cast_fp16)[name = string("op_7750_cast_fp16")]; fp16 const_213_promoted_to_fp16 = const()[name = string("const_213_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7752_cast_fp16 = mul(x = var_7750_cast_fp16_1, y = const_213_promoted_to_fp16)[name = string("op_7752_cast_fp16")]; int32 var_7754 = const()[name = string("op_7754"), val = int32(-2)]; bool var_7755_interleave_0 = const()[name = string("op_7755_interleave_0"), val = bool(false)]; tensor var_7755_cast_fp16 = concat(axis = var_7754, interleave = var_7755_interleave_0, values = (var_7752_cast_fp16, var_7750_cast_fp16_0))[name = string("op_7755_cast_fp16")]; tensor var_7756_cast_fp16 = mul(x = var_7755_cast_fp16, y = var_878_cast_fp16)[name = string("op_7756_cast_fp16")]; tensor key_states_215_cast_fp16 = add(x = var_7749_cast_fp16, y = var_7756_cast_fp16)[name = string("key_states_215_cast_fp16")]; tensor expand_dims_252 = const()[name = string("expand_dims_252"), val = tensor([21])]; tensor expand_dims_253 = const()[name = string("expand_dims_253"), val = tensor([0])]; tensor expand_dims_255 = const()[name = string("expand_dims_255"), val = tensor([0])]; int32 concat_257_axis_0 = const()[name = string("concat_257_axis_0"), val = int32(0)]; bool concat_257_interleave_0 = const()[name = string("concat_257_interleave_0"), val = bool(false)]; tensor concat_257 = concat(axis = concat_257_axis_0, interleave = concat_257_interleave_0, values = (expand_dims_252, expand_dims_253, position_id, expand_dims_255))[name = string("concat_257")]; tensor expand_dims_256 = const()[name = string("expand_dims_256"), val = tensor([22])]; tensor concat_258_values1_0 = const()[name = string("concat_258_values1_0"), val = tensor([0])]; tensor concat_258_values3_0 = const()[name = string("concat_258_values3_0"), val = tensor([0])]; int32 concat_258_axis_0 = const()[name = string("concat_258_axis_0"), val = int32(0)]; bool concat_258_interleave_0 = const()[name = string("concat_258_interleave_0"), val = bool(false)]; tensor concat_258 = concat(axis = concat_258_axis_0, interleave = concat_258_interleave_0, values = (expand_dims_256, concat_258_values1_0, cache_position_end, concat_258_values3_0))[name = string("concat_258")]; tensor key_states_217_perm_0 = const()[name = string("key_states_217_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_22_stride_0 = const()[name = string("key_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_217_cast_fp16 = transpose(perm = key_states_217_perm_0, x = key_states_215_cast_fp16)[name = string("transpose_622")]; tensor key_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = key_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = key_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_22_squeeze_mask_0, stride = key_cache_internal_tensor_assign_22_stride_0, update = key_states_217_cast_fp16, x = coreml_update_state_432)[name = string("key_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_22_cast_fp16, input = key_cache)[name = string("coreml_update_state_434_write_state")]; tensor coreml_update_state_434 = read_state(input = key_cache)[name = string("coreml_update_state_434")]; tensor value_states_129_perm_0 = const()[name = string("value_states_129_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_22_stride_0 = const()[name = string("value_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_129_cast_fp16 = transpose(perm = value_states_129_perm_0, x = var_7732_cast_fp16)[name = string("transpose_621")]; tensor value_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = value_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = value_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_22_squeeze_mask_0, stride = value_cache_internal_tensor_assign_22_stride_0, update = value_states_129_cast_fp16, x = coreml_update_state_433)[name = string("value_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_22_cast_fp16, input = value_cache)[name = string("coreml_update_state_435_write_state")]; tensor coreml_update_state_435 = read_state(input = value_cache)[name = string("coreml_update_state_435")]; tensor var_7826_begin_0 = const()[name = string("op_7826_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7826_end_0 = const()[name = string("op_7826_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7826_end_mask_0 = const()[name = string("op_7826_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7826_cast_fp16 = slice_by_index(begin = var_7826_begin_0, end = var_7826_end_0, end_mask = var_7826_end_mask_0, x = coreml_update_state_434)[name = string("op_7826_cast_fp16")]; tensor tile_42 = const()[name = string("tile_42"), val = tensor([1, 1])]; int32 var_7829_axis_0 = const()[name = string("op_7829_axis_0"), val = int32(1)]; tensor var_7829_cast_fp16_0, tensor var_7829_cast_fp16_1 = split(axis = var_7829_axis_0, split_sizes = tile_42, x = var_7826_cast_fp16)[name = string("op_7829_cast_fp16")]; tensor var_7836_begin_0 = const()[name = string("op_7836_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7836_end_0 = const()[name = string("op_7836_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7836_end_mask_0 = const()[name = string("op_7836_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7836_cast_fp16 = slice_by_index(begin = var_7836_begin_0, end = var_7836_end_0, end_mask = var_7836_end_mask_0, x = coreml_update_state_435)[name = string("op_7836_cast_fp16")]; tensor tile_43 = const()[name = string("tile_43"), val = tensor([1, 1])]; int32 var_7839_axis_0 = const()[name = string("op_7839_axis_0"), val = int32(1)]; tensor var_7839_cast_fp16_0, tensor var_7839_cast_fp16_1 = split(axis = var_7839_axis_0, split_sizes = tile_43, x = var_7836_cast_fp16)[name = string("op_7839_cast_fp16")]; tensor var_7842_split_sizes_0 = const()[name = string("op_7842_split_sizes_0"), val = tensor([8, 8])]; int32 var_7842_axis_0 = const()[name = string("op_7842_axis_0"), val = int32(1)]; tensor var_7842_0, tensor var_7842_1 = split(axis = var_7842_axis_0, split_sizes = var_7842_split_sizes_0, x = query_states_129_cast_fp16)[name = string("op_7842")]; bool attn_weights_337_transpose_x_0 = const()[name = string("attn_weights_337_transpose_x_0"), val = bool(false)]; bool attn_weights_337_transpose_y_0 = const()[name = string("attn_weights_337_transpose_y_0"), val = bool(false)]; tensor attn_weights_337_cast_fp16 = matmul(transpose_x = attn_weights_337_transpose_x_0, transpose_y = attn_weights_337_transpose_y_0, x = var_7829_cast_fp16_0, y = var_7842_0)[name = string("attn_weights_337_cast_fp16")]; fp16 var_7845_to_fp16 = const()[name = string("op_7845_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_339_cast_fp16 = mul(x = attn_weights_337_cast_fp16, y = var_7845_to_fp16)[name = string("attn_weights_339_cast_fp16")]; tensor attn_weights_341_cast_fp16 = add(x = attn_weights_339_cast_fp16, y = attn_mask_1)[name = string("attn_weights_341_cast_fp16")]; int32 var_7849 = const()[name = string("op_7849"), val = int32(-2)]; tensor attn_weights_343_cast_fp16 = softmax(axis = var_7849, x = attn_weights_341_cast_fp16)[name = string("attn_weights_343_cast_fp16")]; bool var_7855_transpose_x_1 = const()[name = string("op_7855_transpose_x_1"), val = bool(true)]; bool var_7855_transpose_y_1 = const()[name = string("op_7855_transpose_y_1"), val = bool(false)]; tensor var_7855_cast_fp16 = matmul(transpose_x = var_7855_transpose_x_1, transpose_y = var_7855_transpose_y_1, x = attn_weights_343_cast_fp16, y = var_7839_cast_fp16_0)[name = string("op_7855_cast_fp16")]; bool attn_weights_345_transpose_x_0 = const()[name = string("attn_weights_345_transpose_x_0"), val = bool(false)]; bool attn_weights_345_transpose_y_0 = const()[name = string("attn_weights_345_transpose_y_0"), val = bool(false)]; tensor attn_weights_345_cast_fp16 = matmul(transpose_x = attn_weights_345_transpose_x_0, transpose_y = attn_weights_345_transpose_y_0, x = var_7829_cast_fp16_1, y = var_7842_1)[name = string("attn_weights_345_cast_fp16")]; fp16 var_7857_to_fp16 = const()[name = string("op_7857_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_347_cast_fp16 = mul(x = attn_weights_345_cast_fp16, y = var_7857_to_fp16)[name = string("attn_weights_347_cast_fp16")]; tensor attn_weights_349_cast_fp16 = add(x = attn_weights_347_cast_fp16, y = attn_mask_1)[name = string("attn_weights_349_cast_fp16")]; int32 var_7861 = const()[name = string("op_7861"), val = int32(-2)]; tensor attn_weights_351_cast_fp16 = softmax(axis = var_7861, x = attn_weights_349_cast_fp16)[name = string("attn_weights_351_cast_fp16")]; bool attn_output_169_transpose_x_1 = const()[name = string("attn_output_169_transpose_x_1"), val = bool(true)]; bool attn_output_169_transpose_y_1 = const()[name = string("attn_output_169_transpose_y_1"), val = bool(false)]; tensor attn_output_169_cast_fp16 = matmul(transpose_x = attn_output_169_transpose_x_1, transpose_y = attn_output_169_transpose_y_1, x = attn_weights_351_cast_fp16, y = var_7839_cast_fp16_1)[name = string("attn_output_169_cast_fp16")]; int32 var_7869 = const()[name = string("op_7869"), val = int32(1)]; bool attn_output_171_interleave_0 = const()[name = string("attn_output_171_interleave_0"), val = bool(false)]; tensor attn_output_171_cast_fp16 = concat(axis = var_7869, interleave = attn_output_171_interleave_0, values = (var_7855_cast_fp16, attn_output_169_cast_fp16))[name = string("attn_output_171_cast_fp16")]; tensor var_7873_perm_0 = const()[name = string("op_7873_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_263x = const()[name = string("concat_263x"), val = tensor([1, 2048, 1, -1])]; tensor var_7873_cast_fp16 = transpose(perm = var_7873_perm_0, x = attn_output_171_cast_fp16)[name = string("transpose_620")]; tensor attn_output_175_cast_fp16 = reshape(shape = concat_263x, x = var_7873_cast_fp16)[name = string("attn_output_175_cast_fp16")]; tensor hidden_states_213_strides_0 = const()[name = string("hidden_states_213_strides_0"), val = tensor([1, 1])]; string hidden_states_213_pad_type_0 = const()[name = string("hidden_states_213_pad_type_0"), val = string("valid")]; tensor hidden_states_213_pad_0 = const()[name = string("hidden_states_213_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_213_dilations_0 = const()[name = string("hidden_states_213_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_213_groups_0 = const()[name = string("hidden_states_213_groups_0"), val = int32(1)]; tensor hidden_states_213_cast_fp16 = conv(dilations = hidden_states_213_dilations_0, groups = hidden_states_213_groups_0, pad = hidden_states_213_pad_0, pad_type = hidden_states_213_pad_type_0, strides = hidden_states_213_strides_0, weight = layers_21_self_attn_o_proj_weight_cast_fp16, x = attn_output_175_cast_fp16)[name = string("hidden_states_213_cast_fp16")]; tensor hidden_states_215_cast_fp16 = add(x = hidden_states_209_cast_fp16, y = hidden_states_213_cast_fp16)[name = string("hidden_states_215_cast_fp16")]; fp16 const_218_promoted_to_fp16 = const()[name = string("const_218_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7906_cast_fp16 = mul(x = hidden_states_215_cast_fp16, y = const_218_promoted_to_fp16)[name = string("op_7906_cast_fp16")]; int32 var_7904 = const()[name = string("op_7904"), val = int32(1)]; bool doubled_173_interleave_0 = const()[name = string("doubled_173_interleave_0"), val = bool(false)]; tensor doubled_173_cast_fp16 = concat(axis = var_7904, interleave = doubled_173_interleave_0, values = (hidden_states_215_cast_fp16, var_7906_cast_fp16))[name = string("doubled_173_cast_fp16")]; tensor out_87_axes_0 = const()[name = string("out_87_axes_0"), val = tensor([1])]; tensor out_87_gamma_0_to_fp16 = const()[name = string("out_87_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528778304)))]; fp16 var_7916_to_fp16 = const()[name = string("op_7916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_87_cast_fp16 = layer_norm(axes = out_87_axes_0, epsilon = var_7916_to_fp16, gamma = out_87_gamma_0_to_fp16, x = doubled_173_cast_fp16)[name = string("out_87_cast_fp16")]; tensor var_7927_split_sizes_0 = const()[name = string("op_7927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7927_axis_0 = const()[name = string("op_7927_axis_0"), val = int32(1)]; tensor var_7927_cast_fp16_0, tensor var_7927_cast_fp16_1 = split(axis = var_7927_axis_0, split_sizes = var_7927_split_sizes_0, x = out_87_cast_fp16)[name = string("op_7927_cast_fp16")]; tensor input_43_strides_0 = const()[name = string("input_43_strides_0"), val = tensor([1, 1])]; string input_43_pad_type_0 = const()[name = string("input_43_pad_type_0"), val = string("valid")]; tensor input_43_pad_0 = const()[name = string("input_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_43_dilations_0 = const()[name = string("input_43_dilations_0"), val = tensor([1, 1])]; int32 input_43_groups_0 = const()[name = string("input_43_groups_0"), val = int32(1)]; tensor input_43_cast_fp16 = conv(dilations = input_43_dilations_0, groups = input_43_groups_0, pad = input_43_pad_0, pad_type = input_43_pad_type_0, strides = input_43_strides_0, weight = layers_21_mlp_gate_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("input_43_cast_fp16")]; tensor var_7944_cast_fp16 = silu(x = input_43_cast_fp16)[name = string("op_7944_cast_fp16")]; tensor var_7950_strides_0 = const()[name = string("op_7950_strides_0"), val = tensor([1, 1])]; string var_7950_pad_type_0 = const()[name = string("op_7950_pad_type_0"), val = string("valid")]; tensor var_7950_pad_0 = const()[name = string("op_7950_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7950_dilations_0 = const()[name = string("op_7950_dilations_0"), val = tensor([1, 1])]; int32 var_7950_groups_0 = const()[name = string("op_7950_groups_0"), val = int32(1)]; tensor var_7950_cast_fp16 = conv(dilations = var_7950_dilations_0, groups = var_7950_groups_0, pad = var_7950_pad_0, pad_type = var_7950_pad_type_0, strides = var_7950_strides_0, weight = layers_21_mlp_up_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("op_7950_cast_fp16")]; tensor x_219_cast_fp16 = mul(x = var_7944_cast_fp16, y = var_7950_cast_fp16)[name = string("x_219_cast_fp16")]; tensor hidden_states_217_strides_0 = const()[name = string("hidden_states_217_strides_0"), val = tensor([1, 1])]; string hidden_states_217_pad_type_0 = const()[name = string("hidden_states_217_pad_type_0"), val = string("valid")]; tensor hidden_states_217_pad_0 = const()[name = string("hidden_states_217_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_217_dilations_0 = const()[name = string("hidden_states_217_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_217_groups_0 = const()[name = string("hidden_states_217_groups_0"), val = int32(1)]; tensor hidden_states_217_cast_fp16 = conv(dilations = hidden_states_217_dilations_0, groups = hidden_states_217_groups_0, pad = hidden_states_217_pad_0, pad_type = hidden_states_217_pad_type_0, strides = hidden_states_217_strides_0, weight = layers_21_mlp_down_proj_weight_cast_fp16, x = x_219_cast_fp16)[name = string("hidden_states_217_cast_fp16")]; tensor hidden_states_219_cast_fp16 = add(x = hidden_states_215_cast_fp16, y = hidden_states_217_cast_fp16)[name = string("hidden_states_219_cast_fp16")]; fp16 const_220_promoted_to_fp16 = const()[name = string("const_220_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7968_cast_fp16 = mul(x = hidden_states_219_cast_fp16, y = const_220_promoted_to_fp16)[name = string("op_7968_cast_fp16")]; int32 var_7966 = const()[name = string("op_7966"), val = int32(1)]; bool doubled_177_interleave_0 = const()[name = string("doubled_177_interleave_0"), val = bool(false)]; tensor doubled_177_cast_fp16 = concat(axis = var_7966, interleave = doubled_177_interleave_0, values = (hidden_states_219_cast_fp16, var_7968_cast_fp16))[name = string("doubled_177_cast_fp16")]; tensor out_89_axes_0 = const()[name = string("out_89_axes_0"), val = tensor([1])]; tensor out_89_gamma_0_to_fp16 = const()[name = string("out_89_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528786560)))]; fp16 var_7978_to_fp16 = const()[name = string("op_7978_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_89_cast_fp16 = layer_norm(axes = out_89_axes_0, epsilon = var_7978_to_fp16, gamma = out_89_gamma_0_to_fp16, x = doubled_177_cast_fp16)[name = string("out_89_cast_fp16")]; tensor var_7989_split_sizes_0 = const()[name = string("op_7989_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7989_axis_0 = const()[name = string("op_7989_axis_0"), val = int32(1)]; tensor var_7989_cast_fp16_0, tensor var_7989_cast_fp16_1 = split(axis = var_7989_axis_0, split_sizes = var_7989_split_sizes_0, x = out_89_cast_fp16)[name = string("op_7989_cast_fp16")]; tensor query_states_133_strides_0 = const()[name = string("query_states_133_strides_0"), val = tensor([1, 1])]; string query_states_133_pad_type_0 = const()[name = string("query_states_133_pad_type_0"), val = string("valid")]; tensor query_states_133_pad_0 = const()[name = string("query_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_133_dilations_0 = const()[name = string("query_states_133_dilations_0"), val = tensor([1, 1])]; int32 query_states_133_groups_0 = const()[name = string("query_states_133_groups_0"), val = int32(1)]; tensor query_states_133_cast_fp16 = conv(dilations = query_states_133_dilations_0, groups = query_states_133_groups_0, pad = query_states_133_pad_0, pad_type = query_states_133_pad_type_0, strides = query_states_133_strides_0, weight = layers_22_self_attn_q_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("query_states_133_cast_fp16")]; tensor key_states_221_strides_0 = const()[name = string("key_states_221_strides_0"), val = tensor([1, 1])]; string key_states_221_pad_type_0 = const()[name = string("key_states_221_pad_type_0"), val = string("valid")]; tensor key_states_221_pad_0 = const()[name = string("key_states_221_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_221_dilations_0 = const()[name = string("key_states_221_dilations_0"), val = tensor([1, 1])]; int32 key_states_221_groups_0 = const()[name = string("key_states_221_groups_0"), val = int32(1)]; tensor key_states_221_cast_fp16 = conv(dilations = key_states_221_dilations_0, groups = key_states_221_groups_0, pad = key_states_221_pad_0, pad_type = key_states_221_pad_type_0, strides = key_states_221_strides_0, weight = layers_22_self_attn_k_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("key_states_221_cast_fp16")]; tensor layers_22_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528794816)))]; tensor value_states_133_strides_0 = const()[name = string("value_states_133_strides_0"), val = tensor([1, 1])]; string value_states_133_pad_type_0 = const()[name = string("value_states_133_pad_type_0"), val = string("valid")]; tensor value_states_133_pad_0 = const()[name = string("value_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_133_dilations_0 = const()[name = string("value_states_133_dilations_0"), val = tensor([1, 1])]; int32 value_states_133_groups_0 = const()[name = string("value_states_133_groups_0"), val = int32(1)]; tensor value_states_133_cast_fp16 = conv(dilations = value_states_133_dilations_0, groups = value_states_133_groups_0, pad = value_states_133_pad_0, pad_type = value_states_133_pad_type_0, strides = value_states_133_strides_0, weight = layers_22_self_attn_v_proj_weight_to_fp16, x = var_7989_cast_fp16_0)[name = string("value_states_133_cast_fp16")]; tensor concat_264x = const()[name = string("concat_264x"), val = tensor([1, 16, 128, -1])]; tensor x_221_cast_fp16 = reshape(shape = concat_264x, x = query_states_133_cast_fp16)[name = string("x_221_cast_fp16")]; tensor concat_265x = const()[name = string("concat_265x"), val = tensor([1, 2, 128, -1])]; tensor var_8046_cast_fp16 = reshape(shape = concat_265x, x = key_states_221_cast_fp16)[name = string("op_8046_cast_fp16")]; tensor concat_266x = const()[name = string("concat_266x"), val = tensor([1, 2, 128, -1])]; tensor var_8053_cast_fp16 = reshape(shape = concat_266x, x = value_states_133_cast_fp16)[name = string("op_8053_cast_fp16")]; tensor var_8057_cast_fp16 = mul(x = x_221_cast_fp16, y = var_869_cast_fp16)[name = string("op_8057_cast_fp16")]; tensor var_8058_split_sizes_0 = const()[name = string("op_8058_split_sizes_0"), val = tensor([64, 64])]; int32 var_8058_axis_0 = const()[name = string("op_8058_axis_0"), val = int32(-2)]; tensor var_8058_cast_fp16_0, tensor var_8058_cast_fp16_1 = split(axis = var_8058_axis_0, split_sizes = var_8058_split_sizes_0, x = x_221_cast_fp16)[name = string("op_8058_cast_fp16")]; fp16 const_222_promoted_to_fp16 = const()[name = string("const_222_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8060_cast_fp16 = mul(x = var_8058_cast_fp16_1, y = const_222_promoted_to_fp16)[name = string("op_8060_cast_fp16")]; int32 var_8062 = const()[name = string("op_8062"), val = int32(-2)]; bool var_8063_interleave_0 = const()[name = string("op_8063_interleave_0"), val = bool(false)]; tensor var_8063_cast_fp16 = concat(axis = var_8062, interleave = var_8063_interleave_0, values = (var_8060_cast_fp16, var_8058_cast_fp16_0))[name = string("op_8063_cast_fp16")]; tensor var_8064_cast_fp16 = mul(x = var_8063_cast_fp16, y = var_878_cast_fp16)[name = string("op_8064_cast_fp16")]; tensor query_states_135_cast_fp16 = add(x = var_8057_cast_fp16, y = var_8064_cast_fp16)[name = string("query_states_135_cast_fp16")]; tensor var_8070_cast_fp16 = mul(x = var_8046_cast_fp16, y = var_869_cast_fp16)[name = string("op_8070_cast_fp16")]; tensor var_8071_split_sizes_0 = const()[name = string("op_8071_split_sizes_0"), val = tensor([64, 64])]; int32 var_8071_axis_0 = const()[name = string("op_8071_axis_0"), val = int32(-2)]; tensor var_8071_cast_fp16_0, tensor var_8071_cast_fp16_1 = split(axis = var_8071_axis_0, split_sizes = var_8071_split_sizes_0, x = var_8046_cast_fp16)[name = string("op_8071_cast_fp16")]; fp16 const_223_promoted_to_fp16 = const()[name = string("const_223_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8073_cast_fp16 = mul(x = var_8071_cast_fp16_1, y = const_223_promoted_to_fp16)[name = string("op_8073_cast_fp16")]; int32 var_8075 = const()[name = string("op_8075"), val = int32(-2)]; bool var_8076_interleave_0 = const()[name = string("op_8076_interleave_0"), val = bool(false)]; tensor var_8076_cast_fp16 = concat(axis = var_8075, interleave = var_8076_interleave_0, values = (var_8073_cast_fp16, var_8071_cast_fp16_0))[name = string("op_8076_cast_fp16")]; tensor var_8077_cast_fp16 = mul(x = var_8076_cast_fp16, y = var_878_cast_fp16)[name = string("op_8077_cast_fp16")]; tensor key_states_225_cast_fp16 = add(x = var_8070_cast_fp16, y = var_8077_cast_fp16)[name = string("key_states_225_cast_fp16")]; tensor expand_dims_264 = const()[name = string("expand_dims_264"), val = tensor([22])]; tensor expand_dims_265 = const()[name = string("expand_dims_265"), val = tensor([0])]; tensor expand_dims_267 = const()[name = string("expand_dims_267"), val = tensor([0])]; int32 concat_269_axis_0 = const()[name = string("concat_269_axis_0"), val = int32(0)]; bool concat_269_interleave_0 = const()[name = string("concat_269_interleave_0"), val = bool(false)]; tensor concat_269 = concat(axis = concat_269_axis_0, interleave = concat_269_interleave_0, values = (expand_dims_264, expand_dims_265, position_id, expand_dims_267))[name = string("concat_269")]; tensor expand_dims_268 = const()[name = string("expand_dims_268"), val = tensor([23])]; tensor concat_270_values1_0 = const()[name = string("concat_270_values1_0"), val = tensor([0])]; tensor concat_270_values3_0 = const()[name = string("concat_270_values3_0"), val = tensor([0])]; int32 concat_270_axis_0 = const()[name = string("concat_270_axis_0"), val = int32(0)]; bool concat_270_interleave_0 = const()[name = string("concat_270_interleave_0"), val = bool(false)]; tensor concat_270 = concat(axis = concat_270_axis_0, interleave = concat_270_interleave_0, values = (expand_dims_268, concat_270_values1_0, cache_position_end, concat_270_values3_0))[name = string("concat_270")]; tensor key_states_227_perm_0 = const()[name = string("key_states_227_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_23_stride_0 = const()[name = string("key_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_227_cast_fp16 = transpose(perm = key_states_227_perm_0, x = key_states_225_cast_fp16)[name = string("transpose_619")]; tensor key_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = key_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = key_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_23_squeeze_mask_0, stride = key_cache_internal_tensor_assign_23_stride_0, update = key_states_227_cast_fp16, x = coreml_update_state_434)[name = string("key_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_23_cast_fp16, input = key_cache)[name = string("coreml_update_state_436_write_state")]; tensor coreml_update_state_436 = read_state(input = key_cache)[name = string("coreml_update_state_436")]; tensor value_states_135_perm_0 = const()[name = string("value_states_135_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_23_stride_0 = const()[name = string("value_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_135_cast_fp16 = transpose(perm = value_states_135_perm_0, x = var_8053_cast_fp16)[name = string("transpose_618")]; tensor value_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = value_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = value_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_23_squeeze_mask_0, stride = value_cache_internal_tensor_assign_23_stride_0, update = value_states_135_cast_fp16, x = coreml_update_state_435)[name = string("value_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_23_cast_fp16, input = value_cache)[name = string("coreml_update_state_437_write_state")]; tensor coreml_update_state_437 = read_state(input = value_cache)[name = string("coreml_update_state_437")]; tensor var_8147_begin_0 = const()[name = string("op_8147_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8147_end_0 = const()[name = string("op_8147_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8147_end_mask_0 = const()[name = string("op_8147_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8147_cast_fp16 = slice_by_index(begin = var_8147_begin_0, end = var_8147_end_0, end_mask = var_8147_end_mask_0, x = coreml_update_state_436)[name = string("op_8147_cast_fp16")]; tensor tile_44 = const()[name = string("tile_44"), val = tensor([1, 1])]; int32 var_8150_axis_0 = const()[name = string("op_8150_axis_0"), val = int32(1)]; tensor var_8150_cast_fp16_0, tensor var_8150_cast_fp16_1 = split(axis = var_8150_axis_0, split_sizes = tile_44, x = var_8147_cast_fp16)[name = string("op_8150_cast_fp16")]; tensor var_8157_begin_0 = const()[name = string("op_8157_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8157_end_0 = const()[name = string("op_8157_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8157_end_mask_0 = const()[name = string("op_8157_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8157_cast_fp16 = slice_by_index(begin = var_8157_begin_0, end = var_8157_end_0, end_mask = var_8157_end_mask_0, x = coreml_update_state_437)[name = string("op_8157_cast_fp16")]; tensor tile_45 = const()[name = string("tile_45"), val = tensor([1, 1])]; int32 var_8160_axis_0 = const()[name = string("op_8160_axis_0"), val = int32(1)]; tensor var_8160_cast_fp16_0, tensor var_8160_cast_fp16_1 = split(axis = var_8160_axis_0, split_sizes = tile_45, x = var_8157_cast_fp16)[name = string("op_8160_cast_fp16")]; tensor var_8163_split_sizes_0 = const()[name = string("op_8163_split_sizes_0"), val = tensor([8, 8])]; int32 var_8163_axis_0 = const()[name = string("op_8163_axis_0"), val = int32(1)]; tensor var_8163_0, tensor var_8163_1 = split(axis = var_8163_axis_0, split_sizes = var_8163_split_sizes_0, x = query_states_135_cast_fp16)[name = string("op_8163")]; bool attn_weights_353_transpose_x_0 = const()[name = string("attn_weights_353_transpose_x_0"), val = bool(false)]; bool attn_weights_353_transpose_y_0 = const()[name = string("attn_weights_353_transpose_y_0"), val = bool(false)]; tensor attn_weights_353_cast_fp16 = matmul(transpose_x = attn_weights_353_transpose_x_0, transpose_y = attn_weights_353_transpose_y_0, x = var_8150_cast_fp16_0, y = var_8163_0)[name = string("attn_weights_353_cast_fp16")]; fp16 var_8166_to_fp16 = const()[name = string("op_8166_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_355_cast_fp16 = mul(x = attn_weights_353_cast_fp16, y = var_8166_to_fp16)[name = string("attn_weights_355_cast_fp16")]; tensor attn_weights_357_cast_fp16 = add(x = attn_weights_355_cast_fp16, y = attn_mask_1)[name = string("attn_weights_357_cast_fp16")]; int32 var_8170 = const()[name = string("op_8170"), val = int32(-2)]; tensor attn_weights_359_cast_fp16 = softmax(axis = var_8170, x = attn_weights_357_cast_fp16)[name = string("attn_weights_359_cast_fp16")]; bool var_8176_transpose_x_1 = const()[name = string("op_8176_transpose_x_1"), val = bool(true)]; bool var_8176_transpose_y_1 = const()[name = string("op_8176_transpose_y_1"), val = bool(false)]; tensor var_8176_cast_fp16 = matmul(transpose_x = var_8176_transpose_x_1, transpose_y = var_8176_transpose_y_1, x = attn_weights_359_cast_fp16, y = var_8160_cast_fp16_0)[name = string("op_8176_cast_fp16")]; bool attn_weights_361_transpose_x_0 = const()[name = string("attn_weights_361_transpose_x_0"), val = bool(false)]; bool attn_weights_361_transpose_y_0 = const()[name = string("attn_weights_361_transpose_y_0"), val = bool(false)]; tensor attn_weights_361_cast_fp16 = matmul(transpose_x = attn_weights_361_transpose_x_0, transpose_y = attn_weights_361_transpose_y_0, x = var_8150_cast_fp16_1, y = var_8163_1)[name = string("attn_weights_361_cast_fp16")]; fp16 var_8178_to_fp16 = const()[name = string("op_8178_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_363_cast_fp16 = mul(x = attn_weights_361_cast_fp16, y = var_8178_to_fp16)[name = string("attn_weights_363_cast_fp16")]; tensor attn_weights_365_cast_fp16 = add(x = attn_weights_363_cast_fp16, y = attn_mask_1)[name = string("attn_weights_365_cast_fp16")]; int32 var_8182 = const()[name = string("op_8182"), val = int32(-2)]; tensor attn_weights_367_cast_fp16 = softmax(axis = var_8182, x = attn_weights_365_cast_fp16)[name = string("attn_weights_367_cast_fp16")]; bool attn_output_177_transpose_x_1 = const()[name = string("attn_output_177_transpose_x_1"), val = bool(true)]; bool attn_output_177_transpose_y_1 = const()[name = string("attn_output_177_transpose_y_1"), val = bool(false)]; tensor attn_output_177_cast_fp16 = matmul(transpose_x = attn_output_177_transpose_x_1, transpose_y = attn_output_177_transpose_y_1, x = attn_weights_367_cast_fp16, y = var_8160_cast_fp16_1)[name = string("attn_output_177_cast_fp16")]; int32 var_8190 = const()[name = string("op_8190"), val = int32(1)]; bool attn_output_179_interleave_0 = const()[name = string("attn_output_179_interleave_0"), val = bool(false)]; tensor attn_output_179_cast_fp16 = concat(axis = var_8190, interleave = attn_output_179_interleave_0, values = (var_8176_cast_fp16, attn_output_177_cast_fp16))[name = string("attn_output_179_cast_fp16")]; tensor var_8194_perm_0 = const()[name = string("op_8194_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_275x = const()[name = string("concat_275x"), val = tensor([1, 2048, 1, -1])]; tensor var_8194_cast_fp16 = transpose(perm = var_8194_perm_0, x = attn_output_179_cast_fp16)[name = string("transpose_617")]; tensor attn_output_183_cast_fp16 = reshape(shape = concat_275x, x = var_8194_cast_fp16)[name = string("attn_output_183_cast_fp16")]; tensor layers_22_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1529843456)))]; tensor hidden_states_223_strides_0 = const()[name = string("hidden_states_223_strides_0"), val = tensor([1, 1])]; string hidden_states_223_pad_type_0 = const()[name = string("hidden_states_223_pad_type_0"), val = string("valid")]; tensor hidden_states_223_pad_0 = const()[name = string("hidden_states_223_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_223_dilations_0 = const()[name = string("hidden_states_223_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_223_groups_0 = const()[name = string("hidden_states_223_groups_0"), val = int32(1)]; tensor hidden_states_223_cast_fp16 = conv(dilations = hidden_states_223_dilations_0, groups = hidden_states_223_groups_0, pad = hidden_states_223_pad_0, pad_type = hidden_states_223_pad_type_0, strides = hidden_states_223_strides_0, weight = layers_22_self_attn_o_proj_weight_to_fp16, x = attn_output_183_cast_fp16)[name = string("hidden_states_223_cast_fp16")]; tensor hidden_states_225_cast_fp16 = add(x = hidden_states_219_cast_fp16, y = hidden_states_223_cast_fp16)[name = string("hidden_states_225_cast_fp16")]; fp16 const_228_promoted_to_fp16 = const()[name = string("const_228_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8227_cast_fp16 = mul(x = hidden_states_225_cast_fp16, y = const_228_promoted_to_fp16)[name = string("op_8227_cast_fp16")]; int32 var_8225 = const()[name = string("op_8225"), val = int32(1)]; bool doubled_181_interleave_0 = const()[name = string("doubled_181_interleave_0"), val = bool(false)]; tensor doubled_181_cast_fp16 = concat(axis = var_8225, interleave = doubled_181_interleave_0, values = (hidden_states_225_cast_fp16, var_8227_cast_fp16))[name = string("doubled_181_cast_fp16")]; tensor out_91_axes_0 = const()[name = string("out_91_axes_0"), val = tensor([1])]; tensor out_91_gamma_0_to_fp16 = const()[name = string("out_91_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538232128)))]; fp16 var_8237_to_fp16 = const()[name = string("op_8237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_91_cast_fp16 = layer_norm(axes = out_91_axes_0, epsilon = var_8237_to_fp16, gamma = out_91_gamma_0_to_fp16, x = doubled_181_cast_fp16)[name = string("out_91_cast_fp16")]; tensor var_8248_split_sizes_0 = const()[name = string("op_8248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8248_axis_0 = const()[name = string("op_8248_axis_0"), val = int32(1)]; tensor var_8248_cast_fp16_0, tensor var_8248_cast_fp16_1 = split(axis = var_8248_axis_0, split_sizes = var_8248_split_sizes_0, x = out_91_cast_fp16)[name = string("op_8248_cast_fp16")]; tensor input_45_strides_0 = const()[name = string("input_45_strides_0"), val = tensor([1, 1])]; string input_45_pad_type_0 = const()[name = string("input_45_pad_type_0"), val = string("valid")]; tensor input_45_pad_0 = const()[name = string("input_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_45_dilations_0 = const()[name = string("input_45_dilations_0"), val = tensor([1, 1])]; int32 input_45_groups_0 = const()[name = string("input_45_groups_0"), val = int32(1)]; tensor input_45_cast_fp16 = conv(dilations = input_45_dilations_0, groups = input_45_groups_0, pad = input_45_pad_0, pad_type = input_45_pad_type_0, strides = input_45_strides_0, weight = layers_22_mlp_gate_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("input_45_cast_fp16")]; tensor var_8265_cast_fp16 = silu(x = input_45_cast_fp16)[name = string("op_8265_cast_fp16")]; tensor var_8271_strides_0 = const()[name = string("op_8271_strides_0"), val = tensor([1, 1])]; string var_8271_pad_type_0 = const()[name = string("op_8271_pad_type_0"), val = string("valid")]; tensor var_8271_pad_0 = const()[name = string("op_8271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8271_dilations_0 = const()[name = string("op_8271_dilations_0"), val = tensor([1, 1])]; int32 var_8271_groups_0 = const()[name = string("op_8271_groups_0"), val = int32(1)]; tensor var_8271_cast_fp16 = conv(dilations = var_8271_dilations_0, groups = var_8271_groups_0, pad = var_8271_pad_0, pad_type = var_8271_pad_type_0, strides = var_8271_strides_0, weight = layers_22_mlp_up_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("op_8271_cast_fp16")]; tensor x_229_cast_fp16 = mul(x = var_8265_cast_fp16, y = var_8271_cast_fp16)[name = string("x_229_cast_fp16")]; tensor hidden_states_227_strides_0 = const()[name = string("hidden_states_227_strides_0"), val = tensor([1, 1])]; string hidden_states_227_pad_type_0 = const()[name = string("hidden_states_227_pad_type_0"), val = string("valid")]; tensor hidden_states_227_pad_0 = const()[name = string("hidden_states_227_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_227_dilations_0 = const()[name = string("hidden_states_227_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_227_groups_0 = const()[name = string("hidden_states_227_groups_0"), val = int32(1)]; tensor hidden_states_227_cast_fp16 = conv(dilations = hidden_states_227_dilations_0, groups = hidden_states_227_groups_0, pad = hidden_states_227_pad_0, pad_type = hidden_states_227_pad_type_0, strides = hidden_states_227_strides_0, weight = layers_22_mlp_down_proj_weight_cast_fp16, x = x_229_cast_fp16)[name = string("hidden_states_227_cast_fp16")]; tensor hidden_states_229_cast_fp16 = add(x = hidden_states_225_cast_fp16, y = hidden_states_227_cast_fp16)[name = string("hidden_states_229_cast_fp16")]; fp16 const_230_promoted_to_fp16 = const()[name = string("const_230_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8289_cast_fp16 = mul(x = hidden_states_229_cast_fp16, y = const_230_promoted_to_fp16)[name = string("op_8289_cast_fp16")]; int32 var_8287 = const()[name = string("op_8287"), val = int32(1)]; bool doubled_185_interleave_0 = const()[name = string("doubled_185_interleave_0"), val = bool(false)]; tensor doubled_185_cast_fp16 = concat(axis = var_8287, interleave = doubled_185_interleave_0, values = (hidden_states_229_cast_fp16, var_8289_cast_fp16))[name = string("doubled_185_cast_fp16")]; tensor out_93_axes_0 = const()[name = string("out_93_axes_0"), val = tensor([1])]; tensor out_93_gamma_0_to_fp16 = const()[name = string("out_93_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538240384)))]; fp16 var_8299_to_fp16 = const()[name = string("op_8299_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_93_cast_fp16 = layer_norm(axes = out_93_axes_0, epsilon = var_8299_to_fp16, gamma = out_93_gamma_0_to_fp16, x = doubled_185_cast_fp16)[name = string("out_93_cast_fp16")]; tensor var_8310_split_sizes_0 = const()[name = string("op_8310_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8310_axis_0 = const()[name = string("op_8310_axis_0"), val = int32(1)]; tensor var_8310_cast_fp16_0, tensor var_8310_cast_fp16_1 = split(axis = var_8310_axis_0, split_sizes = var_8310_split_sizes_0, x = out_93_cast_fp16)[name = string("op_8310_cast_fp16")]; tensor query_states_139_strides_0 = const()[name = string("query_states_139_strides_0"), val = tensor([1, 1])]; string query_states_139_pad_type_0 = const()[name = string("query_states_139_pad_type_0"), val = string("valid")]; tensor query_states_139_pad_0 = const()[name = string("query_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_139_dilations_0 = const()[name = string("query_states_139_dilations_0"), val = tensor([1, 1])]; int32 query_states_139_groups_0 = const()[name = string("query_states_139_groups_0"), val = int32(1)]; tensor query_states_139_cast_fp16 = conv(dilations = query_states_139_dilations_0, groups = query_states_139_groups_0, pad = query_states_139_pad_0, pad_type = query_states_139_pad_type_0, strides = query_states_139_strides_0, weight = layers_23_self_attn_q_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("query_states_139_cast_fp16")]; tensor key_states_231_strides_0 = const()[name = string("key_states_231_strides_0"), val = tensor([1, 1])]; string key_states_231_pad_type_0 = const()[name = string("key_states_231_pad_type_0"), val = string("valid")]; tensor key_states_231_pad_0 = const()[name = string("key_states_231_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_231_dilations_0 = const()[name = string("key_states_231_dilations_0"), val = tensor([1, 1])]; int32 key_states_231_groups_0 = const()[name = string("key_states_231_groups_0"), val = int32(1)]; tensor key_states_231_cast_fp16 = conv(dilations = key_states_231_dilations_0, groups = key_states_231_groups_0, pad = key_states_231_pad_0, pad_type = key_states_231_pad_type_0, strides = key_states_231_strides_0, weight = layers_23_self_attn_k_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("key_states_231_cast_fp16")]; tensor layers_23_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_23_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538248640)))]; tensor value_states_139_strides_0 = const()[name = string("value_states_139_strides_0"), val = tensor([1, 1])]; string value_states_139_pad_type_0 = const()[name = string("value_states_139_pad_type_0"), val = string("valid")]; tensor value_states_139_pad_0 = const()[name = string("value_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_139_dilations_0 = const()[name = string("value_states_139_dilations_0"), val = tensor([1, 1])]; int32 value_states_139_groups_0 = const()[name = string("value_states_139_groups_0"), val = int32(1)]; tensor value_states_139_cast_fp16 = conv(dilations = value_states_139_dilations_0, groups = value_states_139_groups_0, pad = value_states_139_pad_0, pad_type = value_states_139_pad_type_0, strides = value_states_139_strides_0, weight = layers_23_self_attn_v_proj_weight_to_fp16, x = var_8310_cast_fp16_0)[name = string("value_states_139_cast_fp16")]; tensor concat_276x = const()[name = string("concat_276x"), val = tensor([1, 16, 128, -1])]; tensor x_231_cast_fp16 = reshape(shape = concat_276x, x = query_states_139_cast_fp16)[name = string("x_231_cast_fp16")]; tensor concat_277x = const()[name = string("concat_277x"), val = tensor([1, 2, 128, -1])]; tensor var_8367_cast_fp16 = reshape(shape = concat_277x, x = key_states_231_cast_fp16)[name = string("op_8367_cast_fp16")]; tensor concat_278x = const()[name = string("concat_278x"), val = tensor([1, 2, 128, -1])]; tensor var_8374_cast_fp16 = reshape(shape = concat_278x, x = value_states_139_cast_fp16)[name = string("op_8374_cast_fp16")]; tensor var_8378_cast_fp16 = mul(x = x_231_cast_fp16, y = var_869_cast_fp16)[name = string("op_8378_cast_fp16")]; tensor var_8379_split_sizes_0 = const()[name = string("op_8379_split_sizes_0"), val = tensor([64, 64])]; int32 var_8379_axis_0 = const()[name = string("op_8379_axis_0"), val = int32(-2)]; tensor var_8379_cast_fp16_0, tensor var_8379_cast_fp16_1 = split(axis = var_8379_axis_0, split_sizes = var_8379_split_sizes_0, x = x_231_cast_fp16)[name = string("op_8379_cast_fp16")]; fp16 const_232_promoted_to_fp16 = const()[name = string("const_232_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8381_cast_fp16 = mul(x = var_8379_cast_fp16_1, y = const_232_promoted_to_fp16)[name = string("op_8381_cast_fp16")]; int32 var_8383 = const()[name = string("op_8383"), val = int32(-2)]; bool var_8384_interleave_0 = const()[name = string("op_8384_interleave_0"), val = bool(false)]; tensor var_8384_cast_fp16 = concat(axis = var_8383, interleave = var_8384_interleave_0, values = (var_8381_cast_fp16, var_8379_cast_fp16_0))[name = string("op_8384_cast_fp16")]; tensor var_8385_cast_fp16 = mul(x = var_8384_cast_fp16, y = var_878_cast_fp16)[name = string("op_8385_cast_fp16")]; tensor query_states_141_cast_fp16 = add(x = var_8378_cast_fp16, y = var_8385_cast_fp16)[name = string("query_states_141_cast_fp16")]; tensor var_8391_cast_fp16 = mul(x = var_8367_cast_fp16, y = var_869_cast_fp16)[name = string("op_8391_cast_fp16")]; tensor var_8392_split_sizes_0 = const()[name = string("op_8392_split_sizes_0"), val = tensor([64, 64])]; int32 var_8392_axis_0 = const()[name = string("op_8392_axis_0"), val = int32(-2)]; tensor var_8392_cast_fp16_0, tensor var_8392_cast_fp16_1 = split(axis = var_8392_axis_0, split_sizes = var_8392_split_sizes_0, x = var_8367_cast_fp16)[name = string("op_8392_cast_fp16")]; fp16 const_233_promoted_to_fp16 = const()[name = string("const_233_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8394_cast_fp16 = mul(x = var_8392_cast_fp16_1, y = const_233_promoted_to_fp16)[name = string("op_8394_cast_fp16")]; int32 var_8396 = const()[name = string("op_8396"), val = int32(-2)]; bool var_8397_interleave_0 = const()[name = string("op_8397_interleave_0"), val = bool(false)]; tensor var_8397_cast_fp16 = concat(axis = var_8396, interleave = var_8397_interleave_0, values = (var_8394_cast_fp16, var_8392_cast_fp16_0))[name = string("op_8397_cast_fp16")]; tensor var_8398_cast_fp16 = mul(x = var_8397_cast_fp16, y = var_878_cast_fp16)[name = string("op_8398_cast_fp16")]; tensor key_states_235_cast_fp16 = add(x = var_8391_cast_fp16, y = var_8398_cast_fp16)[name = string("key_states_235_cast_fp16")]; tensor expand_dims_276 = const()[name = string("expand_dims_276"), val = tensor([23])]; tensor expand_dims_277 = const()[name = string("expand_dims_277"), val = tensor([0])]; tensor expand_dims_279 = const()[name = string("expand_dims_279"), val = tensor([0])]; int32 concat_281_axis_0 = const()[name = string("concat_281_axis_0"), val = int32(0)]; bool concat_281_interleave_0 = const()[name = string("concat_281_interleave_0"), val = bool(false)]; tensor concat_281 = concat(axis = concat_281_axis_0, interleave = concat_281_interleave_0, values = (expand_dims_276, expand_dims_277, position_id, expand_dims_279))[name = string("concat_281")]; tensor expand_dims_280 = const()[name = string("expand_dims_280"), val = tensor([24])]; tensor concat_282_values1_0 = const()[name = string("concat_282_values1_0"), val = tensor([0])]; tensor concat_282_values3_0 = const()[name = string("concat_282_values3_0"), val = tensor([0])]; int32 concat_282_axis_0 = const()[name = string("concat_282_axis_0"), val = int32(0)]; bool concat_282_interleave_0 = const()[name = string("concat_282_interleave_0"), val = bool(false)]; tensor concat_282 = concat(axis = concat_282_axis_0, interleave = concat_282_interleave_0, values = (expand_dims_280, concat_282_values1_0, cache_position_end, concat_282_values3_0))[name = string("concat_282")]; tensor key_states_237_perm_0 = const()[name = string("key_states_237_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_24_stride_0 = const()[name = string("key_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_237_cast_fp16 = transpose(perm = key_states_237_perm_0, x = key_states_235_cast_fp16)[name = string("transpose_616")]; tensor key_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = key_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = key_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_24_squeeze_mask_0, stride = key_cache_internal_tensor_assign_24_stride_0, update = key_states_237_cast_fp16, x = coreml_update_state_436)[name = string("key_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_24_cast_fp16, input = key_cache)[name = string("coreml_update_state_438_write_state")]; tensor coreml_update_state_438 = read_state(input = key_cache)[name = string("coreml_update_state_438")]; tensor value_states_141_perm_0 = const()[name = string("value_states_141_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_24_stride_0 = const()[name = string("value_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_141_cast_fp16 = transpose(perm = value_states_141_perm_0, x = var_8374_cast_fp16)[name = string("transpose_615")]; tensor value_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = value_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = value_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_24_squeeze_mask_0, stride = value_cache_internal_tensor_assign_24_stride_0, update = value_states_141_cast_fp16, x = coreml_update_state_437)[name = string("value_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_24_cast_fp16, input = value_cache)[name = string("coreml_update_state_439_write_state")]; tensor coreml_update_state_439 = read_state(input = value_cache)[name = string("coreml_update_state_439")]; tensor var_8468_begin_0 = const()[name = string("op_8468_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8468_end_0 = const()[name = string("op_8468_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8468_end_mask_0 = const()[name = string("op_8468_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8468_cast_fp16 = slice_by_index(begin = var_8468_begin_0, end = var_8468_end_0, end_mask = var_8468_end_mask_0, x = coreml_update_state_438)[name = string("op_8468_cast_fp16")]; tensor tile_46 = const()[name = string("tile_46"), val = tensor([1, 1])]; int32 var_8471_axis_0 = const()[name = string("op_8471_axis_0"), val = int32(1)]; tensor var_8471_cast_fp16_0, tensor var_8471_cast_fp16_1 = split(axis = var_8471_axis_0, split_sizes = tile_46, x = var_8468_cast_fp16)[name = string("op_8471_cast_fp16")]; tensor var_8478_begin_0 = const()[name = string("op_8478_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8478_end_0 = const()[name = string("op_8478_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8478_end_mask_0 = const()[name = string("op_8478_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8478_cast_fp16 = slice_by_index(begin = var_8478_begin_0, end = var_8478_end_0, end_mask = var_8478_end_mask_0, x = coreml_update_state_439)[name = string("op_8478_cast_fp16")]; tensor tile_47 = const()[name = string("tile_47"), val = tensor([1, 1])]; int32 var_8481_axis_0 = const()[name = string("op_8481_axis_0"), val = int32(1)]; tensor var_8481_cast_fp16_0, tensor var_8481_cast_fp16_1 = split(axis = var_8481_axis_0, split_sizes = tile_47, x = var_8478_cast_fp16)[name = string("op_8481_cast_fp16")]; tensor var_8484_split_sizes_0 = const()[name = string("op_8484_split_sizes_0"), val = tensor([8, 8])]; int32 var_8484_axis_0 = const()[name = string("op_8484_axis_0"), val = int32(1)]; tensor var_8484_0, tensor var_8484_1 = split(axis = var_8484_axis_0, split_sizes = var_8484_split_sizes_0, x = query_states_141_cast_fp16)[name = string("op_8484")]; bool attn_weights_369_transpose_x_0 = const()[name = string("attn_weights_369_transpose_x_0"), val = bool(false)]; bool attn_weights_369_transpose_y_0 = const()[name = string("attn_weights_369_transpose_y_0"), val = bool(false)]; tensor attn_weights_369_cast_fp16 = matmul(transpose_x = attn_weights_369_transpose_x_0, transpose_y = attn_weights_369_transpose_y_0, x = var_8471_cast_fp16_0, y = var_8484_0)[name = string("attn_weights_369_cast_fp16")]; fp16 var_8487_to_fp16 = const()[name = string("op_8487_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_371_cast_fp16 = mul(x = attn_weights_369_cast_fp16, y = var_8487_to_fp16)[name = string("attn_weights_371_cast_fp16")]; tensor attn_weights_373_cast_fp16 = add(x = attn_weights_371_cast_fp16, y = attn_mask_1)[name = string("attn_weights_373_cast_fp16")]; int32 var_8491 = const()[name = string("op_8491"), val = int32(-2)]; tensor attn_weights_375_cast_fp16 = softmax(axis = var_8491, x = attn_weights_373_cast_fp16)[name = string("attn_weights_375_cast_fp16")]; bool var_8497_transpose_x_1 = const()[name = string("op_8497_transpose_x_1"), val = bool(true)]; bool var_8497_transpose_y_1 = const()[name = string("op_8497_transpose_y_1"), val = bool(false)]; tensor var_8497_cast_fp16 = matmul(transpose_x = var_8497_transpose_x_1, transpose_y = var_8497_transpose_y_1, x = attn_weights_375_cast_fp16, y = var_8481_cast_fp16_0)[name = string("op_8497_cast_fp16")]; bool attn_weights_377_transpose_x_0 = const()[name = string("attn_weights_377_transpose_x_0"), val = bool(false)]; bool attn_weights_377_transpose_y_0 = const()[name = string("attn_weights_377_transpose_y_0"), val = bool(false)]; tensor attn_weights_377_cast_fp16 = matmul(transpose_x = attn_weights_377_transpose_x_0, transpose_y = attn_weights_377_transpose_y_0, x = var_8471_cast_fp16_1, y = var_8484_1)[name = string("attn_weights_377_cast_fp16")]; fp16 var_8499_to_fp16 = const()[name = string("op_8499_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_379_cast_fp16 = mul(x = attn_weights_377_cast_fp16, y = var_8499_to_fp16)[name = string("attn_weights_379_cast_fp16")]; tensor attn_weights_381_cast_fp16 = add(x = attn_weights_379_cast_fp16, y = attn_mask_1)[name = string("attn_weights_381_cast_fp16")]; int32 var_8503 = const()[name = string("op_8503"), val = int32(-2)]; tensor attn_weights_383_cast_fp16 = softmax(axis = var_8503, x = attn_weights_381_cast_fp16)[name = string("attn_weights_383_cast_fp16")]; bool attn_output_185_transpose_x_1 = const()[name = string("attn_output_185_transpose_x_1"), val = bool(true)]; bool attn_output_185_transpose_y_1 = const()[name = string("attn_output_185_transpose_y_1"), val = bool(false)]; tensor attn_output_185_cast_fp16 = matmul(transpose_x = attn_output_185_transpose_x_1, transpose_y = attn_output_185_transpose_y_1, x = attn_weights_383_cast_fp16, y = var_8481_cast_fp16_1)[name = string("attn_output_185_cast_fp16")]; int32 var_8511 = const()[name = string("op_8511"), val = int32(1)]; bool attn_output_187_interleave_0 = const()[name = string("attn_output_187_interleave_0"), val = bool(false)]; tensor attn_output_187_cast_fp16 = concat(axis = var_8511, interleave = attn_output_187_interleave_0, values = (var_8497_cast_fp16, attn_output_185_cast_fp16))[name = string("attn_output_187_cast_fp16")]; tensor var_8515_perm_0 = const()[name = string("op_8515_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_287x = const()[name = string("concat_287x"), val = tensor([1, 2048, 1, -1])]; tensor var_8515_cast_fp16 = transpose(perm = var_8515_perm_0, x = attn_output_187_cast_fp16)[name = string("transpose_614")]; tensor attn_output_191_cast_fp16 = reshape(shape = concat_287x, x = var_8515_cast_fp16)[name = string("attn_output_191_cast_fp16")]; tensor hidden_states_233_strides_0 = const()[name = string("hidden_states_233_strides_0"), val = tensor([1, 1])]; string hidden_states_233_pad_type_0 = const()[name = string("hidden_states_233_pad_type_0"), val = string("valid")]; tensor hidden_states_233_pad_0 = const()[name = string("hidden_states_233_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_233_dilations_0 = const()[name = string("hidden_states_233_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_233_groups_0 = const()[name = string("hidden_states_233_groups_0"), val = int32(1)]; tensor hidden_states_233_cast_fp16 = conv(dilations = hidden_states_233_dilations_0, groups = hidden_states_233_groups_0, pad = hidden_states_233_pad_0, pad_type = hidden_states_233_pad_type_0, strides = hidden_states_233_strides_0, weight = layers_23_self_attn_o_proj_weight_cast_fp16, x = attn_output_191_cast_fp16)[name = string("hidden_states_233_cast_fp16")]; tensor hidden_states_235_cast_fp16 = add(x = hidden_states_229_cast_fp16, y = hidden_states_233_cast_fp16)[name = string("hidden_states_235_cast_fp16")]; fp16 const_238_promoted_to_fp16 = const()[name = string("const_238_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8548_cast_fp16 = mul(x = hidden_states_235_cast_fp16, y = const_238_promoted_to_fp16)[name = string("op_8548_cast_fp16")]; int32 var_8546 = const()[name = string("op_8546"), val = int32(1)]; bool doubled_189_interleave_0 = const()[name = string("doubled_189_interleave_0"), val = bool(false)]; tensor doubled_189_cast_fp16 = concat(axis = var_8546, interleave = doubled_189_interleave_0, values = (hidden_states_235_cast_fp16, var_8548_cast_fp16))[name = string("doubled_189_cast_fp16")]; tensor out_95_axes_0 = const()[name = string("out_95_axes_0"), val = tensor([1])]; tensor out_95_gamma_0_to_fp16 = const()[name = string("out_95_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539297280)))]; fp16 var_8558_to_fp16 = const()[name = string("op_8558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_95_cast_fp16 = layer_norm(axes = out_95_axes_0, epsilon = var_8558_to_fp16, gamma = out_95_gamma_0_to_fp16, x = doubled_189_cast_fp16)[name = string("out_95_cast_fp16")]; tensor var_8569_split_sizes_0 = const()[name = string("op_8569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8569_axis_0 = const()[name = string("op_8569_axis_0"), val = int32(1)]; tensor var_8569_cast_fp16_0, tensor var_8569_cast_fp16_1 = split(axis = var_8569_axis_0, split_sizes = var_8569_split_sizes_0, x = out_95_cast_fp16)[name = string("op_8569_cast_fp16")]; tensor input_47_strides_0 = const()[name = string("input_47_strides_0"), val = tensor([1, 1])]; string input_47_pad_type_0 = const()[name = string("input_47_pad_type_0"), val = string("valid")]; tensor input_47_pad_0 = const()[name = string("input_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_47_dilations_0 = const()[name = string("input_47_dilations_0"), val = tensor([1, 1])]; int32 input_47_groups_0 = const()[name = string("input_47_groups_0"), val = int32(1)]; tensor input_47_cast_fp16 = conv(dilations = input_47_dilations_0, groups = input_47_groups_0, pad = input_47_pad_0, pad_type = input_47_pad_type_0, strides = input_47_strides_0, weight = layers_23_mlp_gate_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("input_47_cast_fp16")]; tensor var_8586_cast_fp16 = silu(x = input_47_cast_fp16)[name = string("op_8586_cast_fp16")]; tensor var_8592_strides_0 = const()[name = string("op_8592_strides_0"), val = tensor([1, 1])]; string var_8592_pad_type_0 = const()[name = string("op_8592_pad_type_0"), val = string("valid")]; tensor var_8592_pad_0 = const()[name = string("op_8592_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8592_dilations_0 = const()[name = string("op_8592_dilations_0"), val = tensor([1, 1])]; int32 var_8592_groups_0 = const()[name = string("op_8592_groups_0"), val = int32(1)]; tensor var_8592_cast_fp16 = conv(dilations = var_8592_dilations_0, groups = var_8592_groups_0, pad = var_8592_pad_0, pad_type = var_8592_pad_type_0, strides = var_8592_strides_0, weight = layers_23_mlp_up_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("op_8592_cast_fp16")]; tensor x_239_cast_fp16 = mul(x = var_8586_cast_fp16, y = var_8592_cast_fp16)[name = string("x_239_cast_fp16")]; tensor hidden_states_237_strides_0 = const()[name = string("hidden_states_237_strides_0"), val = tensor([1, 1])]; string hidden_states_237_pad_type_0 = const()[name = string("hidden_states_237_pad_type_0"), val = string("valid")]; tensor hidden_states_237_pad_0 = const()[name = string("hidden_states_237_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_237_dilations_0 = const()[name = string("hidden_states_237_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_237_groups_0 = const()[name = string("hidden_states_237_groups_0"), val = int32(1)]; tensor hidden_states_237_cast_fp16 = conv(dilations = hidden_states_237_dilations_0, groups = hidden_states_237_groups_0, pad = hidden_states_237_pad_0, pad_type = hidden_states_237_pad_type_0, strides = hidden_states_237_strides_0, weight = layers_23_mlp_down_proj_weight_cast_fp16, x = x_239_cast_fp16)[name = string("hidden_states_237_cast_fp16")]; tensor hidden_states_239_cast_fp16 = add(x = hidden_states_235_cast_fp16, y = hidden_states_237_cast_fp16)[name = string("hidden_states_239_cast_fp16")]; fp16 const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8610_cast_fp16 = mul(x = hidden_states_239_cast_fp16, y = const_240_promoted_to_fp16)[name = string("op_8610_cast_fp16")]; int32 var_8608 = const()[name = string("op_8608"), val = int32(1)]; bool doubled_193_interleave_0 = const()[name = string("doubled_193_interleave_0"), val = bool(false)]; tensor doubled_193_cast_fp16 = concat(axis = var_8608, interleave = doubled_193_interleave_0, values = (hidden_states_239_cast_fp16, var_8610_cast_fp16))[name = string("doubled_193_cast_fp16")]; tensor out_97_axes_0 = const()[name = string("out_97_axes_0"), val = tensor([1])]; tensor out_97_gamma_0_to_fp16 = const()[name = string("out_97_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539305536)))]; fp16 var_8620_to_fp16 = const()[name = string("op_8620_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_97_cast_fp16 = layer_norm(axes = out_97_axes_0, epsilon = var_8620_to_fp16, gamma = out_97_gamma_0_to_fp16, x = doubled_193_cast_fp16)[name = string("out_97_cast_fp16")]; tensor var_8631_split_sizes_0 = const()[name = string("op_8631_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8631_axis_0 = const()[name = string("op_8631_axis_0"), val = int32(1)]; tensor var_8631_cast_fp16_0, tensor var_8631_cast_fp16_1 = split(axis = var_8631_axis_0, split_sizes = var_8631_split_sizes_0, x = out_97_cast_fp16)[name = string("op_8631_cast_fp16")]; tensor query_states_145_strides_0 = const()[name = string("query_states_145_strides_0"), val = tensor([1, 1])]; string query_states_145_pad_type_0 = const()[name = string("query_states_145_pad_type_0"), val = string("valid")]; tensor query_states_145_pad_0 = const()[name = string("query_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_145_dilations_0 = const()[name = string("query_states_145_dilations_0"), val = tensor([1, 1])]; int32 query_states_145_groups_0 = const()[name = string("query_states_145_groups_0"), val = int32(1)]; tensor query_states_145_cast_fp16 = conv(dilations = query_states_145_dilations_0, groups = query_states_145_groups_0, pad = query_states_145_pad_0, pad_type = query_states_145_pad_type_0, strides = query_states_145_strides_0, weight = layers_24_self_attn_q_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("query_states_145_cast_fp16")]; tensor key_states_241_strides_0 = const()[name = string("key_states_241_strides_0"), val = tensor([1, 1])]; string key_states_241_pad_type_0 = const()[name = string("key_states_241_pad_type_0"), val = string("valid")]; tensor key_states_241_pad_0 = const()[name = string("key_states_241_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_241_dilations_0 = const()[name = string("key_states_241_dilations_0"), val = tensor([1, 1])]; int32 key_states_241_groups_0 = const()[name = string("key_states_241_groups_0"), val = int32(1)]; tensor key_states_241_cast_fp16 = conv(dilations = key_states_241_dilations_0, groups = key_states_241_groups_0, pad = key_states_241_pad_0, pad_type = key_states_241_pad_type_0, strides = key_states_241_strides_0, weight = layers_24_self_attn_k_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("key_states_241_cast_fp16")]; tensor layers_24_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_24_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539313792)))]; tensor value_states_145_strides_0 = const()[name = string("value_states_145_strides_0"), val = tensor([1, 1])]; string value_states_145_pad_type_0 = const()[name = string("value_states_145_pad_type_0"), val = string("valid")]; tensor value_states_145_pad_0 = const()[name = string("value_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_145_dilations_0 = const()[name = string("value_states_145_dilations_0"), val = tensor([1, 1])]; int32 value_states_145_groups_0 = const()[name = string("value_states_145_groups_0"), val = int32(1)]; tensor value_states_145_cast_fp16 = conv(dilations = value_states_145_dilations_0, groups = value_states_145_groups_0, pad = value_states_145_pad_0, pad_type = value_states_145_pad_type_0, strides = value_states_145_strides_0, weight = layers_24_self_attn_v_proj_weight_to_fp16, x = var_8631_cast_fp16_0)[name = string("value_states_145_cast_fp16")]; tensor concat_288x = const()[name = string("concat_288x"), val = tensor([1, 16, 128, -1])]; tensor x_241_cast_fp16 = reshape(shape = concat_288x, x = query_states_145_cast_fp16)[name = string("x_241_cast_fp16")]; tensor concat_289x = const()[name = string("concat_289x"), val = tensor([1, 2, 128, -1])]; tensor var_8688_cast_fp16 = reshape(shape = concat_289x, x = key_states_241_cast_fp16)[name = string("op_8688_cast_fp16")]; tensor concat_290x = const()[name = string("concat_290x"), val = tensor([1, 2, 128, -1])]; tensor var_8695_cast_fp16 = reshape(shape = concat_290x, x = value_states_145_cast_fp16)[name = string("op_8695_cast_fp16")]; tensor var_8699_cast_fp16 = mul(x = x_241_cast_fp16, y = var_869_cast_fp16)[name = string("op_8699_cast_fp16")]; tensor var_8700_split_sizes_0 = const()[name = string("op_8700_split_sizes_0"), val = tensor([64, 64])]; int32 var_8700_axis_0 = const()[name = string("op_8700_axis_0"), val = int32(-2)]; tensor var_8700_cast_fp16_0, tensor var_8700_cast_fp16_1 = split(axis = var_8700_axis_0, split_sizes = var_8700_split_sizes_0, x = x_241_cast_fp16)[name = string("op_8700_cast_fp16")]; fp16 const_242_promoted_to_fp16 = const()[name = string("const_242_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8702_cast_fp16 = mul(x = var_8700_cast_fp16_1, y = const_242_promoted_to_fp16)[name = string("op_8702_cast_fp16")]; int32 var_8704 = const()[name = string("op_8704"), val = int32(-2)]; bool var_8705_interleave_0 = const()[name = string("op_8705_interleave_0"), val = bool(false)]; tensor var_8705_cast_fp16 = concat(axis = var_8704, interleave = var_8705_interleave_0, values = (var_8702_cast_fp16, var_8700_cast_fp16_0))[name = string("op_8705_cast_fp16")]; tensor var_8706_cast_fp16 = mul(x = var_8705_cast_fp16, y = var_878_cast_fp16)[name = string("op_8706_cast_fp16")]; tensor query_states_147_cast_fp16 = add(x = var_8699_cast_fp16, y = var_8706_cast_fp16)[name = string("query_states_147_cast_fp16")]; tensor var_8712_cast_fp16 = mul(x = var_8688_cast_fp16, y = var_869_cast_fp16)[name = string("op_8712_cast_fp16")]; tensor var_8713_split_sizes_0 = const()[name = string("op_8713_split_sizes_0"), val = tensor([64, 64])]; int32 var_8713_axis_0 = const()[name = string("op_8713_axis_0"), val = int32(-2)]; tensor var_8713_cast_fp16_0, tensor var_8713_cast_fp16_1 = split(axis = var_8713_axis_0, split_sizes = var_8713_split_sizes_0, x = var_8688_cast_fp16)[name = string("op_8713_cast_fp16")]; fp16 const_243_promoted_to_fp16 = const()[name = string("const_243_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8715_cast_fp16 = mul(x = var_8713_cast_fp16_1, y = const_243_promoted_to_fp16)[name = string("op_8715_cast_fp16")]; int32 var_8717 = const()[name = string("op_8717"), val = int32(-2)]; bool var_8718_interleave_0 = const()[name = string("op_8718_interleave_0"), val = bool(false)]; tensor var_8718_cast_fp16 = concat(axis = var_8717, interleave = var_8718_interleave_0, values = (var_8715_cast_fp16, var_8713_cast_fp16_0))[name = string("op_8718_cast_fp16")]; tensor var_8719_cast_fp16 = mul(x = var_8718_cast_fp16, y = var_878_cast_fp16)[name = string("op_8719_cast_fp16")]; tensor key_states_245_cast_fp16 = add(x = var_8712_cast_fp16, y = var_8719_cast_fp16)[name = string("key_states_245_cast_fp16")]; tensor expand_dims_288 = const()[name = string("expand_dims_288"), val = tensor([24])]; tensor expand_dims_289 = const()[name = string("expand_dims_289"), val = tensor([0])]; tensor expand_dims_291 = const()[name = string("expand_dims_291"), val = tensor([0])]; int32 concat_293_axis_0 = const()[name = string("concat_293_axis_0"), val = int32(0)]; bool concat_293_interleave_0 = const()[name = string("concat_293_interleave_0"), val = bool(false)]; tensor concat_293 = concat(axis = concat_293_axis_0, interleave = concat_293_interleave_0, values = (expand_dims_288, expand_dims_289, position_id, expand_dims_291))[name = string("concat_293")]; tensor expand_dims_292 = const()[name = string("expand_dims_292"), val = tensor([25])]; tensor concat_294_values1_0 = const()[name = string("concat_294_values1_0"), val = tensor([0])]; tensor concat_294_values3_0 = const()[name = string("concat_294_values3_0"), val = tensor([0])]; int32 concat_294_axis_0 = const()[name = string("concat_294_axis_0"), val = int32(0)]; bool concat_294_interleave_0 = const()[name = string("concat_294_interleave_0"), val = bool(false)]; tensor concat_294 = concat(axis = concat_294_axis_0, interleave = concat_294_interleave_0, values = (expand_dims_292, concat_294_values1_0, cache_position_end, concat_294_values3_0))[name = string("concat_294")]; tensor key_states_247_perm_0 = const()[name = string("key_states_247_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_25_stride_0 = const()[name = string("key_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_247_cast_fp16 = transpose(perm = key_states_247_perm_0, x = key_states_245_cast_fp16)[name = string("transpose_613")]; tensor key_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = key_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = key_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_25_squeeze_mask_0, stride = key_cache_internal_tensor_assign_25_stride_0, update = key_states_247_cast_fp16, x = coreml_update_state_438)[name = string("key_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_25_cast_fp16, input = key_cache)[name = string("coreml_update_state_440_write_state")]; tensor coreml_update_state_440 = read_state(input = key_cache)[name = string("coreml_update_state_440")]; tensor value_states_147_perm_0 = const()[name = string("value_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_25_stride_0 = const()[name = string("value_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_147_cast_fp16 = transpose(perm = value_states_147_perm_0, x = var_8695_cast_fp16)[name = string("transpose_612")]; tensor value_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = value_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = value_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_25_squeeze_mask_0, stride = value_cache_internal_tensor_assign_25_stride_0, update = value_states_147_cast_fp16, x = coreml_update_state_439)[name = string("value_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_25_cast_fp16, input = value_cache)[name = string("coreml_update_state_441_write_state")]; tensor coreml_update_state_441 = read_state(input = value_cache)[name = string("coreml_update_state_441")]; tensor var_8789_begin_0 = const()[name = string("op_8789_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8789_end_0 = const()[name = string("op_8789_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8789_end_mask_0 = const()[name = string("op_8789_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8789_cast_fp16 = slice_by_index(begin = var_8789_begin_0, end = var_8789_end_0, end_mask = var_8789_end_mask_0, x = coreml_update_state_440)[name = string("op_8789_cast_fp16")]; tensor tile_48 = const()[name = string("tile_48"), val = tensor([1, 1])]; int32 var_8792_axis_0 = const()[name = string("op_8792_axis_0"), val = int32(1)]; tensor var_8792_cast_fp16_0, tensor var_8792_cast_fp16_1 = split(axis = var_8792_axis_0, split_sizes = tile_48, x = var_8789_cast_fp16)[name = string("op_8792_cast_fp16")]; tensor var_8799_begin_0 = const()[name = string("op_8799_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8799_end_0 = const()[name = string("op_8799_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8799_end_mask_0 = const()[name = string("op_8799_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8799_cast_fp16 = slice_by_index(begin = var_8799_begin_0, end = var_8799_end_0, end_mask = var_8799_end_mask_0, x = coreml_update_state_441)[name = string("op_8799_cast_fp16")]; tensor tile_49 = const()[name = string("tile_49"), val = tensor([1, 1])]; int32 var_8802_axis_0 = const()[name = string("op_8802_axis_0"), val = int32(1)]; tensor var_8802_cast_fp16_0, tensor var_8802_cast_fp16_1 = split(axis = var_8802_axis_0, split_sizes = tile_49, x = var_8799_cast_fp16)[name = string("op_8802_cast_fp16")]; tensor var_8805_split_sizes_0 = const()[name = string("op_8805_split_sizes_0"), val = tensor([8, 8])]; int32 var_8805_axis_0 = const()[name = string("op_8805_axis_0"), val = int32(1)]; tensor var_8805_0, tensor var_8805_1 = split(axis = var_8805_axis_0, split_sizes = var_8805_split_sizes_0, x = query_states_147_cast_fp16)[name = string("op_8805")]; bool attn_weights_385_transpose_x_0 = const()[name = string("attn_weights_385_transpose_x_0"), val = bool(false)]; bool attn_weights_385_transpose_y_0 = const()[name = string("attn_weights_385_transpose_y_0"), val = bool(false)]; tensor attn_weights_385_cast_fp16 = matmul(transpose_x = attn_weights_385_transpose_x_0, transpose_y = attn_weights_385_transpose_y_0, x = var_8792_cast_fp16_0, y = var_8805_0)[name = string("attn_weights_385_cast_fp16")]; fp16 var_8808_to_fp16 = const()[name = string("op_8808_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_387_cast_fp16 = mul(x = attn_weights_385_cast_fp16, y = var_8808_to_fp16)[name = string("attn_weights_387_cast_fp16")]; tensor attn_weights_389_cast_fp16 = add(x = attn_weights_387_cast_fp16, y = attn_mask_1)[name = string("attn_weights_389_cast_fp16")]; int32 var_8812 = const()[name = string("op_8812"), val = int32(-2)]; tensor attn_weights_391_cast_fp16 = softmax(axis = var_8812, x = attn_weights_389_cast_fp16)[name = string("attn_weights_391_cast_fp16")]; bool var_8818_transpose_x_1 = const()[name = string("op_8818_transpose_x_1"), val = bool(true)]; bool var_8818_transpose_y_1 = const()[name = string("op_8818_transpose_y_1"), val = bool(false)]; tensor var_8818_cast_fp16 = matmul(transpose_x = var_8818_transpose_x_1, transpose_y = var_8818_transpose_y_1, x = attn_weights_391_cast_fp16, y = var_8802_cast_fp16_0)[name = string("op_8818_cast_fp16")]; bool attn_weights_393_transpose_x_0 = const()[name = string("attn_weights_393_transpose_x_0"), val = bool(false)]; bool attn_weights_393_transpose_y_0 = const()[name = string("attn_weights_393_transpose_y_0"), val = bool(false)]; tensor attn_weights_393_cast_fp16 = matmul(transpose_x = attn_weights_393_transpose_x_0, transpose_y = attn_weights_393_transpose_y_0, x = var_8792_cast_fp16_1, y = var_8805_1)[name = string("attn_weights_393_cast_fp16")]; fp16 var_8820_to_fp16 = const()[name = string("op_8820_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_395_cast_fp16 = mul(x = attn_weights_393_cast_fp16, y = var_8820_to_fp16)[name = string("attn_weights_395_cast_fp16")]; tensor attn_weights_397_cast_fp16 = add(x = attn_weights_395_cast_fp16, y = attn_mask_1)[name = string("attn_weights_397_cast_fp16")]; int32 var_8824 = const()[name = string("op_8824"), val = int32(-2)]; tensor attn_weights_399_cast_fp16 = softmax(axis = var_8824, x = attn_weights_397_cast_fp16)[name = string("attn_weights_399_cast_fp16")]; bool attn_output_193_transpose_x_1 = const()[name = string("attn_output_193_transpose_x_1"), val = bool(true)]; bool attn_output_193_transpose_y_1 = const()[name = string("attn_output_193_transpose_y_1"), val = bool(false)]; tensor attn_output_193_cast_fp16 = matmul(transpose_x = attn_output_193_transpose_x_1, transpose_y = attn_output_193_transpose_y_1, x = attn_weights_399_cast_fp16, y = var_8802_cast_fp16_1)[name = string("attn_output_193_cast_fp16")]; int32 var_8832 = const()[name = string("op_8832"), val = int32(1)]; bool attn_output_195_interleave_0 = const()[name = string("attn_output_195_interleave_0"), val = bool(false)]; tensor attn_output_195_cast_fp16 = concat(axis = var_8832, interleave = attn_output_195_interleave_0, values = (var_8818_cast_fp16, attn_output_193_cast_fp16))[name = string("attn_output_195_cast_fp16")]; tensor var_8836_perm_0 = const()[name = string("op_8836_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_299x = const()[name = string("concat_299x"), val = tensor([1, 2048, 1, -1])]; tensor var_8836_cast_fp16 = transpose(perm = var_8836_perm_0, x = attn_output_195_cast_fp16)[name = string("transpose_611")]; tensor attn_output_199_cast_fp16 = reshape(shape = concat_299x, x = var_8836_cast_fp16)[name = string("attn_output_199_cast_fp16")]; tensor hidden_states_243_strides_0 = const()[name = string("hidden_states_243_strides_0"), val = tensor([1, 1])]; string hidden_states_243_pad_type_0 = const()[name = string("hidden_states_243_pad_type_0"), val = string("valid")]; tensor hidden_states_243_pad_0 = const()[name = string("hidden_states_243_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_243_dilations_0 = const()[name = string("hidden_states_243_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_243_groups_0 = const()[name = string("hidden_states_243_groups_0"), val = int32(1)]; tensor hidden_states_243_cast_fp16 = conv(dilations = hidden_states_243_dilations_0, groups = hidden_states_243_groups_0, pad = hidden_states_243_pad_0, pad_type = hidden_states_243_pad_type_0, strides = hidden_states_243_strides_0, weight = layers_24_self_attn_o_proj_weight_cast_fp16, x = attn_output_199_cast_fp16)[name = string("hidden_states_243_cast_fp16")]; tensor hidden_states_245_cast_fp16 = add(x = hidden_states_239_cast_fp16, y = hidden_states_243_cast_fp16)[name = string("hidden_states_245_cast_fp16")]; fp16 const_248_promoted_to_fp16 = const()[name = string("const_248_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8869_cast_fp16 = mul(x = hidden_states_245_cast_fp16, y = const_248_promoted_to_fp16)[name = string("op_8869_cast_fp16")]; int32 var_8867 = const()[name = string("op_8867"), val = int32(1)]; bool doubled_197_interleave_0 = const()[name = string("doubled_197_interleave_0"), val = bool(false)]; tensor doubled_197_cast_fp16 = concat(axis = var_8867, interleave = doubled_197_interleave_0, values = (hidden_states_245_cast_fp16, var_8869_cast_fp16))[name = string("doubled_197_cast_fp16")]; tensor out_99_axes_0 = const()[name = string("out_99_axes_0"), val = tensor([1])]; tensor out_99_gamma_0_to_fp16 = const()[name = string("out_99_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540362432)))]; fp16 var_8879_to_fp16 = const()[name = string("op_8879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_99_cast_fp16 = layer_norm(axes = out_99_axes_0, epsilon = var_8879_to_fp16, gamma = out_99_gamma_0_to_fp16, x = doubled_197_cast_fp16)[name = string("out_99_cast_fp16")]; tensor var_8890_split_sizes_0 = const()[name = string("op_8890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8890_axis_0 = const()[name = string("op_8890_axis_0"), val = int32(1)]; tensor var_8890_cast_fp16_0, tensor var_8890_cast_fp16_1 = split(axis = var_8890_axis_0, split_sizes = var_8890_split_sizes_0, x = out_99_cast_fp16)[name = string("op_8890_cast_fp16")]; tensor input_49_strides_0 = const()[name = string("input_49_strides_0"), val = tensor([1, 1])]; string input_49_pad_type_0 = const()[name = string("input_49_pad_type_0"), val = string("valid")]; tensor input_49_pad_0 = const()[name = string("input_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_49_dilations_0 = const()[name = string("input_49_dilations_0"), val = tensor([1, 1])]; int32 input_49_groups_0 = const()[name = string("input_49_groups_0"), val = int32(1)]; tensor input_49_cast_fp16 = conv(dilations = input_49_dilations_0, groups = input_49_groups_0, pad = input_49_pad_0, pad_type = input_49_pad_type_0, strides = input_49_strides_0, weight = layers_24_mlp_gate_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("input_49_cast_fp16")]; tensor var_8907_cast_fp16 = silu(x = input_49_cast_fp16)[name = string("op_8907_cast_fp16")]; tensor var_8913_strides_0 = const()[name = string("op_8913_strides_0"), val = tensor([1, 1])]; string var_8913_pad_type_0 = const()[name = string("op_8913_pad_type_0"), val = string("valid")]; tensor var_8913_pad_0 = const()[name = string("op_8913_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8913_dilations_0 = const()[name = string("op_8913_dilations_0"), val = tensor([1, 1])]; int32 var_8913_groups_0 = const()[name = string("op_8913_groups_0"), val = int32(1)]; tensor var_8913_cast_fp16 = conv(dilations = var_8913_dilations_0, groups = var_8913_groups_0, pad = var_8913_pad_0, pad_type = var_8913_pad_type_0, strides = var_8913_strides_0, weight = layers_24_mlp_up_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("op_8913_cast_fp16")]; tensor x_249_cast_fp16 = mul(x = var_8907_cast_fp16, y = var_8913_cast_fp16)[name = string("x_249_cast_fp16")]; tensor hidden_states_247_strides_0 = const()[name = string("hidden_states_247_strides_0"), val = tensor([1, 1])]; string hidden_states_247_pad_type_0 = const()[name = string("hidden_states_247_pad_type_0"), val = string("valid")]; tensor hidden_states_247_pad_0 = const()[name = string("hidden_states_247_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_247_dilations_0 = const()[name = string("hidden_states_247_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_247_groups_0 = const()[name = string("hidden_states_247_groups_0"), val = int32(1)]; tensor hidden_states_247_cast_fp16 = conv(dilations = hidden_states_247_dilations_0, groups = hidden_states_247_groups_0, pad = hidden_states_247_pad_0, pad_type = hidden_states_247_pad_type_0, strides = hidden_states_247_strides_0, weight = layers_24_mlp_down_proj_weight_cast_fp16, x = x_249_cast_fp16)[name = string("hidden_states_247_cast_fp16")]; tensor hidden_states_249_cast_fp16 = add(x = hidden_states_245_cast_fp16, y = hidden_states_247_cast_fp16)[name = string("hidden_states_249_cast_fp16")]; fp16 const_250_promoted_to_fp16 = const()[name = string("const_250_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8931_cast_fp16 = mul(x = hidden_states_249_cast_fp16, y = const_250_promoted_to_fp16)[name = string("op_8931_cast_fp16")]; int32 var_8929 = const()[name = string("op_8929"), val = int32(1)]; bool doubled_201_interleave_0 = const()[name = string("doubled_201_interleave_0"), val = bool(false)]; tensor doubled_201_cast_fp16 = concat(axis = var_8929, interleave = doubled_201_interleave_0, values = (hidden_states_249_cast_fp16, var_8931_cast_fp16))[name = string("doubled_201_cast_fp16")]; tensor out_101_axes_0 = const()[name = string("out_101_axes_0"), val = tensor([1])]; tensor out_101_gamma_0_to_fp16 = const()[name = string("out_101_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540370688)))]; fp16 var_8941_to_fp16 = const()[name = string("op_8941_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_101_cast_fp16 = layer_norm(axes = out_101_axes_0, epsilon = var_8941_to_fp16, gamma = out_101_gamma_0_to_fp16, x = doubled_201_cast_fp16)[name = string("out_101_cast_fp16")]; tensor var_8952_split_sizes_0 = const()[name = string("op_8952_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8952_axis_0 = const()[name = string("op_8952_axis_0"), val = int32(1)]; tensor var_8952_cast_fp16_0, tensor var_8952_cast_fp16_1 = split(axis = var_8952_axis_0, split_sizes = var_8952_split_sizes_0, x = out_101_cast_fp16)[name = string("op_8952_cast_fp16")]; tensor query_states_151_strides_0 = const()[name = string("query_states_151_strides_0"), val = tensor([1, 1])]; string query_states_151_pad_type_0 = const()[name = string("query_states_151_pad_type_0"), val = string("valid")]; tensor query_states_151_pad_0 = const()[name = string("query_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_151_dilations_0 = const()[name = string("query_states_151_dilations_0"), val = tensor([1, 1])]; int32 query_states_151_groups_0 = const()[name = string("query_states_151_groups_0"), val = int32(1)]; tensor query_states_151_cast_fp16 = conv(dilations = query_states_151_dilations_0, groups = query_states_151_groups_0, pad = query_states_151_pad_0, pad_type = query_states_151_pad_type_0, strides = query_states_151_strides_0, weight = layers_25_self_attn_q_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("query_states_151_cast_fp16")]; tensor key_states_251_strides_0 = const()[name = string("key_states_251_strides_0"), val = tensor([1, 1])]; string key_states_251_pad_type_0 = const()[name = string("key_states_251_pad_type_0"), val = string("valid")]; tensor key_states_251_pad_0 = const()[name = string("key_states_251_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_251_dilations_0 = const()[name = string("key_states_251_dilations_0"), val = tensor([1, 1])]; int32 key_states_251_groups_0 = const()[name = string("key_states_251_groups_0"), val = int32(1)]; tensor key_states_251_cast_fp16 = conv(dilations = key_states_251_dilations_0, groups = key_states_251_groups_0, pad = key_states_251_pad_0, pad_type = key_states_251_pad_type_0, strides = key_states_251_strides_0, weight = layers_25_self_attn_k_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("key_states_251_cast_fp16")]; tensor layers_25_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_25_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540378944)))]; tensor value_states_151_strides_0 = const()[name = string("value_states_151_strides_0"), val = tensor([1, 1])]; string value_states_151_pad_type_0 = const()[name = string("value_states_151_pad_type_0"), val = string("valid")]; tensor value_states_151_pad_0 = const()[name = string("value_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_151_dilations_0 = const()[name = string("value_states_151_dilations_0"), val = tensor([1, 1])]; int32 value_states_151_groups_0 = const()[name = string("value_states_151_groups_0"), val = int32(1)]; tensor value_states_151_cast_fp16 = conv(dilations = value_states_151_dilations_0, groups = value_states_151_groups_0, pad = value_states_151_pad_0, pad_type = value_states_151_pad_type_0, strides = value_states_151_strides_0, weight = layers_25_self_attn_v_proj_weight_to_fp16, x = var_8952_cast_fp16_0)[name = string("value_states_151_cast_fp16")]; tensor concat_300x = const()[name = string("concat_300x"), val = tensor([1, 16, 128, -1])]; tensor x_251_cast_fp16 = reshape(shape = concat_300x, x = query_states_151_cast_fp16)[name = string("x_251_cast_fp16")]; tensor concat_301x = const()[name = string("concat_301x"), val = tensor([1, 2, 128, -1])]; tensor var_9009_cast_fp16 = reshape(shape = concat_301x, x = key_states_251_cast_fp16)[name = string("op_9009_cast_fp16")]; tensor concat_302x = const()[name = string("concat_302x"), val = tensor([1, 2, 128, -1])]; tensor var_9016_cast_fp16 = reshape(shape = concat_302x, x = value_states_151_cast_fp16)[name = string("op_9016_cast_fp16")]; tensor var_9020_cast_fp16 = mul(x = x_251_cast_fp16, y = var_869_cast_fp16)[name = string("op_9020_cast_fp16")]; tensor var_9021_split_sizes_0 = const()[name = string("op_9021_split_sizes_0"), val = tensor([64, 64])]; int32 var_9021_axis_0 = const()[name = string("op_9021_axis_0"), val = int32(-2)]; tensor var_9021_cast_fp16_0, tensor var_9021_cast_fp16_1 = split(axis = var_9021_axis_0, split_sizes = var_9021_split_sizes_0, x = x_251_cast_fp16)[name = string("op_9021_cast_fp16")]; fp16 const_252_promoted_to_fp16 = const()[name = string("const_252_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9023_cast_fp16 = mul(x = var_9021_cast_fp16_1, y = const_252_promoted_to_fp16)[name = string("op_9023_cast_fp16")]; int32 var_9025 = const()[name = string("op_9025"), val = int32(-2)]; bool var_9026_interleave_0 = const()[name = string("op_9026_interleave_0"), val = bool(false)]; tensor var_9026_cast_fp16 = concat(axis = var_9025, interleave = var_9026_interleave_0, values = (var_9023_cast_fp16, var_9021_cast_fp16_0))[name = string("op_9026_cast_fp16")]; tensor var_9027_cast_fp16 = mul(x = var_9026_cast_fp16, y = var_878_cast_fp16)[name = string("op_9027_cast_fp16")]; tensor query_states_153_cast_fp16 = add(x = var_9020_cast_fp16, y = var_9027_cast_fp16)[name = string("query_states_153_cast_fp16")]; tensor var_9033_cast_fp16 = mul(x = var_9009_cast_fp16, y = var_869_cast_fp16)[name = string("op_9033_cast_fp16")]; tensor var_9034_split_sizes_0 = const()[name = string("op_9034_split_sizes_0"), val = tensor([64, 64])]; int32 var_9034_axis_0 = const()[name = string("op_9034_axis_0"), val = int32(-2)]; tensor var_9034_cast_fp16_0, tensor var_9034_cast_fp16_1 = split(axis = var_9034_axis_0, split_sizes = var_9034_split_sizes_0, x = var_9009_cast_fp16)[name = string("op_9034_cast_fp16")]; fp16 const_253_promoted_to_fp16 = const()[name = string("const_253_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9036_cast_fp16 = mul(x = var_9034_cast_fp16_1, y = const_253_promoted_to_fp16)[name = string("op_9036_cast_fp16")]; int32 var_9038 = const()[name = string("op_9038"), val = int32(-2)]; bool var_9039_interleave_0 = const()[name = string("op_9039_interleave_0"), val = bool(false)]; tensor var_9039_cast_fp16 = concat(axis = var_9038, interleave = var_9039_interleave_0, values = (var_9036_cast_fp16, var_9034_cast_fp16_0))[name = string("op_9039_cast_fp16")]; tensor var_9040_cast_fp16 = mul(x = var_9039_cast_fp16, y = var_878_cast_fp16)[name = string("op_9040_cast_fp16")]; tensor key_states_255_cast_fp16 = add(x = var_9033_cast_fp16, y = var_9040_cast_fp16)[name = string("key_states_255_cast_fp16")]; tensor expand_dims_300 = const()[name = string("expand_dims_300"), val = tensor([25])]; tensor expand_dims_301 = const()[name = string("expand_dims_301"), val = tensor([0])]; tensor expand_dims_303 = const()[name = string("expand_dims_303"), val = tensor([0])]; int32 concat_305_axis_0 = const()[name = string("concat_305_axis_0"), val = int32(0)]; bool concat_305_interleave_0 = const()[name = string("concat_305_interleave_0"), val = bool(false)]; tensor concat_305 = concat(axis = concat_305_axis_0, interleave = concat_305_interleave_0, values = (expand_dims_300, expand_dims_301, position_id, expand_dims_303))[name = string("concat_305")]; tensor expand_dims_304 = const()[name = string("expand_dims_304"), val = tensor([26])]; tensor concat_306_values1_0 = const()[name = string("concat_306_values1_0"), val = tensor([0])]; tensor concat_306_values3_0 = const()[name = string("concat_306_values3_0"), val = tensor([0])]; int32 concat_306_axis_0 = const()[name = string("concat_306_axis_0"), val = int32(0)]; bool concat_306_interleave_0 = const()[name = string("concat_306_interleave_0"), val = bool(false)]; tensor concat_306 = concat(axis = concat_306_axis_0, interleave = concat_306_interleave_0, values = (expand_dims_304, concat_306_values1_0, cache_position_end, concat_306_values3_0))[name = string("concat_306")]; tensor key_states_257_perm_0 = const()[name = string("key_states_257_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_26_stride_0 = const()[name = string("key_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_257_cast_fp16 = transpose(perm = key_states_257_perm_0, x = key_states_255_cast_fp16)[name = string("transpose_610")]; tensor key_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = key_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = key_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_26_squeeze_mask_0, stride = key_cache_internal_tensor_assign_26_stride_0, update = key_states_257_cast_fp16, x = coreml_update_state_440)[name = string("key_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_26_cast_fp16, input = key_cache)[name = string("coreml_update_state_442_write_state")]; tensor coreml_update_state_442 = read_state(input = key_cache)[name = string("coreml_update_state_442")]; tensor value_states_153_perm_0 = const()[name = string("value_states_153_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_26_stride_0 = const()[name = string("value_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_153_cast_fp16 = transpose(perm = value_states_153_perm_0, x = var_9016_cast_fp16)[name = string("transpose_609")]; tensor value_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = value_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = value_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_26_squeeze_mask_0, stride = value_cache_internal_tensor_assign_26_stride_0, update = value_states_153_cast_fp16, x = coreml_update_state_441)[name = string("value_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_26_cast_fp16, input = value_cache)[name = string("coreml_update_state_443_write_state")]; tensor coreml_update_state_443 = read_state(input = value_cache)[name = string("coreml_update_state_443")]; tensor var_9110_begin_0 = const()[name = string("op_9110_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9110_end_0 = const()[name = string("op_9110_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9110_end_mask_0 = const()[name = string("op_9110_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9110_cast_fp16 = slice_by_index(begin = var_9110_begin_0, end = var_9110_end_0, end_mask = var_9110_end_mask_0, x = coreml_update_state_442)[name = string("op_9110_cast_fp16")]; tensor tile_50 = const()[name = string("tile_50"), val = tensor([1, 1])]; int32 var_9113_axis_0 = const()[name = string("op_9113_axis_0"), val = int32(1)]; tensor var_9113_cast_fp16_0, tensor var_9113_cast_fp16_1 = split(axis = var_9113_axis_0, split_sizes = tile_50, x = var_9110_cast_fp16)[name = string("op_9113_cast_fp16")]; tensor var_9120_begin_0 = const()[name = string("op_9120_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9120_end_0 = const()[name = string("op_9120_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9120_end_mask_0 = const()[name = string("op_9120_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9120_cast_fp16 = slice_by_index(begin = var_9120_begin_0, end = var_9120_end_0, end_mask = var_9120_end_mask_0, x = coreml_update_state_443)[name = string("op_9120_cast_fp16")]; tensor tile_51 = const()[name = string("tile_51"), val = tensor([1, 1])]; int32 var_9123_axis_0 = const()[name = string("op_9123_axis_0"), val = int32(1)]; tensor var_9123_cast_fp16_0, tensor var_9123_cast_fp16_1 = split(axis = var_9123_axis_0, split_sizes = tile_51, x = var_9120_cast_fp16)[name = string("op_9123_cast_fp16")]; tensor var_9126_split_sizes_0 = const()[name = string("op_9126_split_sizes_0"), val = tensor([8, 8])]; int32 var_9126_axis_0 = const()[name = string("op_9126_axis_0"), val = int32(1)]; tensor var_9126_0, tensor var_9126_1 = split(axis = var_9126_axis_0, split_sizes = var_9126_split_sizes_0, x = query_states_153_cast_fp16)[name = string("op_9126")]; bool attn_weights_401_transpose_x_0 = const()[name = string("attn_weights_401_transpose_x_0"), val = bool(false)]; bool attn_weights_401_transpose_y_0 = const()[name = string("attn_weights_401_transpose_y_0"), val = bool(false)]; tensor attn_weights_401_cast_fp16 = matmul(transpose_x = attn_weights_401_transpose_x_0, transpose_y = attn_weights_401_transpose_y_0, x = var_9113_cast_fp16_0, y = var_9126_0)[name = string("attn_weights_401_cast_fp16")]; fp16 var_9129_to_fp16 = const()[name = string("op_9129_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_403_cast_fp16 = mul(x = attn_weights_401_cast_fp16, y = var_9129_to_fp16)[name = string("attn_weights_403_cast_fp16")]; tensor attn_weights_405_cast_fp16 = add(x = attn_weights_403_cast_fp16, y = attn_mask_1)[name = string("attn_weights_405_cast_fp16")]; int32 var_9133 = const()[name = string("op_9133"), val = int32(-2)]; tensor attn_weights_407_cast_fp16 = softmax(axis = var_9133, x = attn_weights_405_cast_fp16)[name = string("attn_weights_407_cast_fp16")]; bool var_9139_transpose_x_1 = const()[name = string("op_9139_transpose_x_1"), val = bool(true)]; bool var_9139_transpose_y_1 = const()[name = string("op_9139_transpose_y_1"), val = bool(false)]; tensor var_9139_cast_fp16 = matmul(transpose_x = var_9139_transpose_x_1, transpose_y = var_9139_transpose_y_1, x = attn_weights_407_cast_fp16, y = var_9123_cast_fp16_0)[name = string("op_9139_cast_fp16")]; bool attn_weights_409_transpose_x_0 = const()[name = string("attn_weights_409_transpose_x_0"), val = bool(false)]; bool attn_weights_409_transpose_y_0 = const()[name = string("attn_weights_409_transpose_y_0"), val = bool(false)]; tensor attn_weights_409_cast_fp16 = matmul(transpose_x = attn_weights_409_transpose_x_0, transpose_y = attn_weights_409_transpose_y_0, x = var_9113_cast_fp16_1, y = var_9126_1)[name = string("attn_weights_409_cast_fp16")]; fp16 var_9141_to_fp16 = const()[name = string("op_9141_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_411_cast_fp16 = mul(x = attn_weights_409_cast_fp16, y = var_9141_to_fp16)[name = string("attn_weights_411_cast_fp16")]; tensor attn_weights_413_cast_fp16 = add(x = attn_weights_411_cast_fp16, y = attn_mask_1)[name = string("attn_weights_413_cast_fp16")]; int32 var_9145 = const()[name = string("op_9145"), val = int32(-2)]; tensor attn_weights_415_cast_fp16 = softmax(axis = var_9145, x = attn_weights_413_cast_fp16)[name = string("attn_weights_415_cast_fp16")]; bool attn_output_201_transpose_x_1 = const()[name = string("attn_output_201_transpose_x_1"), val = bool(true)]; bool attn_output_201_transpose_y_1 = const()[name = string("attn_output_201_transpose_y_1"), val = bool(false)]; tensor attn_output_201_cast_fp16 = matmul(transpose_x = attn_output_201_transpose_x_1, transpose_y = attn_output_201_transpose_y_1, x = attn_weights_415_cast_fp16, y = var_9123_cast_fp16_1)[name = string("attn_output_201_cast_fp16")]; int32 var_9153 = const()[name = string("op_9153"), val = int32(1)]; bool attn_output_203_interleave_0 = const()[name = string("attn_output_203_interleave_0"), val = bool(false)]; tensor attn_output_203_cast_fp16 = concat(axis = var_9153, interleave = attn_output_203_interleave_0, values = (var_9139_cast_fp16, attn_output_201_cast_fp16))[name = string("attn_output_203_cast_fp16")]; tensor var_9157_perm_0 = const()[name = string("op_9157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_311x = const()[name = string("concat_311x"), val = tensor([1, 2048, 1, -1])]; tensor var_9157_cast_fp16 = transpose(perm = var_9157_perm_0, x = attn_output_203_cast_fp16)[name = string("transpose_608")]; tensor attn_output_207_cast_fp16 = reshape(shape = concat_311x, x = var_9157_cast_fp16)[name = string("attn_output_207_cast_fp16")]; tensor hidden_states_253_strides_0 = const()[name = string("hidden_states_253_strides_0"), val = tensor([1, 1])]; string hidden_states_253_pad_type_0 = const()[name = string("hidden_states_253_pad_type_0"), val = string("valid")]; tensor hidden_states_253_pad_0 = const()[name = string("hidden_states_253_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_253_dilations_0 = const()[name = string("hidden_states_253_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_253_groups_0 = const()[name = string("hidden_states_253_groups_0"), val = int32(1)]; tensor hidden_states_253_cast_fp16 = conv(dilations = hidden_states_253_dilations_0, groups = hidden_states_253_groups_0, pad = hidden_states_253_pad_0, pad_type = hidden_states_253_pad_type_0, strides = hidden_states_253_strides_0, weight = layers_25_self_attn_o_proj_weight_cast_fp16, x = attn_output_207_cast_fp16)[name = string("hidden_states_253_cast_fp16")]; tensor hidden_states_255_cast_fp16 = add(x = hidden_states_249_cast_fp16, y = hidden_states_253_cast_fp16)[name = string("hidden_states_255_cast_fp16")]; fp16 const_258_promoted_to_fp16 = const()[name = string("const_258_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9190_cast_fp16 = mul(x = hidden_states_255_cast_fp16, y = const_258_promoted_to_fp16)[name = string("op_9190_cast_fp16")]; int32 var_9188 = const()[name = string("op_9188"), val = int32(1)]; bool doubled_205_interleave_0 = const()[name = string("doubled_205_interleave_0"), val = bool(false)]; tensor doubled_205_cast_fp16 = concat(axis = var_9188, interleave = doubled_205_interleave_0, values = (hidden_states_255_cast_fp16, var_9190_cast_fp16))[name = string("doubled_205_cast_fp16")]; tensor out_103_axes_0 = const()[name = string("out_103_axes_0"), val = tensor([1])]; tensor out_103_gamma_0_to_fp16 = const()[name = string("out_103_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541427584)))]; fp16 var_9200_to_fp16 = const()[name = string("op_9200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_103_cast_fp16 = layer_norm(axes = out_103_axes_0, epsilon = var_9200_to_fp16, gamma = out_103_gamma_0_to_fp16, x = doubled_205_cast_fp16)[name = string("out_103_cast_fp16")]; tensor var_9211_split_sizes_0 = const()[name = string("op_9211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9211_axis_0 = const()[name = string("op_9211_axis_0"), val = int32(1)]; tensor var_9211_cast_fp16_0, tensor var_9211_cast_fp16_1 = split(axis = var_9211_axis_0, split_sizes = var_9211_split_sizes_0, x = out_103_cast_fp16)[name = string("op_9211_cast_fp16")]; tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; tensor input_51_cast_fp16 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = layers_25_mlp_gate_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("input_51_cast_fp16")]; tensor var_9228_cast_fp16 = silu(x = input_51_cast_fp16)[name = string("op_9228_cast_fp16")]; tensor var_9234_strides_0 = const()[name = string("op_9234_strides_0"), val = tensor([1, 1])]; string var_9234_pad_type_0 = const()[name = string("op_9234_pad_type_0"), val = string("valid")]; tensor var_9234_pad_0 = const()[name = string("op_9234_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9234_dilations_0 = const()[name = string("op_9234_dilations_0"), val = tensor([1, 1])]; int32 var_9234_groups_0 = const()[name = string("op_9234_groups_0"), val = int32(1)]; tensor var_9234_cast_fp16 = conv(dilations = var_9234_dilations_0, groups = var_9234_groups_0, pad = var_9234_pad_0, pad_type = var_9234_pad_type_0, strides = var_9234_strides_0, weight = layers_25_mlp_up_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("op_9234_cast_fp16")]; tensor x_259_cast_fp16 = mul(x = var_9228_cast_fp16, y = var_9234_cast_fp16)[name = string("x_259_cast_fp16")]; tensor layers_25_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_25_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541435840)))]; tensor hidden_states_257_strides_0 = const()[name = string("hidden_states_257_strides_0"), val = tensor([1, 1])]; string hidden_states_257_pad_type_0 = const()[name = string("hidden_states_257_pad_type_0"), val = string("valid")]; tensor hidden_states_257_pad_0 = const()[name = string("hidden_states_257_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_257_dilations_0 = const()[name = string("hidden_states_257_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_257_groups_0 = const()[name = string("hidden_states_257_groups_0"), val = int32(1)]; tensor hidden_states_257_cast_fp16 = conv(dilations = hidden_states_257_dilations_0, groups = hidden_states_257_groups_0, pad = hidden_states_257_pad_0, pad_type = hidden_states_257_pad_type_0, strides = hidden_states_257_strides_0, weight = layers_25_mlp_down_proj_weight_to_fp16, x = x_259_cast_fp16)[name = string("hidden_states_257_cast_fp16")]; tensor hidden_states_259_cast_fp16 = add(x = hidden_states_255_cast_fp16, y = hidden_states_257_cast_fp16)[name = string("hidden_states_259_cast_fp16")]; fp16 const_260_promoted_to_fp16 = const()[name = string("const_260_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9252_cast_fp16 = mul(x = hidden_states_259_cast_fp16, y = const_260_promoted_to_fp16)[name = string("op_9252_cast_fp16")]; int32 var_9250 = const()[name = string("op_9250"), val = int32(1)]; bool doubled_209_interleave_0 = const()[name = string("doubled_209_interleave_0"), val = bool(false)]; tensor doubled_209_cast_fp16 = concat(axis = var_9250, interleave = doubled_209_interleave_0, values = (hidden_states_259_cast_fp16, var_9252_cast_fp16))[name = string("doubled_209_cast_fp16")]; tensor out_105_axes_0 = const()[name = string("out_105_axes_0"), val = tensor([1])]; tensor out_105_gamma_0_to_fp16 = const()[name = string("out_105_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566601728)))]; fp16 var_9262_to_fp16 = const()[name = string("op_9262_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_105_cast_fp16 = layer_norm(axes = out_105_axes_0, epsilon = var_9262_to_fp16, gamma = out_105_gamma_0_to_fp16, x = doubled_209_cast_fp16)[name = string("out_105_cast_fp16")]; tensor var_9273_split_sizes_0 = const()[name = string("op_9273_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9273_axis_0 = const()[name = string("op_9273_axis_0"), val = int32(1)]; tensor var_9273_cast_fp16_0, tensor var_9273_cast_fp16_1 = split(axis = var_9273_axis_0, split_sizes = var_9273_split_sizes_0, x = out_105_cast_fp16)[name = string("op_9273_cast_fp16")]; tensor query_states_157_strides_0 = const()[name = string("query_states_157_strides_0"), val = tensor([1, 1])]; string query_states_157_pad_type_0 = const()[name = string("query_states_157_pad_type_0"), val = string("valid")]; tensor query_states_157_pad_0 = const()[name = string("query_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_157_dilations_0 = const()[name = string("query_states_157_dilations_0"), val = tensor([1, 1])]; int32 query_states_157_groups_0 = const()[name = string("query_states_157_groups_0"), val = int32(1)]; tensor query_states_157_cast_fp16 = conv(dilations = query_states_157_dilations_0, groups = query_states_157_groups_0, pad = query_states_157_pad_0, pad_type = query_states_157_pad_type_0, strides = query_states_157_strides_0, weight = layers_26_self_attn_q_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("query_states_157_cast_fp16")]; tensor key_states_261_strides_0 = const()[name = string("key_states_261_strides_0"), val = tensor([1, 1])]; string key_states_261_pad_type_0 = const()[name = string("key_states_261_pad_type_0"), val = string("valid")]; tensor key_states_261_pad_0 = const()[name = string("key_states_261_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_261_dilations_0 = const()[name = string("key_states_261_dilations_0"), val = tensor([1, 1])]; int32 key_states_261_groups_0 = const()[name = string("key_states_261_groups_0"), val = int32(1)]; tensor key_states_261_cast_fp16 = conv(dilations = key_states_261_dilations_0, groups = key_states_261_groups_0, pad = key_states_261_pad_0, pad_type = key_states_261_pad_type_0, strides = key_states_261_strides_0, weight = layers_26_self_attn_k_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("key_states_261_cast_fp16")]; tensor layers_26_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_26_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566609984)))]; tensor value_states_157_strides_0 = const()[name = string("value_states_157_strides_0"), val = tensor([1, 1])]; string value_states_157_pad_type_0 = const()[name = string("value_states_157_pad_type_0"), val = string("valid")]; tensor value_states_157_pad_0 = const()[name = string("value_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_157_dilations_0 = const()[name = string("value_states_157_dilations_0"), val = tensor([1, 1])]; int32 value_states_157_groups_0 = const()[name = string("value_states_157_groups_0"), val = int32(1)]; tensor value_states_157_cast_fp16 = conv(dilations = value_states_157_dilations_0, groups = value_states_157_groups_0, pad = value_states_157_pad_0, pad_type = value_states_157_pad_type_0, strides = value_states_157_strides_0, weight = layers_26_self_attn_v_proj_weight_to_fp16, x = var_9273_cast_fp16_0)[name = string("value_states_157_cast_fp16")]; tensor concat_312x = const()[name = string("concat_312x"), val = tensor([1, 16, 128, -1])]; tensor x_261_cast_fp16 = reshape(shape = concat_312x, x = query_states_157_cast_fp16)[name = string("x_261_cast_fp16")]; tensor concat_313x = const()[name = string("concat_313x"), val = tensor([1, 2, 128, -1])]; tensor var_9330_cast_fp16 = reshape(shape = concat_313x, x = key_states_261_cast_fp16)[name = string("op_9330_cast_fp16")]; tensor concat_314x = const()[name = string("concat_314x"), val = tensor([1, 2, 128, -1])]; tensor var_9337_cast_fp16 = reshape(shape = concat_314x, x = value_states_157_cast_fp16)[name = string("op_9337_cast_fp16")]; tensor var_9341_cast_fp16 = mul(x = x_261_cast_fp16, y = var_869_cast_fp16)[name = string("op_9341_cast_fp16")]; tensor var_9342_split_sizes_0 = const()[name = string("op_9342_split_sizes_0"), val = tensor([64, 64])]; int32 var_9342_axis_0 = const()[name = string("op_9342_axis_0"), val = int32(-2)]; tensor var_9342_cast_fp16_0, tensor var_9342_cast_fp16_1 = split(axis = var_9342_axis_0, split_sizes = var_9342_split_sizes_0, x = x_261_cast_fp16)[name = string("op_9342_cast_fp16")]; fp16 const_262_promoted_to_fp16 = const()[name = string("const_262_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9344_cast_fp16 = mul(x = var_9342_cast_fp16_1, y = const_262_promoted_to_fp16)[name = string("op_9344_cast_fp16")]; int32 var_9346 = const()[name = string("op_9346"), val = int32(-2)]; bool var_9347_interleave_0 = const()[name = string("op_9347_interleave_0"), val = bool(false)]; tensor var_9347_cast_fp16 = concat(axis = var_9346, interleave = var_9347_interleave_0, values = (var_9344_cast_fp16, var_9342_cast_fp16_0))[name = string("op_9347_cast_fp16")]; tensor var_9348_cast_fp16 = mul(x = var_9347_cast_fp16, y = var_878_cast_fp16)[name = string("op_9348_cast_fp16")]; tensor query_states_159_cast_fp16 = add(x = var_9341_cast_fp16, y = var_9348_cast_fp16)[name = string("query_states_159_cast_fp16")]; tensor var_9354_cast_fp16 = mul(x = var_9330_cast_fp16, y = var_869_cast_fp16)[name = string("op_9354_cast_fp16")]; tensor var_9355_split_sizes_0 = const()[name = string("op_9355_split_sizes_0"), val = tensor([64, 64])]; int32 var_9355_axis_0 = const()[name = string("op_9355_axis_0"), val = int32(-2)]; tensor var_9355_cast_fp16_0, tensor var_9355_cast_fp16_1 = split(axis = var_9355_axis_0, split_sizes = var_9355_split_sizes_0, x = var_9330_cast_fp16)[name = string("op_9355_cast_fp16")]; fp16 const_263_promoted_to_fp16 = const()[name = string("const_263_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9357_cast_fp16 = mul(x = var_9355_cast_fp16_1, y = const_263_promoted_to_fp16)[name = string("op_9357_cast_fp16")]; int32 var_9359 = const()[name = string("op_9359"), val = int32(-2)]; bool var_9360_interleave_0 = const()[name = string("op_9360_interleave_0"), val = bool(false)]; tensor var_9360_cast_fp16 = concat(axis = var_9359, interleave = var_9360_interleave_0, values = (var_9357_cast_fp16, var_9355_cast_fp16_0))[name = string("op_9360_cast_fp16")]; tensor var_9361_cast_fp16 = mul(x = var_9360_cast_fp16, y = var_878_cast_fp16)[name = string("op_9361_cast_fp16")]; tensor key_states_265_cast_fp16 = add(x = var_9354_cast_fp16, y = var_9361_cast_fp16)[name = string("key_states_265_cast_fp16")]; tensor expand_dims_312 = const()[name = string("expand_dims_312"), val = tensor([26])]; tensor expand_dims_313 = const()[name = string("expand_dims_313"), val = tensor([0])]; tensor expand_dims_315 = const()[name = string("expand_dims_315"), val = tensor([0])]; int32 concat_317_axis_0 = const()[name = string("concat_317_axis_0"), val = int32(0)]; bool concat_317_interleave_0 = const()[name = string("concat_317_interleave_0"), val = bool(false)]; tensor concat_317 = concat(axis = concat_317_axis_0, interleave = concat_317_interleave_0, values = (expand_dims_312, expand_dims_313, position_id, expand_dims_315))[name = string("concat_317")]; tensor expand_dims_316 = const()[name = string("expand_dims_316"), val = tensor([27])]; tensor concat_318_values1_0 = const()[name = string("concat_318_values1_0"), val = tensor([0])]; tensor concat_318_values3_0 = const()[name = string("concat_318_values3_0"), val = tensor([0])]; int32 concat_318_axis_0 = const()[name = string("concat_318_axis_0"), val = int32(0)]; bool concat_318_interleave_0 = const()[name = string("concat_318_interleave_0"), val = bool(false)]; tensor concat_318 = concat(axis = concat_318_axis_0, interleave = concat_318_interleave_0, values = (expand_dims_316, concat_318_values1_0, cache_position_end, concat_318_values3_0))[name = string("concat_318")]; tensor key_states_267_perm_0 = const()[name = string("key_states_267_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_27_stride_0 = const()[name = string("key_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_267_cast_fp16 = transpose(perm = key_states_267_perm_0, x = key_states_265_cast_fp16)[name = string("transpose_607")]; tensor key_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = key_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = key_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_27_squeeze_mask_0, stride = key_cache_internal_tensor_assign_27_stride_0, update = key_states_267_cast_fp16, x = coreml_update_state_442)[name = string("key_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_27_cast_fp16, input = key_cache)[name = string("coreml_update_state_444_write_state")]; tensor coreml_update_state_444 = read_state(input = key_cache)[name = string("coreml_update_state_444")]; tensor value_states_159_perm_0 = const()[name = string("value_states_159_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_27_stride_0 = const()[name = string("value_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_159_cast_fp16 = transpose(perm = value_states_159_perm_0, x = var_9337_cast_fp16)[name = string("transpose_606")]; tensor value_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = value_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = value_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_27_squeeze_mask_0, stride = value_cache_internal_tensor_assign_27_stride_0, update = value_states_159_cast_fp16, x = coreml_update_state_443)[name = string("value_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_27_cast_fp16, input = value_cache)[name = string("coreml_update_state_445_write_state")]; tensor coreml_update_state_445 = read_state(input = value_cache)[name = string("coreml_update_state_445")]; tensor var_9431_begin_0 = const()[name = string("op_9431_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9431_end_0 = const()[name = string("op_9431_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9431_end_mask_0 = const()[name = string("op_9431_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9431_cast_fp16 = slice_by_index(begin = var_9431_begin_0, end = var_9431_end_0, end_mask = var_9431_end_mask_0, x = coreml_update_state_444)[name = string("op_9431_cast_fp16")]; tensor tile_52 = const()[name = string("tile_52"), val = tensor([1, 1])]; int32 var_9434_axis_0 = const()[name = string("op_9434_axis_0"), val = int32(1)]; tensor var_9434_cast_fp16_0, tensor var_9434_cast_fp16_1 = split(axis = var_9434_axis_0, split_sizes = tile_52, x = var_9431_cast_fp16)[name = string("op_9434_cast_fp16")]; tensor var_9441_begin_0 = const()[name = string("op_9441_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9441_end_0 = const()[name = string("op_9441_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9441_end_mask_0 = const()[name = string("op_9441_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9441_cast_fp16 = slice_by_index(begin = var_9441_begin_0, end = var_9441_end_0, end_mask = var_9441_end_mask_0, x = coreml_update_state_445)[name = string("op_9441_cast_fp16")]; tensor tile_53 = const()[name = string("tile_53"), val = tensor([1, 1])]; int32 var_9444_axis_0 = const()[name = string("op_9444_axis_0"), val = int32(1)]; tensor var_9444_cast_fp16_0, tensor var_9444_cast_fp16_1 = split(axis = var_9444_axis_0, split_sizes = tile_53, x = var_9441_cast_fp16)[name = string("op_9444_cast_fp16")]; tensor var_9447_split_sizes_0 = const()[name = string("op_9447_split_sizes_0"), val = tensor([8, 8])]; int32 var_9447_axis_0 = const()[name = string("op_9447_axis_0"), val = int32(1)]; tensor var_9447_0, tensor var_9447_1 = split(axis = var_9447_axis_0, split_sizes = var_9447_split_sizes_0, x = query_states_159_cast_fp16)[name = string("op_9447")]; bool attn_weights_417_transpose_x_0 = const()[name = string("attn_weights_417_transpose_x_0"), val = bool(false)]; bool attn_weights_417_transpose_y_0 = const()[name = string("attn_weights_417_transpose_y_0"), val = bool(false)]; tensor attn_weights_417_cast_fp16 = matmul(transpose_x = attn_weights_417_transpose_x_0, transpose_y = attn_weights_417_transpose_y_0, x = var_9434_cast_fp16_0, y = var_9447_0)[name = string("attn_weights_417_cast_fp16")]; fp16 var_9450_to_fp16 = const()[name = string("op_9450_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_419_cast_fp16 = mul(x = attn_weights_417_cast_fp16, y = var_9450_to_fp16)[name = string("attn_weights_419_cast_fp16")]; tensor attn_weights_421_cast_fp16 = add(x = attn_weights_419_cast_fp16, y = attn_mask_1)[name = string("attn_weights_421_cast_fp16")]; int32 var_9454 = const()[name = string("op_9454"), val = int32(-2)]; tensor attn_weights_423_cast_fp16 = softmax(axis = var_9454, x = attn_weights_421_cast_fp16)[name = string("attn_weights_423_cast_fp16")]; bool var_9460_transpose_x_1 = const()[name = string("op_9460_transpose_x_1"), val = bool(true)]; bool var_9460_transpose_y_1 = const()[name = string("op_9460_transpose_y_1"), val = bool(false)]; tensor var_9460_cast_fp16 = matmul(transpose_x = var_9460_transpose_x_1, transpose_y = var_9460_transpose_y_1, x = attn_weights_423_cast_fp16, y = var_9444_cast_fp16_0)[name = string("op_9460_cast_fp16")]; bool attn_weights_425_transpose_x_0 = const()[name = string("attn_weights_425_transpose_x_0"), val = bool(false)]; bool attn_weights_425_transpose_y_0 = const()[name = string("attn_weights_425_transpose_y_0"), val = bool(false)]; tensor attn_weights_425_cast_fp16 = matmul(transpose_x = attn_weights_425_transpose_x_0, transpose_y = attn_weights_425_transpose_y_0, x = var_9434_cast_fp16_1, y = var_9447_1)[name = string("attn_weights_425_cast_fp16")]; fp16 var_9462_to_fp16 = const()[name = string("op_9462_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_427_cast_fp16 = mul(x = attn_weights_425_cast_fp16, y = var_9462_to_fp16)[name = string("attn_weights_427_cast_fp16")]; tensor attn_weights_429_cast_fp16 = add(x = attn_weights_427_cast_fp16, y = attn_mask_1)[name = string("attn_weights_429_cast_fp16")]; int32 var_9466 = const()[name = string("op_9466"), val = int32(-2)]; tensor attn_weights_431_cast_fp16 = softmax(axis = var_9466, x = attn_weights_429_cast_fp16)[name = string("attn_weights_431_cast_fp16")]; bool attn_output_209_transpose_x_1 = const()[name = string("attn_output_209_transpose_x_1"), val = bool(true)]; bool attn_output_209_transpose_y_1 = const()[name = string("attn_output_209_transpose_y_1"), val = bool(false)]; tensor attn_output_209_cast_fp16 = matmul(transpose_x = attn_output_209_transpose_x_1, transpose_y = attn_output_209_transpose_y_1, x = attn_weights_431_cast_fp16, y = var_9444_cast_fp16_1)[name = string("attn_output_209_cast_fp16")]; int32 var_9474 = const()[name = string("op_9474"), val = int32(1)]; bool attn_output_211_interleave_0 = const()[name = string("attn_output_211_interleave_0"), val = bool(false)]; tensor attn_output_211_cast_fp16 = concat(axis = var_9474, interleave = attn_output_211_interleave_0, values = (var_9460_cast_fp16, attn_output_209_cast_fp16))[name = string("attn_output_211_cast_fp16")]; tensor var_9478_perm_0 = const()[name = string("op_9478_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_323x = const()[name = string("concat_323x"), val = tensor([1, 2048, 1, -1])]; tensor var_9478_cast_fp16 = transpose(perm = var_9478_perm_0, x = attn_output_211_cast_fp16)[name = string("transpose_605")]; tensor attn_output_215_cast_fp16 = reshape(shape = concat_323x, x = var_9478_cast_fp16)[name = string("attn_output_215_cast_fp16")]; tensor hidden_states_263_strides_0 = const()[name = string("hidden_states_263_strides_0"), val = tensor([1, 1])]; string hidden_states_263_pad_type_0 = const()[name = string("hidden_states_263_pad_type_0"), val = string("valid")]; tensor hidden_states_263_pad_0 = const()[name = string("hidden_states_263_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_263_dilations_0 = const()[name = string("hidden_states_263_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_263_groups_0 = const()[name = string("hidden_states_263_groups_0"), val = int32(1)]; tensor hidden_states_263_cast_fp16 = conv(dilations = hidden_states_263_dilations_0, groups = hidden_states_263_groups_0, pad = hidden_states_263_pad_0, pad_type = hidden_states_263_pad_type_0, strides = hidden_states_263_strides_0, weight = layers_26_self_attn_o_proj_weight_cast_fp16, x = attn_output_215_cast_fp16)[name = string("hidden_states_263_cast_fp16")]; tensor hidden_states_265_cast_fp16 = add(x = hidden_states_259_cast_fp16, y = hidden_states_263_cast_fp16)[name = string("hidden_states_265_cast_fp16")]; fp16 const_268_promoted_to_fp16 = const()[name = string("const_268_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9511_cast_fp16 = mul(x = hidden_states_265_cast_fp16, y = const_268_promoted_to_fp16)[name = string("op_9511_cast_fp16")]; int32 var_9509 = const()[name = string("op_9509"), val = int32(1)]; bool doubled_213_interleave_0 = const()[name = string("doubled_213_interleave_0"), val = bool(false)]; tensor doubled_213_cast_fp16 = concat(axis = var_9509, interleave = doubled_213_interleave_0, values = (hidden_states_265_cast_fp16, var_9511_cast_fp16))[name = string("doubled_213_cast_fp16")]; tensor out_107_axes_0 = const()[name = string("out_107_axes_0"), val = tensor([1])]; tensor out_107_gamma_0_to_fp16 = const()[name = string("out_107_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567658624)))]; fp16 var_9521_to_fp16 = const()[name = string("op_9521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_107_cast_fp16 = layer_norm(axes = out_107_axes_0, epsilon = var_9521_to_fp16, gamma = out_107_gamma_0_to_fp16, x = doubled_213_cast_fp16)[name = string("out_107_cast_fp16")]; tensor var_9532_split_sizes_0 = const()[name = string("op_9532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9532_axis_0 = const()[name = string("op_9532_axis_0"), val = int32(1)]; tensor var_9532_cast_fp16_0, tensor var_9532_cast_fp16_1 = split(axis = var_9532_axis_0, split_sizes = var_9532_split_sizes_0, x = out_107_cast_fp16)[name = string("op_9532_cast_fp16")]; tensor input_53_strides_0 = const()[name = string("input_53_strides_0"), val = tensor([1, 1])]; string input_53_pad_type_0 = const()[name = string("input_53_pad_type_0"), val = string("valid")]; tensor input_53_pad_0 = const()[name = string("input_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_53_dilations_0 = const()[name = string("input_53_dilations_0"), val = tensor([1, 1])]; int32 input_53_groups_0 = const()[name = string("input_53_groups_0"), val = int32(1)]; tensor input_53_cast_fp16 = conv(dilations = input_53_dilations_0, groups = input_53_groups_0, pad = input_53_pad_0, pad_type = input_53_pad_type_0, strides = input_53_strides_0, weight = layers_26_mlp_gate_proj_weight_cast_fp16, x = var_9532_cast_fp16_0)[name = string("input_53_cast_fp16")]; tensor var_9549_cast_fp16 = silu(x = input_53_cast_fp16)[name = string("op_9549_cast_fp16")]; tensor layers_26_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567666880)))]; tensor var_9555_strides_0 = const()[name = string("op_9555_strides_0"), val = tensor([1, 1])]; string var_9555_pad_type_0 = const()[name = string("op_9555_pad_type_0"), val = string("valid")]; tensor var_9555_pad_0 = const()[name = string("op_9555_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9555_dilations_0 = const()[name = string("op_9555_dilations_0"), val = tensor([1, 1])]; int32 var_9555_groups_0 = const()[name = string("op_9555_groups_0"), val = int32(1)]; tensor var_9555_cast_fp16 = conv(dilations = var_9555_dilations_0, groups = var_9555_groups_0, pad = var_9555_pad_0, pad_type = var_9555_pad_type_0, strides = var_9555_strides_0, weight = layers_26_mlp_up_proj_weight_to_fp16, x = var_9532_cast_fp16_0)[name = string("op_9555_cast_fp16")]; tensor x_269_cast_fp16 = mul(x = var_9549_cast_fp16, y = var_9555_cast_fp16)[name = string("x_269_cast_fp16")]; tensor layers_26_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1592832768)))]; tensor hidden_states_267_strides_0 = const()[name = string("hidden_states_267_strides_0"), val = tensor([1, 1])]; string hidden_states_267_pad_type_0 = const()[name = string("hidden_states_267_pad_type_0"), val = string("valid")]; tensor hidden_states_267_pad_0 = const()[name = string("hidden_states_267_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_267_dilations_0 = const()[name = string("hidden_states_267_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_267_groups_0 = const()[name = string("hidden_states_267_groups_0"), val = int32(1)]; tensor hidden_states_267_cast_fp16 = conv(dilations = hidden_states_267_dilations_0, groups = hidden_states_267_groups_0, pad = hidden_states_267_pad_0, pad_type = hidden_states_267_pad_type_0, strides = hidden_states_267_strides_0, weight = layers_26_mlp_down_proj_weight_to_fp16, x = x_269_cast_fp16)[name = string("hidden_states_267_cast_fp16")]; tensor hidden_states_269_cast_fp16 = add(x = hidden_states_265_cast_fp16, y = hidden_states_267_cast_fp16)[name = string("hidden_states_269_cast_fp16")]; fp16 const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9573_cast_fp16 = mul(x = hidden_states_269_cast_fp16, y = const_270_promoted_to_fp16)[name = string("op_9573_cast_fp16")]; int32 var_9571 = const()[name = string("op_9571"), val = int32(1)]; bool doubled_217_interleave_0 = const()[name = string("doubled_217_interleave_0"), val = bool(false)]; tensor doubled_217_cast_fp16 = concat(axis = var_9571, interleave = doubled_217_interleave_0, values = (hidden_states_269_cast_fp16, var_9573_cast_fp16))[name = string("doubled_217_cast_fp16")]; tensor out_109_axes_0 = const()[name = string("out_109_axes_0"), val = tensor([1])]; tensor out_109_gamma_0_to_fp16 = const()[name = string("out_109_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1617998656)))]; fp16 var_9583_to_fp16 = const()[name = string("op_9583_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_109_cast_fp16 = layer_norm(axes = out_109_axes_0, epsilon = var_9583_to_fp16, gamma = out_109_gamma_0_to_fp16, x = doubled_217_cast_fp16)[name = string("out_109_cast_fp16")]; tensor var_9594_split_sizes_0 = const()[name = string("op_9594_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9594_axis_0 = const()[name = string("op_9594_axis_0"), val = int32(1)]; tensor var_9594_cast_fp16_0, tensor var_9594_cast_fp16_1 = split(axis = var_9594_axis_0, split_sizes = var_9594_split_sizes_0, x = out_109_cast_fp16)[name = string("op_9594_cast_fp16")]; tensor layers_27_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1618006912)))]; tensor query_states_163_strides_0 = const()[name = string("query_states_163_strides_0"), val = tensor([1, 1])]; string query_states_163_pad_type_0 = const()[name = string("query_states_163_pad_type_0"), val = string("valid")]; tensor query_states_163_pad_0 = const()[name = string("query_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_163_dilations_0 = const()[name = string("query_states_163_dilations_0"), val = tensor([1, 1])]; int32 query_states_163_groups_0 = const()[name = string("query_states_163_groups_0"), val = int32(1)]; tensor query_states_163_cast_fp16 = conv(dilations = query_states_163_dilations_0, groups = query_states_163_groups_0, pad = query_states_163_pad_0, pad_type = query_states_163_pad_type_0, strides = query_states_163_strides_0, weight = layers_27_self_attn_q_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("query_states_163_cast_fp16")]; tensor layers_27_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1626395584)))]; tensor key_states_271_strides_0 = const()[name = string("key_states_271_strides_0"), val = tensor([1, 1])]; string key_states_271_pad_type_0 = const()[name = string("key_states_271_pad_type_0"), val = string("valid")]; tensor key_states_271_pad_0 = const()[name = string("key_states_271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_271_dilations_0 = const()[name = string("key_states_271_dilations_0"), val = tensor([1, 1])]; int32 key_states_271_groups_0 = const()[name = string("key_states_271_groups_0"), val = int32(1)]; tensor key_states_271_cast_fp16 = conv(dilations = key_states_271_dilations_0, groups = key_states_271_groups_0, pad = key_states_271_pad_0, pad_type = key_states_271_pad_type_0, strides = key_states_271_strides_0, weight = layers_27_self_attn_k_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("key_states_271_cast_fp16")]; tensor layers_27_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1627444224)))]; tensor value_states_163_strides_0 = const()[name = string("value_states_163_strides_0"), val = tensor([1, 1])]; string value_states_163_pad_type_0 = const()[name = string("value_states_163_pad_type_0"), val = string("valid")]; tensor value_states_163_pad_0 = const()[name = string("value_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_163_dilations_0 = const()[name = string("value_states_163_dilations_0"), val = tensor([1, 1])]; int32 value_states_163_groups_0 = const()[name = string("value_states_163_groups_0"), val = int32(1)]; tensor value_states_163_cast_fp16 = conv(dilations = value_states_163_dilations_0, groups = value_states_163_groups_0, pad = value_states_163_pad_0, pad_type = value_states_163_pad_type_0, strides = value_states_163_strides_0, weight = layers_27_self_attn_v_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("value_states_163_cast_fp16")]; tensor concat_324x = const()[name = string("concat_324x"), val = tensor([1, 16, 128, -1])]; tensor x_271_cast_fp16 = reshape(shape = concat_324x, x = query_states_163_cast_fp16)[name = string("x_271_cast_fp16")]; tensor concat_325x = const()[name = string("concat_325x"), val = tensor([1, 2, 128, -1])]; tensor var_9651_cast_fp16 = reshape(shape = concat_325x, x = key_states_271_cast_fp16)[name = string("op_9651_cast_fp16")]; tensor concat_326x = const()[name = string("concat_326x"), val = tensor([1, 2, 128, -1])]; tensor var_9658_cast_fp16 = reshape(shape = concat_326x, x = value_states_163_cast_fp16)[name = string("op_9658_cast_fp16")]; tensor var_9662_cast_fp16 = mul(x = x_271_cast_fp16, y = var_869_cast_fp16)[name = string("op_9662_cast_fp16")]; tensor var_9663_split_sizes_0 = const()[name = string("op_9663_split_sizes_0"), val = tensor([64, 64])]; int32 var_9663_axis_0 = const()[name = string("op_9663_axis_0"), val = int32(-2)]; tensor var_9663_cast_fp16_0, tensor var_9663_cast_fp16_1 = split(axis = var_9663_axis_0, split_sizes = var_9663_split_sizes_0, x = x_271_cast_fp16)[name = string("op_9663_cast_fp16")]; fp16 const_272_promoted_to_fp16 = const()[name = string("const_272_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9665_cast_fp16 = mul(x = var_9663_cast_fp16_1, y = const_272_promoted_to_fp16)[name = string("op_9665_cast_fp16")]; int32 var_9667 = const()[name = string("op_9667"), val = int32(-2)]; bool var_9668_interleave_0 = const()[name = string("op_9668_interleave_0"), val = bool(false)]; tensor var_9668_cast_fp16 = concat(axis = var_9667, interleave = var_9668_interleave_0, values = (var_9665_cast_fp16, var_9663_cast_fp16_0))[name = string("op_9668_cast_fp16")]; tensor var_9669_cast_fp16 = mul(x = var_9668_cast_fp16, y = var_878_cast_fp16)[name = string("op_9669_cast_fp16")]; tensor query_states_165_cast_fp16 = add(x = var_9662_cast_fp16, y = var_9669_cast_fp16)[name = string("query_states_165_cast_fp16")]; tensor var_9675_cast_fp16 = mul(x = var_9651_cast_fp16, y = var_869_cast_fp16)[name = string("op_9675_cast_fp16")]; tensor var_9676_split_sizes_0 = const()[name = string("op_9676_split_sizes_0"), val = tensor([64, 64])]; int32 var_9676_axis_0 = const()[name = string("op_9676_axis_0"), val = int32(-2)]; tensor var_9676_cast_fp16_0, tensor var_9676_cast_fp16_1 = split(axis = var_9676_axis_0, split_sizes = var_9676_split_sizes_0, x = var_9651_cast_fp16)[name = string("op_9676_cast_fp16")]; fp16 const_273_promoted_to_fp16 = const()[name = string("const_273_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9678_cast_fp16 = mul(x = var_9676_cast_fp16_1, y = const_273_promoted_to_fp16)[name = string("op_9678_cast_fp16")]; int32 var_9680 = const()[name = string("op_9680"), val = int32(-2)]; bool var_9681_interleave_0 = const()[name = string("op_9681_interleave_0"), val = bool(false)]; tensor var_9681_cast_fp16 = concat(axis = var_9680, interleave = var_9681_interleave_0, values = (var_9678_cast_fp16, var_9676_cast_fp16_0))[name = string("op_9681_cast_fp16")]; tensor var_9682_cast_fp16 = mul(x = var_9681_cast_fp16, y = var_878_cast_fp16)[name = string("op_9682_cast_fp16")]; tensor key_states_275_cast_fp16 = add(x = var_9675_cast_fp16, y = var_9682_cast_fp16)[name = string("key_states_275_cast_fp16")]; tensor expand_dims_324 = const()[name = string("expand_dims_324"), val = tensor([27])]; tensor expand_dims_325 = const()[name = string("expand_dims_325"), val = tensor([0])]; tensor expand_dims_327 = const()[name = string("expand_dims_327"), val = tensor([0])]; int32 concat_329_axis_0 = const()[name = string("concat_329_axis_0"), val = int32(0)]; bool concat_329_interleave_0 = const()[name = string("concat_329_interleave_0"), val = bool(false)]; tensor concat_329 = concat(axis = concat_329_axis_0, interleave = concat_329_interleave_0, values = (expand_dims_324, expand_dims_325, position_id, expand_dims_327))[name = string("concat_329")]; tensor expand_dims_328 = const()[name = string("expand_dims_328"), val = tensor([28])]; tensor concat_330_values1_0 = const()[name = string("concat_330_values1_0"), val = tensor([0])]; tensor concat_330_values3_0 = const()[name = string("concat_330_values3_0"), val = tensor([0])]; int32 concat_330_axis_0 = const()[name = string("concat_330_axis_0"), val = int32(0)]; bool concat_330_interleave_0 = const()[name = string("concat_330_interleave_0"), val = bool(false)]; tensor concat_330 = concat(axis = concat_330_axis_0, interleave = concat_330_interleave_0, values = (expand_dims_328, concat_330_values1_0, cache_position_end, concat_330_values3_0))[name = string("concat_330")]; tensor key_states_277_perm_0 = const()[name = string("key_states_277_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_28_stride_0 = const()[name = string("key_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_277_cast_fp16 = transpose(perm = key_states_277_perm_0, x = key_states_275_cast_fp16)[name = string("transpose_604")]; tensor key_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = key_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = key_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_28_squeeze_mask_0, stride = key_cache_internal_tensor_assign_28_stride_0, update = key_states_277_cast_fp16, x = coreml_update_state_444)[name = string("key_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_28_cast_fp16, input = key_cache)[name = string("coreml_update_state_446_write_state")]; tensor coreml_update_state_446 = read_state(input = key_cache)[name = string("coreml_update_state_446")]; tensor value_states_165_perm_0 = const()[name = string("value_states_165_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_28_stride_0 = const()[name = string("value_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_165_cast_fp16 = transpose(perm = value_states_165_perm_0, x = var_9658_cast_fp16)[name = string("transpose_603")]; tensor value_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = value_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = value_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_28_squeeze_mask_0, stride = value_cache_internal_tensor_assign_28_stride_0, update = value_states_165_cast_fp16, x = coreml_update_state_445)[name = string("value_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_28_cast_fp16, input = value_cache)[name = string("coreml_update_state_447_write_state")]; tensor coreml_update_state_447 = read_state(input = value_cache)[name = string("coreml_update_state_447")]; tensor var_9752_begin_0 = const()[name = string("op_9752_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9752_end_0 = const()[name = string("op_9752_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9752_end_mask_0 = const()[name = string("op_9752_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9752_cast_fp16 = slice_by_index(begin = var_9752_begin_0, end = var_9752_end_0, end_mask = var_9752_end_mask_0, x = coreml_update_state_446)[name = string("op_9752_cast_fp16")]; tensor tile_54 = const()[name = string("tile_54"), val = tensor([1, 1])]; int32 var_9755_axis_0 = const()[name = string("op_9755_axis_0"), val = int32(1)]; tensor var_9755_cast_fp16_0, tensor var_9755_cast_fp16_1 = split(axis = var_9755_axis_0, split_sizes = tile_54, x = var_9752_cast_fp16)[name = string("op_9755_cast_fp16")]; tensor var_9762_begin_0 = const()[name = string("op_9762_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9762_end_0 = const()[name = string("op_9762_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9762_end_mask_0 = const()[name = string("op_9762_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9762_cast_fp16 = slice_by_index(begin = var_9762_begin_0, end = var_9762_end_0, end_mask = var_9762_end_mask_0, x = coreml_update_state_447)[name = string("op_9762_cast_fp16")]; tensor tile_55 = const()[name = string("tile_55"), val = tensor([1, 1])]; int32 var_9765_axis_0 = const()[name = string("op_9765_axis_0"), val = int32(1)]; tensor var_9765_cast_fp16_0, tensor var_9765_cast_fp16_1 = split(axis = var_9765_axis_0, split_sizes = tile_55, x = var_9762_cast_fp16)[name = string("op_9765_cast_fp16")]; tensor var_9768_split_sizes_0 = const()[name = string("op_9768_split_sizes_0"), val = tensor([8, 8])]; int32 var_9768_axis_0 = const()[name = string("op_9768_axis_0"), val = int32(1)]; tensor var_9768_0, tensor var_9768_1 = split(axis = var_9768_axis_0, split_sizes = var_9768_split_sizes_0, x = query_states_165_cast_fp16)[name = string("op_9768")]; bool attn_weights_433_transpose_x_0 = const()[name = string("attn_weights_433_transpose_x_0"), val = bool(false)]; bool attn_weights_433_transpose_y_0 = const()[name = string("attn_weights_433_transpose_y_0"), val = bool(false)]; tensor attn_weights_433_cast_fp16 = matmul(transpose_x = attn_weights_433_transpose_x_0, transpose_y = attn_weights_433_transpose_y_0, x = var_9755_cast_fp16_0, y = var_9768_0)[name = string("attn_weights_433_cast_fp16")]; fp16 var_9771_to_fp16 = const()[name = string("op_9771_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_435_cast_fp16 = mul(x = attn_weights_433_cast_fp16, y = var_9771_to_fp16)[name = string("attn_weights_435_cast_fp16")]; tensor attn_weights_437_cast_fp16 = add(x = attn_weights_435_cast_fp16, y = attn_mask_1)[name = string("attn_weights_437_cast_fp16")]; int32 var_9775 = const()[name = string("op_9775"), val = int32(-2)]; tensor attn_weights_439_cast_fp16 = softmax(axis = var_9775, x = attn_weights_437_cast_fp16)[name = string("attn_weights_439_cast_fp16")]; bool var_9781_transpose_x_1 = const()[name = string("op_9781_transpose_x_1"), val = bool(true)]; bool var_9781_transpose_y_1 = const()[name = string("op_9781_transpose_y_1"), val = bool(false)]; tensor var_9781_cast_fp16 = matmul(transpose_x = var_9781_transpose_x_1, transpose_y = var_9781_transpose_y_1, x = attn_weights_439_cast_fp16, y = var_9765_cast_fp16_0)[name = string("op_9781_cast_fp16")]; bool attn_weights_441_transpose_x_0 = const()[name = string("attn_weights_441_transpose_x_0"), val = bool(false)]; bool attn_weights_441_transpose_y_0 = const()[name = string("attn_weights_441_transpose_y_0"), val = bool(false)]; tensor attn_weights_441_cast_fp16 = matmul(transpose_x = attn_weights_441_transpose_x_0, transpose_y = attn_weights_441_transpose_y_0, x = var_9755_cast_fp16_1, y = var_9768_1)[name = string("attn_weights_441_cast_fp16")]; fp16 var_9783_to_fp16 = const()[name = string("op_9783_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_443_cast_fp16 = mul(x = attn_weights_441_cast_fp16, y = var_9783_to_fp16)[name = string("attn_weights_443_cast_fp16")]; tensor attn_weights_445_cast_fp16 = add(x = attn_weights_443_cast_fp16, y = attn_mask_1)[name = string("attn_weights_445_cast_fp16")]; int32 var_9787 = const()[name = string("op_9787"), val = int32(-2)]; tensor attn_weights_cast_fp16 = softmax(axis = var_9787, x = attn_weights_445_cast_fp16)[name = string("attn_weights_cast_fp16")]; bool attn_output_217_transpose_x_1 = const()[name = string("attn_output_217_transpose_x_1"), val = bool(true)]; bool attn_output_217_transpose_y_1 = const()[name = string("attn_output_217_transpose_y_1"), val = bool(false)]; tensor attn_output_217_cast_fp16 = matmul(transpose_x = attn_output_217_transpose_x_1, transpose_y = attn_output_217_transpose_y_1, x = attn_weights_cast_fp16, y = var_9765_cast_fp16_1)[name = string("attn_output_217_cast_fp16")]; int32 var_9795 = const()[name = string("op_9795"), val = int32(1)]; bool attn_output_219_interleave_0 = const()[name = string("attn_output_219_interleave_0"), val = bool(false)]; tensor attn_output_219_cast_fp16 = concat(axis = var_9795, interleave = attn_output_219_interleave_0, values = (var_9781_cast_fp16, attn_output_217_cast_fp16))[name = string("attn_output_219_cast_fp16")]; tensor var_9799_perm_0 = const()[name = string("op_9799_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_335x = const()[name = string("concat_335x"), val = tensor([1, 2048, 1, -1])]; tensor var_9799_cast_fp16 = transpose(perm = var_9799_perm_0, x = attn_output_219_cast_fp16)[name = string("transpose_602")]; tensor attn_output_cast_fp16 = reshape(shape = concat_335x, x = var_9799_cast_fp16)[name = string("attn_output_cast_fp16")]; tensor layers_27_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1628492864)))]; tensor hidden_states_273_strides_0 = const()[name = string("hidden_states_273_strides_0"), val = tensor([1, 1])]; string hidden_states_273_pad_type_0 = const()[name = string("hidden_states_273_pad_type_0"), val = string("valid")]; tensor hidden_states_273_pad_0 = const()[name = string("hidden_states_273_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_273_dilations_0 = const()[name = string("hidden_states_273_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_273_groups_0 = const()[name = string("hidden_states_273_groups_0"), val = int32(1)]; tensor hidden_states_273_cast_fp16 = conv(dilations = hidden_states_273_dilations_0, groups = hidden_states_273_groups_0, pad = hidden_states_273_pad_0, pad_type = hidden_states_273_pad_type_0, strides = hidden_states_273_strides_0, weight = layers_27_self_attn_o_proj_weight_to_fp16, x = attn_output_cast_fp16)[name = string("hidden_states_273_cast_fp16")]; tensor hidden_states_275_cast_fp16 = add(x = hidden_states_269_cast_fp16, y = hidden_states_273_cast_fp16)[name = string("hidden_states_275_cast_fp16")]; fp16 const_278_promoted_to_fp16 = const()[name = string("const_278_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9832_cast_fp16 = mul(x = hidden_states_275_cast_fp16, y = const_278_promoted_to_fp16)[name = string("op_9832_cast_fp16")]; int32 var_9830 = const()[name = string("op_9830"), val = int32(1)]; bool doubled_221_interleave_0 = const()[name = string("doubled_221_interleave_0"), val = bool(false)]; tensor doubled_221_cast_fp16 = concat(axis = var_9830, interleave = doubled_221_interleave_0, values = (hidden_states_275_cast_fp16, var_9832_cast_fp16))[name = string("doubled_221_cast_fp16")]; tensor out_111_axes_0 = const()[name = string("out_111_axes_0"), val = tensor([1])]; tensor out_111_gamma_0_to_fp16 = const()[name = string("out_111_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636881536)))]; fp16 var_9842_to_fp16 = const()[name = string("op_9842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_111_cast_fp16 = layer_norm(axes = out_111_axes_0, epsilon = var_9842_to_fp16, gamma = out_111_gamma_0_to_fp16, x = doubled_221_cast_fp16)[name = string("out_111_cast_fp16")]; tensor var_9853_split_sizes_0 = const()[name = string("op_9853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9853_axis_0 = const()[name = string("op_9853_axis_0"), val = int32(1)]; tensor var_9853_cast_fp16_0, tensor var_9853_cast_fp16_1 = split(axis = var_9853_axis_0, split_sizes = var_9853_split_sizes_0, x = out_111_cast_fp16)[name = string("op_9853_cast_fp16")]; tensor layers_27_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636889792)))]; tensor input_strides_0 = const()[name = string("input_strides_0"), val = tensor([1, 1])]; string input_pad_type_0 = const()[name = string("input_pad_type_0"), val = string("valid")]; tensor input_pad_0 = const()[name = string("input_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_dilations_0 = const()[name = string("input_dilations_0"), val = tensor([1, 1])]; int32 input_groups_0 = const()[name = string("input_groups_0"), val = int32(1)]; tensor input_cast_fp16 = conv(dilations = input_dilations_0, groups = input_groups_0, pad = input_pad_0, pad_type = input_pad_type_0, strides = input_strides_0, weight = layers_27_mlp_gate_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("input_cast_fp16")]; tensor var_9870_cast_fp16 = silu(x = input_cast_fp16)[name = string("op_9870_cast_fp16")]; tensor layers_27_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1662055680)))]; tensor var_9876_strides_0 = const()[name = string("op_9876_strides_0"), val = tensor([1, 1])]; string var_9876_pad_type_0 = const()[name = string("op_9876_pad_type_0"), val = string("valid")]; tensor var_9876_pad_0 = const()[name = string("op_9876_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9876_dilations_0 = const()[name = string("op_9876_dilations_0"), val = tensor([1, 1])]; int32 var_9876_groups_0 = const()[name = string("op_9876_groups_0"), val = int32(1)]; tensor var_9876_cast_fp16 = conv(dilations = var_9876_dilations_0, groups = var_9876_groups_0, pad = var_9876_pad_0, pad_type = var_9876_pad_type_0, strides = var_9876_strides_0, weight = layers_27_mlp_up_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("op_9876_cast_fp16")]; tensor x_cast_fp16 = mul(x = var_9870_cast_fp16, y = var_9876_cast_fp16)[name = string("x_cast_fp16")]; tensor layers_27_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1687221568)))]; tensor hidden_states_277_strides_0 = const()[name = string("hidden_states_277_strides_0"), val = tensor([1, 1])]; string hidden_states_277_pad_type_0 = const()[name = string("hidden_states_277_pad_type_0"), val = string("valid")]; tensor hidden_states_277_pad_0 = const()[name = string("hidden_states_277_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_277_dilations_0 = const()[name = string("hidden_states_277_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_277_groups_0 = const()[name = string("hidden_states_277_groups_0"), val = int32(1)]; tensor hidden_states_277_cast_fp16 = conv(dilations = hidden_states_277_dilations_0, groups = hidden_states_277_groups_0, pad = hidden_states_277_pad_0, pad_type = hidden_states_277_pad_type_0, strides = hidden_states_277_strides_0, weight = layers_27_mlp_down_proj_weight_to_fp16, x = x_cast_fp16)[name = string("hidden_states_277_cast_fp16")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_275_cast_fp16, y = hidden_states_277_cast_fp16)[name = string("hidden_states_cast_fp16")]; fp16 const_280_promoted_to_fp16 = const()[name = string("const_280_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9894_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_280_promoted_to_fp16)[name = string("op_9894_cast_fp16")]; int32 var_9892 = const()[name = string("op_9892"), val = int32(1)]; bool doubled_225_interleave_0 = const()[name = string("doubled_225_interleave_0"), val = bool(false)]; tensor doubled_225_cast_fp16 = concat(axis = var_9892, interleave = doubled_225_interleave_0, values = (hidden_states_cast_fp16, var_9894_cast_fp16))[name = string("doubled_225_cast_fp16")]; tensor out_axes_0 = const()[name = string("out_axes_0"), val = tensor([1])]; tensor out_gamma_0_to_fp16 = const()[name = string("out_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1712387456)))]; fp16 var_9904_to_fp16 = const()[name = string("op_9904_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_cast_fp16 = layer_norm(axes = out_axes_0, epsilon = var_9904_to_fp16, gamma = out_gamma_0_to_fp16, x = doubled_225_cast_fp16)[name = string("out_cast_fp16")]; tensor var_9915_split_sizes_0 = const()[name = string("op_9915_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9915_axis_0 = const()[name = string("op_9915_axis_0"), val = int32(1)]; tensor hidden_states, tensor var_9915_cast_fp16_1 = split(axis = var_9915_axis_0, split_sizes = var_9915_split_sizes_0, x = out_cast_fp16)[name = string("op_9915_cast_fp16")]; } -> (hidden_states); func length_32(tensor inputs_embeds, state> key_cache, tensor position_id, tensor position_index_seed, state> value_cache) { tensor layers_1_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524992))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524416))))[name = string("layers_1_self_attn_v_proj_weight_cast_fp16")]; tensor layers_1_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(525312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13120640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13108288))))[name = string("layers_1_mlp_up_proj_weight_cast_fp16")]; tensor layers_2_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13126848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651776))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651200))))[name = string("layers_2_self_attn_v_proj_weight_cast_fp16")]; tensor layers_2_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13652096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26247424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26235072))))[name = string("layers_2_mlp_up_proj_weight_cast_fp16")]; tensor layers_3_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26253632))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778560))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26777984))))[name = string("layers_3_self_attn_v_proj_weight_cast_fp16")]; tensor layers_3_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30977408))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30973248))))[name = string("layers_3_self_attn_o_proj_weight_cast_fp16")]; tensor layers_3_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30979520))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43566656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43562496))))[name = string("layers_3_mlp_down_proj_weight_cast_fp16")]; tensor layers_4_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43568768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093696))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093120))))[name = string("layers_4_self_attn_v_proj_weight_cast_fp16")]; tensor layers_4_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44094016))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48292544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48288384))))[name = string("layers_4_self_attn_o_proj_weight_cast_fp16")]; tensor layers_4_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48294656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60877632))))[name = string("layers_4_mlp_gate_proj_weight_cast_fp16")]; tensor layers_4_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60896192))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73491520))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73479168))))[name = string("layers_4_mlp_up_proj_weight_cast_fp16")]; tensor layers_4_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73497728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86084864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86080704))))[name = string("layers_4_mlp_down_proj_weight_cast_fp16")]; tensor layers_5_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86086976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611328))))[name = string("layers_5_self_attn_v_proj_weight_cast_fp16")]; tensor layers_5_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86612224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90810752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90806592))))[name = string("layers_5_self_attn_o_proj_weight_cast_fp16")]; tensor layers_5_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90812864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103408192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103395840))))[name = string("layers_5_mlp_up_proj_weight_cast_fp16")]; tensor layers_5_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103414400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116001536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115997376))))[name = string("layers_5_mlp_down_proj_weight_cast_fp16")]; tensor layers_6_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116003648))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528576))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528000))))[name = string("layers_6_self_attn_v_proj_weight_cast_fp16")]; tensor layers_6_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528896))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120727424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120723264))))[name = string("layers_6_self_attn_o_proj_weight_cast_fp16")]; tensor layers_6_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120729536))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133324864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133312512))))[name = string("layers_6_mlp_gate_proj_weight_cast_fp16")]; tensor layers_6_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133331072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145926400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145914048))))[name = string("layers_6_mlp_up_proj_weight_cast_fp16")]; tensor layers_6_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145932608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158519744))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158515584))))[name = string("layers_6_mlp_down_proj_weight_cast_fp16")]; tensor layers_7_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158521856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046208))))[name = string("layers_7_self_attn_v_proj_weight_cast_fp16")]; tensor layers_7_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159047104))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163245632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163241472))))[name = string("layers_7_self_attn_o_proj_weight_cast_fp16")]; tensor layers_7_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163247744))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175843072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175830720))))[name = string("layers_7_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175849280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374208))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176373632))))[name = string("layers_8_self_attn_v_proj_weight_cast_fp16")]; tensor layers_8_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374528))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180573056))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180568896))))[name = string("layers_8_self_attn_o_proj_weight_cast_fp16")]; tensor layers_8_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180575168))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193170496))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193158144))))[name = string("layers_8_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193176704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205772032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205759680))))[name = string("layers_8_mlp_up_proj_weight_cast_fp16")]; tensor layers_8_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205778240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218365376))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218361216))))[name = string("layers_8_mlp_down_proj_weight_cast_fp16")]; tensor layers_9_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218367488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218891840))))[name = string("layers_9_self_attn_v_proj_weight_cast_fp16")]; tensor layers_9_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223091264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223087104))))[name = string("layers_9_self_attn_o_proj_weight_cast_fp16")]; tensor layers_9_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223093376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235688704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235676352))))[name = string("layers_9_mlp_gate_proj_weight_cast_fp16")]; tensor layers_9_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235694912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248290240))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248277888))))[name = string("layers_9_mlp_up_proj_weight_cast_fp16")]; tensor layers_9_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248296448))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260883584))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260879424))))[name = string("layers_9_mlp_down_proj_weight_cast_fp16")]; tensor layers_10_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260885696))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410048))))[name = string("layers_10_self_attn_v_proj_weight_cast_fp16")]; tensor layers_10_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410944))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265609472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265605312))))[name = string("layers_10_self_attn_o_proj_weight_cast_fp16")]; tensor layers_10_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265611584))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278206912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278194560))))[name = string("layers_10_mlp_gate_proj_weight_cast_fp16")]; tensor layers_10_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278213120))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290808448))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290796096))))[name = string("layers_10_mlp_up_proj_weight_cast_fp16")]; tensor layers_10_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290814656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303401792))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303397632))))[name = string("layers_10_mlp_down_proj_weight_cast_fp16")]; tensor layers_11_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303403904))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307602432))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307598272))))[name = string("layers_11_self_attn_q_proj_weight_cast_fp16")]; tensor layers_11_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307604544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308128896))))[name = string("layers_11_self_attn_k_proj_weight_cast_fp16")]; tensor layers_11_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129792))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654720))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654144))))[name = string("layers_11_self_attn_v_proj_weight_cast_fp16")]; tensor layers_11_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308655040))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312853568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312849408))))[name = string("layers_11_self_attn_o_proj_weight_cast_fp16")]; tensor layers_11_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312855680))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325451008))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325438656))))[name = string("layers_11_mlp_gate_proj_weight_cast_fp16")]; tensor layers_11_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325457216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338052544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338040192))))[name = string("layers_11_mlp_up_proj_weight_cast_fp16")]; tensor layers_11_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338058752))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350645888))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350641728))))[name = string("layers_11_mlp_down_proj_weight_cast_fp16")]; tensor layers_12_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350648000))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354846528))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354842368))))[name = string("layers_12_self_attn_q_proj_weight_cast_fp16")]; tensor layers_12_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354848640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355372992))))[name = string("layers_12_self_attn_k_proj_weight_cast_fp16")]; tensor layers_12_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373888))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898816))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898240))))[name = string("layers_12_self_attn_v_proj_weight_cast_fp16")]; tensor layers_12_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355899136))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360097664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360093504))))[name = string("layers_12_self_attn_o_proj_weight_cast_fp16")]; tensor layers_12_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360099776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372695104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372682752))))[name = string("layers_12_mlp_gate_proj_weight_cast_fp16")]; tensor layers_12_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372701312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385296640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385284288))))[name = string("layers_12_mlp_up_proj_weight_cast_fp16")]; tensor layers_12_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385302848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397885824))))[name = string("layers_12_mlp_down_proj_weight_cast_fp16")]; tensor layers_13_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397892096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402090624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402086464))))[name = string("layers_13_self_attn_q_proj_weight_cast_fp16")]; tensor layers_13_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402092736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617088))))[name = string("layers_13_self_attn_k_proj_weight_cast_fp16")]; tensor layers_13_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617984))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142336))))[name = string("layers_13_self_attn_v_proj_weight_cast_fp16")]; tensor layers_13_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403143232))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407341760))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407337600))))[name = string("layers_13_self_attn_o_proj_weight_cast_fp16")]; tensor layers_13_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407343872))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419939200))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419926848))))[name = string("layers_13_mlp_gate_proj_weight_cast_fp16")]; tensor layers_13_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419945408))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432532544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432528384))))[name = string("layers_13_mlp_down_proj_weight_cast_fp16")]; tensor layers_14_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432534656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436733184))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436729024))))[name = string("layers_14_self_attn_q_proj_weight_cast_fp16")]; tensor layers_14_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436735296))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437259648))))[name = string("layers_14_self_attn_v_proj_weight_cast_fp16")]; tensor layers_14_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441459072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441454912))))[name = string("layers_14_self_attn_o_proj_weight_cast_fp16")]; tensor layers_14_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441461184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454056512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454044160))))[name = string("layers_14_mlp_gate_proj_weight_cast_fp16")]; tensor layers_14_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454062720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466658048))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466645696))))[name = string("layers_14_mlp_up_proj_weight_cast_fp16")]; tensor layers_14_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466664256))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479251392))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479247232))))[name = string("layers_14_mlp_down_proj_weight_cast_fp16")]; tensor layers_15_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479253504))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483452032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483447872))))[name = string("layers_15_self_attn_q_proj_weight_cast_fp16")]; tensor layers_15_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483454144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483978496))))[name = string("layers_15_self_attn_k_proj_weight_cast_fp16")]; tensor layers_15_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504320))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484503744))))[name = string("layers_15_self_attn_v_proj_weight_cast_fp16")]; tensor layers_15_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488703168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488699008))))[name = string("layers_15_self_attn_o_proj_weight_cast_fp16")]; tensor layers_15_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488705280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501300608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501288256))))[name = string("layers_15_mlp_gate_proj_weight_cast_fp16")]; tensor layers_15_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501306816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513902144))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513889792))))[name = string("layers_15_mlp_up_proj_weight_cast_fp16")]; tensor layers_15_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513908352))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526495488))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526491328))))[name = string("layers_15_mlp_down_proj_weight_cast_fp16")]; tensor layers_16_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526497600))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530696128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530691968))))[name = string("layers_16_self_attn_q_proj_weight_cast_fp16")]; tensor layers_16_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530698240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531222592))))[name = string("layers_16_self_attn_k_proj_weight_cast_fp16")]; tensor layers_16_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531747840))))[name = string("layers_16_self_attn_v_proj_weight_cast_fp16")]; tensor layers_16_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535947264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535943104))))[name = string("layers_16_self_attn_o_proj_weight_cast_fp16")]; tensor layers_16_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535949376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548536512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548532352))))[name = string("layers_16_mlp_down_proj_weight_cast_fp16")]; tensor layers_17_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548538624))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552737152))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552732992))))[name = string("layers_17_self_attn_q_proj_weight_cast_fp16")]; tensor layers_17_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552739264))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553263616))))[name = string("layers_17_self_attn_k_proj_weight_cast_fp16")]; tensor layers_17_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264512))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553788864))))[name = string("layers_17_self_attn_v_proj_weight_cast_fp16")]; tensor layers_17_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789760))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557988288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557984128))))[name = string("layers_17_self_attn_o_proj_weight_cast_fp16")]; tensor layers_17_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557990400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570585728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570573376))))[name = string("layers_17_mlp_gate_proj_weight_cast_fp16")]; tensor layers_17_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570591936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583187264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583174912))))[name = string("layers_17_mlp_up_proj_weight_cast_fp16")]; tensor layers_17_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583193472))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595780608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595776448))))[name = string("layers_17_mlp_down_proj_weight_cast_fp16")]; tensor layers_18_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595782720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599981248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599977088))))[name = string("layers_18_self_attn_q_proj_weight_cast_fp16")]; tensor layers_18_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599983360))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600507712))))[name = string("layers_18_self_attn_k_proj_weight_cast_fp16")]; tensor layers_18_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601032960))))[name = string("layers_18_self_attn_v_proj_weight_cast_fp16")]; tensor layers_18_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605232384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605228224))))[name = string("layers_18_self_attn_o_proj_weight_cast_fp16")]; tensor layers_18_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605234496))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617829824))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617817472))))[name = string("layers_18_mlp_gate_proj_weight_cast_fp16")]; tensor layers_18_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617836032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630431360))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630419008))))[name = string("layers_18_mlp_up_proj_weight_cast_fp16")]; tensor layers_18_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630437568))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643024704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643020544))))[name = string("layers_18_mlp_down_proj_weight_cast_fp16")]; tensor layers_19_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643026816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647225344))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647221184))))[name = string("layers_19_self_attn_q_proj_weight_cast_fp16")]; tensor layers_19_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647227456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647751808))))[name = string("layers_19_self_attn_k_proj_weight_cast_fp16")]; tensor layers_19_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660348032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660335680))))[name = string("layers_19_mlp_gate_proj_weight_cast_fp16")]; tensor layers_19_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660354240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672949568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672937216))))[name = string("layers_19_mlp_up_proj_weight_cast_fp16")]; tensor layers_19_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672955776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685542912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685538752))))[name = string("layers_19_mlp_down_proj_weight_cast_fp16")]; tensor layers_20_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685545024))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689743552))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689739392))))[name = string("layers_20_self_attn_q_proj_weight_cast_fp16")]; tensor layers_20_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689745664))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270592))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270016))))[name = string("layers_20_self_attn_k_proj_weight_cast_fp16")]; tensor layers_20_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694469440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694465280))))[name = string("layers_20_self_attn_o_proj_weight_cast_fp16")]; tensor layers_20_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694471552))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707066880))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707054528))))[name = string("layers_20_mlp_gate_proj_weight_cast_fp16")]; tensor layers_20_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707073088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719660224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719656064))))[name = string("layers_20_mlp_down_proj_weight_cast_fp16")]; tensor layers_21_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719662336))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723860864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723856704))))[name = string("layers_21_self_attn_q_proj_weight_cast_fp16")]; tensor layers_21_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723862976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387328))))[name = string("layers_21_self_attn_k_proj_weight_cast_fp16")]; tensor layers_21_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724388224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728586752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728582592))))[name = string("layers_21_self_attn_o_proj_weight_cast_fp16")]; tensor layers_21_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728588864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741184192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741171840))))[name = string("layers_21_mlp_gate_proj_weight_cast_fp16")]; tensor layers_21_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741190400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753785728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753773376))))[name = string("layers_21_mlp_up_proj_weight_cast_fp16")]; tensor layers_21_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753791936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766379072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766374912))))[name = string("layers_21_mlp_down_proj_weight_cast_fp16")]; tensor layers_22_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766381184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770579712))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770575552))))[name = string("layers_22_self_attn_q_proj_weight_cast_fp16")]; tensor layers_22_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770581824))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106176))))[name = string("layers_22_self_attn_k_proj_weight_cast_fp16")]; tensor layers_22_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771107072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783702400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783690048))))[name = string("layers_22_mlp_gate_proj_weight_cast_fp16")]; tensor layers_22_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783708608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796303936))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796291584))))[name = string("layers_22_mlp_up_proj_weight_cast_fp16")]; tensor layers_22_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796310144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808897280))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808893120))))[name = string("layers_22_mlp_down_proj_weight_cast_fp16")]; tensor layers_23_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808899392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813097920))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813093760))))[name = string("layers_23_self_attn_q_proj_weight_cast_fp16")]; tensor layers_23_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813100032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624960))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624384))))[name = string("layers_23_self_attn_k_proj_weight_cast_fp16")]; tensor layers_23_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813625280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817823808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817819648))))[name = string("layers_23_self_attn_o_proj_weight_cast_fp16")]; tensor layers_23_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817825920))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830421248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830408896))))[name = string("layers_23_mlp_gate_proj_weight_cast_fp16")]; tensor layers_23_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830427456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843022784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843010432))))[name = string("layers_23_mlp_up_proj_weight_cast_fp16")]; tensor layers_23_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843028992))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855616128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855611968))))[name = string("layers_23_mlp_down_proj_weight_cast_fp16")]; tensor layers_24_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855618240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859816768))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859812608))))[name = string("layers_24_self_attn_q_proj_weight_cast_fp16")]; tensor layers_24_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859818880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343232))))[name = string("layers_24_self_attn_k_proj_weight_cast_fp16")]; tensor layers_24_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860344128))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864542656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864538496))))[name = string("layers_24_self_attn_o_proj_weight_cast_fp16")]; tensor layers_24_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864544768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877140096))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877127744))))[name = string("layers_24_mlp_gate_proj_weight_cast_fp16")]; tensor layers_24_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877146304))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889741632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889729280))))[name = string("layers_24_mlp_up_proj_weight_cast_fp16")]; tensor layers_24_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889747840))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902334976))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902330816))))[name = string("layers_24_mlp_down_proj_weight_cast_fp16")]; tensor layers_25_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902337088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906535616))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906531456))))[name = string("layers_25_self_attn_q_proj_weight_cast_fp16")]; tensor layers_25_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906537728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062080))))[name = string("layers_25_self_attn_k_proj_weight_cast_fp16")]; tensor layers_25_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911261504))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911257344))))[name = string("layers_25_self_attn_o_proj_weight_cast_fp16")]; tensor layers_25_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911263616))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923858944))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923846592))))[name = string("layers_25_mlp_gate_proj_weight_cast_fp16")]; tensor layers_25_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923865152))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936460480))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936448128))))[name = string("layers_25_mlp_up_proj_weight_cast_fp16")]; tensor layers_26_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936466688))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940665216))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940661056))))[name = string("layers_26_self_attn_q_proj_weight_cast_fp16")]; tensor layers_26_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940667328))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192256))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941191680))))[name = string("layers_26_self_attn_k_proj_weight_cast_fp16")]; tensor layers_26_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192576))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945391104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945386944))))[name = string("layers_26_self_attn_o_proj_weight_cast_fp16")]; tensor layers_26_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945393216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957988544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957976192))))[name = string("layers_26_mlp_gate_proj_weight_cast_fp16")]; int32 var_765 = const()[name = string("op_765"), val = int32(0)]; tensor var_766 = mul(x = position_index_seed, y = var_765)[name = string("op_766")]; int32 var_768 = const()[name = string("op_768"), val = int32(1)]; tensor ones = add(x = var_766, y = var_768)[name = string("ones")]; int32 var_770 = const()[name = string("op_770"), val = int32(0)]; bool var_772_exclusive_0 = const()[name = string("op_772_exclusive_0"), val = bool(false)]; bool var_772_reverse_0 = const()[name = string("op_772_reverse_0"), val = bool(false)]; tensor var_772 = cumsum(axis = var_770, exclusive = var_772_exclusive_0, reverse = var_772_reverse_0, x = ones)[name = string("op_772")]; int32 var_774 = const()[name = string("op_774"), val = int32(1)]; tensor position_offsets = sub(x = var_772, y = var_774)[name = string("position_offsets")]; tensor position_ids_1 = add(x = position_offsets, y = position_id)[name = string("position_ids_1")]; bool var_784_keep_dims_0 = const()[name = string("op_784_keep_dims_0"), val = bool(false)]; int32 var_784 = reduce_sum(keep_dims = var_784_keep_dims_0, x = ones)[name = string("op_784")]; int32 var_786 = const()[name = string("op_786"), val = int32(1)]; int32 offset = sub(x = var_784, y = var_786)[name = string("offset")]; tensor var_789 = add(x = position_id, y = offset)[name = string("op_789")]; int32 var_791 = const()[name = string("op_791"), val = int32(1)]; tensor cache_position_end = add(x = var_789, y = var_791)[name = string("cache_position_end")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor greater_equal_0 = greater_equal(x = position_ids_1, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(32768)]; tensor add_0 = add(x = position_ids_1, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = position_ids_1, b = add_0, cond = greater_equal_0)[name = string("select_0")]; tensor rope_emb_cos_cached_to_fp16 = const()[name = string("rope_emb_cos_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957994752)))]; int32 cos_1_batch_dims_0 = const()[name = string("cos_1_batch_dims_0"), val = int32(0)]; bool cos_1_validate_indices_0 = const()[name = string("cos_1_validate_indices_0"), val = bool(false)]; int32 greater_equal_8_y_0 = const()[name = string("greater_equal_8_y_0"), val = int32(0)]; tensor greater_equal_8 = greater_equal(x = select_0, y = greater_equal_8_y_0)[name = string("greater_equal_8")]; int32 slice_by_index_8 = const()[name = string("slice_by_index_8"), val = int32(32768)]; tensor add_8 = add(x = select_0, y = slice_by_index_8)[name = string("add_8")]; tensor select_8 = select(a = select_0, b = add_8, cond = greater_equal_8)[name = string("select_8")]; int32 cos_1_cast_fp16_axis_4 = const()[name = string("cos_1_cast_fp16_axis_4"), val = int32(0)]; tensor cos_1_cast_fp16 = gather(axis = cos_1_cast_fp16_axis_4, batch_dims = cos_1_batch_dims_0, indices = select_8, validate_indices = cos_1_validate_indices_0, x = rope_emb_cos_cached_to_fp16)[name = string("cos_1_cast_fp16")]; tensor rope_emb_sin_cached_to_fp16 = const()[name = string("rope_emb_sin_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(966383424)))]; int32 sin_1_batch_dims_0 = const()[name = string("sin_1_batch_dims_0"), val = int32(0)]; bool sin_1_validate_indices_0 = const()[name = string("sin_1_validate_indices_0"), val = bool(false)]; int32 sin_1_cast_fp16_axis_4 = const()[name = string("sin_1_cast_fp16_axis_4"), val = int32(0)]; tensor sin_1_cast_fp16 = gather(axis = sin_1_cast_fp16_axis_4, batch_dims = sin_1_batch_dims_0, indices = select_8, validate_indices = sin_1_validate_indices_0, x = rope_emb_sin_cached_to_fp16)[name = string("sin_1_cast_fp16")]; tensor var_865_perm_0 = const()[name = string("op_865_perm_0"), val = tensor([-1, -2])]; tensor var_867_axes_0 = const()[name = string("op_867_axes_0"), val = tensor([0])]; tensor var_865_cast_fp16 = transpose(perm = var_865_perm_0, x = cos_1_cast_fp16)[name = string("transpose_429")]; tensor var_867_cast_fp16 = expand_dims(axes = var_867_axes_0, x = var_865_cast_fp16)[name = string("op_867_cast_fp16")]; tensor var_869_axes_0 = const()[name = string("op_869_axes_0"), val = tensor([0])]; tensor var_869_cast_fp16 = expand_dims(axes = var_869_axes_0, x = var_867_cast_fp16)[name = string("op_869_cast_fp16")]; tensor var_874_perm_0 = const()[name = string("op_874_perm_0"), val = tensor([-1, -2])]; tensor var_876_axes_0 = const()[name = string("op_876_axes_0"), val = tensor([0])]; tensor var_874_cast_fp16 = transpose(perm = var_874_perm_0, x = sin_1_cast_fp16)[name = string("transpose_428")]; tensor var_876_cast_fp16 = expand_dims(axes = var_876_axes_0, x = var_874_cast_fp16)[name = string("op_876_cast_fp16")]; tensor var_878_axes_0 = const()[name = string("op_878_axes_0"), val = tensor([0])]; tensor var_878_cast_fp16 = expand_dims(axes = var_878_axes_0, x = var_876_cast_fp16)[name = string("op_878_cast_fp16")]; string position_ids_1_to_uint16_dtype_0 = const()[name = string("position_ids_1_to_uint16_dtype_0"), val = string("uint16")]; tensor causal_mask = const()[name = string("causal_mask"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(974772096)))]; int32 mask_axis_0 = const()[name = string("mask_axis_0"), val = int32(1)]; int32 mask_batch_dims_0 = const()[name = string("mask_batch_dims_0"), val = int32(0)]; bool mask_validate_indices_0 = const()[name = string("mask_validate_indices_0"), val = bool(false)]; tensor position_ids_1_to_uint16 = cast(dtype = position_ids_1_to_uint16_dtype_0, x = position_ids_1)[name = string("cast_9")]; tensor mask_cast_uint16 = gather(axis = mask_axis_0, batch_dims = mask_batch_dims_0, indices = position_ids_1_to_uint16, validate_indices = mask_validate_indices_0, x = causal_mask)[name = string("mask_cast_uint16")]; tensor var_895_axes_0 = const()[name = string("op_895_axes_0"), val = tensor([0])]; tensor var_895 = expand_dims(axes = var_895_axes_0, x = mask_cast_uint16)[name = string("op_895")]; tensor attn_mask_1_axes_0 = const()[name = string("attn_mask_1_axes_0"), val = tensor([0])]; tensor attn_mask_1 = expand_dims(axes = attn_mask_1_axes_0, x = var_895)[name = string("attn_mask_1")]; string inputs_embeds_to_fp16_dtype_0 = const()[name = string("inputs_embeds_to_fp16_dtype_0"), val = string("fp16")]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor inputs_embeds_to_fp16 = cast(dtype = inputs_embeds_to_fp16_dtype_0, x = inputs_embeds)[name = string("cast_8")]; tensor var_906_cast_fp16 = mul(x = inputs_embeds_to_fp16, y = const_0_promoted_to_fp16)[name = string("op_906_cast_fp16")]; int32 var_904 = const()[name = string("op_904"), val = int32(1)]; bool doubled_1_interleave_0 = const()[name = string("doubled_1_interleave_0"), val = bool(false)]; tensor doubled_1_cast_fp16 = concat(axis = var_904, interleave = doubled_1_interleave_0, values = (inputs_embeds_to_fp16, var_906_cast_fp16))[name = string("doubled_1_cast_fp16")]; tensor out_1_axes_0 = const()[name = string("out_1_axes_0"), val = tensor([1])]; tensor out_1_gamma_0_to_fp16 = const()[name = string("out_1_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983160768)))]; fp16 var_916_to_fp16 = const()[name = string("op_916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_1_cast_fp16 = layer_norm(axes = out_1_axes_0, epsilon = var_916_to_fp16, gamma = out_1_gamma_0_to_fp16, x = doubled_1_cast_fp16)[name = string("out_1_cast_fp16")]; tensor var_927_split_sizes_0 = const()[name = string("op_927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_927_axis_0 = const()[name = string("op_927_axis_0"), val = int32(1)]; tensor var_927_cast_fp16_0, tensor var_927_cast_fp16_1 = split(axis = var_927_axis_0, split_sizes = var_927_split_sizes_0, x = out_1_cast_fp16)[name = string("op_927_cast_fp16")]; tensor layers_0_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983169024)))]; tensor query_states_1_strides_0 = const()[name = string("query_states_1_strides_0"), val = tensor([1, 1])]; string query_states_1_pad_type_0 = const()[name = string("query_states_1_pad_type_0"), val = string("valid")]; tensor query_states_1_pad_0 = const()[name = string("query_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_1_dilations_0 = const()[name = string("query_states_1_dilations_0"), val = tensor([1, 1])]; int32 query_states_1_groups_0 = const()[name = string("query_states_1_groups_0"), val = int32(1)]; tensor query_states_1_cast_fp16 = conv(dilations = query_states_1_dilations_0, groups = query_states_1_groups_0, pad = query_states_1_pad_0, pad_type = query_states_1_pad_type_0, strides = query_states_1_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("query_states_1_cast_fp16")]; tensor layers_0_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(991557696)))]; tensor key_states_1_strides_0 = const()[name = string("key_states_1_strides_0"), val = tensor([1, 1])]; string key_states_1_pad_type_0 = const()[name = string("key_states_1_pad_type_0"), val = string("valid")]; tensor key_states_1_pad_0 = const()[name = string("key_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_1_dilations_0 = const()[name = string("key_states_1_dilations_0"), val = tensor([1, 1])]; int32 key_states_1_groups_0 = const()[name = string("key_states_1_groups_0"), val = int32(1)]; tensor key_states_1_cast_fp16 = conv(dilations = key_states_1_dilations_0, groups = key_states_1_groups_0, pad = key_states_1_pad_0, pad_type = key_states_1_pad_type_0, strides = key_states_1_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("key_states_1_cast_fp16")]; tensor layers_0_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(992606336)))]; tensor value_states_1_strides_0 = const()[name = string("value_states_1_strides_0"), val = tensor([1, 1])]; string value_states_1_pad_type_0 = const()[name = string("value_states_1_pad_type_0"), val = string("valid")]; tensor value_states_1_pad_0 = const()[name = string("value_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_1_dilations_0 = const()[name = string("value_states_1_dilations_0"), val = tensor([1, 1])]; int32 value_states_1_groups_0 = const()[name = string("value_states_1_groups_0"), val = int32(1)]; tensor value_states_1_cast_fp16 = conv(dilations = value_states_1_dilations_0, groups = value_states_1_groups_0, pad = value_states_1_pad_0, pad_type = value_states_1_pad_type_0, strides = value_states_1_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("value_states_1_cast_fp16")]; tensor concat_0x = const()[name = string("concat_0x"), val = tensor([1, 16, 128, -1])]; tensor x_1_cast_fp16 = reshape(shape = concat_0x, x = query_states_1_cast_fp16)[name = string("x_1_cast_fp16")]; tensor concat_1x = const()[name = string("concat_1x"), val = tensor([1, 2, 128, -1])]; tensor var_984_cast_fp16 = reshape(shape = concat_1x, x = key_states_1_cast_fp16)[name = string("op_984_cast_fp16")]; tensor concat_2x = const()[name = string("concat_2x"), val = tensor([1, 2, 128, -1])]; tensor var_991_cast_fp16 = reshape(shape = concat_2x, x = value_states_1_cast_fp16)[name = string("op_991_cast_fp16")]; tensor var_995_cast_fp16 = mul(x = x_1_cast_fp16, y = var_869_cast_fp16)[name = string("op_995_cast_fp16")]; tensor var_996_split_sizes_0 = const()[name = string("op_996_split_sizes_0"), val = tensor([64, 64])]; int32 var_996_axis_0 = const()[name = string("op_996_axis_0"), val = int32(-2)]; tensor var_996_cast_fp16_0, tensor var_996_cast_fp16_1 = split(axis = var_996_axis_0, split_sizes = var_996_split_sizes_0, x = x_1_cast_fp16)[name = string("op_996_cast_fp16")]; fp16 const_2_promoted_to_fp16 = const()[name = string("const_2_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_998_cast_fp16 = mul(x = var_996_cast_fp16_1, y = const_2_promoted_to_fp16)[name = string("op_998_cast_fp16")]; int32 var_1000 = const()[name = string("op_1000"), val = int32(-2)]; bool var_1001_interleave_0 = const()[name = string("op_1001_interleave_0"), val = bool(false)]; tensor var_1001_cast_fp16 = concat(axis = var_1000, interleave = var_1001_interleave_0, values = (var_998_cast_fp16, var_996_cast_fp16_0))[name = string("op_1001_cast_fp16")]; tensor var_1002_cast_fp16 = mul(x = var_1001_cast_fp16, y = var_878_cast_fp16)[name = string("op_1002_cast_fp16")]; tensor query_states_3_cast_fp16 = add(x = var_995_cast_fp16, y = var_1002_cast_fp16)[name = string("query_states_3_cast_fp16")]; tensor var_1008_cast_fp16 = mul(x = var_984_cast_fp16, y = var_869_cast_fp16)[name = string("op_1008_cast_fp16")]; tensor var_1009_split_sizes_0 = const()[name = string("op_1009_split_sizes_0"), val = tensor([64, 64])]; int32 var_1009_axis_0 = const()[name = string("op_1009_axis_0"), val = int32(-2)]; tensor var_1009_cast_fp16_0, tensor var_1009_cast_fp16_1 = split(axis = var_1009_axis_0, split_sizes = var_1009_split_sizes_0, x = var_984_cast_fp16)[name = string("op_1009_cast_fp16")]; fp16 const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1011_cast_fp16 = mul(x = var_1009_cast_fp16_1, y = const_3_promoted_to_fp16)[name = string("op_1011_cast_fp16")]; int32 var_1013 = const()[name = string("op_1013"), val = int32(-2)]; bool var_1014_interleave_0 = const()[name = string("op_1014_interleave_0"), val = bool(false)]; tensor var_1014_cast_fp16 = concat(axis = var_1013, interleave = var_1014_interleave_0, values = (var_1011_cast_fp16, var_1009_cast_fp16_0))[name = string("op_1014_cast_fp16")]; tensor var_1015_cast_fp16 = mul(x = var_1014_cast_fp16, y = var_878_cast_fp16)[name = string("op_1015_cast_fp16")]; tensor key_states_5_cast_fp16 = add(x = var_1008_cast_fp16, y = var_1015_cast_fp16)[name = string("key_states_5_cast_fp16")]; tensor read_state_0 = read_state(input = key_cache)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; int32 concat_5_axis_0 = const()[name = string("concat_5_axis_0"), val = int32(0)]; bool concat_5_interleave_0 = const()[name = string("concat_5_interleave_0"), val = bool(false)]; tensor concat_5 = concat(axis = concat_5_axis_0, interleave = concat_5_interleave_0, values = (expand_dims_0, expand_dims_1, position_id, expand_dims_3))[name = string("concat_5")]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; tensor concat_6_values1_0 = const()[name = string("concat_6_values1_0"), val = tensor([0])]; tensor concat_6_values3_0 = const()[name = string("concat_6_values3_0"), val = tensor([0])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_4, concat_6_values1_0, cache_position_end, concat_6_values3_0))[name = string("concat_6")]; tensor key_states_7_perm_0 = const()[name = string("key_states_7_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_1_stride_0 = const()[name = string("key_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_7_cast_fp16 = transpose(perm = key_states_7_perm_0, x = key_states_5_cast_fp16)[name = string("transpose_427")]; tensor key_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = key_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = key_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_1_squeeze_mask_0, stride = key_cache_internal_tensor_assign_1_stride_0, update = key_states_7_cast_fp16, x = read_state_0)[name = string("key_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_1_cast_fp16, input = key_cache)[name = string("coreml_update_state_224_write_state")]; tensor coreml_update_state_224 = read_state(input = key_cache)[name = string("coreml_update_state_224")]; tensor read_state_1 = read_state(input = value_cache)[name = string("read_state_1")]; tensor value_states_3_perm_0 = const()[name = string("value_states_3_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_1_stride_0 = const()[name = string("value_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_3_cast_fp16 = transpose(perm = value_states_3_perm_0, x = var_991_cast_fp16)[name = string("transpose_426")]; tensor value_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = value_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = value_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_1_squeeze_mask_0, stride = value_cache_internal_tensor_assign_1_stride_0, update = value_states_3_cast_fp16, x = read_state_1)[name = string("value_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_1_cast_fp16, input = value_cache)[name = string("coreml_update_state_225_write_state")]; tensor coreml_update_state_225 = read_state(input = value_cache)[name = string("coreml_update_state_225")]; tensor var_1085_begin_0 = const()[name = string("op_1085_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1085_end_0 = const()[name = string("op_1085_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1085_end_mask_0 = const()[name = string("op_1085_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1085_cast_fp16 = slice_by_index(begin = var_1085_begin_0, end = var_1085_end_0, end_mask = var_1085_end_mask_0, x = coreml_update_state_224)[name = string("op_1085_cast_fp16")]; tensor tile_0 = const()[name = string("tile_0"), val = tensor([1, 1])]; int32 var_1088_axis_0 = const()[name = string("op_1088_axis_0"), val = int32(1)]; tensor var_1088_cast_fp16_0, tensor var_1088_cast_fp16_1 = split(axis = var_1088_axis_0, split_sizes = tile_0, x = var_1085_cast_fp16)[name = string("op_1088_cast_fp16")]; tensor var_1095_begin_0 = const()[name = string("op_1095_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1095_end_0 = const()[name = string("op_1095_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1095_end_mask_0 = const()[name = string("op_1095_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1095_cast_fp16 = slice_by_index(begin = var_1095_begin_0, end = var_1095_end_0, end_mask = var_1095_end_mask_0, x = coreml_update_state_225)[name = string("op_1095_cast_fp16")]; tensor tile_1 = const()[name = string("tile_1"), val = tensor([1, 1])]; int32 var_1098_axis_0 = const()[name = string("op_1098_axis_0"), val = int32(1)]; tensor var_1098_cast_fp16_0, tensor var_1098_cast_fp16_1 = split(axis = var_1098_axis_0, split_sizes = tile_1, x = var_1095_cast_fp16)[name = string("op_1098_cast_fp16")]; tensor var_1101_split_sizes_0 = const()[name = string("op_1101_split_sizes_0"), val = tensor([8, 8])]; int32 var_1101_axis_0 = const()[name = string("op_1101_axis_0"), val = int32(1)]; tensor var_1101_0, tensor var_1101_1 = split(axis = var_1101_axis_0, split_sizes = var_1101_split_sizes_0, x = query_states_3_cast_fp16)[name = string("op_1101")]; bool attn_weights_1_transpose_x_0 = const()[name = string("attn_weights_1_transpose_x_0"), val = bool(false)]; bool attn_weights_1_transpose_y_0 = const()[name = string("attn_weights_1_transpose_y_0"), val = bool(false)]; tensor attn_weights_1_cast_fp16 = matmul(transpose_x = attn_weights_1_transpose_x_0, transpose_y = attn_weights_1_transpose_y_0, x = var_1088_cast_fp16_0, y = var_1101_0)[name = string("attn_weights_1_cast_fp16")]; fp16 var_1104_to_fp16 = const()[name = string("op_1104_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_3_cast_fp16 = mul(x = attn_weights_1_cast_fp16, y = var_1104_to_fp16)[name = string("attn_weights_3_cast_fp16")]; tensor attn_weights_5_cast_fp16 = add(x = attn_weights_3_cast_fp16, y = attn_mask_1)[name = string("attn_weights_5_cast_fp16")]; int32 var_1108 = const()[name = string("op_1108"), val = int32(-2)]; tensor attn_weights_7_cast_fp16 = softmax(axis = var_1108, x = attn_weights_5_cast_fp16)[name = string("attn_weights_7_cast_fp16")]; bool var_1114_transpose_x_1 = const()[name = string("op_1114_transpose_x_1"), val = bool(true)]; bool var_1114_transpose_y_1 = const()[name = string("op_1114_transpose_y_1"), val = bool(false)]; tensor var_1114_cast_fp16 = matmul(transpose_x = var_1114_transpose_x_1, transpose_y = var_1114_transpose_y_1, x = attn_weights_7_cast_fp16, y = var_1098_cast_fp16_0)[name = string("op_1114_cast_fp16")]; bool attn_weights_9_transpose_x_0 = const()[name = string("attn_weights_9_transpose_x_0"), val = bool(false)]; bool attn_weights_9_transpose_y_0 = const()[name = string("attn_weights_9_transpose_y_0"), val = bool(false)]; tensor attn_weights_9_cast_fp16 = matmul(transpose_x = attn_weights_9_transpose_x_0, transpose_y = attn_weights_9_transpose_y_0, x = var_1088_cast_fp16_1, y = var_1101_1)[name = string("attn_weights_9_cast_fp16")]; fp16 var_1116_to_fp16 = const()[name = string("op_1116_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_11_cast_fp16 = mul(x = attn_weights_9_cast_fp16, y = var_1116_to_fp16)[name = string("attn_weights_11_cast_fp16")]; tensor attn_weights_13_cast_fp16 = add(x = attn_weights_11_cast_fp16, y = attn_mask_1)[name = string("attn_weights_13_cast_fp16")]; int32 var_1120 = const()[name = string("op_1120"), val = int32(-2)]; tensor attn_weights_15_cast_fp16 = softmax(axis = var_1120, x = attn_weights_13_cast_fp16)[name = string("attn_weights_15_cast_fp16")]; bool attn_output_1_transpose_x_1 = const()[name = string("attn_output_1_transpose_x_1"), val = bool(true)]; bool attn_output_1_transpose_y_1 = const()[name = string("attn_output_1_transpose_y_1"), val = bool(false)]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_1, transpose_y = attn_output_1_transpose_y_1, x = attn_weights_15_cast_fp16, y = var_1098_cast_fp16_1)[name = string("attn_output_1_cast_fp16")]; int32 var_1128 = const()[name = string("op_1128"), val = int32(1)]; bool attn_output_3_interleave_0 = const()[name = string("attn_output_3_interleave_0"), val = bool(false)]; tensor attn_output_3_cast_fp16 = concat(axis = var_1128, interleave = attn_output_3_interleave_0, values = (var_1114_cast_fp16, attn_output_1_cast_fp16))[name = string("attn_output_3_cast_fp16")]; tensor var_1132_perm_0 = const()[name = string("op_1132_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_11x = const()[name = string("concat_11x"), val = tensor([1, 2048, 1, -1])]; tensor var_1132_cast_fp16 = transpose(perm = var_1132_perm_0, x = attn_output_3_cast_fp16)[name = string("transpose_425")]; tensor attn_output_7_cast_fp16 = reshape(shape = concat_11x, x = var_1132_cast_fp16)[name = string("attn_output_7_cast_fp16")]; tensor layers_0_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(993654976)))]; tensor hidden_states_3_strides_0 = const()[name = string("hidden_states_3_strides_0"), val = tensor([1, 1])]; string hidden_states_3_pad_type_0 = const()[name = string("hidden_states_3_pad_type_0"), val = string("valid")]; tensor hidden_states_3_pad_0 = const()[name = string("hidden_states_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_3_dilations_0 = const()[name = string("hidden_states_3_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_3_groups_0 = const()[name = string("hidden_states_3_groups_0"), val = int32(1)]; tensor hidden_states_3_cast_fp16 = conv(dilations = hidden_states_3_dilations_0, groups = hidden_states_3_groups_0, pad = hidden_states_3_pad_0, pad_type = hidden_states_3_pad_type_0, strides = hidden_states_3_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16, x = attn_output_7_cast_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor hidden_states_5_cast_fp16 = add(x = inputs_embeds_to_fp16, y = hidden_states_3_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1165_cast_fp16 = mul(x = hidden_states_5_cast_fp16, y = const_8_promoted_to_fp16)[name = string("op_1165_cast_fp16")]; int32 var_1163 = const()[name = string("op_1163"), val = int32(1)]; bool doubled_5_interleave_0 = const()[name = string("doubled_5_interleave_0"), val = bool(false)]; tensor doubled_5_cast_fp16 = concat(axis = var_1163, interleave = doubled_5_interleave_0, values = (hidden_states_5_cast_fp16, var_1165_cast_fp16))[name = string("doubled_5_cast_fp16")]; tensor out_3_axes_0 = const()[name = string("out_3_axes_0"), val = tensor([1])]; tensor out_3_gamma_0_to_fp16 = const()[name = string("out_3_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002043648)))]; fp16 var_1175_to_fp16 = const()[name = string("op_1175_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_3_cast_fp16 = layer_norm(axes = out_3_axes_0, epsilon = var_1175_to_fp16, gamma = out_3_gamma_0_to_fp16, x = doubled_5_cast_fp16)[name = string("out_3_cast_fp16")]; tensor var_1186_split_sizes_0 = const()[name = string("op_1186_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1186_axis_0 = const()[name = string("op_1186_axis_0"), val = int32(1)]; tensor var_1186_cast_fp16_0, tensor var_1186_cast_fp16_1 = split(axis = var_1186_axis_0, split_sizes = var_1186_split_sizes_0, x = out_3_cast_fp16)[name = string("op_1186_cast_fp16")]; tensor layers_0_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002051904)))]; tensor input_1_strides_0 = const()[name = string("input_1_strides_0"), val = tensor([1, 1])]; string input_1_pad_type_0 = const()[name = string("input_1_pad_type_0"), val = string("valid")]; tensor input_1_pad_0 = const()[name = string("input_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_1_dilations_0 = const()[name = string("input_1_dilations_0"), val = tensor([1, 1])]; int32 input_1_groups_0 = const()[name = string("input_1_groups_0"), val = int32(1)]; tensor input_1_cast_fp16 = conv(dilations = input_1_dilations_0, groups = input_1_groups_0, pad = input_1_pad_0, pad_type = input_1_pad_type_0, strides = input_1_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("input_1_cast_fp16")]; tensor var_1203_cast_fp16 = silu(x = input_1_cast_fp16)[name = string("op_1203_cast_fp16")]; tensor layers_0_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1027217792)))]; tensor var_1209_strides_0 = const()[name = string("op_1209_strides_0"), val = tensor([1, 1])]; string var_1209_pad_type_0 = const()[name = string("op_1209_pad_type_0"), val = string("valid")]; tensor var_1209_pad_0 = const()[name = string("op_1209_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1209_dilations_0 = const()[name = string("op_1209_dilations_0"), val = tensor([1, 1])]; int32 var_1209_groups_0 = const()[name = string("op_1209_groups_0"), val = int32(1)]; tensor var_1209_cast_fp16 = conv(dilations = var_1209_dilations_0, groups = var_1209_groups_0, pad = var_1209_pad_0, pad_type = var_1209_pad_type_0, strides = var_1209_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("op_1209_cast_fp16")]; tensor x_9_cast_fp16 = mul(x = var_1203_cast_fp16, y = var_1209_cast_fp16)[name = string("x_9_cast_fp16")]; tensor layers_0_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1052383680)))]; tensor hidden_states_7_strides_0 = const()[name = string("hidden_states_7_strides_0"), val = tensor([1, 1])]; string hidden_states_7_pad_type_0 = const()[name = string("hidden_states_7_pad_type_0"), val = string("valid")]; tensor hidden_states_7_pad_0 = const()[name = string("hidden_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_7_dilations_0 = const()[name = string("hidden_states_7_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_7_groups_0 = const()[name = string("hidden_states_7_groups_0"), val = int32(1)]; tensor hidden_states_7_cast_fp16 = conv(dilations = hidden_states_7_dilations_0, groups = hidden_states_7_groups_0, pad = hidden_states_7_pad_0, pad_type = hidden_states_7_pad_type_0, strides = hidden_states_7_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16, x = x_9_cast_fp16)[name = string("hidden_states_7_cast_fp16")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = hidden_states_7_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; fp16 const_10_promoted_to_fp16 = const()[name = string("const_10_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1227_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_10_promoted_to_fp16)[name = string("op_1227_cast_fp16")]; int32 var_1225 = const()[name = string("op_1225"), val = int32(1)]; bool doubled_9_interleave_0 = const()[name = string("doubled_9_interleave_0"), val = bool(false)]; tensor doubled_9_cast_fp16 = concat(axis = var_1225, interleave = doubled_9_interleave_0, values = (hidden_states_9_cast_fp16, var_1227_cast_fp16))[name = string("doubled_9_cast_fp16")]; tensor out_5_axes_0 = const()[name = string("out_5_axes_0"), val = tensor([1])]; tensor out_5_gamma_0_to_fp16 = const()[name = string("out_5_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077549568)))]; fp16 var_1237_to_fp16 = const()[name = string("op_1237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_5_cast_fp16 = layer_norm(axes = out_5_axes_0, epsilon = var_1237_to_fp16, gamma = out_5_gamma_0_to_fp16, x = doubled_9_cast_fp16)[name = string("out_5_cast_fp16")]; tensor var_1248_split_sizes_0 = const()[name = string("op_1248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1248_axis_0 = const()[name = string("op_1248_axis_0"), val = int32(1)]; tensor var_1248_cast_fp16_0, tensor var_1248_cast_fp16_1 = split(axis = var_1248_axis_0, split_sizes = var_1248_split_sizes_0, x = out_5_cast_fp16)[name = string("op_1248_cast_fp16")]; tensor layers_1_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077557824)))]; tensor query_states_7_strides_0 = const()[name = string("query_states_7_strides_0"), val = tensor([1, 1])]; string query_states_7_pad_type_0 = const()[name = string("query_states_7_pad_type_0"), val = string("valid")]; tensor query_states_7_pad_0 = const()[name = string("query_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_7_dilations_0 = const()[name = string("query_states_7_dilations_0"), val = tensor([1, 1])]; int32 query_states_7_groups_0 = const()[name = string("query_states_7_groups_0"), val = int32(1)]; tensor query_states_7_cast_fp16 = conv(dilations = query_states_7_dilations_0, groups = query_states_7_groups_0, pad = query_states_7_pad_0, pad_type = query_states_7_pad_type_0, strides = query_states_7_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("query_states_7_cast_fp16")]; tensor layers_1_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1085946496)))]; tensor key_states_11_strides_0 = const()[name = string("key_states_11_strides_0"), val = tensor([1, 1])]; string key_states_11_pad_type_0 = const()[name = string("key_states_11_pad_type_0"), val = string("valid")]; tensor key_states_11_pad_0 = const()[name = string("key_states_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_11_dilations_0 = const()[name = string("key_states_11_dilations_0"), val = tensor([1, 1])]; int32 key_states_11_groups_0 = const()[name = string("key_states_11_groups_0"), val = int32(1)]; tensor key_states_11_cast_fp16 = conv(dilations = key_states_11_dilations_0, groups = key_states_11_groups_0, pad = key_states_11_pad_0, pad_type = key_states_11_pad_type_0, strides = key_states_11_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("key_states_11_cast_fp16")]; tensor value_states_7_strides_0 = const()[name = string("value_states_7_strides_0"), val = tensor([1, 1])]; string value_states_7_pad_type_0 = const()[name = string("value_states_7_pad_type_0"), val = string("valid")]; tensor value_states_7_pad_0 = const()[name = string("value_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_7_dilations_0 = const()[name = string("value_states_7_dilations_0"), val = tensor([1, 1])]; int32 value_states_7_groups_0 = const()[name = string("value_states_7_groups_0"), val = int32(1)]; tensor value_states_7_cast_fp16 = conv(dilations = value_states_7_dilations_0, groups = value_states_7_groups_0, pad = value_states_7_pad_0, pad_type = value_states_7_pad_type_0, strides = value_states_7_strides_0, weight = layers_1_self_attn_v_proj_weight_cast_fp16, x = var_1248_cast_fp16_0)[name = string("value_states_7_cast_fp16")]; tensor concat_12x = const()[name = string("concat_12x"), val = tensor([1, 16, 128, -1])]; tensor x_11_cast_fp16 = reshape(shape = concat_12x, x = query_states_7_cast_fp16)[name = string("x_11_cast_fp16")]; tensor concat_13x = const()[name = string("concat_13x"), val = tensor([1, 2, 128, -1])]; tensor var_1305_cast_fp16 = reshape(shape = concat_13x, x = key_states_11_cast_fp16)[name = string("op_1305_cast_fp16")]; tensor concat_14x = const()[name = string("concat_14x"), val = tensor([1, 2, 128, -1])]; tensor var_1312_cast_fp16 = reshape(shape = concat_14x, x = value_states_7_cast_fp16)[name = string("op_1312_cast_fp16")]; tensor var_1316_cast_fp16 = mul(x = x_11_cast_fp16, y = var_869_cast_fp16)[name = string("op_1316_cast_fp16")]; tensor var_1317_split_sizes_0 = const()[name = string("op_1317_split_sizes_0"), val = tensor([64, 64])]; int32 var_1317_axis_0 = const()[name = string("op_1317_axis_0"), val = int32(-2)]; tensor var_1317_cast_fp16_0, tensor var_1317_cast_fp16_1 = split(axis = var_1317_axis_0, split_sizes = var_1317_split_sizes_0, x = x_11_cast_fp16)[name = string("op_1317_cast_fp16")]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1319_cast_fp16 = mul(x = var_1317_cast_fp16_1, y = const_12_promoted_to_fp16)[name = string("op_1319_cast_fp16")]; int32 var_1321 = const()[name = string("op_1321"), val = int32(-2)]; bool var_1322_interleave_0 = const()[name = string("op_1322_interleave_0"), val = bool(false)]; tensor var_1322_cast_fp16 = concat(axis = var_1321, interleave = var_1322_interleave_0, values = (var_1319_cast_fp16, var_1317_cast_fp16_0))[name = string("op_1322_cast_fp16")]; tensor var_1323_cast_fp16 = mul(x = var_1322_cast_fp16, y = var_878_cast_fp16)[name = string("op_1323_cast_fp16")]; tensor query_states_9_cast_fp16 = add(x = var_1316_cast_fp16, y = var_1323_cast_fp16)[name = string("query_states_9_cast_fp16")]; tensor var_1329_cast_fp16 = mul(x = var_1305_cast_fp16, y = var_869_cast_fp16)[name = string("op_1329_cast_fp16")]; tensor var_1330_split_sizes_0 = const()[name = string("op_1330_split_sizes_0"), val = tensor([64, 64])]; int32 var_1330_axis_0 = const()[name = string("op_1330_axis_0"), val = int32(-2)]; tensor var_1330_cast_fp16_0, tensor var_1330_cast_fp16_1 = split(axis = var_1330_axis_0, split_sizes = var_1330_split_sizes_0, x = var_1305_cast_fp16)[name = string("op_1330_cast_fp16")]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1332_cast_fp16 = mul(x = var_1330_cast_fp16_1, y = const_13_promoted_to_fp16)[name = string("op_1332_cast_fp16")]; int32 var_1334 = const()[name = string("op_1334"), val = int32(-2)]; bool var_1335_interleave_0 = const()[name = string("op_1335_interleave_0"), val = bool(false)]; tensor var_1335_cast_fp16 = concat(axis = var_1334, interleave = var_1335_interleave_0, values = (var_1332_cast_fp16, var_1330_cast_fp16_0))[name = string("op_1335_cast_fp16")]; tensor var_1336_cast_fp16 = mul(x = var_1335_cast_fp16, y = var_878_cast_fp16)[name = string("op_1336_cast_fp16")]; tensor key_states_15_cast_fp16 = add(x = var_1329_cast_fp16, y = var_1336_cast_fp16)[name = string("key_states_15_cast_fp16")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; int32 concat_17_axis_0 = const()[name = string("concat_17_axis_0"), val = int32(0)]; bool concat_17_interleave_0 = const()[name = string("concat_17_interleave_0"), val = bool(false)]; tensor concat_17 = concat(axis = concat_17_axis_0, interleave = concat_17_interleave_0, values = (expand_dims_12, expand_dims_13, position_id, expand_dims_15))[name = string("concat_17")]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; tensor concat_18_values1_0 = const()[name = string("concat_18_values1_0"), val = tensor([0])]; tensor concat_18_values3_0 = const()[name = string("concat_18_values3_0"), val = tensor([0])]; int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_16, concat_18_values1_0, cache_position_end, concat_18_values3_0))[name = string("concat_18")]; tensor key_states_17_perm_0 = const()[name = string("key_states_17_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_2_stride_0 = const()[name = string("key_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_17_cast_fp16 = transpose(perm = key_states_17_perm_0, x = key_states_15_cast_fp16)[name = string("transpose_424")]; tensor key_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = key_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = key_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_2_squeeze_mask_0, stride = key_cache_internal_tensor_assign_2_stride_0, update = key_states_17_cast_fp16, x = coreml_update_state_224)[name = string("key_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_2_cast_fp16, input = key_cache)[name = string("coreml_update_state_226_write_state")]; tensor coreml_update_state_226 = read_state(input = key_cache)[name = string("coreml_update_state_226")]; tensor value_states_9_perm_0 = const()[name = string("value_states_9_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_2_stride_0 = const()[name = string("value_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_9_cast_fp16 = transpose(perm = value_states_9_perm_0, x = var_1312_cast_fp16)[name = string("transpose_423")]; tensor value_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = value_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = value_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_2_squeeze_mask_0, stride = value_cache_internal_tensor_assign_2_stride_0, update = value_states_9_cast_fp16, x = coreml_update_state_225)[name = string("value_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_2_cast_fp16, input = value_cache)[name = string("coreml_update_state_227_write_state")]; tensor coreml_update_state_227 = read_state(input = value_cache)[name = string("coreml_update_state_227")]; tensor var_1406_begin_0 = const()[name = string("op_1406_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1406_end_0 = const()[name = string("op_1406_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1406_end_mask_0 = const()[name = string("op_1406_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1406_cast_fp16 = slice_by_index(begin = var_1406_begin_0, end = var_1406_end_0, end_mask = var_1406_end_mask_0, x = coreml_update_state_226)[name = string("op_1406_cast_fp16")]; tensor tile_2 = const()[name = string("tile_2"), val = tensor([1, 1])]; int32 var_1409_axis_0 = const()[name = string("op_1409_axis_0"), val = int32(1)]; tensor var_1409_cast_fp16_0, tensor var_1409_cast_fp16_1 = split(axis = var_1409_axis_0, split_sizes = tile_2, x = var_1406_cast_fp16)[name = string("op_1409_cast_fp16")]; tensor var_1416_begin_0 = const()[name = string("op_1416_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1416_end_0 = const()[name = string("op_1416_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1416_end_mask_0 = const()[name = string("op_1416_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1416_cast_fp16 = slice_by_index(begin = var_1416_begin_0, end = var_1416_end_0, end_mask = var_1416_end_mask_0, x = coreml_update_state_227)[name = string("op_1416_cast_fp16")]; tensor tile_3 = const()[name = string("tile_3"), val = tensor([1, 1])]; int32 var_1419_axis_0 = const()[name = string("op_1419_axis_0"), val = int32(1)]; tensor var_1419_cast_fp16_0, tensor var_1419_cast_fp16_1 = split(axis = var_1419_axis_0, split_sizes = tile_3, x = var_1416_cast_fp16)[name = string("op_1419_cast_fp16")]; tensor var_1422_split_sizes_0 = const()[name = string("op_1422_split_sizes_0"), val = tensor([8, 8])]; int32 var_1422_axis_0 = const()[name = string("op_1422_axis_0"), val = int32(1)]; tensor var_1422_0, tensor var_1422_1 = split(axis = var_1422_axis_0, split_sizes = var_1422_split_sizes_0, x = query_states_9_cast_fp16)[name = string("op_1422")]; bool attn_weights_17_transpose_x_0 = const()[name = string("attn_weights_17_transpose_x_0"), val = bool(false)]; bool attn_weights_17_transpose_y_0 = const()[name = string("attn_weights_17_transpose_y_0"), val = bool(false)]; tensor attn_weights_17_cast_fp16 = matmul(transpose_x = attn_weights_17_transpose_x_0, transpose_y = attn_weights_17_transpose_y_0, x = var_1409_cast_fp16_0, y = var_1422_0)[name = string("attn_weights_17_cast_fp16")]; fp16 var_1425_to_fp16 = const()[name = string("op_1425_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_19_cast_fp16 = mul(x = attn_weights_17_cast_fp16, y = var_1425_to_fp16)[name = string("attn_weights_19_cast_fp16")]; tensor attn_weights_21_cast_fp16 = add(x = attn_weights_19_cast_fp16, y = attn_mask_1)[name = string("attn_weights_21_cast_fp16")]; int32 var_1429 = const()[name = string("op_1429"), val = int32(-2)]; tensor attn_weights_23_cast_fp16 = softmax(axis = var_1429, x = attn_weights_21_cast_fp16)[name = string("attn_weights_23_cast_fp16")]; bool var_1435_transpose_x_1 = const()[name = string("op_1435_transpose_x_1"), val = bool(true)]; bool var_1435_transpose_y_1 = const()[name = string("op_1435_transpose_y_1"), val = bool(false)]; tensor var_1435_cast_fp16 = matmul(transpose_x = var_1435_transpose_x_1, transpose_y = var_1435_transpose_y_1, x = attn_weights_23_cast_fp16, y = var_1419_cast_fp16_0)[name = string("op_1435_cast_fp16")]; bool attn_weights_25_transpose_x_0 = const()[name = string("attn_weights_25_transpose_x_0"), val = bool(false)]; bool attn_weights_25_transpose_y_0 = const()[name = string("attn_weights_25_transpose_y_0"), val = bool(false)]; tensor attn_weights_25_cast_fp16 = matmul(transpose_x = attn_weights_25_transpose_x_0, transpose_y = attn_weights_25_transpose_y_0, x = var_1409_cast_fp16_1, y = var_1422_1)[name = string("attn_weights_25_cast_fp16")]; fp16 var_1437_to_fp16 = const()[name = string("op_1437_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_27_cast_fp16 = mul(x = attn_weights_25_cast_fp16, y = var_1437_to_fp16)[name = string("attn_weights_27_cast_fp16")]; tensor attn_weights_29_cast_fp16 = add(x = attn_weights_27_cast_fp16, y = attn_mask_1)[name = string("attn_weights_29_cast_fp16")]; int32 var_1441 = const()[name = string("op_1441"), val = int32(-2)]; tensor attn_weights_31_cast_fp16 = softmax(axis = var_1441, x = attn_weights_29_cast_fp16)[name = string("attn_weights_31_cast_fp16")]; bool attn_output_9_transpose_x_1 = const()[name = string("attn_output_9_transpose_x_1"), val = bool(true)]; bool attn_output_9_transpose_y_1 = const()[name = string("attn_output_9_transpose_y_1"), val = bool(false)]; tensor attn_output_9_cast_fp16 = matmul(transpose_x = attn_output_9_transpose_x_1, transpose_y = attn_output_9_transpose_y_1, x = attn_weights_31_cast_fp16, y = var_1419_cast_fp16_1)[name = string("attn_output_9_cast_fp16")]; int32 var_1449 = const()[name = string("op_1449"), val = int32(1)]; bool attn_output_11_interleave_0 = const()[name = string("attn_output_11_interleave_0"), val = bool(false)]; tensor attn_output_11_cast_fp16 = concat(axis = var_1449, interleave = attn_output_11_interleave_0, values = (var_1435_cast_fp16, attn_output_9_cast_fp16))[name = string("attn_output_11_cast_fp16")]; tensor var_1453_perm_0 = const()[name = string("op_1453_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_23x = const()[name = string("concat_23x"), val = tensor([1, 2048, 1, -1])]; tensor var_1453_cast_fp16 = transpose(perm = var_1453_perm_0, x = attn_output_11_cast_fp16)[name = string("transpose_422")]; tensor attn_output_15_cast_fp16 = reshape(shape = concat_23x, x = var_1453_cast_fp16)[name = string("attn_output_15_cast_fp16")]; tensor layers_1_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086995136)))]; tensor hidden_states_13_strides_0 = const()[name = string("hidden_states_13_strides_0"), val = tensor([1, 1])]; string hidden_states_13_pad_type_0 = const()[name = string("hidden_states_13_pad_type_0"), val = string("valid")]; tensor hidden_states_13_pad_0 = const()[name = string("hidden_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_13_dilations_0 = const()[name = string("hidden_states_13_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_13_groups_0 = const()[name = string("hidden_states_13_groups_0"), val = int32(1)]; tensor hidden_states_13_cast_fp16 = conv(dilations = hidden_states_13_dilations_0, groups = hidden_states_13_groups_0, pad = hidden_states_13_pad_0, pad_type = hidden_states_13_pad_type_0, strides = hidden_states_13_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16, x = attn_output_15_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; tensor hidden_states_15_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = hidden_states_13_cast_fp16)[name = string("hidden_states_15_cast_fp16")]; fp16 const_18_promoted_to_fp16 = const()[name = string("const_18_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1486_cast_fp16 = mul(x = hidden_states_15_cast_fp16, y = const_18_promoted_to_fp16)[name = string("op_1486_cast_fp16")]; int32 var_1484 = const()[name = string("op_1484"), val = int32(1)]; bool doubled_13_interleave_0 = const()[name = string("doubled_13_interleave_0"), val = bool(false)]; tensor doubled_13_cast_fp16 = concat(axis = var_1484, interleave = doubled_13_interleave_0, values = (hidden_states_15_cast_fp16, var_1486_cast_fp16))[name = string("doubled_13_cast_fp16")]; tensor out_7_axes_0 = const()[name = string("out_7_axes_0"), val = tensor([1])]; tensor out_7_gamma_0_to_fp16 = const()[name = string("out_7_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095383808)))]; fp16 var_1496_to_fp16 = const()[name = string("op_1496_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_7_cast_fp16 = layer_norm(axes = out_7_axes_0, epsilon = var_1496_to_fp16, gamma = out_7_gamma_0_to_fp16, x = doubled_13_cast_fp16)[name = string("out_7_cast_fp16")]; tensor var_1507_split_sizes_0 = const()[name = string("op_1507_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1507_axis_0 = const()[name = string("op_1507_axis_0"), val = int32(1)]; tensor var_1507_cast_fp16_0, tensor var_1507_cast_fp16_1 = split(axis = var_1507_axis_0, split_sizes = var_1507_split_sizes_0, x = out_7_cast_fp16)[name = string("op_1507_cast_fp16")]; tensor layers_1_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095392064)))]; tensor input_3_strides_0 = const()[name = string("input_3_strides_0"), val = tensor([1, 1])]; string input_3_pad_type_0 = const()[name = string("input_3_pad_type_0"), val = string("valid")]; tensor input_3_pad_0 = const()[name = string("input_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_3_dilations_0 = const()[name = string("input_3_dilations_0"), val = tensor([1, 1])]; int32 input_3_groups_0 = const()[name = string("input_3_groups_0"), val = int32(1)]; tensor input_3_cast_fp16 = conv(dilations = input_3_dilations_0, groups = input_3_groups_0, pad = input_3_pad_0, pad_type = input_3_pad_type_0, strides = input_3_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16, x = var_1507_cast_fp16_0)[name = string("input_3_cast_fp16")]; tensor var_1524_cast_fp16 = silu(x = input_3_cast_fp16)[name = string("op_1524_cast_fp16")]; tensor var_1530_strides_0 = const()[name = string("op_1530_strides_0"), val = tensor([1, 1])]; string var_1530_pad_type_0 = const()[name = string("op_1530_pad_type_0"), val = string("valid")]; tensor var_1530_pad_0 = const()[name = string("op_1530_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1530_dilations_0 = const()[name = string("op_1530_dilations_0"), val = tensor([1, 1])]; int32 var_1530_groups_0 = const()[name = string("op_1530_groups_0"), val = int32(1)]; tensor var_1530_cast_fp16 = conv(dilations = var_1530_dilations_0, groups = var_1530_groups_0, pad = var_1530_pad_0, pad_type = var_1530_pad_type_0, strides = var_1530_strides_0, weight = layers_1_mlp_up_proj_weight_cast_fp16, x = var_1507_cast_fp16_0)[name = string("op_1530_cast_fp16")]; tensor x_19_cast_fp16 = mul(x = var_1524_cast_fp16, y = var_1530_cast_fp16)[name = string("x_19_cast_fp16")]; tensor layers_1_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1120557952)))]; tensor hidden_states_17_strides_0 = const()[name = string("hidden_states_17_strides_0"), val = tensor([1, 1])]; string hidden_states_17_pad_type_0 = const()[name = string("hidden_states_17_pad_type_0"), val = string("valid")]; tensor hidden_states_17_pad_0 = const()[name = string("hidden_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_17_dilations_0 = const()[name = string("hidden_states_17_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_17_groups_0 = const()[name = string("hidden_states_17_groups_0"), val = int32(1)]; tensor hidden_states_17_cast_fp16 = conv(dilations = hidden_states_17_dilations_0, groups = hidden_states_17_groups_0, pad = hidden_states_17_pad_0, pad_type = hidden_states_17_pad_type_0, strides = hidden_states_17_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16, x = x_19_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; tensor hidden_states_19_cast_fp16 = add(x = hidden_states_15_cast_fp16, y = hidden_states_17_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1548_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_1548_cast_fp16")]; int32 var_1546 = const()[name = string("op_1546"), val = int32(1)]; bool doubled_17_interleave_0 = const()[name = string("doubled_17_interleave_0"), val = bool(false)]; tensor doubled_17_cast_fp16 = concat(axis = var_1546, interleave = doubled_17_interleave_0, values = (hidden_states_19_cast_fp16, var_1548_cast_fp16))[name = string("doubled_17_cast_fp16")]; tensor out_9_axes_0 = const()[name = string("out_9_axes_0"), val = tensor([1])]; tensor out_9_gamma_0_to_fp16 = const()[name = string("out_9_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145723840)))]; fp16 var_1558_to_fp16 = const()[name = string("op_1558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_9_cast_fp16 = layer_norm(axes = out_9_axes_0, epsilon = var_1558_to_fp16, gamma = out_9_gamma_0_to_fp16, x = doubled_17_cast_fp16)[name = string("out_9_cast_fp16")]; tensor var_1569_split_sizes_0 = const()[name = string("op_1569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1569_axis_0 = const()[name = string("op_1569_axis_0"), val = int32(1)]; tensor var_1569_cast_fp16_0, tensor var_1569_cast_fp16_1 = split(axis = var_1569_axis_0, split_sizes = var_1569_split_sizes_0, x = out_9_cast_fp16)[name = string("op_1569_cast_fp16")]; tensor layers_2_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145732096)))]; tensor query_states_13_strides_0 = const()[name = string("query_states_13_strides_0"), val = tensor([1, 1])]; string query_states_13_pad_type_0 = const()[name = string("query_states_13_pad_type_0"), val = string("valid")]; tensor query_states_13_pad_0 = const()[name = string("query_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_13_dilations_0 = const()[name = string("query_states_13_dilations_0"), val = tensor([1, 1])]; int32 query_states_13_groups_0 = const()[name = string("query_states_13_groups_0"), val = int32(1)]; tensor query_states_13_cast_fp16 = conv(dilations = query_states_13_dilations_0, groups = query_states_13_groups_0, pad = query_states_13_pad_0, pad_type = query_states_13_pad_type_0, strides = query_states_13_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("query_states_13_cast_fp16")]; tensor layers_2_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1154120768)))]; tensor key_states_21_strides_0 = const()[name = string("key_states_21_strides_0"), val = tensor([1, 1])]; string key_states_21_pad_type_0 = const()[name = string("key_states_21_pad_type_0"), val = string("valid")]; tensor key_states_21_pad_0 = const()[name = string("key_states_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_21_dilations_0 = const()[name = string("key_states_21_dilations_0"), val = tensor([1, 1])]; int32 key_states_21_groups_0 = const()[name = string("key_states_21_groups_0"), val = int32(1)]; tensor key_states_21_cast_fp16 = conv(dilations = key_states_21_dilations_0, groups = key_states_21_groups_0, pad = key_states_21_pad_0, pad_type = key_states_21_pad_type_0, strides = key_states_21_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("key_states_21_cast_fp16")]; tensor value_states_13_strides_0 = const()[name = string("value_states_13_strides_0"), val = tensor([1, 1])]; string value_states_13_pad_type_0 = const()[name = string("value_states_13_pad_type_0"), val = string("valid")]; tensor value_states_13_pad_0 = const()[name = string("value_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_13_dilations_0 = const()[name = string("value_states_13_dilations_0"), val = tensor([1, 1])]; int32 value_states_13_groups_0 = const()[name = string("value_states_13_groups_0"), val = int32(1)]; tensor value_states_13_cast_fp16 = conv(dilations = value_states_13_dilations_0, groups = value_states_13_groups_0, pad = value_states_13_pad_0, pad_type = value_states_13_pad_type_0, strides = value_states_13_strides_0, weight = layers_2_self_attn_v_proj_weight_cast_fp16, x = var_1569_cast_fp16_0)[name = string("value_states_13_cast_fp16")]; tensor concat_24x = const()[name = string("concat_24x"), val = tensor([1, 16, 128, -1])]; tensor x_21_cast_fp16 = reshape(shape = concat_24x, x = query_states_13_cast_fp16)[name = string("x_21_cast_fp16")]; tensor concat_25x = const()[name = string("concat_25x"), val = tensor([1, 2, 128, -1])]; tensor var_1626_cast_fp16 = reshape(shape = concat_25x, x = key_states_21_cast_fp16)[name = string("op_1626_cast_fp16")]; tensor concat_26x = const()[name = string("concat_26x"), val = tensor([1, 2, 128, -1])]; tensor var_1633_cast_fp16 = reshape(shape = concat_26x, x = value_states_13_cast_fp16)[name = string("op_1633_cast_fp16")]; tensor var_1637_cast_fp16 = mul(x = x_21_cast_fp16, y = var_869_cast_fp16)[name = string("op_1637_cast_fp16")]; tensor var_1638_split_sizes_0 = const()[name = string("op_1638_split_sizes_0"), val = tensor([64, 64])]; int32 var_1638_axis_0 = const()[name = string("op_1638_axis_0"), val = int32(-2)]; tensor var_1638_cast_fp16_0, tensor var_1638_cast_fp16_1 = split(axis = var_1638_axis_0, split_sizes = var_1638_split_sizes_0, x = x_21_cast_fp16)[name = string("op_1638_cast_fp16")]; fp16 const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1640_cast_fp16 = mul(x = var_1638_cast_fp16_1, y = const_22_promoted_to_fp16)[name = string("op_1640_cast_fp16")]; int32 var_1642 = const()[name = string("op_1642"), val = int32(-2)]; bool var_1643_interleave_0 = const()[name = string("op_1643_interleave_0"), val = bool(false)]; tensor var_1643_cast_fp16 = concat(axis = var_1642, interleave = var_1643_interleave_0, values = (var_1640_cast_fp16, var_1638_cast_fp16_0))[name = string("op_1643_cast_fp16")]; tensor var_1644_cast_fp16 = mul(x = var_1643_cast_fp16, y = var_878_cast_fp16)[name = string("op_1644_cast_fp16")]; tensor query_states_15_cast_fp16 = add(x = var_1637_cast_fp16, y = var_1644_cast_fp16)[name = string("query_states_15_cast_fp16")]; tensor var_1650_cast_fp16 = mul(x = var_1626_cast_fp16, y = var_869_cast_fp16)[name = string("op_1650_cast_fp16")]; tensor var_1651_split_sizes_0 = const()[name = string("op_1651_split_sizes_0"), val = tensor([64, 64])]; int32 var_1651_axis_0 = const()[name = string("op_1651_axis_0"), val = int32(-2)]; tensor var_1651_cast_fp16_0, tensor var_1651_cast_fp16_1 = split(axis = var_1651_axis_0, split_sizes = var_1651_split_sizes_0, x = var_1626_cast_fp16)[name = string("op_1651_cast_fp16")]; fp16 const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1653_cast_fp16 = mul(x = var_1651_cast_fp16_1, y = const_23_promoted_to_fp16)[name = string("op_1653_cast_fp16")]; int32 var_1655 = const()[name = string("op_1655"), val = int32(-2)]; bool var_1656_interleave_0 = const()[name = string("op_1656_interleave_0"), val = bool(false)]; tensor var_1656_cast_fp16 = concat(axis = var_1655, interleave = var_1656_interleave_0, values = (var_1653_cast_fp16, var_1651_cast_fp16_0))[name = string("op_1656_cast_fp16")]; tensor var_1657_cast_fp16 = mul(x = var_1656_cast_fp16, y = var_878_cast_fp16)[name = string("op_1657_cast_fp16")]; tensor key_states_25_cast_fp16 = add(x = var_1650_cast_fp16, y = var_1657_cast_fp16)[name = string("key_states_25_cast_fp16")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; int32 concat_29_axis_0 = const()[name = string("concat_29_axis_0"), val = int32(0)]; bool concat_29_interleave_0 = const()[name = string("concat_29_interleave_0"), val = bool(false)]; tensor concat_29 = concat(axis = concat_29_axis_0, interleave = concat_29_interleave_0, values = (expand_dims_24, expand_dims_25, position_id, expand_dims_27))[name = string("concat_29")]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; tensor concat_30_values1_0 = const()[name = string("concat_30_values1_0"), val = tensor([0])]; tensor concat_30_values3_0 = const()[name = string("concat_30_values3_0"), val = tensor([0])]; int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_28, concat_30_values1_0, cache_position_end, concat_30_values3_0))[name = string("concat_30")]; tensor key_states_27_perm_0 = const()[name = string("key_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_3_stride_0 = const()[name = string("key_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_27_cast_fp16 = transpose(perm = key_states_27_perm_0, x = key_states_25_cast_fp16)[name = string("transpose_421")]; tensor key_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = key_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = key_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_3_squeeze_mask_0, stride = key_cache_internal_tensor_assign_3_stride_0, update = key_states_27_cast_fp16, x = coreml_update_state_226)[name = string("key_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_3_cast_fp16, input = key_cache)[name = string("coreml_update_state_228_write_state")]; tensor coreml_update_state_228 = read_state(input = key_cache)[name = string("coreml_update_state_228")]; tensor value_states_15_perm_0 = const()[name = string("value_states_15_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_3_stride_0 = const()[name = string("value_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_15_cast_fp16 = transpose(perm = value_states_15_perm_0, x = var_1633_cast_fp16)[name = string("transpose_420")]; tensor value_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = value_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = value_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_3_squeeze_mask_0, stride = value_cache_internal_tensor_assign_3_stride_0, update = value_states_15_cast_fp16, x = coreml_update_state_227)[name = string("value_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_3_cast_fp16, input = value_cache)[name = string("coreml_update_state_229_write_state")]; tensor coreml_update_state_229 = read_state(input = value_cache)[name = string("coreml_update_state_229")]; tensor var_1727_begin_0 = const()[name = string("op_1727_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1727_end_0 = const()[name = string("op_1727_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1727_end_mask_0 = const()[name = string("op_1727_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1727_cast_fp16 = slice_by_index(begin = var_1727_begin_0, end = var_1727_end_0, end_mask = var_1727_end_mask_0, x = coreml_update_state_228)[name = string("op_1727_cast_fp16")]; tensor tile_4 = const()[name = string("tile_4"), val = tensor([1, 1])]; int32 var_1730_axis_0 = const()[name = string("op_1730_axis_0"), val = int32(1)]; tensor var_1730_cast_fp16_0, tensor var_1730_cast_fp16_1 = split(axis = var_1730_axis_0, split_sizes = tile_4, x = var_1727_cast_fp16)[name = string("op_1730_cast_fp16")]; tensor var_1737_begin_0 = const()[name = string("op_1737_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1737_end_0 = const()[name = string("op_1737_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1737_end_mask_0 = const()[name = string("op_1737_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1737_cast_fp16 = slice_by_index(begin = var_1737_begin_0, end = var_1737_end_0, end_mask = var_1737_end_mask_0, x = coreml_update_state_229)[name = string("op_1737_cast_fp16")]; tensor tile_5 = const()[name = string("tile_5"), val = tensor([1, 1])]; int32 var_1740_axis_0 = const()[name = string("op_1740_axis_0"), val = int32(1)]; tensor var_1740_cast_fp16_0, tensor var_1740_cast_fp16_1 = split(axis = var_1740_axis_0, split_sizes = tile_5, x = var_1737_cast_fp16)[name = string("op_1740_cast_fp16")]; tensor var_1743_split_sizes_0 = const()[name = string("op_1743_split_sizes_0"), val = tensor([8, 8])]; int32 var_1743_axis_0 = const()[name = string("op_1743_axis_0"), val = int32(1)]; tensor var_1743_0, tensor var_1743_1 = split(axis = var_1743_axis_0, split_sizes = var_1743_split_sizes_0, x = query_states_15_cast_fp16)[name = string("op_1743")]; bool attn_weights_33_transpose_x_0 = const()[name = string("attn_weights_33_transpose_x_0"), val = bool(false)]; bool attn_weights_33_transpose_y_0 = const()[name = string("attn_weights_33_transpose_y_0"), val = bool(false)]; tensor attn_weights_33_cast_fp16 = matmul(transpose_x = attn_weights_33_transpose_x_0, transpose_y = attn_weights_33_transpose_y_0, x = var_1730_cast_fp16_0, y = var_1743_0)[name = string("attn_weights_33_cast_fp16")]; fp16 var_1746_to_fp16 = const()[name = string("op_1746_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_35_cast_fp16 = mul(x = attn_weights_33_cast_fp16, y = var_1746_to_fp16)[name = string("attn_weights_35_cast_fp16")]; tensor attn_weights_37_cast_fp16 = add(x = attn_weights_35_cast_fp16, y = attn_mask_1)[name = string("attn_weights_37_cast_fp16")]; int32 var_1750 = const()[name = string("op_1750"), val = int32(-2)]; tensor attn_weights_39_cast_fp16 = softmax(axis = var_1750, x = attn_weights_37_cast_fp16)[name = string("attn_weights_39_cast_fp16")]; bool var_1756_transpose_x_1 = const()[name = string("op_1756_transpose_x_1"), val = bool(true)]; bool var_1756_transpose_y_1 = const()[name = string("op_1756_transpose_y_1"), val = bool(false)]; tensor var_1756_cast_fp16 = matmul(transpose_x = var_1756_transpose_x_1, transpose_y = var_1756_transpose_y_1, x = attn_weights_39_cast_fp16, y = var_1740_cast_fp16_0)[name = string("op_1756_cast_fp16")]; bool attn_weights_41_transpose_x_0 = const()[name = string("attn_weights_41_transpose_x_0"), val = bool(false)]; bool attn_weights_41_transpose_y_0 = const()[name = string("attn_weights_41_transpose_y_0"), val = bool(false)]; tensor attn_weights_41_cast_fp16 = matmul(transpose_x = attn_weights_41_transpose_x_0, transpose_y = attn_weights_41_transpose_y_0, x = var_1730_cast_fp16_1, y = var_1743_1)[name = string("attn_weights_41_cast_fp16")]; fp16 var_1758_to_fp16 = const()[name = string("op_1758_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_43_cast_fp16 = mul(x = attn_weights_41_cast_fp16, y = var_1758_to_fp16)[name = string("attn_weights_43_cast_fp16")]; tensor attn_weights_45_cast_fp16 = add(x = attn_weights_43_cast_fp16, y = attn_mask_1)[name = string("attn_weights_45_cast_fp16")]; int32 var_1762 = const()[name = string("op_1762"), val = int32(-2)]; tensor attn_weights_47_cast_fp16 = softmax(axis = var_1762, x = attn_weights_45_cast_fp16)[name = string("attn_weights_47_cast_fp16")]; bool attn_output_17_transpose_x_1 = const()[name = string("attn_output_17_transpose_x_1"), val = bool(true)]; bool attn_output_17_transpose_y_1 = const()[name = string("attn_output_17_transpose_y_1"), val = bool(false)]; tensor attn_output_17_cast_fp16 = matmul(transpose_x = attn_output_17_transpose_x_1, transpose_y = attn_output_17_transpose_y_1, x = attn_weights_47_cast_fp16, y = var_1740_cast_fp16_1)[name = string("attn_output_17_cast_fp16")]; int32 var_1770 = const()[name = string("op_1770"), val = int32(1)]; bool attn_output_19_interleave_0 = const()[name = string("attn_output_19_interleave_0"), val = bool(false)]; tensor attn_output_19_cast_fp16 = concat(axis = var_1770, interleave = attn_output_19_interleave_0, values = (var_1756_cast_fp16, attn_output_17_cast_fp16))[name = string("attn_output_19_cast_fp16")]; tensor var_1774_perm_0 = const()[name = string("op_1774_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_35x = const()[name = string("concat_35x"), val = tensor([1, 2048, 1, -1])]; tensor var_1774_cast_fp16 = transpose(perm = var_1774_perm_0, x = attn_output_19_cast_fp16)[name = string("transpose_419")]; tensor attn_output_23_cast_fp16 = reshape(shape = concat_35x, x = var_1774_cast_fp16)[name = string("attn_output_23_cast_fp16")]; tensor layers_2_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1155169408)))]; tensor hidden_states_23_strides_0 = const()[name = string("hidden_states_23_strides_0"), val = tensor([1, 1])]; string hidden_states_23_pad_type_0 = const()[name = string("hidden_states_23_pad_type_0"), val = string("valid")]; tensor hidden_states_23_pad_0 = const()[name = string("hidden_states_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_23_dilations_0 = const()[name = string("hidden_states_23_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_23_groups_0 = const()[name = string("hidden_states_23_groups_0"), val = int32(1)]; tensor hidden_states_23_cast_fp16 = conv(dilations = hidden_states_23_dilations_0, groups = hidden_states_23_groups_0, pad = hidden_states_23_pad_0, pad_type = hidden_states_23_pad_type_0, strides = hidden_states_23_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16, x = attn_output_23_cast_fp16)[name = string("hidden_states_23_cast_fp16")]; tensor hidden_states_25_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = hidden_states_23_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1807_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_1807_cast_fp16")]; int32 var_1805 = const()[name = string("op_1805"), val = int32(1)]; bool doubled_21_interleave_0 = const()[name = string("doubled_21_interleave_0"), val = bool(false)]; tensor doubled_21_cast_fp16 = concat(axis = var_1805, interleave = doubled_21_interleave_0, values = (hidden_states_25_cast_fp16, var_1807_cast_fp16))[name = string("doubled_21_cast_fp16")]; tensor out_11_axes_0 = const()[name = string("out_11_axes_0"), val = tensor([1])]; tensor out_11_gamma_0_to_fp16 = const()[name = string("out_11_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163558080)))]; fp16 var_1817_to_fp16 = const()[name = string("op_1817_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_11_cast_fp16 = layer_norm(axes = out_11_axes_0, epsilon = var_1817_to_fp16, gamma = out_11_gamma_0_to_fp16, x = doubled_21_cast_fp16)[name = string("out_11_cast_fp16")]; tensor var_1828_split_sizes_0 = const()[name = string("op_1828_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1828_axis_0 = const()[name = string("op_1828_axis_0"), val = int32(1)]; tensor var_1828_cast_fp16_0, tensor var_1828_cast_fp16_1 = split(axis = var_1828_axis_0, split_sizes = var_1828_split_sizes_0, x = out_11_cast_fp16)[name = string("op_1828_cast_fp16")]; tensor layers_2_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163566336)))]; tensor input_5_strides_0 = const()[name = string("input_5_strides_0"), val = tensor([1, 1])]; string input_5_pad_type_0 = const()[name = string("input_5_pad_type_0"), val = string("valid")]; tensor input_5_pad_0 = const()[name = string("input_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_5_dilations_0 = const()[name = string("input_5_dilations_0"), val = tensor([1, 1])]; int32 input_5_groups_0 = const()[name = string("input_5_groups_0"), val = int32(1)]; tensor input_5_cast_fp16 = conv(dilations = input_5_dilations_0, groups = input_5_groups_0, pad = input_5_pad_0, pad_type = input_5_pad_type_0, strides = input_5_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16, x = var_1828_cast_fp16_0)[name = string("input_5_cast_fp16")]; tensor var_1845_cast_fp16 = silu(x = input_5_cast_fp16)[name = string("op_1845_cast_fp16")]; tensor var_1851_strides_0 = const()[name = string("op_1851_strides_0"), val = tensor([1, 1])]; string var_1851_pad_type_0 = const()[name = string("op_1851_pad_type_0"), val = string("valid")]; tensor var_1851_pad_0 = const()[name = string("op_1851_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1851_dilations_0 = const()[name = string("op_1851_dilations_0"), val = tensor([1, 1])]; int32 var_1851_groups_0 = const()[name = string("op_1851_groups_0"), val = int32(1)]; tensor var_1851_cast_fp16 = conv(dilations = var_1851_dilations_0, groups = var_1851_groups_0, pad = var_1851_pad_0, pad_type = var_1851_pad_type_0, strides = var_1851_strides_0, weight = layers_2_mlp_up_proj_weight_cast_fp16, x = var_1828_cast_fp16_0)[name = string("op_1851_cast_fp16")]; tensor x_29_cast_fp16 = mul(x = var_1845_cast_fp16, y = var_1851_cast_fp16)[name = string("x_29_cast_fp16")]; tensor layers_2_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1188732224)))]; tensor hidden_states_27_strides_0 = const()[name = string("hidden_states_27_strides_0"), val = tensor([1, 1])]; string hidden_states_27_pad_type_0 = const()[name = string("hidden_states_27_pad_type_0"), val = string("valid")]; tensor hidden_states_27_pad_0 = const()[name = string("hidden_states_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_27_dilations_0 = const()[name = string("hidden_states_27_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_27_groups_0 = const()[name = string("hidden_states_27_groups_0"), val = int32(1)]; tensor hidden_states_27_cast_fp16 = conv(dilations = hidden_states_27_dilations_0, groups = hidden_states_27_groups_0, pad = hidden_states_27_pad_0, pad_type = hidden_states_27_pad_type_0, strides = hidden_states_27_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16, x = x_29_cast_fp16)[name = string("hidden_states_27_cast_fp16")]; tensor hidden_states_29_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = hidden_states_27_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; fp16 const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1869_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_30_promoted_to_fp16)[name = string("op_1869_cast_fp16")]; int32 var_1867 = const()[name = string("op_1867"), val = int32(1)]; bool doubled_25_interleave_0 = const()[name = string("doubled_25_interleave_0"), val = bool(false)]; tensor doubled_25_cast_fp16 = concat(axis = var_1867, interleave = doubled_25_interleave_0, values = (hidden_states_29_cast_fp16, var_1869_cast_fp16))[name = string("doubled_25_cast_fp16")]; tensor out_13_axes_0 = const()[name = string("out_13_axes_0"), val = tensor([1])]; tensor out_13_gamma_0_to_fp16 = const()[name = string("out_13_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213898112)))]; fp16 var_1879_to_fp16 = const()[name = string("op_1879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_13_cast_fp16 = layer_norm(axes = out_13_axes_0, epsilon = var_1879_to_fp16, gamma = out_13_gamma_0_to_fp16, x = doubled_25_cast_fp16)[name = string("out_13_cast_fp16")]; tensor var_1890_split_sizes_0 = const()[name = string("op_1890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1890_axis_0 = const()[name = string("op_1890_axis_0"), val = int32(1)]; tensor var_1890_cast_fp16_0, tensor var_1890_cast_fp16_1 = split(axis = var_1890_axis_0, split_sizes = var_1890_split_sizes_0, x = out_13_cast_fp16)[name = string("op_1890_cast_fp16")]; tensor layers_3_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213906368)))]; tensor query_states_19_strides_0 = const()[name = string("query_states_19_strides_0"), val = tensor([1, 1])]; string query_states_19_pad_type_0 = const()[name = string("query_states_19_pad_type_0"), val = string("valid")]; tensor query_states_19_pad_0 = const()[name = string("query_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_19_dilations_0 = const()[name = string("query_states_19_dilations_0"), val = tensor([1, 1])]; int32 query_states_19_groups_0 = const()[name = string("query_states_19_groups_0"), val = int32(1)]; tensor query_states_19_cast_fp16 = conv(dilations = query_states_19_dilations_0, groups = query_states_19_groups_0, pad = query_states_19_pad_0, pad_type = query_states_19_pad_type_0, strides = query_states_19_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("query_states_19_cast_fp16")]; tensor layers_3_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1222295040)))]; tensor key_states_31_strides_0 = const()[name = string("key_states_31_strides_0"), val = tensor([1, 1])]; string key_states_31_pad_type_0 = const()[name = string("key_states_31_pad_type_0"), val = string("valid")]; tensor key_states_31_pad_0 = const()[name = string("key_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_31_dilations_0 = const()[name = string("key_states_31_dilations_0"), val = tensor([1, 1])]; int32 key_states_31_groups_0 = const()[name = string("key_states_31_groups_0"), val = int32(1)]; tensor key_states_31_cast_fp16 = conv(dilations = key_states_31_dilations_0, groups = key_states_31_groups_0, pad = key_states_31_pad_0, pad_type = key_states_31_pad_type_0, strides = key_states_31_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("key_states_31_cast_fp16")]; tensor value_states_19_strides_0 = const()[name = string("value_states_19_strides_0"), val = tensor([1, 1])]; string value_states_19_pad_type_0 = const()[name = string("value_states_19_pad_type_0"), val = string("valid")]; tensor value_states_19_pad_0 = const()[name = string("value_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_19_dilations_0 = const()[name = string("value_states_19_dilations_0"), val = tensor([1, 1])]; int32 value_states_19_groups_0 = const()[name = string("value_states_19_groups_0"), val = int32(1)]; tensor value_states_19_cast_fp16 = conv(dilations = value_states_19_dilations_0, groups = value_states_19_groups_0, pad = value_states_19_pad_0, pad_type = value_states_19_pad_type_0, strides = value_states_19_strides_0, weight = layers_3_self_attn_v_proj_weight_cast_fp16, x = var_1890_cast_fp16_0)[name = string("value_states_19_cast_fp16")]; tensor concat_36x = const()[name = string("concat_36x"), val = tensor([1, 16, 128, -1])]; tensor x_31_cast_fp16 = reshape(shape = concat_36x, x = query_states_19_cast_fp16)[name = string("x_31_cast_fp16")]; tensor concat_37x = const()[name = string("concat_37x"), val = tensor([1, 2, 128, -1])]; tensor var_1947_cast_fp16 = reshape(shape = concat_37x, x = key_states_31_cast_fp16)[name = string("op_1947_cast_fp16")]; tensor concat_38x = const()[name = string("concat_38x"), val = tensor([1, 2, 128, -1])]; tensor var_1954_cast_fp16 = reshape(shape = concat_38x, x = value_states_19_cast_fp16)[name = string("op_1954_cast_fp16")]; tensor var_1958_cast_fp16 = mul(x = x_31_cast_fp16, y = var_869_cast_fp16)[name = string("op_1958_cast_fp16")]; tensor var_1959_split_sizes_0 = const()[name = string("op_1959_split_sizes_0"), val = tensor([64, 64])]; int32 var_1959_axis_0 = const()[name = string("op_1959_axis_0"), val = int32(-2)]; tensor var_1959_cast_fp16_0, tensor var_1959_cast_fp16_1 = split(axis = var_1959_axis_0, split_sizes = var_1959_split_sizes_0, x = x_31_cast_fp16)[name = string("op_1959_cast_fp16")]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1961_cast_fp16 = mul(x = var_1959_cast_fp16_1, y = const_32_promoted_to_fp16)[name = string("op_1961_cast_fp16")]; int32 var_1963 = const()[name = string("op_1963"), val = int32(-2)]; bool var_1964_interleave_0 = const()[name = string("op_1964_interleave_0"), val = bool(false)]; tensor var_1964_cast_fp16 = concat(axis = var_1963, interleave = var_1964_interleave_0, values = (var_1961_cast_fp16, var_1959_cast_fp16_0))[name = string("op_1964_cast_fp16")]; tensor var_1965_cast_fp16 = mul(x = var_1964_cast_fp16, y = var_878_cast_fp16)[name = string("op_1965_cast_fp16")]; tensor query_states_21_cast_fp16 = add(x = var_1958_cast_fp16, y = var_1965_cast_fp16)[name = string("query_states_21_cast_fp16")]; tensor var_1971_cast_fp16 = mul(x = var_1947_cast_fp16, y = var_869_cast_fp16)[name = string("op_1971_cast_fp16")]; tensor var_1972_split_sizes_0 = const()[name = string("op_1972_split_sizes_0"), val = tensor([64, 64])]; int32 var_1972_axis_0 = const()[name = string("op_1972_axis_0"), val = int32(-2)]; tensor var_1972_cast_fp16_0, tensor var_1972_cast_fp16_1 = split(axis = var_1972_axis_0, split_sizes = var_1972_split_sizes_0, x = var_1947_cast_fp16)[name = string("op_1972_cast_fp16")]; fp16 const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1974_cast_fp16 = mul(x = var_1972_cast_fp16_1, y = const_33_promoted_to_fp16)[name = string("op_1974_cast_fp16")]; int32 var_1976 = const()[name = string("op_1976"), val = int32(-2)]; bool var_1977_interleave_0 = const()[name = string("op_1977_interleave_0"), val = bool(false)]; tensor var_1977_cast_fp16 = concat(axis = var_1976, interleave = var_1977_interleave_0, values = (var_1974_cast_fp16, var_1972_cast_fp16_0))[name = string("op_1977_cast_fp16")]; tensor var_1978_cast_fp16 = mul(x = var_1977_cast_fp16, y = var_878_cast_fp16)[name = string("op_1978_cast_fp16")]; tensor key_states_35_cast_fp16 = add(x = var_1971_cast_fp16, y = var_1978_cast_fp16)[name = string("key_states_35_cast_fp16")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; int32 concat_41_axis_0 = const()[name = string("concat_41_axis_0"), val = int32(0)]; bool concat_41_interleave_0 = const()[name = string("concat_41_interleave_0"), val = bool(false)]; tensor concat_41 = concat(axis = concat_41_axis_0, interleave = concat_41_interleave_0, values = (expand_dims_36, expand_dims_37, position_id, expand_dims_39))[name = string("concat_41")]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; tensor concat_42_values1_0 = const()[name = string("concat_42_values1_0"), val = tensor([0])]; tensor concat_42_values3_0 = const()[name = string("concat_42_values3_0"), val = tensor([0])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_40, concat_42_values1_0, cache_position_end, concat_42_values3_0))[name = string("concat_42")]; tensor key_states_37_perm_0 = const()[name = string("key_states_37_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_4_stride_0 = const()[name = string("key_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_37_cast_fp16 = transpose(perm = key_states_37_perm_0, x = key_states_35_cast_fp16)[name = string("transpose_418")]; tensor key_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = key_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = key_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_4_squeeze_mask_0, stride = key_cache_internal_tensor_assign_4_stride_0, update = key_states_37_cast_fp16, x = coreml_update_state_228)[name = string("key_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_4_cast_fp16, input = key_cache)[name = string("coreml_update_state_230_write_state")]; tensor coreml_update_state_230 = read_state(input = key_cache)[name = string("coreml_update_state_230")]; tensor value_states_21_perm_0 = const()[name = string("value_states_21_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_4_stride_0 = const()[name = string("value_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_21_cast_fp16 = transpose(perm = value_states_21_perm_0, x = var_1954_cast_fp16)[name = string("transpose_417")]; tensor value_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = value_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = value_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_4_squeeze_mask_0, stride = value_cache_internal_tensor_assign_4_stride_0, update = value_states_21_cast_fp16, x = coreml_update_state_229)[name = string("value_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_4_cast_fp16, input = value_cache)[name = string("coreml_update_state_231_write_state")]; tensor coreml_update_state_231 = read_state(input = value_cache)[name = string("coreml_update_state_231")]; tensor var_2048_begin_0 = const()[name = string("op_2048_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2048_end_0 = const()[name = string("op_2048_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2048_end_mask_0 = const()[name = string("op_2048_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2048_cast_fp16 = slice_by_index(begin = var_2048_begin_0, end = var_2048_end_0, end_mask = var_2048_end_mask_0, x = coreml_update_state_230)[name = string("op_2048_cast_fp16")]; tensor tile_6 = const()[name = string("tile_6"), val = tensor([1, 1])]; int32 var_2051_axis_0 = const()[name = string("op_2051_axis_0"), val = int32(1)]; tensor var_2051_cast_fp16_0, tensor var_2051_cast_fp16_1 = split(axis = var_2051_axis_0, split_sizes = tile_6, x = var_2048_cast_fp16)[name = string("op_2051_cast_fp16")]; tensor var_2058_begin_0 = const()[name = string("op_2058_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2058_end_0 = const()[name = string("op_2058_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2058_end_mask_0 = const()[name = string("op_2058_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2058_cast_fp16 = slice_by_index(begin = var_2058_begin_0, end = var_2058_end_0, end_mask = var_2058_end_mask_0, x = coreml_update_state_231)[name = string("op_2058_cast_fp16")]; tensor tile_7 = const()[name = string("tile_7"), val = tensor([1, 1])]; int32 var_2061_axis_0 = const()[name = string("op_2061_axis_0"), val = int32(1)]; tensor var_2061_cast_fp16_0, tensor var_2061_cast_fp16_1 = split(axis = var_2061_axis_0, split_sizes = tile_7, x = var_2058_cast_fp16)[name = string("op_2061_cast_fp16")]; tensor var_2064_split_sizes_0 = const()[name = string("op_2064_split_sizes_0"), val = tensor([8, 8])]; int32 var_2064_axis_0 = const()[name = string("op_2064_axis_0"), val = int32(1)]; tensor var_2064_0, tensor var_2064_1 = split(axis = var_2064_axis_0, split_sizes = var_2064_split_sizes_0, x = query_states_21_cast_fp16)[name = string("op_2064")]; bool attn_weights_49_transpose_x_0 = const()[name = string("attn_weights_49_transpose_x_0"), val = bool(false)]; bool attn_weights_49_transpose_y_0 = const()[name = string("attn_weights_49_transpose_y_0"), val = bool(false)]; tensor attn_weights_49_cast_fp16 = matmul(transpose_x = attn_weights_49_transpose_x_0, transpose_y = attn_weights_49_transpose_y_0, x = var_2051_cast_fp16_0, y = var_2064_0)[name = string("attn_weights_49_cast_fp16")]; fp16 var_2067_to_fp16 = const()[name = string("op_2067_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_51_cast_fp16 = mul(x = attn_weights_49_cast_fp16, y = var_2067_to_fp16)[name = string("attn_weights_51_cast_fp16")]; tensor attn_weights_53_cast_fp16 = add(x = attn_weights_51_cast_fp16, y = attn_mask_1)[name = string("attn_weights_53_cast_fp16")]; int32 var_2071 = const()[name = string("op_2071"), val = int32(-2)]; tensor attn_weights_55_cast_fp16 = softmax(axis = var_2071, x = attn_weights_53_cast_fp16)[name = string("attn_weights_55_cast_fp16")]; bool var_2077_transpose_x_1 = const()[name = string("op_2077_transpose_x_1"), val = bool(true)]; bool var_2077_transpose_y_1 = const()[name = string("op_2077_transpose_y_1"), val = bool(false)]; tensor var_2077_cast_fp16 = matmul(transpose_x = var_2077_transpose_x_1, transpose_y = var_2077_transpose_y_1, x = attn_weights_55_cast_fp16, y = var_2061_cast_fp16_0)[name = string("op_2077_cast_fp16")]; bool attn_weights_57_transpose_x_0 = const()[name = string("attn_weights_57_transpose_x_0"), val = bool(false)]; bool attn_weights_57_transpose_y_0 = const()[name = string("attn_weights_57_transpose_y_0"), val = bool(false)]; tensor attn_weights_57_cast_fp16 = matmul(transpose_x = attn_weights_57_transpose_x_0, transpose_y = attn_weights_57_transpose_y_0, x = var_2051_cast_fp16_1, y = var_2064_1)[name = string("attn_weights_57_cast_fp16")]; fp16 var_2079_to_fp16 = const()[name = string("op_2079_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_59_cast_fp16 = mul(x = attn_weights_57_cast_fp16, y = var_2079_to_fp16)[name = string("attn_weights_59_cast_fp16")]; tensor attn_weights_61_cast_fp16 = add(x = attn_weights_59_cast_fp16, y = attn_mask_1)[name = string("attn_weights_61_cast_fp16")]; int32 var_2083 = const()[name = string("op_2083"), val = int32(-2)]; tensor attn_weights_63_cast_fp16 = softmax(axis = var_2083, x = attn_weights_61_cast_fp16)[name = string("attn_weights_63_cast_fp16")]; bool attn_output_25_transpose_x_1 = const()[name = string("attn_output_25_transpose_x_1"), val = bool(true)]; bool attn_output_25_transpose_y_1 = const()[name = string("attn_output_25_transpose_y_1"), val = bool(false)]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_1, transpose_y = attn_output_25_transpose_y_1, x = attn_weights_63_cast_fp16, y = var_2061_cast_fp16_1)[name = string("attn_output_25_cast_fp16")]; int32 var_2091 = const()[name = string("op_2091"), val = int32(1)]; bool attn_output_27_interleave_0 = const()[name = string("attn_output_27_interleave_0"), val = bool(false)]; tensor attn_output_27_cast_fp16 = concat(axis = var_2091, interleave = attn_output_27_interleave_0, values = (var_2077_cast_fp16, attn_output_25_cast_fp16))[name = string("attn_output_27_cast_fp16")]; tensor var_2095_perm_0 = const()[name = string("op_2095_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_47x = const()[name = string("concat_47x"), val = tensor([1, 2048, 1, -1])]; tensor var_2095_cast_fp16 = transpose(perm = var_2095_perm_0, x = attn_output_27_cast_fp16)[name = string("transpose_416")]; tensor attn_output_31_cast_fp16 = reshape(shape = concat_47x, x = var_2095_cast_fp16)[name = string("attn_output_31_cast_fp16")]; tensor hidden_states_33_strides_0 = const()[name = string("hidden_states_33_strides_0"), val = tensor([1, 1])]; string hidden_states_33_pad_type_0 = const()[name = string("hidden_states_33_pad_type_0"), val = string("valid")]; tensor hidden_states_33_pad_0 = const()[name = string("hidden_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_33_dilations_0 = const()[name = string("hidden_states_33_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_33_groups_0 = const()[name = string("hidden_states_33_groups_0"), val = int32(1)]; tensor hidden_states_33_cast_fp16 = conv(dilations = hidden_states_33_dilations_0, groups = hidden_states_33_groups_0, pad = hidden_states_33_pad_0, pad_type = hidden_states_33_pad_type_0, strides = hidden_states_33_strides_0, weight = layers_3_self_attn_o_proj_weight_cast_fp16, x = attn_output_31_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; tensor hidden_states_35_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = hidden_states_33_cast_fp16)[name = string("hidden_states_35_cast_fp16")]; fp16 const_38_promoted_to_fp16 = const()[name = string("const_38_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2128_cast_fp16 = mul(x = hidden_states_35_cast_fp16, y = const_38_promoted_to_fp16)[name = string("op_2128_cast_fp16")]; int32 var_2126 = const()[name = string("op_2126"), val = int32(1)]; bool doubled_29_interleave_0 = const()[name = string("doubled_29_interleave_0"), val = bool(false)]; tensor doubled_29_cast_fp16 = concat(axis = var_2126, interleave = doubled_29_interleave_0, values = (hidden_states_35_cast_fp16, var_2128_cast_fp16))[name = string("doubled_29_cast_fp16")]; tensor out_15_axes_0 = const()[name = string("out_15_axes_0"), val = tensor([1])]; tensor out_15_gamma_0_to_fp16 = const()[name = string("out_15_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223343680)))]; fp16 var_2138_to_fp16 = const()[name = string("op_2138_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_15_cast_fp16 = layer_norm(axes = out_15_axes_0, epsilon = var_2138_to_fp16, gamma = out_15_gamma_0_to_fp16, x = doubled_29_cast_fp16)[name = string("out_15_cast_fp16")]; tensor var_2149_split_sizes_0 = const()[name = string("op_2149_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2149_axis_0 = const()[name = string("op_2149_axis_0"), val = int32(1)]; tensor var_2149_cast_fp16_0, tensor var_2149_cast_fp16_1 = split(axis = var_2149_axis_0, split_sizes = var_2149_split_sizes_0, x = out_15_cast_fp16)[name = string("op_2149_cast_fp16")]; tensor layers_3_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223351936)))]; tensor input_7_strides_0 = const()[name = string("input_7_strides_0"), val = tensor([1, 1])]; string input_7_pad_type_0 = const()[name = string("input_7_pad_type_0"), val = string("valid")]; tensor input_7_pad_0 = const()[name = string("input_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_7_dilations_0 = const()[name = string("input_7_dilations_0"), val = tensor([1, 1])]; int32 input_7_groups_0 = const()[name = string("input_7_groups_0"), val = int32(1)]; tensor input_7_cast_fp16 = conv(dilations = input_7_dilations_0, groups = input_7_groups_0, pad = input_7_pad_0, pad_type = input_7_pad_type_0, strides = input_7_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("input_7_cast_fp16")]; tensor var_2166_cast_fp16 = silu(x = input_7_cast_fp16)[name = string("op_2166_cast_fp16")]; tensor layers_3_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1248517824)))]; tensor var_2172_strides_0 = const()[name = string("op_2172_strides_0"), val = tensor([1, 1])]; string var_2172_pad_type_0 = const()[name = string("op_2172_pad_type_0"), val = string("valid")]; tensor var_2172_pad_0 = const()[name = string("op_2172_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2172_dilations_0 = const()[name = string("op_2172_dilations_0"), val = tensor([1, 1])]; int32 var_2172_groups_0 = const()[name = string("op_2172_groups_0"), val = int32(1)]; tensor var_2172_cast_fp16 = conv(dilations = var_2172_dilations_0, groups = var_2172_groups_0, pad = var_2172_pad_0, pad_type = var_2172_pad_type_0, strides = var_2172_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("op_2172_cast_fp16")]; tensor x_39_cast_fp16 = mul(x = var_2166_cast_fp16, y = var_2172_cast_fp16)[name = string("x_39_cast_fp16")]; tensor hidden_states_37_strides_0 = const()[name = string("hidden_states_37_strides_0"), val = tensor([1, 1])]; string hidden_states_37_pad_type_0 = const()[name = string("hidden_states_37_pad_type_0"), val = string("valid")]; tensor hidden_states_37_pad_0 = const()[name = string("hidden_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_37_dilations_0 = const()[name = string("hidden_states_37_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_37_groups_0 = const()[name = string("hidden_states_37_groups_0"), val = int32(1)]; tensor hidden_states_37_cast_fp16 = conv(dilations = hidden_states_37_dilations_0, groups = hidden_states_37_groups_0, pad = hidden_states_37_pad_0, pad_type = hidden_states_37_pad_type_0, strides = hidden_states_37_strides_0, weight = layers_3_mlp_down_proj_weight_cast_fp16, x = x_39_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; tensor hidden_states_39_cast_fp16 = add(x = hidden_states_35_cast_fp16, y = hidden_states_37_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; fp16 const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2190_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_40_promoted_to_fp16)[name = string("op_2190_cast_fp16")]; int32 var_2188 = const()[name = string("op_2188"), val = int32(1)]; bool doubled_33_interleave_0 = const()[name = string("doubled_33_interleave_0"), val = bool(false)]; tensor doubled_33_cast_fp16 = concat(axis = var_2188, interleave = doubled_33_interleave_0, values = (hidden_states_39_cast_fp16, var_2190_cast_fp16))[name = string("doubled_33_cast_fp16")]; tensor out_17_axes_0 = const()[name = string("out_17_axes_0"), val = tensor([1])]; tensor out_17_gamma_0_to_fp16 = const()[name = string("out_17_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273683712)))]; fp16 var_2200_to_fp16 = const()[name = string("op_2200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_17_cast_fp16 = layer_norm(axes = out_17_axes_0, epsilon = var_2200_to_fp16, gamma = out_17_gamma_0_to_fp16, x = doubled_33_cast_fp16)[name = string("out_17_cast_fp16")]; tensor var_2211_split_sizes_0 = const()[name = string("op_2211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2211_axis_0 = const()[name = string("op_2211_axis_0"), val = int32(1)]; tensor var_2211_cast_fp16_0, tensor var_2211_cast_fp16_1 = split(axis = var_2211_axis_0, split_sizes = var_2211_split_sizes_0, x = out_17_cast_fp16)[name = string("op_2211_cast_fp16")]; tensor layers_4_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273691968)))]; tensor query_states_25_strides_0 = const()[name = string("query_states_25_strides_0"), val = tensor([1, 1])]; string query_states_25_pad_type_0 = const()[name = string("query_states_25_pad_type_0"), val = string("valid")]; tensor query_states_25_pad_0 = const()[name = string("query_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_25_dilations_0 = const()[name = string("query_states_25_dilations_0"), val = tensor([1, 1])]; int32 query_states_25_groups_0 = const()[name = string("query_states_25_groups_0"), val = int32(1)]; tensor query_states_25_cast_fp16 = conv(dilations = query_states_25_dilations_0, groups = query_states_25_groups_0, pad = query_states_25_pad_0, pad_type = query_states_25_pad_type_0, strides = query_states_25_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("query_states_25_cast_fp16")]; tensor layers_4_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1282080640)))]; tensor key_states_41_strides_0 = const()[name = string("key_states_41_strides_0"), val = tensor([1, 1])]; string key_states_41_pad_type_0 = const()[name = string("key_states_41_pad_type_0"), val = string("valid")]; tensor key_states_41_pad_0 = const()[name = string("key_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_41_dilations_0 = const()[name = string("key_states_41_dilations_0"), val = tensor([1, 1])]; int32 key_states_41_groups_0 = const()[name = string("key_states_41_groups_0"), val = int32(1)]; tensor key_states_41_cast_fp16 = conv(dilations = key_states_41_dilations_0, groups = key_states_41_groups_0, pad = key_states_41_pad_0, pad_type = key_states_41_pad_type_0, strides = key_states_41_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("key_states_41_cast_fp16")]; tensor value_states_25_strides_0 = const()[name = string("value_states_25_strides_0"), val = tensor([1, 1])]; string value_states_25_pad_type_0 = const()[name = string("value_states_25_pad_type_0"), val = string("valid")]; tensor value_states_25_pad_0 = const()[name = string("value_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_25_dilations_0 = const()[name = string("value_states_25_dilations_0"), val = tensor([1, 1])]; int32 value_states_25_groups_0 = const()[name = string("value_states_25_groups_0"), val = int32(1)]; tensor value_states_25_cast_fp16 = conv(dilations = value_states_25_dilations_0, groups = value_states_25_groups_0, pad = value_states_25_pad_0, pad_type = value_states_25_pad_type_0, strides = value_states_25_strides_0, weight = layers_4_self_attn_v_proj_weight_cast_fp16, x = var_2211_cast_fp16_0)[name = string("value_states_25_cast_fp16")]; tensor concat_48x = const()[name = string("concat_48x"), val = tensor([1, 16, 128, -1])]; tensor x_41_cast_fp16 = reshape(shape = concat_48x, x = query_states_25_cast_fp16)[name = string("x_41_cast_fp16")]; tensor concat_49x = const()[name = string("concat_49x"), val = tensor([1, 2, 128, -1])]; tensor var_2268_cast_fp16 = reshape(shape = concat_49x, x = key_states_41_cast_fp16)[name = string("op_2268_cast_fp16")]; tensor concat_50x = const()[name = string("concat_50x"), val = tensor([1, 2, 128, -1])]; tensor var_2275_cast_fp16 = reshape(shape = concat_50x, x = value_states_25_cast_fp16)[name = string("op_2275_cast_fp16")]; tensor var_2279_cast_fp16 = mul(x = x_41_cast_fp16, y = var_869_cast_fp16)[name = string("op_2279_cast_fp16")]; tensor var_2280_split_sizes_0 = const()[name = string("op_2280_split_sizes_0"), val = tensor([64, 64])]; int32 var_2280_axis_0 = const()[name = string("op_2280_axis_0"), val = int32(-2)]; tensor var_2280_cast_fp16_0, tensor var_2280_cast_fp16_1 = split(axis = var_2280_axis_0, split_sizes = var_2280_split_sizes_0, x = x_41_cast_fp16)[name = string("op_2280_cast_fp16")]; fp16 const_42_promoted_to_fp16 = const()[name = string("const_42_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2282_cast_fp16 = mul(x = var_2280_cast_fp16_1, y = const_42_promoted_to_fp16)[name = string("op_2282_cast_fp16")]; int32 var_2284 = const()[name = string("op_2284"), val = int32(-2)]; bool var_2285_interleave_0 = const()[name = string("op_2285_interleave_0"), val = bool(false)]; tensor var_2285_cast_fp16 = concat(axis = var_2284, interleave = var_2285_interleave_0, values = (var_2282_cast_fp16, var_2280_cast_fp16_0))[name = string("op_2285_cast_fp16")]; tensor var_2286_cast_fp16 = mul(x = var_2285_cast_fp16, y = var_878_cast_fp16)[name = string("op_2286_cast_fp16")]; tensor query_states_27_cast_fp16 = add(x = var_2279_cast_fp16, y = var_2286_cast_fp16)[name = string("query_states_27_cast_fp16")]; tensor var_2292_cast_fp16 = mul(x = var_2268_cast_fp16, y = var_869_cast_fp16)[name = string("op_2292_cast_fp16")]; tensor var_2293_split_sizes_0 = const()[name = string("op_2293_split_sizes_0"), val = tensor([64, 64])]; int32 var_2293_axis_0 = const()[name = string("op_2293_axis_0"), val = int32(-2)]; tensor var_2293_cast_fp16_0, tensor var_2293_cast_fp16_1 = split(axis = var_2293_axis_0, split_sizes = var_2293_split_sizes_0, x = var_2268_cast_fp16)[name = string("op_2293_cast_fp16")]; fp16 const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2295_cast_fp16 = mul(x = var_2293_cast_fp16_1, y = const_43_promoted_to_fp16)[name = string("op_2295_cast_fp16")]; int32 var_2297 = const()[name = string("op_2297"), val = int32(-2)]; bool var_2298_interleave_0 = const()[name = string("op_2298_interleave_0"), val = bool(false)]; tensor var_2298_cast_fp16 = concat(axis = var_2297, interleave = var_2298_interleave_0, values = (var_2295_cast_fp16, var_2293_cast_fp16_0))[name = string("op_2298_cast_fp16")]; tensor var_2299_cast_fp16 = mul(x = var_2298_cast_fp16, y = var_878_cast_fp16)[name = string("op_2299_cast_fp16")]; tensor key_states_45_cast_fp16 = add(x = var_2292_cast_fp16, y = var_2299_cast_fp16)[name = string("key_states_45_cast_fp16")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; int32 concat_53_axis_0 = const()[name = string("concat_53_axis_0"), val = int32(0)]; bool concat_53_interleave_0 = const()[name = string("concat_53_interleave_0"), val = bool(false)]; tensor concat_53 = concat(axis = concat_53_axis_0, interleave = concat_53_interleave_0, values = (expand_dims_48, expand_dims_49, position_id, expand_dims_51))[name = string("concat_53")]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; tensor concat_54_values1_0 = const()[name = string("concat_54_values1_0"), val = tensor([0])]; tensor concat_54_values3_0 = const()[name = string("concat_54_values3_0"), val = tensor([0])]; int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_52, concat_54_values1_0, cache_position_end, concat_54_values3_0))[name = string("concat_54")]; tensor key_states_47_perm_0 = const()[name = string("key_states_47_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_5_stride_0 = const()[name = string("key_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_47_cast_fp16 = transpose(perm = key_states_47_perm_0, x = key_states_45_cast_fp16)[name = string("transpose_415")]; tensor key_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = key_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = key_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_5_squeeze_mask_0, stride = key_cache_internal_tensor_assign_5_stride_0, update = key_states_47_cast_fp16, x = coreml_update_state_230)[name = string("key_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_5_cast_fp16, input = key_cache)[name = string("coreml_update_state_232_write_state")]; tensor coreml_update_state_232 = read_state(input = key_cache)[name = string("coreml_update_state_232")]; tensor value_states_27_perm_0 = const()[name = string("value_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_5_stride_0 = const()[name = string("value_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_27_cast_fp16 = transpose(perm = value_states_27_perm_0, x = var_2275_cast_fp16)[name = string("transpose_414")]; tensor value_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = value_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = value_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_5_squeeze_mask_0, stride = value_cache_internal_tensor_assign_5_stride_0, update = value_states_27_cast_fp16, x = coreml_update_state_231)[name = string("value_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_5_cast_fp16, input = value_cache)[name = string("coreml_update_state_233_write_state")]; tensor coreml_update_state_233 = read_state(input = value_cache)[name = string("coreml_update_state_233")]; tensor var_2369_begin_0 = const()[name = string("op_2369_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2369_end_0 = const()[name = string("op_2369_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2369_end_mask_0 = const()[name = string("op_2369_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2369_cast_fp16 = slice_by_index(begin = var_2369_begin_0, end = var_2369_end_0, end_mask = var_2369_end_mask_0, x = coreml_update_state_232)[name = string("op_2369_cast_fp16")]; tensor tile_8 = const()[name = string("tile_8"), val = tensor([1, 1])]; int32 var_2372_axis_0 = const()[name = string("op_2372_axis_0"), val = int32(1)]; tensor var_2372_cast_fp16_0, tensor var_2372_cast_fp16_1 = split(axis = var_2372_axis_0, split_sizes = tile_8, x = var_2369_cast_fp16)[name = string("op_2372_cast_fp16")]; tensor var_2379_begin_0 = const()[name = string("op_2379_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2379_end_0 = const()[name = string("op_2379_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2379_end_mask_0 = const()[name = string("op_2379_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2379_cast_fp16 = slice_by_index(begin = var_2379_begin_0, end = var_2379_end_0, end_mask = var_2379_end_mask_0, x = coreml_update_state_233)[name = string("op_2379_cast_fp16")]; tensor tile_9 = const()[name = string("tile_9"), val = tensor([1, 1])]; int32 var_2382_axis_0 = const()[name = string("op_2382_axis_0"), val = int32(1)]; tensor var_2382_cast_fp16_0, tensor var_2382_cast_fp16_1 = split(axis = var_2382_axis_0, split_sizes = tile_9, x = var_2379_cast_fp16)[name = string("op_2382_cast_fp16")]; tensor var_2385_split_sizes_0 = const()[name = string("op_2385_split_sizes_0"), val = tensor([8, 8])]; int32 var_2385_axis_0 = const()[name = string("op_2385_axis_0"), val = int32(1)]; tensor var_2385_0, tensor var_2385_1 = split(axis = var_2385_axis_0, split_sizes = var_2385_split_sizes_0, x = query_states_27_cast_fp16)[name = string("op_2385")]; bool attn_weights_65_transpose_x_0 = const()[name = string("attn_weights_65_transpose_x_0"), val = bool(false)]; bool attn_weights_65_transpose_y_0 = const()[name = string("attn_weights_65_transpose_y_0"), val = bool(false)]; tensor attn_weights_65_cast_fp16 = matmul(transpose_x = attn_weights_65_transpose_x_0, transpose_y = attn_weights_65_transpose_y_0, x = var_2372_cast_fp16_0, y = var_2385_0)[name = string("attn_weights_65_cast_fp16")]; fp16 var_2388_to_fp16 = const()[name = string("op_2388_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_67_cast_fp16 = mul(x = attn_weights_65_cast_fp16, y = var_2388_to_fp16)[name = string("attn_weights_67_cast_fp16")]; tensor attn_weights_69_cast_fp16 = add(x = attn_weights_67_cast_fp16, y = attn_mask_1)[name = string("attn_weights_69_cast_fp16")]; int32 var_2392 = const()[name = string("op_2392"), val = int32(-2)]; tensor attn_weights_71_cast_fp16 = softmax(axis = var_2392, x = attn_weights_69_cast_fp16)[name = string("attn_weights_71_cast_fp16")]; bool var_2398_transpose_x_1 = const()[name = string("op_2398_transpose_x_1"), val = bool(true)]; bool var_2398_transpose_y_1 = const()[name = string("op_2398_transpose_y_1"), val = bool(false)]; tensor var_2398_cast_fp16 = matmul(transpose_x = var_2398_transpose_x_1, transpose_y = var_2398_transpose_y_1, x = attn_weights_71_cast_fp16, y = var_2382_cast_fp16_0)[name = string("op_2398_cast_fp16")]; bool attn_weights_73_transpose_x_0 = const()[name = string("attn_weights_73_transpose_x_0"), val = bool(false)]; bool attn_weights_73_transpose_y_0 = const()[name = string("attn_weights_73_transpose_y_0"), val = bool(false)]; tensor attn_weights_73_cast_fp16 = matmul(transpose_x = attn_weights_73_transpose_x_0, transpose_y = attn_weights_73_transpose_y_0, x = var_2372_cast_fp16_1, y = var_2385_1)[name = string("attn_weights_73_cast_fp16")]; fp16 var_2400_to_fp16 = const()[name = string("op_2400_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_75_cast_fp16 = mul(x = attn_weights_73_cast_fp16, y = var_2400_to_fp16)[name = string("attn_weights_75_cast_fp16")]; tensor attn_weights_77_cast_fp16 = add(x = attn_weights_75_cast_fp16, y = attn_mask_1)[name = string("attn_weights_77_cast_fp16")]; int32 var_2404 = const()[name = string("op_2404"), val = int32(-2)]; tensor attn_weights_79_cast_fp16 = softmax(axis = var_2404, x = attn_weights_77_cast_fp16)[name = string("attn_weights_79_cast_fp16")]; bool attn_output_33_transpose_x_1 = const()[name = string("attn_output_33_transpose_x_1"), val = bool(true)]; bool attn_output_33_transpose_y_1 = const()[name = string("attn_output_33_transpose_y_1"), val = bool(false)]; tensor attn_output_33_cast_fp16 = matmul(transpose_x = attn_output_33_transpose_x_1, transpose_y = attn_output_33_transpose_y_1, x = attn_weights_79_cast_fp16, y = var_2382_cast_fp16_1)[name = string("attn_output_33_cast_fp16")]; int32 var_2412 = const()[name = string("op_2412"), val = int32(1)]; bool attn_output_35_interleave_0 = const()[name = string("attn_output_35_interleave_0"), val = bool(false)]; tensor attn_output_35_cast_fp16 = concat(axis = var_2412, interleave = attn_output_35_interleave_0, values = (var_2398_cast_fp16, attn_output_33_cast_fp16))[name = string("attn_output_35_cast_fp16")]; tensor var_2416_perm_0 = const()[name = string("op_2416_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_59x = const()[name = string("concat_59x"), val = tensor([1, 2048, 1, -1])]; tensor var_2416_cast_fp16 = transpose(perm = var_2416_perm_0, x = attn_output_35_cast_fp16)[name = string("transpose_413")]; tensor attn_output_39_cast_fp16 = reshape(shape = concat_59x, x = var_2416_cast_fp16)[name = string("attn_output_39_cast_fp16")]; tensor hidden_states_43_strides_0 = const()[name = string("hidden_states_43_strides_0"), val = tensor([1, 1])]; string hidden_states_43_pad_type_0 = const()[name = string("hidden_states_43_pad_type_0"), val = string("valid")]; tensor hidden_states_43_pad_0 = const()[name = string("hidden_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_43_dilations_0 = const()[name = string("hidden_states_43_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_43_groups_0 = const()[name = string("hidden_states_43_groups_0"), val = int32(1)]; tensor hidden_states_43_cast_fp16 = conv(dilations = hidden_states_43_dilations_0, groups = hidden_states_43_groups_0, pad = hidden_states_43_pad_0, pad_type = hidden_states_43_pad_type_0, strides = hidden_states_43_strides_0, weight = layers_4_self_attn_o_proj_weight_cast_fp16, x = attn_output_39_cast_fp16)[name = string("hidden_states_43_cast_fp16")]; tensor hidden_states_45_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = hidden_states_43_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; fp16 const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2449_cast_fp16 = mul(x = hidden_states_45_cast_fp16, y = const_48_promoted_to_fp16)[name = string("op_2449_cast_fp16")]; int32 var_2447 = const()[name = string("op_2447"), val = int32(1)]; bool doubled_37_interleave_0 = const()[name = string("doubled_37_interleave_0"), val = bool(false)]; tensor doubled_37_cast_fp16 = concat(axis = var_2447, interleave = doubled_37_interleave_0, values = (hidden_states_45_cast_fp16, var_2449_cast_fp16))[name = string("doubled_37_cast_fp16")]; tensor out_19_axes_0 = const()[name = string("out_19_axes_0"), val = tensor([1])]; tensor out_19_gamma_0_to_fp16 = const()[name = string("out_19_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283129280)))]; fp16 var_2459_to_fp16 = const()[name = string("op_2459_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_19_cast_fp16 = layer_norm(axes = out_19_axes_0, epsilon = var_2459_to_fp16, gamma = out_19_gamma_0_to_fp16, x = doubled_37_cast_fp16)[name = string("out_19_cast_fp16")]; tensor var_2470_split_sizes_0 = const()[name = string("op_2470_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2470_axis_0 = const()[name = string("op_2470_axis_0"), val = int32(1)]; tensor var_2470_cast_fp16_0, tensor var_2470_cast_fp16_1 = split(axis = var_2470_axis_0, split_sizes = var_2470_split_sizes_0, x = out_19_cast_fp16)[name = string("op_2470_cast_fp16")]; tensor input_9_strides_0 = const()[name = string("input_9_strides_0"), val = tensor([1, 1])]; string input_9_pad_type_0 = const()[name = string("input_9_pad_type_0"), val = string("valid")]; tensor input_9_pad_0 = const()[name = string("input_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_9_dilations_0 = const()[name = string("input_9_dilations_0"), val = tensor([1, 1])]; int32 input_9_groups_0 = const()[name = string("input_9_groups_0"), val = int32(1)]; tensor input_9_cast_fp16 = conv(dilations = input_9_dilations_0, groups = input_9_groups_0, pad = input_9_pad_0, pad_type = input_9_pad_type_0, strides = input_9_strides_0, weight = layers_4_mlp_gate_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("input_9_cast_fp16")]; tensor var_2487_cast_fp16 = silu(x = input_9_cast_fp16)[name = string("op_2487_cast_fp16")]; tensor var_2493_strides_0 = const()[name = string("op_2493_strides_0"), val = tensor([1, 1])]; string var_2493_pad_type_0 = const()[name = string("op_2493_pad_type_0"), val = string("valid")]; tensor var_2493_pad_0 = const()[name = string("op_2493_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2493_dilations_0 = const()[name = string("op_2493_dilations_0"), val = tensor([1, 1])]; int32 var_2493_groups_0 = const()[name = string("op_2493_groups_0"), val = int32(1)]; tensor var_2493_cast_fp16 = conv(dilations = var_2493_dilations_0, groups = var_2493_groups_0, pad = var_2493_pad_0, pad_type = var_2493_pad_type_0, strides = var_2493_strides_0, weight = layers_4_mlp_up_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("op_2493_cast_fp16")]; tensor x_49_cast_fp16 = mul(x = var_2487_cast_fp16, y = var_2493_cast_fp16)[name = string("x_49_cast_fp16")]; tensor hidden_states_47_strides_0 = const()[name = string("hidden_states_47_strides_0"), val = tensor([1, 1])]; string hidden_states_47_pad_type_0 = const()[name = string("hidden_states_47_pad_type_0"), val = string("valid")]; tensor hidden_states_47_pad_0 = const()[name = string("hidden_states_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_47_dilations_0 = const()[name = string("hidden_states_47_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_47_groups_0 = const()[name = string("hidden_states_47_groups_0"), val = int32(1)]; tensor hidden_states_47_cast_fp16 = conv(dilations = hidden_states_47_dilations_0, groups = hidden_states_47_groups_0, pad = hidden_states_47_pad_0, pad_type = hidden_states_47_pad_type_0, strides = hidden_states_47_strides_0, weight = layers_4_mlp_down_proj_weight_cast_fp16, x = x_49_cast_fp16)[name = string("hidden_states_47_cast_fp16")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = hidden_states_47_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; fp16 const_50_promoted_to_fp16 = const()[name = string("const_50_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2511_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_50_promoted_to_fp16)[name = string("op_2511_cast_fp16")]; int32 var_2509 = const()[name = string("op_2509"), val = int32(1)]; bool doubled_41_interleave_0 = const()[name = string("doubled_41_interleave_0"), val = bool(false)]; tensor doubled_41_cast_fp16 = concat(axis = var_2509, interleave = doubled_41_interleave_0, values = (hidden_states_49_cast_fp16, var_2511_cast_fp16))[name = string("doubled_41_cast_fp16")]; tensor out_21_axes_0 = const()[name = string("out_21_axes_0"), val = tensor([1])]; tensor out_21_gamma_0_to_fp16 = const()[name = string("out_21_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283137536)))]; fp16 var_2521_to_fp16 = const()[name = string("op_2521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_21_cast_fp16 = layer_norm(axes = out_21_axes_0, epsilon = var_2521_to_fp16, gamma = out_21_gamma_0_to_fp16, x = doubled_41_cast_fp16)[name = string("out_21_cast_fp16")]; tensor var_2532_split_sizes_0 = const()[name = string("op_2532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2532_axis_0 = const()[name = string("op_2532_axis_0"), val = int32(1)]; tensor var_2532_cast_fp16_0, tensor var_2532_cast_fp16_1 = split(axis = var_2532_axis_0, split_sizes = var_2532_split_sizes_0, x = out_21_cast_fp16)[name = string("op_2532_cast_fp16")]; tensor layers_5_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283145792)))]; tensor query_states_31_strides_0 = const()[name = string("query_states_31_strides_0"), val = tensor([1, 1])]; string query_states_31_pad_type_0 = const()[name = string("query_states_31_pad_type_0"), val = string("valid")]; tensor query_states_31_pad_0 = const()[name = string("query_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_31_dilations_0 = const()[name = string("query_states_31_dilations_0"), val = tensor([1, 1])]; int32 query_states_31_groups_0 = const()[name = string("query_states_31_groups_0"), val = int32(1)]; tensor query_states_31_cast_fp16 = conv(dilations = query_states_31_dilations_0, groups = query_states_31_groups_0, pad = query_states_31_pad_0, pad_type = query_states_31_pad_type_0, strides = query_states_31_strides_0, weight = layers_5_self_attn_q_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("query_states_31_cast_fp16")]; tensor layers_5_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1291534464)))]; tensor key_states_51_strides_0 = const()[name = string("key_states_51_strides_0"), val = tensor([1, 1])]; string key_states_51_pad_type_0 = const()[name = string("key_states_51_pad_type_0"), val = string("valid")]; tensor key_states_51_pad_0 = const()[name = string("key_states_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_51_dilations_0 = const()[name = string("key_states_51_dilations_0"), val = tensor([1, 1])]; int32 key_states_51_groups_0 = const()[name = string("key_states_51_groups_0"), val = int32(1)]; tensor key_states_51_cast_fp16 = conv(dilations = key_states_51_dilations_0, groups = key_states_51_groups_0, pad = key_states_51_pad_0, pad_type = key_states_51_pad_type_0, strides = key_states_51_strides_0, weight = layers_5_self_attn_k_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("key_states_51_cast_fp16")]; tensor value_states_31_strides_0 = const()[name = string("value_states_31_strides_0"), val = tensor([1, 1])]; string value_states_31_pad_type_0 = const()[name = string("value_states_31_pad_type_0"), val = string("valid")]; tensor value_states_31_pad_0 = const()[name = string("value_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_31_dilations_0 = const()[name = string("value_states_31_dilations_0"), val = tensor([1, 1])]; int32 value_states_31_groups_0 = const()[name = string("value_states_31_groups_0"), val = int32(1)]; tensor value_states_31_cast_fp16 = conv(dilations = value_states_31_dilations_0, groups = value_states_31_groups_0, pad = value_states_31_pad_0, pad_type = value_states_31_pad_type_0, strides = value_states_31_strides_0, weight = layers_5_self_attn_v_proj_weight_cast_fp16, x = var_2532_cast_fp16_0)[name = string("value_states_31_cast_fp16")]; tensor concat_60x = const()[name = string("concat_60x"), val = tensor([1, 16, 128, -1])]; tensor x_51_cast_fp16 = reshape(shape = concat_60x, x = query_states_31_cast_fp16)[name = string("x_51_cast_fp16")]; tensor concat_61x = const()[name = string("concat_61x"), val = tensor([1, 2, 128, -1])]; tensor var_2589_cast_fp16 = reshape(shape = concat_61x, x = key_states_51_cast_fp16)[name = string("op_2589_cast_fp16")]; tensor concat_62x = const()[name = string("concat_62x"), val = tensor([1, 2, 128, -1])]; tensor var_2596_cast_fp16 = reshape(shape = concat_62x, x = value_states_31_cast_fp16)[name = string("op_2596_cast_fp16")]; tensor var_2600_cast_fp16 = mul(x = x_51_cast_fp16, y = var_869_cast_fp16)[name = string("op_2600_cast_fp16")]; tensor var_2601_split_sizes_0 = const()[name = string("op_2601_split_sizes_0"), val = tensor([64, 64])]; int32 var_2601_axis_0 = const()[name = string("op_2601_axis_0"), val = int32(-2)]; tensor var_2601_cast_fp16_0, tensor var_2601_cast_fp16_1 = split(axis = var_2601_axis_0, split_sizes = var_2601_split_sizes_0, x = x_51_cast_fp16)[name = string("op_2601_cast_fp16")]; fp16 const_52_promoted_to_fp16 = const()[name = string("const_52_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2603_cast_fp16 = mul(x = var_2601_cast_fp16_1, y = const_52_promoted_to_fp16)[name = string("op_2603_cast_fp16")]; int32 var_2605 = const()[name = string("op_2605"), val = int32(-2)]; bool var_2606_interleave_0 = const()[name = string("op_2606_interleave_0"), val = bool(false)]; tensor var_2606_cast_fp16 = concat(axis = var_2605, interleave = var_2606_interleave_0, values = (var_2603_cast_fp16, var_2601_cast_fp16_0))[name = string("op_2606_cast_fp16")]; tensor var_2607_cast_fp16 = mul(x = var_2606_cast_fp16, y = var_878_cast_fp16)[name = string("op_2607_cast_fp16")]; tensor query_states_33_cast_fp16 = add(x = var_2600_cast_fp16, y = var_2607_cast_fp16)[name = string("query_states_33_cast_fp16")]; tensor var_2613_cast_fp16 = mul(x = var_2589_cast_fp16, y = var_869_cast_fp16)[name = string("op_2613_cast_fp16")]; tensor var_2614_split_sizes_0 = const()[name = string("op_2614_split_sizes_0"), val = tensor([64, 64])]; int32 var_2614_axis_0 = const()[name = string("op_2614_axis_0"), val = int32(-2)]; tensor var_2614_cast_fp16_0, tensor var_2614_cast_fp16_1 = split(axis = var_2614_axis_0, split_sizes = var_2614_split_sizes_0, x = var_2589_cast_fp16)[name = string("op_2614_cast_fp16")]; fp16 const_53_promoted_to_fp16 = const()[name = string("const_53_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2616_cast_fp16 = mul(x = var_2614_cast_fp16_1, y = const_53_promoted_to_fp16)[name = string("op_2616_cast_fp16")]; int32 var_2618 = const()[name = string("op_2618"), val = int32(-2)]; bool var_2619_interleave_0 = const()[name = string("op_2619_interleave_0"), val = bool(false)]; tensor var_2619_cast_fp16 = concat(axis = var_2618, interleave = var_2619_interleave_0, values = (var_2616_cast_fp16, var_2614_cast_fp16_0))[name = string("op_2619_cast_fp16")]; tensor var_2620_cast_fp16 = mul(x = var_2619_cast_fp16, y = var_878_cast_fp16)[name = string("op_2620_cast_fp16")]; tensor key_states_55_cast_fp16 = add(x = var_2613_cast_fp16, y = var_2620_cast_fp16)[name = string("key_states_55_cast_fp16")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; int32 concat_65_axis_0 = const()[name = string("concat_65_axis_0"), val = int32(0)]; bool concat_65_interleave_0 = const()[name = string("concat_65_interleave_0"), val = bool(false)]; tensor concat_65 = concat(axis = concat_65_axis_0, interleave = concat_65_interleave_0, values = (expand_dims_60, expand_dims_61, position_id, expand_dims_63))[name = string("concat_65")]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; tensor concat_66_values1_0 = const()[name = string("concat_66_values1_0"), val = tensor([0])]; tensor concat_66_values3_0 = const()[name = string("concat_66_values3_0"), val = tensor([0])]; int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_64, concat_66_values1_0, cache_position_end, concat_66_values3_0))[name = string("concat_66")]; tensor key_states_57_perm_0 = const()[name = string("key_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_6_stride_0 = const()[name = string("key_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_57_cast_fp16 = transpose(perm = key_states_57_perm_0, x = key_states_55_cast_fp16)[name = string("transpose_412")]; tensor key_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = key_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = key_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_6_squeeze_mask_0, stride = key_cache_internal_tensor_assign_6_stride_0, update = key_states_57_cast_fp16, x = coreml_update_state_232)[name = string("key_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_6_cast_fp16, input = key_cache)[name = string("coreml_update_state_234_write_state")]; tensor coreml_update_state_234 = read_state(input = key_cache)[name = string("coreml_update_state_234")]; tensor value_states_33_perm_0 = const()[name = string("value_states_33_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_6_stride_0 = const()[name = string("value_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_33_cast_fp16 = transpose(perm = value_states_33_perm_0, x = var_2596_cast_fp16)[name = string("transpose_411")]; tensor value_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = value_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = value_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_6_squeeze_mask_0, stride = value_cache_internal_tensor_assign_6_stride_0, update = value_states_33_cast_fp16, x = coreml_update_state_233)[name = string("value_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_6_cast_fp16, input = value_cache)[name = string("coreml_update_state_235_write_state")]; tensor coreml_update_state_235 = read_state(input = value_cache)[name = string("coreml_update_state_235")]; tensor var_2690_begin_0 = const()[name = string("op_2690_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2690_end_0 = const()[name = string("op_2690_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2690_end_mask_0 = const()[name = string("op_2690_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2690_cast_fp16 = slice_by_index(begin = var_2690_begin_0, end = var_2690_end_0, end_mask = var_2690_end_mask_0, x = coreml_update_state_234)[name = string("op_2690_cast_fp16")]; tensor tile_10 = const()[name = string("tile_10"), val = tensor([1, 1])]; int32 var_2693_axis_0 = const()[name = string("op_2693_axis_0"), val = int32(1)]; tensor var_2693_cast_fp16_0, tensor var_2693_cast_fp16_1 = split(axis = var_2693_axis_0, split_sizes = tile_10, x = var_2690_cast_fp16)[name = string("op_2693_cast_fp16")]; tensor var_2700_begin_0 = const()[name = string("op_2700_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2700_end_0 = const()[name = string("op_2700_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2700_end_mask_0 = const()[name = string("op_2700_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2700_cast_fp16 = slice_by_index(begin = var_2700_begin_0, end = var_2700_end_0, end_mask = var_2700_end_mask_0, x = coreml_update_state_235)[name = string("op_2700_cast_fp16")]; tensor tile_11 = const()[name = string("tile_11"), val = tensor([1, 1])]; int32 var_2703_axis_0 = const()[name = string("op_2703_axis_0"), val = int32(1)]; tensor var_2703_cast_fp16_0, tensor var_2703_cast_fp16_1 = split(axis = var_2703_axis_0, split_sizes = tile_11, x = var_2700_cast_fp16)[name = string("op_2703_cast_fp16")]; tensor var_2706_split_sizes_0 = const()[name = string("op_2706_split_sizes_0"), val = tensor([8, 8])]; int32 var_2706_axis_0 = const()[name = string("op_2706_axis_0"), val = int32(1)]; tensor var_2706_0, tensor var_2706_1 = split(axis = var_2706_axis_0, split_sizes = var_2706_split_sizes_0, x = query_states_33_cast_fp16)[name = string("op_2706")]; bool attn_weights_81_transpose_x_0 = const()[name = string("attn_weights_81_transpose_x_0"), val = bool(false)]; bool attn_weights_81_transpose_y_0 = const()[name = string("attn_weights_81_transpose_y_0"), val = bool(false)]; tensor attn_weights_81_cast_fp16 = matmul(transpose_x = attn_weights_81_transpose_x_0, transpose_y = attn_weights_81_transpose_y_0, x = var_2693_cast_fp16_0, y = var_2706_0)[name = string("attn_weights_81_cast_fp16")]; fp16 var_2709_to_fp16 = const()[name = string("op_2709_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_83_cast_fp16 = mul(x = attn_weights_81_cast_fp16, y = var_2709_to_fp16)[name = string("attn_weights_83_cast_fp16")]; tensor attn_weights_85_cast_fp16 = add(x = attn_weights_83_cast_fp16, y = attn_mask_1)[name = string("attn_weights_85_cast_fp16")]; int32 var_2713 = const()[name = string("op_2713"), val = int32(-2)]; tensor attn_weights_87_cast_fp16 = softmax(axis = var_2713, x = attn_weights_85_cast_fp16)[name = string("attn_weights_87_cast_fp16")]; bool var_2719_transpose_x_1 = const()[name = string("op_2719_transpose_x_1"), val = bool(true)]; bool var_2719_transpose_y_1 = const()[name = string("op_2719_transpose_y_1"), val = bool(false)]; tensor var_2719_cast_fp16 = matmul(transpose_x = var_2719_transpose_x_1, transpose_y = var_2719_transpose_y_1, x = attn_weights_87_cast_fp16, y = var_2703_cast_fp16_0)[name = string("op_2719_cast_fp16")]; bool attn_weights_89_transpose_x_0 = const()[name = string("attn_weights_89_transpose_x_0"), val = bool(false)]; bool attn_weights_89_transpose_y_0 = const()[name = string("attn_weights_89_transpose_y_0"), val = bool(false)]; tensor attn_weights_89_cast_fp16 = matmul(transpose_x = attn_weights_89_transpose_x_0, transpose_y = attn_weights_89_transpose_y_0, x = var_2693_cast_fp16_1, y = var_2706_1)[name = string("attn_weights_89_cast_fp16")]; fp16 var_2721_to_fp16 = const()[name = string("op_2721_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_91_cast_fp16 = mul(x = attn_weights_89_cast_fp16, y = var_2721_to_fp16)[name = string("attn_weights_91_cast_fp16")]; tensor attn_weights_93_cast_fp16 = add(x = attn_weights_91_cast_fp16, y = attn_mask_1)[name = string("attn_weights_93_cast_fp16")]; int32 var_2725 = const()[name = string("op_2725"), val = int32(-2)]; tensor attn_weights_95_cast_fp16 = softmax(axis = var_2725, x = attn_weights_93_cast_fp16)[name = string("attn_weights_95_cast_fp16")]; bool attn_output_41_transpose_x_1 = const()[name = string("attn_output_41_transpose_x_1"), val = bool(true)]; bool attn_output_41_transpose_y_1 = const()[name = string("attn_output_41_transpose_y_1"), val = bool(false)]; tensor attn_output_41_cast_fp16 = matmul(transpose_x = attn_output_41_transpose_x_1, transpose_y = attn_output_41_transpose_y_1, x = attn_weights_95_cast_fp16, y = var_2703_cast_fp16_1)[name = string("attn_output_41_cast_fp16")]; int32 var_2733 = const()[name = string("op_2733"), val = int32(1)]; bool attn_output_43_interleave_0 = const()[name = string("attn_output_43_interleave_0"), val = bool(false)]; tensor attn_output_43_cast_fp16 = concat(axis = var_2733, interleave = attn_output_43_interleave_0, values = (var_2719_cast_fp16, attn_output_41_cast_fp16))[name = string("attn_output_43_cast_fp16")]; tensor var_2737_perm_0 = const()[name = string("op_2737_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_71x = const()[name = string("concat_71x"), val = tensor([1, 2048, 1, -1])]; tensor var_2737_cast_fp16 = transpose(perm = var_2737_perm_0, x = attn_output_43_cast_fp16)[name = string("transpose_410")]; tensor attn_output_47_cast_fp16 = reshape(shape = concat_71x, x = var_2737_cast_fp16)[name = string("attn_output_47_cast_fp16")]; tensor hidden_states_53_strides_0 = const()[name = string("hidden_states_53_strides_0"), val = tensor([1, 1])]; string hidden_states_53_pad_type_0 = const()[name = string("hidden_states_53_pad_type_0"), val = string("valid")]; tensor hidden_states_53_pad_0 = const()[name = string("hidden_states_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_53_dilations_0 = const()[name = string("hidden_states_53_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_53_groups_0 = const()[name = string("hidden_states_53_groups_0"), val = int32(1)]; tensor hidden_states_53_cast_fp16 = conv(dilations = hidden_states_53_dilations_0, groups = hidden_states_53_groups_0, pad = hidden_states_53_pad_0, pad_type = hidden_states_53_pad_type_0, strides = hidden_states_53_strides_0, weight = layers_5_self_attn_o_proj_weight_cast_fp16, x = attn_output_47_cast_fp16)[name = string("hidden_states_53_cast_fp16")]; tensor hidden_states_55_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = hidden_states_53_cast_fp16)[name = string("hidden_states_55_cast_fp16")]; fp16 const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2770_cast_fp16 = mul(x = hidden_states_55_cast_fp16, y = const_58_promoted_to_fp16)[name = string("op_2770_cast_fp16")]; int32 var_2768 = const()[name = string("op_2768"), val = int32(1)]; bool doubled_45_interleave_0 = const()[name = string("doubled_45_interleave_0"), val = bool(false)]; tensor doubled_45_cast_fp16 = concat(axis = var_2768, interleave = doubled_45_interleave_0, values = (hidden_states_55_cast_fp16, var_2770_cast_fp16))[name = string("doubled_45_cast_fp16")]; tensor out_23_axes_0 = const()[name = string("out_23_axes_0"), val = tensor([1])]; tensor out_23_gamma_0_to_fp16 = const()[name = string("out_23_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292583104)))]; fp16 var_2780_to_fp16 = const()[name = string("op_2780_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_23_cast_fp16 = layer_norm(axes = out_23_axes_0, epsilon = var_2780_to_fp16, gamma = out_23_gamma_0_to_fp16, x = doubled_45_cast_fp16)[name = string("out_23_cast_fp16")]; tensor var_2791_split_sizes_0 = const()[name = string("op_2791_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2791_axis_0 = const()[name = string("op_2791_axis_0"), val = int32(1)]; tensor var_2791_cast_fp16_0, tensor var_2791_cast_fp16_1 = split(axis = var_2791_axis_0, split_sizes = var_2791_split_sizes_0, x = out_23_cast_fp16)[name = string("op_2791_cast_fp16")]; tensor layers_5_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_5_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292591360)))]; tensor input_11_strides_0 = const()[name = string("input_11_strides_0"), val = tensor([1, 1])]; string input_11_pad_type_0 = const()[name = string("input_11_pad_type_0"), val = string("valid")]; tensor input_11_pad_0 = const()[name = string("input_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_11_dilations_0 = const()[name = string("input_11_dilations_0"), val = tensor([1, 1])]; int32 input_11_groups_0 = const()[name = string("input_11_groups_0"), val = int32(1)]; tensor input_11_cast_fp16 = conv(dilations = input_11_dilations_0, groups = input_11_groups_0, pad = input_11_pad_0, pad_type = input_11_pad_type_0, strides = input_11_strides_0, weight = layers_5_mlp_gate_proj_weight_to_fp16, x = var_2791_cast_fp16_0)[name = string("input_11_cast_fp16")]; tensor var_2808_cast_fp16 = silu(x = input_11_cast_fp16)[name = string("op_2808_cast_fp16")]; tensor var_2814_strides_0 = const()[name = string("op_2814_strides_0"), val = tensor([1, 1])]; string var_2814_pad_type_0 = const()[name = string("op_2814_pad_type_0"), val = string("valid")]; tensor var_2814_pad_0 = const()[name = string("op_2814_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2814_dilations_0 = const()[name = string("op_2814_dilations_0"), val = tensor([1, 1])]; int32 var_2814_groups_0 = const()[name = string("op_2814_groups_0"), val = int32(1)]; tensor var_2814_cast_fp16 = conv(dilations = var_2814_dilations_0, groups = var_2814_groups_0, pad = var_2814_pad_0, pad_type = var_2814_pad_type_0, strides = var_2814_strides_0, weight = layers_5_mlp_up_proj_weight_cast_fp16, x = var_2791_cast_fp16_0)[name = string("op_2814_cast_fp16")]; tensor x_59_cast_fp16 = mul(x = var_2808_cast_fp16, y = var_2814_cast_fp16)[name = string("x_59_cast_fp16")]; tensor hidden_states_57_strides_0 = const()[name = string("hidden_states_57_strides_0"), val = tensor([1, 1])]; string hidden_states_57_pad_type_0 = const()[name = string("hidden_states_57_pad_type_0"), val = string("valid")]; tensor hidden_states_57_pad_0 = const()[name = string("hidden_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_57_dilations_0 = const()[name = string("hidden_states_57_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_57_groups_0 = const()[name = string("hidden_states_57_groups_0"), val = int32(1)]; tensor hidden_states_57_cast_fp16 = conv(dilations = hidden_states_57_dilations_0, groups = hidden_states_57_groups_0, pad = hidden_states_57_pad_0, pad_type = hidden_states_57_pad_type_0, strides = hidden_states_57_strides_0, weight = layers_5_mlp_down_proj_weight_cast_fp16, x = x_59_cast_fp16)[name = string("hidden_states_57_cast_fp16")]; tensor hidden_states_59_cast_fp16 = add(x = hidden_states_55_cast_fp16, y = hidden_states_57_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2832_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_2832_cast_fp16")]; int32 var_2830 = const()[name = string("op_2830"), val = int32(1)]; bool doubled_49_interleave_0 = const()[name = string("doubled_49_interleave_0"), val = bool(false)]; tensor doubled_49_cast_fp16 = concat(axis = var_2830, interleave = doubled_49_interleave_0, values = (hidden_states_59_cast_fp16, var_2832_cast_fp16))[name = string("doubled_49_cast_fp16")]; tensor out_25_axes_0 = const()[name = string("out_25_axes_0"), val = tensor([1])]; tensor out_25_gamma_0_to_fp16 = const()[name = string("out_25_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317757248)))]; fp16 var_2842_to_fp16 = const()[name = string("op_2842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_25_cast_fp16 = layer_norm(axes = out_25_axes_0, epsilon = var_2842_to_fp16, gamma = out_25_gamma_0_to_fp16, x = doubled_49_cast_fp16)[name = string("out_25_cast_fp16")]; tensor var_2853_split_sizes_0 = const()[name = string("op_2853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2853_axis_0 = const()[name = string("op_2853_axis_0"), val = int32(1)]; tensor var_2853_cast_fp16_0, tensor var_2853_cast_fp16_1 = split(axis = var_2853_axis_0, split_sizes = var_2853_split_sizes_0, x = out_25_cast_fp16)[name = string("op_2853_cast_fp16")]; tensor layers_6_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317765504)))]; tensor query_states_37_strides_0 = const()[name = string("query_states_37_strides_0"), val = tensor([1, 1])]; string query_states_37_pad_type_0 = const()[name = string("query_states_37_pad_type_0"), val = string("valid")]; tensor query_states_37_pad_0 = const()[name = string("query_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_37_dilations_0 = const()[name = string("query_states_37_dilations_0"), val = tensor([1, 1])]; int32 query_states_37_groups_0 = const()[name = string("query_states_37_groups_0"), val = int32(1)]; tensor query_states_37_cast_fp16 = conv(dilations = query_states_37_dilations_0, groups = query_states_37_groups_0, pad = query_states_37_pad_0, pad_type = query_states_37_pad_type_0, strides = query_states_37_strides_0, weight = layers_6_self_attn_q_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("query_states_37_cast_fp16")]; tensor layers_6_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1326154176)))]; tensor key_states_61_strides_0 = const()[name = string("key_states_61_strides_0"), val = tensor([1, 1])]; string key_states_61_pad_type_0 = const()[name = string("key_states_61_pad_type_0"), val = string("valid")]; tensor key_states_61_pad_0 = const()[name = string("key_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_61_dilations_0 = const()[name = string("key_states_61_dilations_0"), val = tensor([1, 1])]; int32 key_states_61_groups_0 = const()[name = string("key_states_61_groups_0"), val = int32(1)]; tensor key_states_61_cast_fp16 = conv(dilations = key_states_61_dilations_0, groups = key_states_61_groups_0, pad = key_states_61_pad_0, pad_type = key_states_61_pad_type_0, strides = key_states_61_strides_0, weight = layers_6_self_attn_k_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("key_states_61_cast_fp16")]; tensor value_states_37_strides_0 = const()[name = string("value_states_37_strides_0"), val = tensor([1, 1])]; string value_states_37_pad_type_0 = const()[name = string("value_states_37_pad_type_0"), val = string("valid")]; tensor value_states_37_pad_0 = const()[name = string("value_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_37_dilations_0 = const()[name = string("value_states_37_dilations_0"), val = tensor([1, 1])]; int32 value_states_37_groups_0 = const()[name = string("value_states_37_groups_0"), val = int32(1)]; tensor value_states_37_cast_fp16 = conv(dilations = value_states_37_dilations_0, groups = value_states_37_groups_0, pad = value_states_37_pad_0, pad_type = value_states_37_pad_type_0, strides = value_states_37_strides_0, weight = layers_6_self_attn_v_proj_weight_cast_fp16, x = var_2853_cast_fp16_0)[name = string("value_states_37_cast_fp16")]; tensor concat_72x = const()[name = string("concat_72x"), val = tensor([1, 16, 128, -1])]; tensor x_61_cast_fp16 = reshape(shape = concat_72x, x = query_states_37_cast_fp16)[name = string("x_61_cast_fp16")]; tensor concat_73x = const()[name = string("concat_73x"), val = tensor([1, 2, 128, -1])]; tensor var_2910_cast_fp16 = reshape(shape = concat_73x, x = key_states_61_cast_fp16)[name = string("op_2910_cast_fp16")]; tensor concat_74x = const()[name = string("concat_74x"), val = tensor([1, 2, 128, -1])]; tensor var_2917_cast_fp16 = reshape(shape = concat_74x, x = value_states_37_cast_fp16)[name = string("op_2917_cast_fp16")]; tensor var_2921_cast_fp16 = mul(x = x_61_cast_fp16, y = var_869_cast_fp16)[name = string("op_2921_cast_fp16")]; tensor var_2922_split_sizes_0 = const()[name = string("op_2922_split_sizes_0"), val = tensor([64, 64])]; int32 var_2922_axis_0 = const()[name = string("op_2922_axis_0"), val = int32(-2)]; tensor var_2922_cast_fp16_0, tensor var_2922_cast_fp16_1 = split(axis = var_2922_axis_0, split_sizes = var_2922_split_sizes_0, x = x_61_cast_fp16)[name = string("op_2922_cast_fp16")]; fp16 const_62_promoted_to_fp16 = const()[name = string("const_62_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2924_cast_fp16 = mul(x = var_2922_cast_fp16_1, y = const_62_promoted_to_fp16)[name = string("op_2924_cast_fp16")]; int32 var_2926 = const()[name = string("op_2926"), val = int32(-2)]; bool var_2927_interleave_0 = const()[name = string("op_2927_interleave_0"), val = bool(false)]; tensor var_2927_cast_fp16 = concat(axis = var_2926, interleave = var_2927_interleave_0, values = (var_2924_cast_fp16, var_2922_cast_fp16_0))[name = string("op_2927_cast_fp16")]; tensor var_2928_cast_fp16 = mul(x = var_2927_cast_fp16, y = var_878_cast_fp16)[name = string("op_2928_cast_fp16")]; tensor query_states_39_cast_fp16 = add(x = var_2921_cast_fp16, y = var_2928_cast_fp16)[name = string("query_states_39_cast_fp16")]; tensor var_2934_cast_fp16 = mul(x = var_2910_cast_fp16, y = var_869_cast_fp16)[name = string("op_2934_cast_fp16")]; tensor var_2935_split_sizes_0 = const()[name = string("op_2935_split_sizes_0"), val = tensor([64, 64])]; int32 var_2935_axis_0 = const()[name = string("op_2935_axis_0"), val = int32(-2)]; tensor var_2935_cast_fp16_0, tensor var_2935_cast_fp16_1 = split(axis = var_2935_axis_0, split_sizes = var_2935_split_sizes_0, x = var_2910_cast_fp16)[name = string("op_2935_cast_fp16")]; fp16 const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2937_cast_fp16 = mul(x = var_2935_cast_fp16_1, y = const_63_promoted_to_fp16)[name = string("op_2937_cast_fp16")]; int32 var_2939 = const()[name = string("op_2939"), val = int32(-2)]; bool var_2940_interleave_0 = const()[name = string("op_2940_interleave_0"), val = bool(false)]; tensor var_2940_cast_fp16 = concat(axis = var_2939, interleave = var_2940_interleave_0, values = (var_2937_cast_fp16, var_2935_cast_fp16_0))[name = string("op_2940_cast_fp16")]; tensor var_2941_cast_fp16 = mul(x = var_2940_cast_fp16, y = var_878_cast_fp16)[name = string("op_2941_cast_fp16")]; tensor key_states_65_cast_fp16 = add(x = var_2934_cast_fp16, y = var_2941_cast_fp16)[name = string("key_states_65_cast_fp16")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; int32 concat_77_axis_0 = const()[name = string("concat_77_axis_0"), val = int32(0)]; bool concat_77_interleave_0 = const()[name = string("concat_77_interleave_0"), val = bool(false)]; tensor concat_77 = concat(axis = concat_77_axis_0, interleave = concat_77_interleave_0, values = (expand_dims_72, expand_dims_73, position_id, expand_dims_75))[name = string("concat_77")]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; tensor concat_78_values1_0 = const()[name = string("concat_78_values1_0"), val = tensor([0])]; tensor concat_78_values3_0 = const()[name = string("concat_78_values3_0"), val = tensor([0])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_76, concat_78_values1_0, cache_position_end, concat_78_values3_0))[name = string("concat_78")]; tensor key_states_67_perm_0 = const()[name = string("key_states_67_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_7_stride_0 = const()[name = string("key_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_67_cast_fp16 = transpose(perm = key_states_67_perm_0, x = key_states_65_cast_fp16)[name = string("transpose_409")]; tensor key_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = key_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = key_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_7_squeeze_mask_0, stride = key_cache_internal_tensor_assign_7_stride_0, update = key_states_67_cast_fp16, x = coreml_update_state_234)[name = string("key_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_7_cast_fp16, input = key_cache)[name = string("coreml_update_state_236_write_state")]; tensor coreml_update_state_236 = read_state(input = key_cache)[name = string("coreml_update_state_236")]; tensor value_states_39_perm_0 = const()[name = string("value_states_39_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_7_stride_0 = const()[name = string("value_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_39_cast_fp16 = transpose(perm = value_states_39_perm_0, x = var_2917_cast_fp16)[name = string("transpose_408")]; tensor value_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = value_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = value_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_7_squeeze_mask_0, stride = value_cache_internal_tensor_assign_7_stride_0, update = value_states_39_cast_fp16, x = coreml_update_state_235)[name = string("value_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_7_cast_fp16, input = value_cache)[name = string("coreml_update_state_237_write_state")]; tensor coreml_update_state_237 = read_state(input = value_cache)[name = string("coreml_update_state_237")]; tensor var_3011_begin_0 = const()[name = string("op_3011_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3011_end_0 = const()[name = string("op_3011_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3011_end_mask_0 = const()[name = string("op_3011_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3011_cast_fp16 = slice_by_index(begin = var_3011_begin_0, end = var_3011_end_0, end_mask = var_3011_end_mask_0, x = coreml_update_state_236)[name = string("op_3011_cast_fp16")]; tensor tile_12 = const()[name = string("tile_12"), val = tensor([1, 1])]; int32 var_3014_axis_0 = const()[name = string("op_3014_axis_0"), val = int32(1)]; tensor var_3014_cast_fp16_0, tensor var_3014_cast_fp16_1 = split(axis = var_3014_axis_0, split_sizes = tile_12, x = var_3011_cast_fp16)[name = string("op_3014_cast_fp16")]; tensor var_3021_begin_0 = const()[name = string("op_3021_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3021_end_0 = const()[name = string("op_3021_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3021_end_mask_0 = const()[name = string("op_3021_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3021_cast_fp16 = slice_by_index(begin = var_3021_begin_0, end = var_3021_end_0, end_mask = var_3021_end_mask_0, x = coreml_update_state_237)[name = string("op_3021_cast_fp16")]; tensor tile_13 = const()[name = string("tile_13"), val = tensor([1, 1])]; int32 var_3024_axis_0 = const()[name = string("op_3024_axis_0"), val = int32(1)]; tensor var_3024_cast_fp16_0, tensor var_3024_cast_fp16_1 = split(axis = var_3024_axis_0, split_sizes = tile_13, x = var_3021_cast_fp16)[name = string("op_3024_cast_fp16")]; tensor var_3027_split_sizes_0 = const()[name = string("op_3027_split_sizes_0"), val = tensor([8, 8])]; int32 var_3027_axis_0 = const()[name = string("op_3027_axis_0"), val = int32(1)]; tensor var_3027_0, tensor var_3027_1 = split(axis = var_3027_axis_0, split_sizes = var_3027_split_sizes_0, x = query_states_39_cast_fp16)[name = string("op_3027")]; bool attn_weights_97_transpose_x_0 = const()[name = string("attn_weights_97_transpose_x_0"), val = bool(false)]; bool attn_weights_97_transpose_y_0 = const()[name = string("attn_weights_97_transpose_y_0"), val = bool(false)]; tensor attn_weights_97_cast_fp16 = matmul(transpose_x = attn_weights_97_transpose_x_0, transpose_y = attn_weights_97_transpose_y_0, x = var_3014_cast_fp16_0, y = var_3027_0)[name = string("attn_weights_97_cast_fp16")]; fp16 var_3030_to_fp16 = const()[name = string("op_3030_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_99_cast_fp16 = mul(x = attn_weights_97_cast_fp16, y = var_3030_to_fp16)[name = string("attn_weights_99_cast_fp16")]; tensor attn_weights_101_cast_fp16 = add(x = attn_weights_99_cast_fp16, y = attn_mask_1)[name = string("attn_weights_101_cast_fp16")]; int32 var_3034 = const()[name = string("op_3034"), val = int32(-2)]; tensor attn_weights_103_cast_fp16 = softmax(axis = var_3034, x = attn_weights_101_cast_fp16)[name = string("attn_weights_103_cast_fp16")]; bool var_3040_transpose_x_1 = const()[name = string("op_3040_transpose_x_1"), val = bool(true)]; bool var_3040_transpose_y_1 = const()[name = string("op_3040_transpose_y_1"), val = bool(false)]; tensor var_3040_cast_fp16 = matmul(transpose_x = var_3040_transpose_x_1, transpose_y = var_3040_transpose_y_1, x = attn_weights_103_cast_fp16, y = var_3024_cast_fp16_0)[name = string("op_3040_cast_fp16")]; bool attn_weights_105_transpose_x_0 = const()[name = string("attn_weights_105_transpose_x_0"), val = bool(false)]; bool attn_weights_105_transpose_y_0 = const()[name = string("attn_weights_105_transpose_y_0"), val = bool(false)]; tensor attn_weights_105_cast_fp16 = matmul(transpose_x = attn_weights_105_transpose_x_0, transpose_y = attn_weights_105_transpose_y_0, x = var_3014_cast_fp16_1, y = var_3027_1)[name = string("attn_weights_105_cast_fp16")]; fp16 var_3042_to_fp16 = const()[name = string("op_3042_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_107_cast_fp16 = mul(x = attn_weights_105_cast_fp16, y = var_3042_to_fp16)[name = string("attn_weights_107_cast_fp16")]; tensor attn_weights_109_cast_fp16 = add(x = attn_weights_107_cast_fp16, y = attn_mask_1)[name = string("attn_weights_109_cast_fp16")]; int32 var_3046 = const()[name = string("op_3046"), val = int32(-2)]; tensor attn_weights_111_cast_fp16 = softmax(axis = var_3046, x = attn_weights_109_cast_fp16)[name = string("attn_weights_111_cast_fp16")]; bool attn_output_49_transpose_x_1 = const()[name = string("attn_output_49_transpose_x_1"), val = bool(true)]; bool attn_output_49_transpose_y_1 = const()[name = string("attn_output_49_transpose_y_1"), val = bool(false)]; tensor attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_1, transpose_y = attn_output_49_transpose_y_1, x = attn_weights_111_cast_fp16, y = var_3024_cast_fp16_1)[name = string("attn_output_49_cast_fp16")]; int32 var_3054 = const()[name = string("op_3054"), val = int32(1)]; bool attn_output_51_interleave_0 = const()[name = string("attn_output_51_interleave_0"), val = bool(false)]; tensor attn_output_51_cast_fp16 = concat(axis = var_3054, interleave = attn_output_51_interleave_0, values = (var_3040_cast_fp16, attn_output_49_cast_fp16))[name = string("attn_output_51_cast_fp16")]; tensor var_3058_perm_0 = const()[name = string("op_3058_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_83x = const()[name = string("concat_83x"), val = tensor([1, 2048, 1, -1])]; tensor var_3058_cast_fp16 = transpose(perm = var_3058_perm_0, x = attn_output_51_cast_fp16)[name = string("transpose_407")]; tensor attn_output_55_cast_fp16 = reshape(shape = concat_83x, x = var_3058_cast_fp16)[name = string("attn_output_55_cast_fp16")]; tensor hidden_states_63_strides_0 = const()[name = string("hidden_states_63_strides_0"), val = tensor([1, 1])]; string hidden_states_63_pad_type_0 = const()[name = string("hidden_states_63_pad_type_0"), val = string("valid")]; tensor hidden_states_63_pad_0 = const()[name = string("hidden_states_63_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_63_dilations_0 = const()[name = string("hidden_states_63_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_63_groups_0 = const()[name = string("hidden_states_63_groups_0"), val = int32(1)]; tensor hidden_states_63_cast_fp16 = conv(dilations = hidden_states_63_dilations_0, groups = hidden_states_63_groups_0, pad = hidden_states_63_pad_0, pad_type = hidden_states_63_pad_type_0, strides = hidden_states_63_strides_0, weight = layers_6_self_attn_o_proj_weight_cast_fp16, x = attn_output_55_cast_fp16)[name = string("hidden_states_63_cast_fp16")]; tensor hidden_states_65_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = hidden_states_63_cast_fp16)[name = string("hidden_states_65_cast_fp16")]; fp16 const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3091_cast_fp16 = mul(x = hidden_states_65_cast_fp16, y = const_68_promoted_to_fp16)[name = string("op_3091_cast_fp16")]; int32 var_3089 = const()[name = string("op_3089"), val = int32(1)]; bool doubled_53_interleave_0 = const()[name = string("doubled_53_interleave_0"), val = bool(false)]; tensor doubled_53_cast_fp16 = concat(axis = var_3089, interleave = doubled_53_interleave_0, values = (hidden_states_65_cast_fp16, var_3091_cast_fp16))[name = string("doubled_53_cast_fp16")]; tensor out_27_axes_0 = const()[name = string("out_27_axes_0"), val = tensor([1])]; tensor out_27_gamma_0_to_fp16 = const()[name = string("out_27_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327202816)))]; fp16 var_3101_to_fp16 = const()[name = string("op_3101_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_27_cast_fp16 = layer_norm(axes = out_27_axes_0, epsilon = var_3101_to_fp16, gamma = out_27_gamma_0_to_fp16, x = doubled_53_cast_fp16)[name = string("out_27_cast_fp16")]; tensor var_3112_split_sizes_0 = const()[name = string("op_3112_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3112_axis_0 = const()[name = string("op_3112_axis_0"), val = int32(1)]; tensor var_3112_cast_fp16_0, tensor var_3112_cast_fp16_1 = split(axis = var_3112_axis_0, split_sizes = var_3112_split_sizes_0, x = out_27_cast_fp16)[name = string("op_3112_cast_fp16")]; tensor input_13_strides_0 = const()[name = string("input_13_strides_0"), val = tensor([1, 1])]; string input_13_pad_type_0 = const()[name = string("input_13_pad_type_0"), val = string("valid")]; tensor input_13_pad_0 = const()[name = string("input_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_13_dilations_0 = const()[name = string("input_13_dilations_0"), val = tensor([1, 1])]; int32 input_13_groups_0 = const()[name = string("input_13_groups_0"), val = int32(1)]; tensor input_13_cast_fp16 = conv(dilations = input_13_dilations_0, groups = input_13_groups_0, pad = input_13_pad_0, pad_type = input_13_pad_type_0, strides = input_13_strides_0, weight = layers_6_mlp_gate_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("input_13_cast_fp16")]; tensor var_3129_cast_fp16 = silu(x = input_13_cast_fp16)[name = string("op_3129_cast_fp16")]; tensor var_3135_strides_0 = const()[name = string("op_3135_strides_0"), val = tensor([1, 1])]; string var_3135_pad_type_0 = const()[name = string("op_3135_pad_type_0"), val = string("valid")]; tensor var_3135_pad_0 = const()[name = string("op_3135_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3135_dilations_0 = const()[name = string("op_3135_dilations_0"), val = tensor([1, 1])]; int32 var_3135_groups_0 = const()[name = string("op_3135_groups_0"), val = int32(1)]; tensor var_3135_cast_fp16 = conv(dilations = var_3135_dilations_0, groups = var_3135_groups_0, pad = var_3135_pad_0, pad_type = var_3135_pad_type_0, strides = var_3135_strides_0, weight = layers_6_mlp_up_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("op_3135_cast_fp16")]; tensor x_69_cast_fp16 = mul(x = var_3129_cast_fp16, y = var_3135_cast_fp16)[name = string("x_69_cast_fp16")]; tensor hidden_states_67_strides_0 = const()[name = string("hidden_states_67_strides_0"), val = tensor([1, 1])]; string hidden_states_67_pad_type_0 = const()[name = string("hidden_states_67_pad_type_0"), val = string("valid")]; tensor hidden_states_67_pad_0 = const()[name = string("hidden_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_67_dilations_0 = const()[name = string("hidden_states_67_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_67_groups_0 = const()[name = string("hidden_states_67_groups_0"), val = int32(1)]; tensor hidden_states_67_cast_fp16 = conv(dilations = hidden_states_67_dilations_0, groups = hidden_states_67_groups_0, pad = hidden_states_67_pad_0, pad_type = hidden_states_67_pad_type_0, strides = hidden_states_67_strides_0, weight = layers_6_mlp_down_proj_weight_cast_fp16, x = x_69_cast_fp16)[name = string("hidden_states_67_cast_fp16")]; tensor hidden_states_69_cast_fp16 = add(x = hidden_states_65_cast_fp16, y = hidden_states_67_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3153_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_70_promoted_to_fp16)[name = string("op_3153_cast_fp16")]; int32 var_3151 = const()[name = string("op_3151"), val = int32(1)]; bool doubled_57_interleave_0 = const()[name = string("doubled_57_interleave_0"), val = bool(false)]; tensor doubled_57_cast_fp16 = concat(axis = var_3151, interleave = doubled_57_interleave_0, values = (hidden_states_69_cast_fp16, var_3153_cast_fp16))[name = string("doubled_57_cast_fp16")]; tensor out_29_axes_0 = const()[name = string("out_29_axes_0"), val = tensor([1])]; tensor out_29_gamma_0_to_fp16 = const()[name = string("out_29_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327211072)))]; fp16 var_3163_to_fp16 = const()[name = string("op_3163_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_29_cast_fp16 = layer_norm(axes = out_29_axes_0, epsilon = var_3163_to_fp16, gamma = out_29_gamma_0_to_fp16, x = doubled_57_cast_fp16)[name = string("out_29_cast_fp16")]; tensor var_3174_split_sizes_0 = const()[name = string("op_3174_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3174_axis_0 = const()[name = string("op_3174_axis_0"), val = int32(1)]; tensor var_3174_cast_fp16_0, tensor var_3174_cast_fp16_1 = split(axis = var_3174_axis_0, split_sizes = var_3174_split_sizes_0, x = out_29_cast_fp16)[name = string("op_3174_cast_fp16")]; tensor layers_7_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327219328)))]; tensor query_states_43_strides_0 = const()[name = string("query_states_43_strides_0"), val = tensor([1, 1])]; string query_states_43_pad_type_0 = const()[name = string("query_states_43_pad_type_0"), val = string("valid")]; tensor query_states_43_pad_0 = const()[name = string("query_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_43_dilations_0 = const()[name = string("query_states_43_dilations_0"), val = tensor([1, 1])]; int32 query_states_43_groups_0 = const()[name = string("query_states_43_groups_0"), val = int32(1)]; tensor query_states_43_cast_fp16 = conv(dilations = query_states_43_dilations_0, groups = query_states_43_groups_0, pad = query_states_43_pad_0, pad_type = query_states_43_pad_type_0, strides = query_states_43_strides_0, weight = layers_7_self_attn_q_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("query_states_43_cast_fp16")]; tensor layers_7_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1335608000)))]; tensor key_states_71_strides_0 = const()[name = string("key_states_71_strides_0"), val = tensor([1, 1])]; string key_states_71_pad_type_0 = const()[name = string("key_states_71_pad_type_0"), val = string("valid")]; tensor key_states_71_pad_0 = const()[name = string("key_states_71_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_71_dilations_0 = const()[name = string("key_states_71_dilations_0"), val = tensor([1, 1])]; int32 key_states_71_groups_0 = const()[name = string("key_states_71_groups_0"), val = int32(1)]; tensor key_states_71_cast_fp16 = conv(dilations = key_states_71_dilations_0, groups = key_states_71_groups_0, pad = key_states_71_pad_0, pad_type = key_states_71_pad_type_0, strides = key_states_71_strides_0, weight = layers_7_self_attn_k_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("key_states_71_cast_fp16")]; tensor value_states_43_strides_0 = const()[name = string("value_states_43_strides_0"), val = tensor([1, 1])]; string value_states_43_pad_type_0 = const()[name = string("value_states_43_pad_type_0"), val = string("valid")]; tensor value_states_43_pad_0 = const()[name = string("value_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_43_dilations_0 = const()[name = string("value_states_43_dilations_0"), val = tensor([1, 1])]; int32 value_states_43_groups_0 = const()[name = string("value_states_43_groups_0"), val = int32(1)]; tensor value_states_43_cast_fp16 = conv(dilations = value_states_43_dilations_0, groups = value_states_43_groups_0, pad = value_states_43_pad_0, pad_type = value_states_43_pad_type_0, strides = value_states_43_strides_0, weight = layers_7_self_attn_v_proj_weight_cast_fp16, x = var_3174_cast_fp16_0)[name = string("value_states_43_cast_fp16")]; tensor concat_84x = const()[name = string("concat_84x"), val = tensor([1, 16, 128, -1])]; tensor x_71_cast_fp16 = reshape(shape = concat_84x, x = query_states_43_cast_fp16)[name = string("x_71_cast_fp16")]; tensor concat_85x = const()[name = string("concat_85x"), val = tensor([1, 2, 128, -1])]; tensor var_3231_cast_fp16 = reshape(shape = concat_85x, x = key_states_71_cast_fp16)[name = string("op_3231_cast_fp16")]; tensor concat_86x = const()[name = string("concat_86x"), val = tensor([1, 2, 128, -1])]; tensor var_3238_cast_fp16 = reshape(shape = concat_86x, x = value_states_43_cast_fp16)[name = string("op_3238_cast_fp16")]; tensor var_3242_cast_fp16 = mul(x = x_71_cast_fp16, y = var_869_cast_fp16)[name = string("op_3242_cast_fp16")]; tensor var_3243_split_sizes_0 = const()[name = string("op_3243_split_sizes_0"), val = tensor([64, 64])]; int32 var_3243_axis_0 = const()[name = string("op_3243_axis_0"), val = int32(-2)]; tensor var_3243_cast_fp16_0, tensor var_3243_cast_fp16_1 = split(axis = var_3243_axis_0, split_sizes = var_3243_split_sizes_0, x = x_71_cast_fp16)[name = string("op_3243_cast_fp16")]; fp16 const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3245_cast_fp16 = mul(x = var_3243_cast_fp16_1, y = const_72_promoted_to_fp16)[name = string("op_3245_cast_fp16")]; int32 var_3247 = const()[name = string("op_3247"), val = int32(-2)]; bool var_3248_interleave_0 = const()[name = string("op_3248_interleave_0"), val = bool(false)]; tensor var_3248_cast_fp16 = concat(axis = var_3247, interleave = var_3248_interleave_0, values = (var_3245_cast_fp16, var_3243_cast_fp16_0))[name = string("op_3248_cast_fp16")]; tensor var_3249_cast_fp16 = mul(x = var_3248_cast_fp16, y = var_878_cast_fp16)[name = string("op_3249_cast_fp16")]; tensor query_states_45_cast_fp16 = add(x = var_3242_cast_fp16, y = var_3249_cast_fp16)[name = string("query_states_45_cast_fp16")]; tensor var_3255_cast_fp16 = mul(x = var_3231_cast_fp16, y = var_869_cast_fp16)[name = string("op_3255_cast_fp16")]; tensor var_3256_split_sizes_0 = const()[name = string("op_3256_split_sizes_0"), val = tensor([64, 64])]; int32 var_3256_axis_0 = const()[name = string("op_3256_axis_0"), val = int32(-2)]; tensor var_3256_cast_fp16_0, tensor var_3256_cast_fp16_1 = split(axis = var_3256_axis_0, split_sizes = var_3256_split_sizes_0, x = var_3231_cast_fp16)[name = string("op_3256_cast_fp16")]; fp16 const_73_promoted_to_fp16 = const()[name = string("const_73_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3258_cast_fp16 = mul(x = var_3256_cast_fp16_1, y = const_73_promoted_to_fp16)[name = string("op_3258_cast_fp16")]; int32 var_3260 = const()[name = string("op_3260"), val = int32(-2)]; bool var_3261_interleave_0 = const()[name = string("op_3261_interleave_0"), val = bool(false)]; tensor var_3261_cast_fp16 = concat(axis = var_3260, interleave = var_3261_interleave_0, values = (var_3258_cast_fp16, var_3256_cast_fp16_0))[name = string("op_3261_cast_fp16")]; tensor var_3262_cast_fp16 = mul(x = var_3261_cast_fp16, y = var_878_cast_fp16)[name = string("op_3262_cast_fp16")]; tensor key_states_75_cast_fp16 = add(x = var_3255_cast_fp16, y = var_3262_cast_fp16)[name = string("key_states_75_cast_fp16")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; int32 concat_89_axis_0 = const()[name = string("concat_89_axis_0"), val = int32(0)]; bool concat_89_interleave_0 = const()[name = string("concat_89_interleave_0"), val = bool(false)]; tensor concat_89 = concat(axis = concat_89_axis_0, interleave = concat_89_interleave_0, values = (expand_dims_84, expand_dims_85, position_id, expand_dims_87))[name = string("concat_89")]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; tensor concat_90_values1_0 = const()[name = string("concat_90_values1_0"), val = tensor([0])]; tensor concat_90_values3_0 = const()[name = string("concat_90_values3_0"), val = tensor([0])]; int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_88, concat_90_values1_0, cache_position_end, concat_90_values3_0))[name = string("concat_90")]; tensor key_states_77_perm_0 = const()[name = string("key_states_77_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_8_stride_0 = const()[name = string("key_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_77_cast_fp16 = transpose(perm = key_states_77_perm_0, x = key_states_75_cast_fp16)[name = string("transpose_406")]; tensor key_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = key_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = key_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_8_squeeze_mask_0, stride = key_cache_internal_tensor_assign_8_stride_0, update = key_states_77_cast_fp16, x = coreml_update_state_236)[name = string("key_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_8_cast_fp16, input = key_cache)[name = string("coreml_update_state_238_write_state")]; tensor coreml_update_state_238 = read_state(input = key_cache)[name = string("coreml_update_state_238")]; tensor value_states_45_perm_0 = const()[name = string("value_states_45_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_8_stride_0 = const()[name = string("value_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_45_cast_fp16 = transpose(perm = value_states_45_perm_0, x = var_3238_cast_fp16)[name = string("transpose_405")]; tensor value_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = value_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = value_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_8_squeeze_mask_0, stride = value_cache_internal_tensor_assign_8_stride_0, update = value_states_45_cast_fp16, x = coreml_update_state_237)[name = string("value_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_8_cast_fp16, input = value_cache)[name = string("coreml_update_state_239_write_state")]; tensor coreml_update_state_239 = read_state(input = value_cache)[name = string("coreml_update_state_239")]; tensor var_3332_begin_0 = const()[name = string("op_3332_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3332_end_0 = const()[name = string("op_3332_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3332_end_mask_0 = const()[name = string("op_3332_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3332_cast_fp16 = slice_by_index(begin = var_3332_begin_0, end = var_3332_end_0, end_mask = var_3332_end_mask_0, x = coreml_update_state_238)[name = string("op_3332_cast_fp16")]; tensor tile_14 = const()[name = string("tile_14"), val = tensor([1, 1])]; int32 var_3335_axis_0 = const()[name = string("op_3335_axis_0"), val = int32(1)]; tensor var_3335_cast_fp16_0, tensor var_3335_cast_fp16_1 = split(axis = var_3335_axis_0, split_sizes = tile_14, x = var_3332_cast_fp16)[name = string("op_3335_cast_fp16")]; tensor var_3342_begin_0 = const()[name = string("op_3342_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3342_end_0 = const()[name = string("op_3342_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3342_end_mask_0 = const()[name = string("op_3342_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3342_cast_fp16 = slice_by_index(begin = var_3342_begin_0, end = var_3342_end_0, end_mask = var_3342_end_mask_0, x = coreml_update_state_239)[name = string("op_3342_cast_fp16")]; tensor tile_15 = const()[name = string("tile_15"), val = tensor([1, 1])]; int32 var_3345_axis_0 = const()[name = string("op_3345_axis_0"), val = int32(1)]; tensor var_3345_cast_fp16_0, tensor var_3345_cast_fp16_1 = split(axis = var_3345_axis_0, split_sizes = tile_15, x = var_3342_cast_fp16)[name = string("op_3345_cast_fp16")]; tensor var_3348_split_sizes_0 = const()[name = string("op_3348_split_sizes_0"), val = tensor([8, 8])]; int32 var_3348_axis_0 = const()[name = string("op_3348_axis_0"), val = int32(1)]; tensor var_3348_0, tensor var_3348_1 = split(axis = var_3348_axis_0, split_sizes = var_3348_split_sizes_0, x = query_states_45_cast_fp16)[name = string("op_3348")]; bool attn_weights_113_transpose_x_0 = const()[name = string("attn_weights_113_transpose_x_0"), val = bool(false)]; bool attn_weights_113_transpose_y_0 = const()[name = string("attn_weights_113_transpose_y_0"), val = bool(false)]; tensor attn_weights_113_cast_fp16 = matmul(transpose_x = attn_weights_113_transpose_x_0, transpose_y = attn_weights_113_transpose_y_0, x = var_3335_cast_fp16_0, y = var_3348_0)[name = string("attn_weights_113_cast_fp16")]; fp16 var_3351_to_fp16 = const()[name = string("op_3351_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_115_cast_fp16 = mul(x = attn_weights_113_cast_fp16, y = var_3351_to_fp16)[name = string("attn_weights_115_cast_fp16")]; tensor attn_weights_117_cast_fp16 = add(x = attn_weights_115_cast_fp16, y = attn_mask_1)[name = string("attn_weights_117_cast_fp16")]; int32 var_3355 = const()[name = string("op_3355"), val = int32(-2)]; tensor attn_weights_119_cast_fp16 = softmax(axis = var_3355, x = attn_weights_117_cast_fp16)[name = string("attn_weights_119_cast_fp16")]; bool var_3361_transpose_x_1 = const()[name = string("op_3361_transpose_x_1"), val = bool(true)]; bool var_3361_transpose_y_1 = const()[name = string("op_3361_transpose_y_1"), val = bool(false)]; tensor var_3361_cast_fp16 = matmul(transpose_x = var_3361_transpose_x_1, transpose_y = var_3361_transpose_y_1, x = attn_weights_119_cast_fp16, y = var_3345_cast_fp16_0)[name = string("op_3361_cast_fp16")]; bool attn_weights_121_transpose_x_0 = const()[name = string("attn_weights_121_transpose_x_0"), val = bool(false)]; bool attn_weights_121_transpose_y_0 = const()[name = string("attn_weights_121_transpose_y_0"), val = bool(false)]; tensor attn_weights_121_cast_fp16 = matmul(transpose_x = attn_weights_121_transpose_x_0, transpose_y = attn_weights_121_transpose_y_0, x = var_3335_cast_fp16_1, y = var_3348_1)[name = string("attn_weights_121_cast_fp16")]; fp16 var_3363_to_fp16 = const()[name = string("op_3363_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_123_cast_fp16 = mul(x = attn_weights_121_cast_fp16, y = var_3363_to_fp16)[name = string("attn_weights_123_cast_fp16")]; tensor attn_weights_125_cast_fp16 = add(x = attn_weights_123_cast_fp16, y = attn_mask_1)[name = string("attn_weights_125_cast_fp16")]; int32 var_3367 = const()[name = string("op_3367"), val = int32(-2)]; tensor attn_weights_127_cast_fp16 = softmax(axis = var_3367, x = attn_weights_125_cast_fp16)[name = string("attn_weights_127_cast_fp16")]; bool attn_output_57_transpose_x_1 = const()[name = string("attn_output_57_transpose_x_1"), val = bool(true)]; bool attn_output_57_transpose_y_1 = const()[name = string("attn_output_57_transpose_y_1"), val = bool(false)]; tensor attn_output_57_cast_fp16 = matmul(transpose_x = attn_output_57_transpose_x_1, transpose_y = attn_output_57_transpose_y_1, x = attn_weights_127_cast_fp16, y = var_3345_cast_fp16_1)[name = string("attn_output_57_cast_fp16")]; int32 var_3375 = const()[name = string("op_3375"), val = int32(1)]; bool attn_output_59_interleave_0 = const()[name = string("attn_output_59_interleave_0"), val = bool(false)]; tensor attn_output_59_cast_fp16 = concat(axis = var_3375, interleave = attn_output_59_interleave_0, values = (var_3361_cast_fp16, attn_output_57_cast_fp16))[name = string("attn_output_59_cast_fp16")]; tensor var_3379_perm_0 = const()[name = string("op_3379_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_95x = const()[name = string("concat_95x"), val = tensor([1, 2048, 1, -1])]; tensor var_3379_cast_fp16 = transpose(perm = var_3379_perm_0, x = attn_output_59_cast_fp16)[name = string("transpose_404")]; tensor attn_output_63_cast_fp16 = reshape(shape = concat_95x, x = var_3379_cast_fp16)[name = string("attn_output_63_cast_fp16")]; tensor hidden_states_73_strides_0 = const()[name = string("hidden_states_73_strides_0"), val = tensor([1, 1])]; string hidden_states_73_pad_type_0 = const()[name = string("hidden_states_73_pad_type_0"), val = string("valid")]; tensor hidden_states_73_pad_0 = const()[name = string("hidden_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_73_dilations_0 = const()[name = string("hidden_states_73_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_73_groups_0 = const()[name = string("hidden_states_73_groups_0"), val = int32(1)]; tensor hidden_states_73_cast_fp16 = conv(dilations = hidden_states_73_dilations_0, groups = hidden_states_73_groups_0, pad = hidden_states_73_pad_0, pad_type = hidden_states_73_pad_type_0, strides = hidden_states_73_strides_0, weight = layers_7_self_attn_o_proj_weight_cast_fp16, x = attn_output_63_cast_fp16)[name = string("hidden_states_73_cast_fp16")]; tensor hidden_states_75_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = hidden_states_73_cast_fp16)[name = string("hidden_states_75_cast_fp16")]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3412_cast_fp16 = mul(x = hidden_states_75_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_3412_cast_fp16")]; int32 var_3410 = const()[name = string("op_3410"), val = int32(1)]; bool doubled_61_interleave_0 = const()[name = string("doubled_61_interleave_0"), val = bool(false)]; tensor doubled_61_cast_fp16 = concat(axis = var_3410, interleave = doubled_61_interleave_0, values = (hidden_states_75_cast_fp16, var_3412_cast_fp16))[name = string("doubled_61_cast_fp16")]; tensor out_31_axes_0 = const()[name = string("out_31_axes_0"), val = tensor([1])]; tensor out_31_gamma_0_to_fp16 = const()[name = string("out_31_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336656640)))]; fp16 var_3422_to_fp16 = const()[name = string("op_3422_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_31_cast_fp16 = layer_norm(axes = out_31_axes_0, epsilon = var_3422_to_fp16, gamma = out_31_gamma_0_to_fp16, x = doubled_61_cast_fp16)[name = string("out_31_cast_fp16")]; tensor var_3433_split_sizes_0 = const()[name = string("op_3433_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3433_axis_0 = const()[name = string("op_3433_axis_0"), val = int32(1)]; tensor var_3433_cast_fp16_0, tensor var_3433_cast_fp16_1 = split(axis = var_3433_axis_0, split_sizes = var_3433_split_sizes_0, x = out_31_cast_fp16)[name = string("op_3433_cast_fp16")]; tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; tensor input_15_cast_fp16 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = layers_7_mlp_gate_proj_weight_cast_fp16, x = var_3433_cast_fp16_0)[name = string("input_15_cast_fp16")]; tensor var_3450_cast_fp16 = silu(x = input_15_cast_fp16)[name = string("op_3450_cast_fp16")]; tensor layers_7_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336664896)))]; tensor var_3456_strides_0 = const()[name = string("op_3456_strides_0"), val = tensor([1, 1])]; string var_3456_pad_type_0 = const()[name = string("op_3456_pad_type_0"), val = string("valid")]; tensor var_3456_pad_0 = const()[name = string("op_3456_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3456_dilations_0 = const()[name = string("op_3456_dilations_0"), val = tensor([1, 1])]; int32 var_3456_groups_0 = const()[name = string("op_3456_groups_0"), val = int32(1)]; tensor var_3456_cast_fp16 = conv(dilations = var_3456_dilations_0, groups = var_3456_groups_0, pad = var_3456_pad_0, pad_type = var_3456_pad_type_0, strides = var_3456_strides_0, weight = layers_7_mlp_up_proj_weight_to_fp16, x = var_3433_cast_fp16_0)[name = string("op_3456_cast_fp16")]; tensor x_79_cast_fp16 = mul(x = var_3450_cast_fp16, y = var_3456_cast_fp16)[name = string("x_79_cast_fp16")]; tensor layers_7_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1361830784)))]; tensor hidden_states_77_strides_0 = const()[name = string("hidden_states_77_strides_0"), val = tensor([1, 1])]; string hidden_states_77_pad_type_0 = const()[name = string("hidden_states_77_pad_type_0"), val = string("valid")]; tensor hidden_states_77_pad_0 = const()[name = string("hidden_states_77_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_77_dilations_0 = const()[name = string("hidden_states_77_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_77_groups_0 = const()[name = string("hidden_states_77_groups_0"), val = int32(1)]; tensor hidden_states_77_cast_fp16 = conv(dilations = hidden_states_77_dilations_0, groups = hidden_states_77_groups_0, pad = hidden_states_77_pad_0, pad_type = hidden_states_77_pad_type_0, strides = hidden_states_77_strides_0, weight = layers_7_mlp_down_proj_weight_to_fp16, x = x_79_cast_fp16)[name = string("hidden_states_77_cast_fp16")]; tensor hidden_states_79_cast_fp16 = add(x = hidden_states_75_cast_fp16, y = hidden_states_77_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3474_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_80_promoted_to_fp16)[name = string("op_3474_cast_fp16")]; int32 var_3472 = const()[name = string("op_3472"), val = int32(1)]; bool doubled_65_interleave_0 = const()[name = string("doubled_65_interleave_0"), val = bool(false)]; tensor doubled_65_cast_fp16 = concat(axis = var_3472, interleave = doubled_65_interleave_0, values = (hidden_states_79_cast_fp16, var_3474_cast_fp16))[name = string("doubled_65_cast_fp16")]; tensor out_33_axes_0 = const()[name = string("out_33_axes_0"), val = tensor([1])]; tensor out_33_gamma_0_to_fp16 = const()[name = string("out_33_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1386996672)))]; fp16 var_3484_to_fp16 = const()[name = string("op_3484_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_33_cast_fp16 = layer_norm(axes = out_33_axes_0, epsilon = var_3484_to_fp16, gamma = out_33_gamma_0_to_fp16, x = doubled_65_cast_fp16)[name = string("out_33_cast_fp16")]; tensor var_3495_split_sizes_0 = const()[name = string("op_3495_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3495_axis_0 = const()[name = string("op_3495_axis_0"), val = int32(1)]; tensor var_3495_cast_fp16_0, tensor var_3495_cast_fp16_1 = split(axis = var_3495_axis_0, split_sizes = var_3495_split_sizes_0, x = out_33_cast_fp16)[name = string("op_3495_cast_fp16")]; tensor layers_8_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1387004928)))]; tensor query_states_49_strides_0 = const()[name = string("query_states_49_strides_0"), val = tensor([1, 1])]; string query_states_49_pad_type_0 = const()[name = string("query_states_49_pad_type_0"), val = string("valid")]; tensor query_states_49_pad_0 = const()[name = string("query_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_49_dilations_0 = const()[name = string("query_states_49_dilations_0"), val = tensor([1, 1])]; int32 query_states_49_groups_0 = const()[name = string("query_states_49_groups_0"), val = int32(1)]; tensor query_states_49_cast_fp16 = conv(dilations = query_states_49_dilations_0, groups = query_states_49_groups_0, pad = query_states_49_pad_0, pad_type = query_states_49_pad_type_0, strides = query_states_49_strides_0, weight = layers_8_self_attn_q_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("query_states_49_cast_fp16")]; tensor layers_8_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1395393600)))]; tensor key_states_81_strides_0 = const()[name = string("key_states_81_strides_0"), val = tensor([1, 1])]; string key_states_81_pad_type_0 = const()[name = string("key_states_81_pad_type_0"), val = string("valid")]; tensor key_states_81_pad_0 = const()[name = string("key_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_81_dilations_0 = const()[name = string("key_states_81_dilations_0"), val = tensor([1, 1])]; int32 key_states_81_groups_0 = const()[name = string("key_states_81_groups_0"), val = int32(1)]; tensor key_states_81_cast_fp16 = conv(dilations = key_states_81_dilations_0, groups = key_states_81_groups_0, pad = key_states_81_pad_0, pad_type = key_states_81_pad_type_0, strides = key_states_81_strides_0, weight = layers_8_self_attn_k_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("key_states_81_cast_fp16")]; tensor value_states_49_strides_0 = const()[name = string("value_states_49_strides_0"), val = tensor([1, 1])]; string value_states_49_pad_type_0 = const()[name = string("value_states_49_pad_type_0"), val = string("valid")]; tensor value_states_49_pad_0 = const()[name = string("value_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_49_dilations_0 = const()[name = string("value_states_49_dilations_0"), val = tensor([1, 1])]; int32 value_states_49_groups_0 = const()[name = string("value_states_49_groups_0"), val = int32(1)]; tensor value_states_49_cast_fp16 = conv(dilations = value_states_49_dilations_0, groups = value_states_49_groups_0, pad = value_states_49_pad_0, pad_type = value_states_49_pad_type_0, strides = value_states_49_strides_0, weight = layers_8_self_attn_v_proj_weight_cast_fp16, x = var_3495_cast_fp16_0)[name = string("value_states_49_cast_fp16")]; tensor concat_96x = const()[name = string("concat_96x"), val = tensor([1, 16, 128, -1])]; tensor x_81_cast_fp16 = reshape(shape = concat_96x, x = query_states_49_cast_fp16)[name = string("x_81_cast_fp16")]; tensor concat_97x = const()[name = string("concat_97x"), val = tensor([1, 2, 128, -1])]; tensor var_3552_cast_fp16 = reshape(shape = concat_97x, x = key_states_81_cast_fp16)[name = string("op_3552_cast_fp16")]; tensor concat_98x = const()[name = string("concat_98x"), val = tensor([1, 2, 128, -1])]; tensor var_3559_cast_fp16 = reshape(shape = concat_98x, x = value_states_49_cast_fp16)[name = string("op_3559_cast_fp16")]; tensor var_3563_cast_fp16 = mul(x = x_81_cast_fp16, y = var_869_cast_fp16)[name = string("op_3563_cast_fp16")]; tensor var_3564_split_sizes_0 = const()[name = string("op_3564_split_sizes_0"), val = tensor([64, 64])]; int32 var_3564_axis_0 = const()[name = string("op_3564_axis_0"), val = int32(-2)]; tensor var_3564_cast_fp16_0, tensor var_3564_cast_fp16_1 = split(axis = var_3564_axis_0, split_sizes = var_3564_split_sizes_0, x = x_81_cast_fp16)[name = string("op_3564_cast_fp16")]; fp16 const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3566_cast_fp16 = mul(x = var_3564_cast_fp16_1, y = const_82_promoted_to_fp16)[name = string("op_3566_cast_fp16")]; int32 var_3568 = const()[name = string("op_3568"), val = int32(-2)]; bool var_3569_interleave_0 = const()[name = string("op_3569_interleave_0"), val = bool(false)]; tensor var_3569_cast_fp16 = concat(axis = var_3568, interleave = var_3569_interleave_0, values = (var_3566_cast_fp16, var_3564_cast_fp16_0))[name = string("op_3569_cast_fp16")]; tensor var_3570_cast_fp16 = mul(x = var_3569_cast_fp16, y = var_878_cast_fp16)[name = string("op_3570_cast_fp16")]; tensor query_states_51_cast_fp16 = add(x = var_3563_cast_fp16, y = var_3570_cast_fp16)[name = string("query_states_51_cast_fp16")]; tensor var_3576_cast_fp16 = mul(x = var_3552_cast_fp16, y = var_869_cast_fp16)[name = string("op_3576_cast_fp16")]; tensor var_3577_split_sizes_0 = const()[name = string("op_3577_split_sizes_0"), val = tensor([64, 64])]; int32 var_3577_axis_0 = const()[name = string("op_3577_axis_0"), val = int32(-2)]; tensor var_3577_cast_fp16_0, tensor var_3577_cast_fp16_1 = split(axis = var_3577_axis_0, split_sizes = var_3577_split_sizes_0, x = var_3552_cast_fp16)[name = string("op_3577_cast_fp16")]; fp16 const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3579_cast_fp16 = mul(x = var_3577_cast_fp16_1, y = const_83_promoted_to_fp16)[name = string("op_3579_cast_fp16")]; int32 var_3581 = const()[name = string("op_3581"), val = int32(-2)]; bool var_3582_interleave_0 = const()[name = string("op_3582_interleave_0"), val = bool(false)]; tensor var_3582_cast_fp16 = concat(axis = var_3581, interleave = var_3582_interleave_0, values = (var_3579_cast_fp16, var_3577_cast_fp16_0))[name = string("op_3582_cast_fp16")]; tensor var_3583_cast_fp16 = mul(x = var_3582_cast_fp16, y = var_878_cast_fp16)[name = string("op_3583_cast_fp16")]; tensor key_states_85_cast_fp16 = add(x = var_3576_cast_fp16, y = var_3583_cast_fp16)[name = string("key_states_85_cast_fp16")]; tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; int32 concat_101_axis_0 = const()[name = string("concat_101_axis_0"), val = int32(0)]; bool concat_101_interleave_0 = const()[name = string("concat_101_interleave_0"), val = bool(false)]; tensor concat_101 = concat(axis = concat_101_axis_0, interleave = concat_101_interleave_0, values = (expand_dims_96, expand_dims_97, position_id, expand_dims_99))[name = string("concat_101")]; tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; tensor concat_102_values1_0 = const()[name = string("concat_102_values1_0"), val = tensor([0])]; tensor concat_102_values3_0 = const()[name = string("concat_102_values3_0"), val = tensor([0])]; int32 concat_102_axis_0 = const()[name = string("concat_102_axis_0"), val = int32(0)]; bool concat_102_interleave_0 = const()[name = string("concat_102_interleave_0"), val = bool(false)]; tensor concat_102 = concat(axis = concat_102_axis_0, interleave = concat_102_interleave_0, values = (expand_dims_100, concat_102_values1_0, cache_position_end, concat_102_values3_0))[name = string("concat_102")]; tensor key_states_87_perm_0 = const()[name = string("key_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_9_stride_0 = const()[name = string("key_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_87_cast_fp16 = transpose(perm = key_states_87_perm_0, x = key_states_85_cast_fp16)[name = string("transpose_403")]; tensor key_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = key_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = key_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_9_squeeze_mask_0, stride = key_cache_internal_tensor_assign_9_stride_0, update = key_states_87_cast_fp16, x = coreml_update_state_238)[name = string("key_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_9_cast_fp16, input = key_cache)[name = string("coreml_update_state_240_write_state")]; tensor coreml_update_state_240 = read_state(input = key_cache)[name = string("coreml_update_state_240")]; tensor value_states_51_perm_0 = const()[name = string("value_states_51_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_9_stride_0 = const()[name = string("value_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_51_cast_fp16 = transpose(perm = value_states_51_perm_0, x = var_3559_cast_fp16)[name = string("transpose_402")]; tensor value_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = value_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = value_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_9_squeeze_mask_0, stride = value_cache_internal_tensor_assign_9_stride_0, update = value_states_51_cast_fp16, x = coreml_update_state_239)[name = string("value_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_9_cast_fp16, input = value_cache)[name = string("coreml_update_state_241_write_state")]; tensor coreml_update_state_241 = read_state(input = value_cache)[name = string("coreml_update_state_241")]; tensor var_3653_begin_0 = const()[name = string("op_3653_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3653_end_0 = const()[name = string("op_3653_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3653_end_mask_0 = const()[name = string("op_3653_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3653_cast_fp16 = slice_by_index(begin = var_3653_begin_0, end = var_3653_end_0, end_mask = var_3653_end_mask_0, x = coreml_update_state_240)[name = string("op_3653_cast_fp16")]; tensor tile_16 = const()[name = string("tile_16"), val = tensor([1, 1])]; int32 var_3656_axis_0 = const()[name = string("op_3656_axis_0"), val = int32(1)]; tensor var_3656_cast_fp16_0, tensor var_3656_cast_fp16_1 = split(axis = var_3656_axis_0, split_sizes = tile_16, x = var_3653_cast_fp16)[name = string("op_3656_cast_fp16")]; tensor var_3663_begin_0 = const()[name = string("op_3663_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3663_end_0 = const()[name = string("op_3663_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3663_end_mask_0 = const()[name = string("op_3663_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3663_cast_fp16 = slice_by_index(begin = var_3663_begin_0, end = var_3663_end_0, end_mask = var_3663_end_mask_0, x = coreml_update_state_241)[name = string("op_3663_cast_fp16")]; tensor tile_17 = const()[name = string("tile_17"), val = tensor([1, 1])]; int32 var_3666_axis_0 = const()[name = string("op_3666_axis_0"), val = int32(1)]; tensor var_3666_cast_fp16_0, tensor var_3666_cast_fp16_1 = split(axis = var_3666_axis_0, split_sizes = tile_17, x = var_3663_cast_fp16)[name = string("op_3666_cast_fp16")]; tensor var_3669_split_sizes_0 = const()[name = string("op_3669_split_sizes_0"), val = tensor([8, 8])]; int32 var_3669_axis_0 = const()[name = string("op_3669_axis_0"), val = int32(1)]; tensor var_3669_0, tensor var_3669_1 = split(axis = var_3669_axis_0, split_sizes = var_3669_split_sizes_0, x = query_states_51_cast_fp16)[name = string("op_3669")]; bool attn_weights_129_transpose_x_0 = const()[name = string("attn_weights_129_transpose_x_0"), val = bool(false)]; bool attn_weights_129_transpose_y_0 = const()[name = string("attn_weights_129_transpose_y_0"), val = bool(false)]; tensor attn_weights_129_cast_fp16 = matmul(transpose_x = attn_weights_129_transpose_x_0, transpose_y = attn_weights_129_transpose_y_0, x = var_3656_cast_fp16_0, y = var_3669_0)[name = string("attn_weights_129_cast_fp16")]; fp16 var_3672_to_fp16 = const()[name = string("op_3672_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_131_cast_fp16 = mul(x = attn_weights_129_cast_fp16, y = var_3672_to_fp16)[name = string("attn_weights_131_cast_fp16")]; tensor attn_weights_133_cast_fp16 = add(x = attn_weights_131_cast_fp16, y = attn_mask_1)[name = string("attn_weights_133_cast_fp16")]; int32 var_3676 = const()[name = string("op_3676"), val = int32(-2)]; tensor attn_weights_135_cast_fp16 = softmax(axis = var_3676, x = attn_weights_133_cast_fp16)[name = string("attn_weights_135_cast_fp16")]; bool var_3682_transpose_x_1 = const()[name = string("op_3682_transpose_x_1"), val = bool(true)]; bool var_3682_transpose_y_1 = const()[name = string("op_3682_transpose_y_1"), val = bool(false)]; tensor var_3682_cast_fp16 = matmul(transpose_x = var_3682_transpose_x_1, transpose_y = var_3682_transpose_y_1, x = attn_weights_135_cast_fp16, y = var_3666_cast_fp16_0)[name = string("op_3682_cast_fp16")]; bool attn_weights_137_transpose_x_0 = const()[name = string("attn_weights_137_transpose_x_0"), val = bool(false)]; bool attn_weights_137_transpose_y_0 = const()[name = string("attn_weights_137_transpose_y_0"), val = bool(false)]; tensor attn_weights_137_cast_fp16 = matmul(transpose_x = attn_weights_137_transpose_x_0, transpose_y = attn_weights_137_transpose_y_0, x = var_3656_cast_fp16_1, y = var_3669_1)[name = string("attn_weights_137_cast_fp16")]; fp16 var_3684_to_fp16 = const()[name = string("op_3684_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_139_cast_fp16 = mul(x = attn_weights_137_cast_fp16, y = var_3684_to_fp16)[name = string("attn_weights_139_cast_fp16")]; tensor attn_weights_141_cast_fp16 = add(x = attn_weights_139_cast_fp16, y = attn_mask_1)[name = string("attn_weights_141_cast_fp16")]; int32 var_3688 = const()[name = string("op_3688"), val = int32(-2)]; tensor attn_weights_143_cast_fp16 = softmax(axis = var_3688, x = attn_weights_141_cast_fp16)[name = string("attn_weights_143_cast_fp16")]; bool attn_output_65_transpose_x_1 = const()[name = string("attn_output_65_transpose_x_1"), val = bool(true)]; bool attn_output_65_transpose_y_1 = const()[name = string("attn_output_65_transpose_y_1"), val = bool(false)]; tensor attn_output_65_cast_fp16 = matmul(transpose_x = attn_output_65_transpose_x_1, transpose_y = attn_output_65_transpose_y_1, x = attn_weights_143_cast_fp16, y = var_3666_cast_fp16_1)[name = string("attn_output_65_cast_fp16")]; int32 var_3696 = const()[name = string("op_3696"), val = int32(1)]; bool attn_output_67_interleave_0 = const()[name = string("attn_output_67_interleave_0"), val = bool(false)]; tensor attn_output_67_cast_fp16 = concat(axis = var_3696, interleave = attn_output_67_interleave_0, values = (var_3682_cast_fp16, attn_output_65_cast_fp16))[name = string("attn_output_67_cast_fp16")]; tensor var_3700_perm_0 = const()[name = string("op_3700_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_107x = const()[name = string("concat_107x"), val = tensor([1, 2048, 1, -1])]; tensor var_3700_cast_fp16 = transpose(perm = var_3700_perm_0, x = attn_output_67_cast_fp16)[name = string("transpose_401")]; tensor attn_output_71_cast_fp16 = reshape(shape = concat_107x, x = var_3700_cast_fp16)[name = string("attn_output_71_cast_fp16")]; tensor hidden_states_83_strides_0 = const()[name = string("hidden_states_83_strides_0"), val = tensor([1, 1])]; string hidden_states_83_pad_type_0 = const()[name = string("hidden_states_83_pad_type_0"), val = string("valid")]; tensor hidden_states_83_pad_0 = const()[name = string("hidden_states_83_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_83_dilations_0 = const()[name = string("hidden_states_83_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_83_groups_0 = const()[name = string("hidden_states_83_groups_0"), val = int32(1)]; tensor hidden_states_83_cast_fp16 = conv(dilations = hidden_states_83_dilations_0, groups = hidden_states_83_groups_0, pad = hidden_states_83_pad_0, pad_type = hidden_states_83_pad_type_0, strides = hidden_states_83_strides_0, weight = layers_8_self_attn_o_proj_weight_cast_fp16, x = attn_output_71_cast_fp16)[name = string("hidden_states_83_cast_fp16")]; tensor hidden_states_85_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = hidden_states_83_cast_fp16)[name = string("hidden_states_85_cast_fp16")]; fp16 const_88_promoted_to_fp16 = const()[name = string("const_88_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3733_cast_fp16 = mul(x = hidden_states_85_cast_fp16, y = const_88_promoted_to_fp16)[name = string("op_3733_cast_fp16")]; int32 var_3731 = const()[name = string("op_3731"), val = int32(1)]; bool doubled_69_interleave_0 = const()[name = string("doubled_69_interleave_0"), val = bool(false)]; tensor doubled_69_cast_fp16 = concat(axis = var_3731, interleave = doubled_69_interleave_0, values = (hidden_states_85_cast_fp16, var_3733_cast_fp16))[name = string("doubled_69_cast_fp16")]; tensor out_35_axes_0 = const()[name = string("out_35_axes_0"), val = tensor([1])]; tensor out_35_gamma_0_to_fp16 = const()[name = string("out_35_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396442240)))]; fp16 var_3743_to_fp16 = const()[name = string("op_3743_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_35_cast_fp16 = layer_norm(axes = out_35_axes_0, epsilon = var_3743_to_fp16, gamma = out_35_gamma_0_to_fp16, x = doubled_69_cast_fp16)[name = string("out_35_cast_fp16")]; tensor var_3754_split_sizes_0 = const()[name = string("op_3754_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3754_axis_0 = const()[name = string("op_3754_axis_0"), val = int32(1)]; tensor var_3754_cast_fp16_0, tensor var_3754_cast_fp16_1 = split(axis = var_3754_axis_0, split_sizes = var_3754_split_sizes_0, x = out_35_cast_fp16)[name = string("op_3754_cast_fp16")]; tensor input_17_strides_0 = const()[name = string("input_17_strides_0"), val = tensor([1, 1])]; string input_17_pad_type_0 = const()[name = string("input_17_pad_type_0"), val = string("valid")]; tensor input_17_pad_0 = const()[name = string("input_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_17_dilations_0 = const()[name = string("input_17_dilations_0"), val = tensor([1, 1])]; int32 input_17_groups_0 = const()[name = string("input_17_groups_0"), val = int32(1)]; tensor input_17_cast_fp16 = conv(dilations = input_17_dilations_0, groups = input_17_groups_0, pad = input_17_pad_0, pad_type = input_17_pad_type_0, strides = input_17_strides_0, weight = layers_8_mlp_gate_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("input_17_cast_fp16")]; tensor var_3771_cast_fp16 = silu(x = input_17_cast_fp16)[name = string("op_3771_cast_fp16")]; tensor var_3777_strides_0 = const()[name = string("op_3777_strides_0"), val = tensor([1, 1])]; string var_3777_pad_type_0 = const()[name = string("op_3777_pad_type_0"), val = string("valid")]; tensor var_3777_pad_0 = const()[name = string("op_3777_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3777_dilations_0 = const()[name = string("op_3777_dilations_0"), val = tensor([1, 1])]; int32 var_3777_groups_0 = const()[name = string("op_3777_groups_0"), val = int32(1)]; tensor var_3777_cast_fp16 = conv(dilations = var_3777_dilations_0, groups = var_3777_groups_0, pad = var_3777_pad_0, pad_type = var_3777_pad_type_0, strides = var_3777_strides_0, weight = layers_8_mlp_up_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("op_3777_cast_fp16")]; tensor x_89_cast_fp16 = mul(x = var_3771_cast_fp16, y = var_3777_cast_fp16)[name = string("x_89_cast_fp16")]; tensor hidden_states_87_strides_0 = const()[name = string("hidden_states_87_strides_0"), val = tensor([1, 1])]; string hidden_states_87_pad_type_0 = const()[name = string("hidden_states_87_pad_type_0"), val = string("valid")]; tensor hidden_states_87_pad_0 = const()[name = string("hidden_states_87_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_87_dilations_0 = const()[name = string("hidden_states_87_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_87_groups_0 = const()[name = string("hidden_states_87_groups_0"), val = int32(1)]; tensor hidden_states_87_cast_fp16 = conv(dilations = hidden_states_87_dilations_0, groups = hidden_states_87_groups_0, pad = hidden_states_87_pad_0, pad_type = hidden_states_87_pad_type_0, strides = hidden_states_87_strides_0, weight = layers_8_mlp_down_proj_weight_cast_fp16, x = x_89_cast_fp16)[name = string("hidden_states_87_cast_fp16")]; tensor hidden_states_89_cast_fp16 = add(x = hidden_states_85_cast_fp16, y = hidden_states_87_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3795_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_3795_cast_fp16")]; int32 var_3793 = const()[name = string("op_3793"), val = int32(1)]; bool doubled_73_interleave_0 = const()[name = string("doubled_73_interleave_0"), val = bool(false)]; tensor doubled_73_cast_fp16 = concat(axis = var_3793, interleave = doubled_73_interleave_0, values = (hidden_states_89_cast_fp16, var_3795_cast_fp16))[name = string("doubled_73_cast_fp16")]; tensor out_37_axes_0 = const()[name = string("out_37_axes_0"), val = tensor([1])]; tensor out_37_gamma_0_to_fp16 = const()[name = string("out_37_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396450496)))]; fp16 var_3805_to_fp16 = const()[name = string("op_3805_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_37_cast_fp16 = layer_norm(axes = out_37_axes_0, epsilon = var_3805_to_fp16, gamma = out_37_gamma_0_to_fp16, x = doubled_73_cast_fp16)[name = string("out_37_cast_fp16")]; tensor var_3816_split_sizes_0 = const()[name = string("op_3816_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3816_axis_0 = const()[name = string("op_3816_axis_0"), val = int32(1)]; tensor var_3816_cast_fp16_0, tensor var_3816_cast_fp16_1 = split(axis = var_3816_axis_0, split_sizes = var_3816_split_sizes_0, x = out_37_cast_fp16)[name = string("op_3816_cast_fp16")]; tensor layers_9_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396458752)))]; tensor query_states_55_strides_0 = const()[name = string("query_states_55_strides_0"), val = tensor([1, 1])]; string query_states_55_pad_type_0 = const()[name = string("query_states_55_pad_type_0"), val = string("valid")]; tensor query_states_55_pad_0 = const()[name = string("query_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_55_dilations_0 = const()[name = string("query_states_55_dilations_0"), val = tensor([1, 1])]; int32 query_states_55_groups_0 = const()[name = string("query_states_55_groups_0"), val = int32(1)]; tensor query_states_55_cast_fp16 = conv(dilations = query_states_55_dilations_0, groups = query_states_55_groups_0, pad = query_states_55_pad_0, pad_type = query_states_55_pad_type_0, strides = query_states_55_strides_0, weight = layers_9_self_attn_q_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("query_states_55_cast_fp16")]; tensor layers_9_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1404847424)))]; tensor key_states_91_strides_0 = const()[name = string("key_states_91_strides_0"), val = tensor([1, 1])]; string key_states_91_pad_type_0 = const()[name = string("key_states_91_pad_type_0"), val = string("valid")]; tensor key_states_91_pad_0 = const()[name = string("key_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_91_dilations_0 = const()[name = string("key_states_91_dilations_0"), val = tensor([1, 1])]; int32 key_states_91_groups_0 = const()[name = string("key_states_91_groups_0"), val = int32(1)]; tensor key_states_91_cast_fp16 = conv(dilations = key_states_91_dilations_0, groups = key_states_91_groups_0, pad = key_states_91_pad_0, pad_type = key_states_91_pad_type_0, strides = key_states_91_strides_0, weight = layers_9_self_attn_k_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("key_states_91_cast_fp16")]; tensor value_states_55_strides_0 = const()[name = string("value_states_55_strides_0"), val = tensor([1, 1])]; string value_states_55_pad_type_0 = const()[name = string("value_states_55_pad_type_0"), val = string("valid")]; tensor value_states_55_pad_0 = const()[name = string("value_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_55_dilations_0 = const()[name = string("value_states_55_dilations_0"), val = tensor([1, 1])]; int32 value_states_55_groups_0 = const()[name = string("value_states_55_groups_0"), val = int32(1)]; tensor value_states_55_cast_fp16 = conv(dilations = value_states_55_dilations_0, groups = value_states_55_groups_0, pad = value_states_55_pad_0, pad_type = value_states_55_pad_type_0, strides = value_states_55_strides_0, weight = layers_9_self_attn_v_proj_weight_cast_fp16, x = var_3816_cast_fp16_0)[name = string("value_states_55_cast_fp16")]; tensor concat_108x = const()[name = string("concat_108x"), val = tensor([1, 16, 128, -1])]; tensor x_91_cast_fp16 = reshape(shape = concat_108x, x = query_states_55_cast_fp16)[name = string("x_91_cast_fp16")]; tensor concat_109x = const()[name = string("concat_109x"), val = tensor([1, 2, 128, -1])]; tensor var_3873_cast_fp16 = reshape(shape = concat_109x, x = key_states_91_cast_fp16)[name = string("op_3873_cast_fp16")]; tensor concat_110x = const()[name = string("concat_110x"), val = tensor([1, 2, 128, -1])]; tensor var_3880_cast_fp16 = reshape(shape = concat_110x, x = value_states_55_cast_fp16)[name = string("op_3880_cast_fp16")]; tensor var_3884_cast_fp16 = mul(x = x_91_cast_fp16, y = var_869_cast_fp16)[name = string("op_3884_cast_fp16")]; tensor var_3885_split_sizes_0 = const()[name = string("op_3885_split_sizes_0"), val = tensor([64, 64])]; int32 var_3885_axis_0 = const()[name = string("op_3885_axis_0"), val = int32(-2)]; tensor var_3885_cast_fp16_0, tensor var_3885_cast_fp16_1 = split(axis = var_3885_axis_0, split_sizes = var_3885_split_sizes_0, x = x_91_cast_fp16)[name = string("op_3885_cast_fp16")]; fp16 const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3887_cast_fp16 = mul(x = var_3885_cast_fp16_1, y = const_92_promoted_to_fp16)[name = string("op_3887_cast_fp16")]; int32 var_3889 = const()[name = string("op_3889"), val = int32(-2)]; bool var_3890_interleave_0 = const()[name = string("op_3890_interleave_0"), val = bool(false)]; tensor var_3890_cast_fp16 = concat(axis = var_3889, interleave = var_3890_interleave_0, values = (var_3887_cast_fp16, var_3885_cast_fp16_0))[name = string("op_3890_cast_fp16")]; tensor var_3891_cast_fp16 = mul(x = var_3890_cast_fp16, y = var_878_cast_fp16)[name = string("op_3891_cast_fp16")]; tensor query_states_57_cast_fp16 = add(x = var_3884_cast_fp16, y = var_3891_cast_fp16)[name = string("query_states_57_cast_fp16")]; tensor var_3897_cast_fp16 = mul(x = var_3873_cast_fp16, y = var_869_cast_fp16)[name = string("op_3897_cast_fp16")]; tensor var_3898_split_sizes_0 = const()[name = string("op_3898_split_sizes_0"), val = tensor([64, 64])]; int32 var_3898_axis_0 = const()[name = string("op_3898_axis_0"), val = int32(-2)]; tensor var_3898_cast_fp16_0, tensor var_3898_cast_fp16_1 = split(axis = var_3898_axis_0, split_sizes = var_3898_split_sizes_0, x = var_3873_cast_fp16)[name = string("op_3898_cast_fp16")]; fp16 const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3900_cast_fp16 = mul(x = var_3898_cast_fp16_1, y = const_93_promoted_to_fp16)[name = string("op_3900_cast_fp16")]; int32 var_3902 = const()[name = string("op_3902"), val = int32(-2)]; bool var_3903_interleave_0 = const()[name = string("op_3903_interleave_0"), val = bool(false)]; tensor var_3903_cast_fp16 = concat(axis = var_3902, interleave = var_3903_interleave_0, values = (var_3900_cast_fp16, var_3898_cast_fp16_0))[name = string("op_3903_cast_fp16")]; tensor var_3904_cast_fp16 = mul(x = var_3903_cast_fp16, y = var_878_cast_fp16)[name = string("op_3904_cast_fp16")]; tensor key_states_95_cast_fp16 = add(x = var_3897_cast_fp16, y = var_3904_cast_fp16)[name = string("key_states_95_cast_fp16")]; tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; int32 concat_113_axis_0 = const()[name = string("concat_113_axis_0"), val = int32(0)]; bool concat_113_interleave_0 = const()[name = string("concat_113_interleave_0"), val = bool(false)]; tensor concat_113 = concat(axis = concat_113_axis_0, interleave = concat_113_interleave_0, values = (expand_dims_108, expand_dims_109, position_id, expand_dims_111))[name = string("concat_113")]; tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; tensor concat_114_values1_0 = const()[name = string("concat_114_values1_0"), val = tensor([0])]; tensor concat_114_values3_0 = const()[name = string("concat_114_values3_0"), val = tensor([0])]; int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_112, concat_114_values1_0, cache_position_end, concat_114_values3_0))[name = string("concat_114")]; tensor key_states_97_perm_0 = const()[name = string("key_states_97_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_10_stride_0 = const()[name = string("key_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_97_cast_fp16 = transpose(perm = key_states_97_perm_0, x = key_states_95_cast_fp16)[name = string("transpose_400")]; tensor key_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = key_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = key_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_10_squeeze_mask_0, stride = key_cache_internal_tensor_assign_10_stride_0, update = key_states_97_cast_fp16, x = coreml_update_state_240)[name = string("key_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_10_cast_fp16, input = key_cache)[name = string("coreml_update_state_242_write_state")]; tensor coreml_update_state_242 = read_state(input = key_cache)[name = string("coreml_update_state_242")]; tensor value_states_57_perm_0 = const()[name = string("value_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_10_stride_0 = const()[name = string("value_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_57_cast_fp16 = transpose(perm = value_states_57_perm_0, x = var_3880_cast_fp16)[name = string("transpose_399")]; tensor value_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = value_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = value_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_10_squeeze_mask_0, stride = value_cache_internal_tensor_assign_10_stride_0, update = value_states_57_cast_fp16, x = coreml_update_state_241)[name = string("value_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_10_cast_fp16, input = value_cache)[name = string("coreml_update_state_243_write_state")]; tensor coreml_update_state_243 = read_state(input = value_cache)[name = string("coreml_update_state_243")]; tensor var_3974_begin_0 = const()[name = string("op_3974_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3974_end_0 = const()[name = string("op_3974_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3974_end_mask_0 = const()[name = string("op_3974_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3974_cast_fp16 = slice_by_index(begin = var_3974_begin_0, end = var_3974_end_0, end_mask = var_3974_end_mask_0, x = coreml_update_state_242)[name = string("op_3974_cast_fp16")]; tensor tile_18 = const()[name = string("tile_18"), val = tensor([1, 1])]; int32 var_3977_axis_0 = const()[name = string("op_3977_axis_0"), val = int32(1)]; tensor var_3977_cast_fp16_0, tensor var_3977_cast_fp16_1 = split(axis = var_3977_axis_0, split_sizes = tile_18, x = var_3974_cast_fp16)[name = string("op_3977_cast_fp16")]; tensor var_3984_begin_0 = const()[name = string("op_3984_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3984_end_0 = const()[name = string("op_3984_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3984_end_mask_0 = const()[name = string("op_3984_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3984_cast_fp16 = slice_by_index(begin = var_3984_begin_0, end = var_3984_end_0, end_mask = var_3984_end_mask_0, x = coreml_update_state_243)[name = string("op_3984_cast_fp16")]; tensor tile_19 = const()[name = string("tile_19"), val = tensor([1, 1])]; int32 var_3987_axis_0 = const()[name = string("op_3987_axis_0"), val = int32(1)]; tensor var_3987_cast_fp16_0, tensor var_3987_cast_fp16_1 = split(axis = var_3987_axis_0, split_sizes = tile_19, x = var_3984_cast_fp16)[name = string("op_3987_cast_fp16")]; tensor var_3990_split_sizes_0 = const()[name = string("op_3990_split_sizes_0"), val = tensor([8, 8])]; int32 var_3990_axis_0 = const()[name = string("op_3990_axis_0"), val = int32(1)]; tensor var_3990_0, tensor var_3990_1 = split(axis = var_3990_axis_0, split_sizes = var_3990_split_sizes_0, x = query_states_57_cast_fp16)[name = string("op_3990")]; bool attn_weights_145_transpose_x_0 = const()[name = string("attn_weights_145_transpose_x_0"), val = bool(false)]; bool attn_weights_145_transpose_y_0 = const()[name = string("attn_weights_145_transpose_y_0"), val = bool(false)]; tensor attn_weights_145_cast_fp16 = matmul(transpose_x = attn_weights_145_transpose_x_0, transpose_y = attn_weights_145_transpose_y_0, x = var_3977_cast_fp16_0, y = var_3990_0)[name = string("attn_weights_145_cast_fp16")]; fp16 var_3993_to_fp16 = const()[name = string("op_3993_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_147_cast_fp16 = mul(x = attn_weights_145_cast_fp16, y = var_3993_to_fp16)[name = string("attn_weights_147_cast_fp16")]; tensor attn_weights_149_cast_fp16 = add(x = attn_weights_147_cast_fp16, y = attn_mask_1)[name = string("attn_weights_149_cast_fp16")]; int32 var_3997 = const()[name = string("op_3997"), val = int32(-2)]; tensor attn_weights_151_cast_fp16 = softmax(axis = var_3997, x = attn_weights_149_cast_fp16)[name = string("attn_weights_151_cast_fp16")]; bool var_4003_transpose_x_1 = const()[name = string("op_4003_transpose_x_1"), val = bool(true)]; bool var_4003_transpose_y_1 = const()[name = string("op_4003_transpose_y_1"), val = bool(false)]; tensor var_4003_cast_fp16 = matmul(transpose_x = var_4003_transpose_x_1, transpose_y = var_4003_transpose_y_1, x = attn_weights_151_cast_fp16, y = var_3987_cast_fp16_0)[name = string("op_4003_cast_fp16")]; bool attn_weights_153_transpose_x_0 = const()[name = string("attn_weights_153_transpose_x_0"), val = bool(false)]; bool attn_weights_153_transpose_y_0 = const()[name = string("attn_weights_153_transpose_y_0"), val = bool(false)]; tensor attn_weights_153_cast_fp16 = matmul(transpose_x = attn_weights_153_transpose_x_0, transpose_y = attn_weights_153_transpose_y_0, x = var_3977_cast_fp16_1, y = var_3990_1)[name = string("attn_weights_153_cast_fp16")]; fp16 var_4005_to_fp16 = const()[name = string("op_4005_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_155_cast_fp16 = mul(x = attn_weights_153_cast_fp16, y = var_4005_to_fp16)[name = string("attn_weights_155_cast_fp16")]; tensor attn_weights_157_cast_fp16 = add(x = attn_weights_155_cast_fp16, y = attn_mask_1)[name = string("attn_weights_157_cast_fp16")]; int32 var_4009 = const()[name = string("op_4009"), val = int32(-2)]; tensor attn_weights_159_cast_fp16 = softmax(axis = var_4009, x = attn_weights_157_cast_fp16)[name = string("attn_weights_159_cast_fp16")]; bool attn_output_73_transpose_x_1 = const()[name = string("attn_output_73_transpose_x_1"), val = bool(true)]; bool attn_output_73_transpose_y_1 = const()[name = string("attn_output_73_transpose_y_1"), val = bool(false)]; tensor attn_output_73_cast_fp16 = matmul(transpose_x = attn_output_73_transpose_x_1, transpose_y = attn_output_73_transpose_y_1, x = attn_weights_159_cast_fp16, y = var_3987_cast_fp16_1)[name = string("attn_output_73_cast_fp16")]; int32 var_4017 = const()[name = string("op_4017"), val = int32(1)]; bool attn_output_75_interleave_0 = const()[name = string("attn_output_75_interleave_0"), val = bool(false)]; tensor attn_output_75_cast_fp16 = concat(axis = var_4017, interleave = attn_output_75_interleave_0, values = (var_4003_cast_fp16, attn_output_73_cast_fp16))[name = string("attn_output_75_cast_fp16")]; tensor var_4021_perm_0 = const()[name = string("op_4021_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_119x = const()[name = string("concat_119x"), val = tensor([1, 2048, 1, -1])]; tensor var_4021_cast_fp16 = transpose(perm = var_4021_perm_0, x = attn_output_75_cast_fp16)[name = string("transpose_398")]; tensor attn_output_79_cast_fp16 = reshape(shape = concat_119x, x = var_4021_cast_fp16)[name = string("attn_output_79_cast_fp16")]; tensor hidden_states_93_strides_0 = const()[name = string("hidden_states_93_strides_0"), val = tensor([1, 1])]; string hidden_states_93_pad_type_0 = const()[name = string("hidden_states_93_pad_type_0"), val = string("valid")]; tensor hidden_states_93_pad_0 = const()[name = string("hidden_states_93_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_93_dilations_0 = const()[name = string("hidden_states_93_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_93_groups_0 = const()[name = string("hidden_states_93_groups_0"), val = int32(1)]; tensor hidden_states_93_cast_fp16 = conv(dilations = hidden_states_93_dilations_0, groups = hidden_states_93_groups_0, pad = hidden_states_93_pad_0, pad_type = hidden_states_93_pad_type_0, strides = hidden_states_93_strides_0, weight = layers_9_self_attn_o_proj_weight_cast_fp16, x = attn_output_79_cast_fp16)[name = string("hidden_states_93_cast_fp16")]; tensor hidden_states_95_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = hidden_states_93_cast_fp16)[name = string("hidden_states_95_cast_fp16")]; fp16 const_98_promoted_to_fp16 = const()[name = string("const_98_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4054_cast_fp16 = mul(x = hidden_states_95_cast_fp16, y = const_98_promoted_to_fp16)[name = string("op_4054_cast_fp16")]; int32 var_4052 = const()[name = string("op_4052"), val = int32(1)]; bool doubled_77_interleave_0 = const()[name = string("doubled_77_interleave_0"), val = bool(false)]; tensor doubled_77_cast_fp16 = concat(axis = var_4052, interleave = doubled_77_interleave_0, values = (hidden_states_95_cast_fp16, var_4054_cast_fp16))[name = string("doubled_77_cast_fp16")]; tensor out_39_axes_0 = const()[name = string("out_39_axes_0"), val = tensor([1])]; tensor out_39_gamma_0_to_fp16 = const()[name = string("out_39_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405896064)))]; fp16 var_4064_to_fp16 = const()[name = string("op_4064_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_39_cast_fp16 = layer_norm(axes = out_39_axes_0, epsilon = var_4064_to_fp16, gamma = out_39_gamma_0_to_fp16, x = doubled_77_cast_fp16)[name = string("out_39_cast_fp16")]; tensor var_4075_split_sizes_0 = const()[name = string("op_4075_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4075_axis_0 = const()[name = string("op_4075_axis_0"), val = int32(1)]; tensor var_4075_cast_fp16_0, tensor var_4075_cast_fp16_1 = split(axis = var_4075_axis_0, split_sizes = var_4075_split_sizes_0, x = out_39_cast_fp16)[name = string("op_4075_cast_fp16")]; tensor input_19_strides_0 = const()[name = string("input_19_strides_0"), val = tensor([1, 1])]; string input_19_pad_type_0 = const()[name = string("input_19_pad_type_0"), val = string("valid")]; tensor input_19_pad_0 = const()[name = string("input_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_19_dilations_0 = const()[name = string("input_19_dilations_0"), val = tensor([1, 1])]; int32 input_19_groups_0 = const()[name = string("input_19_groups_0"), val = int32(1)]; tensor input_19_cast_fp16 = conv(dilations = input_19_dilations_0, groups = input_19_groups_0, pad = input_19_pad_0, pad_type = input_19_pad_type_0, strides = input_19_strides_0, weight = layers_9_mlp_gate_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("input_19_cast_fp16")]; tensor var_4092_cast_fp16 = silu(x = input_19_cast_fp16)[name = string("op_4092_cast_fp16")]; tensor var_4098_strides_0 = const()[name = string("op_4098_strides_0"), val = tensor([1, 1])]; string var_4098_pad_type_0 = const()[name = string("op_4098_pad_type_0"), val = string("valid")]; tensor var_4098_pad_0 = const()[name = string("op_4098_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4098_dilations_0 = const()[name = string("op_4098_dilations_0"), val = tensor([1, 1])]; int32 var_4098_groups_0 = const()[name = string("op_4098_groups_0"), val = int32(1)]; tensor var_4098_cast_fp16 = conv(dilations = var_4098_dilations_0, groups = var_4098_groups_0, pad = var_4098_pad_0, pad_type = var_4098_pad_type_0, strides = var_4098_strides_0, weight = layers_9_mlp_up_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("op_4098_cast_fp16")]; tensor x_99_cast_fp16 = mul(x = var_4092_cast_fp16, y = var_4098_cast_fp16)[name = string("x_99_cast_fp16")]; tensor hidden_states_97_strides_0 = const()[name = string("hidden_states_97_strides_0"), val = tensor([1, 1])]; string hidden_states_97_pad_type_0 = const()[name = string("hidden_states_97_pad_type_0"), val = string("valid")]; tensor hidden_states_97_pad_0 = const()[name = string("hidden_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_97_dilations_0 = const()[name = string("hidden_states_97_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_97_groups_0 = const()[name = string("hidden_states_97_groups_0"), val = int32(1)]; tensor hidden_states_97_cast_fp16 = conv(dilations = hidden_states_97_dilations_0, groups = hidden_states_97_groups_0, pad = hidden_states_97_pad_0, pad_type = hidden_states_97_pad_type_0, strides = hidden_states_97_strides_0, weight = layers_9_mlp_down_proj_weight_cast_fp16, x = x_99_cast_fp16)[name = string("hidden_states_97_cast_fp16")]; tensor hidden_states_99_cast_fp16 = add(x = hidden_states_95_cast_fp16, y = hidden_states_97_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; fp16 const_100_promoted_to_fp16 = const()[name = string("const_100_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4116_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_100_promoted_to_fp16)[name = string("op_4116_cast_fp16")]; int32 var_4114 = const()[name = string("op_4114"), val = int32(1)]; bool doubled_81_interleave_0 = const()[name = string("doubled_81_interleave_0"), val = bool(false)]; tensor doubled_81_cast_fp16 = concat(axis = var_4114, interleave = doubled_81_interleave_0, values = (hidden_states_99_cast_fp16, var_4116_cast_fp16))[name = string("doubled_81_cast_fp16")]; tensor out_41_axes_0 = const()[name = string("out_41_axes_0"), val = tensor([1])]; tensor out_41_gamma_0_to_fp16 = const()[name = string("out_41_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405904320)))]; fp16 var_4126_to_fp16 = const()[name = string("op_4126_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_41_cast_fp16 = layer_norm(axes = out_41_axes_0, epsilon = var_4126_to_fp16, gamma = out_41_gamma_0_to_fp16, x = doubled_81_cast_fp16)[name = string("out_41_cast_fp16")]; tensor var_4137_split_sizes_0 = const()[name = string("op_4137_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4137_axis_0 = const()[name = string("op_4137_axis_0"), val = int32(1)]; tensor var_4137_cast_fp16_0, tensor var_4137_cast_fp16_1 = split(axis = var_4137_axis_0, split_sizes = var_4137_split_sizes_0, x = out_41_cast_fp16)[name = string("op_4137_cast_fp16")]; tensor layers_10_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405912576)))]; tensor query_states_61_strides_0 = const()[name = string("query_states_61_strides_0"), val = tensor([1, 1])]; string query_states_61_pad_type_0 = const()[name = string("query_states_61_pad_type_0"), val = string("valid")]; tensor query_states_61_pad_0 = const()[name = string("query_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_61_dilations_0 = const()[name = string("query_states_61_dilations_0"), val = tensor([1, 1])]; int32 query_states_61_groups_0 = const()[name = string("query_states_61_groups_0"), val = int32(1)]; tensor query_states_61_cast_fp16 = conv(dilations = query_states_61_dilations_0, groups = query_states_61_groups_0, pad = query_states_61_pad_0, pad_type = query_states_61_pad_type_0, strides = query_states_61_strides_0, weight = layers_10_self_attn_q_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("query_states_61_cast_fp16")]; tensor layers_10_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1414301248)))]; tensor key_states_101_strides_0 = const()[name = string("key_states_101_strides_0"), val = tensor([1, 1])]; string key_states_101_pad_type_0 = const()[name = string("key_states_101_pad_type_0"), val = string("valid")]; tensor key_states_101_pad_0 = const()[name = string("key_states_101_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_101_dilations_0 = const()[name = string("key_states_101_dilations_0"), val = tensor([1, 1])]; int32 key_states_101_groups_0 = const()[name = string("key_states_101_groups_0"), val = int32(1)]; tensor key_states_101_cast_fp16 = conv(dilations = key_states_101_dilations_0, groups = key_states_101_groups_0, pad = key_states_101_pad_0, pad_type = key_states_101_pad_type_0, strides = key_states_101_strides_0, weight = layers_10_self_attn_k_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("key_states_101_cast_fp16")]; tensor value_states_61_strides_0 = const()[name = string("value_states_61_strides_0"), val = tensor([1, 1])]; string value_states_61_pad_type_0 = const()[name = string("value_states_61_pad_type_0"), val = string("valid")]; tensor value_states_61_pad_0 = const()[name = string("value_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_61_dilations_0 = const()[name = string("value_states_61_dilations_0"), val = tensor([1, 1])]; int32 value_states_61_groups_0 = const()[name = string("value_states_61_groups_0"), val = int32(1)]; tensor value_states_61_cast_fp16 = conv(dilations = value_states_61_dilations_0, groups = value_states_61_groups_0, pad = value_states_61_pad_0, pad_type = value_states_61_pad_type_0, strides = value_states_61_strides_0, weight = layers_10_self_attn_v_proj_weight_cast_fp16, x = var_4137_cast_fp16_0)[name = string("value_states_61_cast_fp16")]; tensor concat_120x = const()[name = string("concat_120x"), val = tensor([1, 16, 128, -1])]; tensor x_101_cast_fp16 = reshape(shape = concat_120x, x = query_states_61_cast_fp16)[name = string("x_101_cast_fp16")]; tensor concat_121x = const()[name = string("concat_121x"), val = tensor([1, 2, 128, -1])]; tensor var_4194_cast_fp16 = reshape(shape = concat_121x, x = key_states_101_cast_fp16)[name = string("op_4194_cast_fp16")]; tensor concat_122x = const()[name = string("concat_122x"), val = tensor([1, 2, 128, -1])]; tensor var_4201_cast_fp16 = reshape(shape = concat_122x, x = value_states_61_cast_fp16)[name = string("op_4201_cast_fp16")]; tensor var_4205_cast_fp16 = mul(x = x_101_cast_fp16, y = var_869_cast_fp16)[name = string("op_4205_cast_fp16")]; tensor var_4206_split_sizes_0 = const()[name = string("op_4206_split_sizes_0"), val = tensor([64, 64])]; int32 var_4206_axis_0 = const()[name = string("op_4206_axis_0"), val = int32(-2)]; tensor var_4206_cast_fp16_0, tensor var_4206_cast_fp16_1 = split(axis = var_4206_axis_0, split_sizes = var_4206_split_sizes_0, x = x_101_cast_fp16)[name = string("op_4206_cast_fp16")]; fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4208_cast_fp16 = mul(x = var_4206_cast_fp16_1, y = const_102_promoted_to_fp16)[name = string("op_4208_cast_fp16")]; int32 var_4210 = const()[name = string("op_4210"), val = int32(-2)]; bool var_4211_interleave_0 = const()[name = string("op_4211_interleave_0"), val = bool(false)]; tensor var_4211_cast_fp16 = concat(axis = var_4210, interleave = var_4211_interleave_0, values = (var_4208_cast_fp16, var_4206_cast_fp16_0))[name = string("op_4211_cast_fp16")]; tensor var_4212_cast_fp16 = mul(x = var_4211_cast_fp16, y = var_878_cast_fp16)[name = string("op_4212_cast_fp16")]; tensor query_states_63_cast_fp16 = add(x = var_4205_cast_fp16, y = var_4212_cast_fp16)[name = string("query_states_63_cast_fp16")]; tensor var_4218_cast_fp16 = mul(x = var_4194_cast_fp16, y = var_869_cast_fp16)[name = string("op_4218_cast_fp16")]; tensor var_4219_split_sizes_0 = const()[name = string("op_4219_split_sizes_0"), val = tensor([64, 64])]; int32 var_4219_axis_0 = const()[name = string("op_4219_axis_0"), val = int32(-2)]; tensor var_4219_cast_fp16_0, tensor var_4219_cast_fp16_1 = split(axis = var_4219_axis_0, split_sizes = var_4219_split_sizes_0, x = var_4194_cast_fp16)[name = string("op_4219_cast_fp16")]; fp16 const_103_promoted_to_fp16 = const()[name = string("const_103_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4221_cast_fp16 = mul(x = var_4219_cast_fp16_1, y = const_103_promoted_to_fp16)[name = string("op_4221_cast_fp16")]; int32 var_4223 = const()[name = string("op_4223"), val = int32(-2)]; bool var_4224_interleave_0 = const()[name = string("op_4224_interleave_0"), val = bool(false)]; tensor var_4224_cast_fp16 = concat(axis = var_4223, interleave = var_4224_interleave_0, values = (var_4221_cast_fp16, var_4219_cast_fp16_0))[name = string("op_4224_cast_fp16")]; tensor var_4225_cast_fp16 = mul(x = var_4224_cast_fp16, y = var_878_cast_fp16)[name = string("op_4225_cast_fp16")]; tensor key_states_105_cast_fp16 = add(x = var_4218_cast_fp16, y = var_4225_cast_fp16)[name = string("key_states_105_cast_fp16")]; tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; int32 concat_125_axis_0 = const()[name = string("concat_125_axis_0"), val = int32(0)]; bool concat_125_interleave_0 = const()[name = string("concat_125_interleave_0"), val = bool(false)]; tensor concat_125 = concat(axis = concat_125_axis_0, interleave = concat_125_interleave_0, values = (expand_dims_120, expand_dims_121, position_id, expand_dims_123))[name = string("concat_125")]; tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; tensor concat_126_values1_0 = const()[name = string("concat_126_values1_0"), val = tensor([0])]; tensor concat_126_values3_0 = const()[name = string("concat_126_values3_0"), val = tensor([0])]; int32 concat_126_axis_0 = const()[name = string("concat_126_axis_0"), val = int32(0)]; bool concat_126_interleave_0 = const()[name = string("concat_126_interleave_0"), val = bool(false)]; tensor concat_126 = concat(axis = concat_126_axis_0, interleave = concat_126_interleave_0, values = (expand_dims_124, concat_126_values1_0, cache_position_end, concat_126_values3_0))[name = string("concat_126")]; tensor key_states_107_perm_0 = const()[name = string("key_states_107_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_11_stride_0 = const()[name = string("key_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_107_cast_fp16 = transpose(perm = key_states_107_perm_0, x = key_states_105_cast_fp16)[name = string("transpose_397")]; tensor key_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = key_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = key_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_11_squeeze_mask_0, stride = key_cache_internal_tensor_assign_11_stride_0, update = key_states_107_cast_fp16, x = coreml_update_state_242)[name = string("key_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_11_cast_fp16, input = key_cache)[name = string("coreml_update_state_244_write_state")]; tensor coreml_update_state_244 = read_state(input = key_cache)[name = string("coreml_update_state_244")]; tensor value_states_63_perm_0 = const()[name = string("value_states_63_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_11_stride_0 = const()[name = string("value_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_63_cast_fp16 = transpose(perm = value_states_63_perm_0, x = var_4201_cast_fp16)[name = string("transpose_396")]; tensor value_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = value_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = value_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_11_squeeze_mask_0, stride = value_cache_internal_tensor_assign_11_stride_0, update = value_states_63_cast_fp16, x = coreml_update_state_243)[name = string("value_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_11_cast_fp16, input = value_cache)[name = string("coreml_update_state_245_write_state")]; tensor coreml_update_state_245 = read_state(input = value_cache)[name = string("coreml_update_state_245")]; tensor var_4295_begin_0 = const()[name = string("op_4295_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4295_end_0 = const()[name = string("op_4295_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4295_end_mask_0 = const()[name = string("op_4295_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4295_cast_fp16 = slice_by_index(begin = var_4295_begin_0, end = var_4295_end_0, end_mask = var_4295_end_mask_0, x = coreml_update_state_244)[name = string("op_4295_cast_fp16")]; tensor tile_20 = const()[name = string("tile_20"), val = tensor([1, 1])]; int32 var_4298_axis_0 = const()[name = string("op_4298_axis_0"), val = int32(1)]; tensor var_4298_cast_fp16_0, tensor var_4298_cast_fp16_1 = split(axis = var_4298_axis_0, split_sizes = tile_20, x = var_4295_cast_fp16)[name = string("op_4298_cast_fp16")]; tensor var_4305_begin_0 = const()[name = string("op_4305_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4305_end_0 = const()[name = string("op_4305_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4305_end_mask_0 = const()[name = string("op_4305_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4305_cast_fp16 = slice_by_index(begin = var_4305_begin_0, end = var_4305_end_0, end_mask = var_4305_end_mask_0, x = coreml_update_state_245)[name = string("op_4305_cast_fp16")]; tensor tile_21 = const()[name = string("tile_21"), val = tensor([1, 1])]; int32 var_4308_axis_0 = const()[name = string("op_4308_axis_0"), val = int32(1)]; tensor var_4308_cast_fp16_0, tensor var_4308_cast_fp16_1 = split(axis = var_4308_axis_0, split_sizes = tile_21, x = var_4305_cast_fp16)[name = string("op_4308_cast_fp16")]; tensor var_4311_split_sizes_0 = const()[name = string("op_4311_split_sizes_0"), val = tensor([8, 8])]; int32 var_4311_axis_0 = const()[name = string("op_4311_axis_0"), val = int32(1)]; tensor var_4311_0, tensor var_4311_1 = split(axis = var_4311_axis_0, split_sizes = var_4311_split_sizes_0, x = query_states_63_cast_fp16)[name = string("op_4311")]; bool attn_weights_161_transpose_x_0 = const()[name = string("attn_weights_161_transpose_x_0"), val = bool(false)]; bool attn_weights_161_transpose_y_0 = const()[name = string("attn_weights_161_transpose_y_0"), val = bool(false)]; tensor attn_weights_161_cast_fp16 = matmul(transpose_x = attn_weights_161_transpose_x_0, transpose_y = attn_weights_161_transpose_y_0, x = var_4298_cast_fp16_0, y = var_4311_0)[name = string("attn_weights_161_cast_fp16")]; fp16 var_4314_to_fp16 = const()[name = string("op_4314_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_163_cast_fp16 = mul(x = attn_weights_161_cast_fp16, y = var_4314_to_fp16)[name = string("attn_weights_163_cast_fp16")]; tensor attn_weights_165_cast_fp16 = add(x = attn_weights_163_cast_fp16, y = attn_mask_1)[name = string("attn_weights_165_cast_fp16")]; int32 var_4318 = const()[name = string("op_4318"), val = int32(-2)]; tensor attn_weights_167_cast_fp16 = softmax(axis = var_4318, x = attn_weights_165_cast_fp16)[name = string("attn_weights_167_cast_fp16")]; bool var_4324_transpose_x_1 = const()[name = string("op_4324_transpose_x_1"), val = bool(true)]; bool var_4324_transpose_y_1 = const()[name = string("op_4324_transpose_y_1"), val = bool(false)]; tensor var_4324_cast_fp16 = matmul(transpose_x = var_4324_transpose_x_1, transpose_y = var_4324_transpose_y_1, x = attn_weights_167_cast_fp16, y = var_4308_cast_fp16_0)[name = string("op_4324_cast_fp16")]; bool attn_weights_169_transpose_x_0 = const()[name = string("attn_weights_169_transpose_x_0"), val = bool(false)]; bool attn_weights_169_transpose_y_0 = const()[name = string("attn_weights_169_transpose_y_0"), val = bool(false)]; tensor attn_weights_169_cast_fp16 = matmul(transpose_x = attn_weights_169_transpose_x_0, transpose_y = attn_weights_169_transpose_y_0, x = var_4298_cast_fp16_1, y = var_4311_1)[name = string("attn_weights_169_cast_fp16")]; fp16 var_4326_to_fp16 = const()[name = string("op_4326_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_171_cast_fp16 = mul(x = attn_weights_169_cast_fp16, y = var_4326_to_fp16)[name = string("attn_weights_171_cast_fp16")]; tensor attn_weights_173_cast_fp16 = add(x = attn_weights_171_cast_fp16, y = attn_mask_1)[name = string("attn_weights_173_cast_fp16")]; int32 var_4330 = const()[name = string("op_4330"), val = int32(-2)]; tensor attn_weights_175_cast_fp16 = softmax(axis = var_4330, x = attn_weights_173_cast_fp16)[name = string("attn_weights_175_cast_fp16")]; bool attn_output_81_transpose_x_1 = const()[name = string("attn_output_81_transpose_x_1"), val = bool(true)]; bool attn_output_81_transpose_y_1 = const()[name = string("attn_output_81_transpose_y_1"), val = bool(false)]; tensor attn_output_81_cast_fp16 = matmul(transpose_x = attn_output_81_transpose_x_1, transpose_y = attn_output_81_transpose_y_1, x = attn_weights_175_cast_fp16, y = var_4308_cast_fp16_1)[name = string("attn_output_81_cast_fp16")]; int32 var_4338 = const()[name = string("op_4338"), val = int32(1)]; bool attn_output_83_interleave_0 = const()[name = string("attn_output_83_interleave_0"), val = bool(false)]; tensor attn_output_83_cast_fp16 = concat(axis = var_4338, interleave = attn_output_83_interleave_0, values = (var_4324_cast_fp16, attn_output_81_cast_fp16))[name = string("attn_output_83_cast_fp16")]; tensor var_4342_perm_0 = const()[name = string("op_4342_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_131x = const()[name = string("concat_131x"), val = tensor([1, 2048, 1, -1])]; tensor var_4342_cast_fp16 = transpose(perm = var_4342_perm_0, x = attn_output_83_cast_fp16)[name = string("transpose_395")]; tensor attn_output_87_cast_fp16 = reshape(shape = concat_131x, x = var_4342_cast_fp16)[name = string("attn_output_87_cast_fp16")]; tensor hidden_states_103_strides_0 = const()[name = string("hidden_states_103_strides_0"), val = tensor([1, 1])]; string hidden_states_103_pad_type_0 = const()[name = string("hidden_states_103_pad_type_0"), val = string("valid")]; tensor hidden_states_103_pad_0 = const()[name = string("hidden_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_103_dilations_0 = const()[name = string("hidden_states_103_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_103_groups_0 = const()[name = string("hidden_states_103_groups_0"), val = int32(1)]; tensor hidden_states_103_cast_fp16 = conv(dilations = hidden_states_103_dilations_0, groups = hidden_states_103_groups_0, pad = hidden_states_103_pad_0, pad_type = hidden_states_103_pad_type_0, strides = hidden_states_103_strides_0, weight = layers_10_self_attn_o_proj_weight_cast_fp16, x = attn_output_87_cast_fp16)[name = string("hidden_states_103_cast_fp16")]; tensor hidden_states_105_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = hidden_states_103_cast_fp16)[name = string("hidden_states_105_cast_fp16")]; fp16 const_108_promoted_to_fp16 = const()[name = string("const_108_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4375_cast_fp16 = mul(x = hidden_states_105_cast_fp16, y = const_108_promoted_to_fp16)[name = string("op_4375_cast_fp16")]; int32 var_4373 = const()[name = string("op_4373"), val = int32(1)]; bool doubled_85_interleave_0 = const()[name = string("doubled_85_interleave_0"), val = bool(false)]; tensor doubled_85_cast_fp16 = concat(axis = var_4373, interleave = doubled_85_interleave_0, values = (hidden_states_105_cast_fp16, var_4375_cast_fp16))[name = string("doubled_85_cast_fp16")]; tensor out_43_axes_0 = const()[name = string("out_43_axes_0"), val = tensor([1])]; tensor out_43_gamma_0_to_fp16 = const()[name = string("out_43_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415349888)))]; fp16 var_4385_to_fp16 = const()[name = string("op_4385_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_43_cast_fp16 = layer_norm(axes = out_43_axes_0, epsilon = var_4385_to_fp16, gamma = out_43_gamma_0_to_fp16, x = doubled_85_cast_fp16)[name = string("out_43_cast_fp16")]; tensor var_4396_split_sizes_0 = const()[name = string("op_4396_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4396_axis_0 = const()[name = string("op_4396_axis_0"), val = int32(1)]; tensor var_4396_cast_fp16_0, tensor var_4396_cast_fp16_1 = split(axis = var_4396_axis_0, split_sizes = var_4396_split_sizes_0, x = out_43_cast_fp16)[name = string("op_4396_cast_fp16")]; tensor input_21_strides_0 = const()[name = string("input_21_strides_0"), val = tensor([1, 1])]; string input_21_pad_type_0 = const()[name = string("input_21_pad_type_0"), val = string("valid")]; tensor input_21_pad_0 = const()[name = string("input_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_21_dilations_0 = const()[name = string("input_21_dilations_0"), val = tensor([1, 1])]; int32 input_21_groups_0 = const()[name = string("input_21_groups_0"), val = int32(1)]; tensor input_21_cast_fp16 = conv(dilations = input_21_dilations_0, groups = input_21_groups_0, pad = input_21_pad_0, pad_type = input_21_pad_type_0, strides = input_21_strides_0, weight = layers_10_mlp_gate_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("input_21_cast_fp16")]; tensor var_4413_cast_fp16 = silu(x = input_21_cast_fp16)[name = string("op_4413_cast_fp16")]; tensor var_4419_strides_0 = const()[name = string("op_4419_strides_0"), val = tensor([1, 1])]; string var_4419_pad_type_0 = const()[name = string("op_4419_pad_type_0"), val = string("valid")]; tensor var_4419_pad_0 = const()[name = string("op_4419_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4419_dilations_0 = const()[name = string("op_4419_dilations_0"), val = tensor([1, 1])]; int32 var_4419_groups_0 = const()[name = string("op_4419_groups_0"), val = int32(1)]; tensor var_4419_cast_fp16 = conv(dilations = var_4419_dilations_0, groups = var_4419_groups_0, pad = var_4419_pad_0, pad_type = var_4419_pad_type_0, strides = var_4419_strides_0, weight = layers_10_mlp_up_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("op_4419_cast_fp16")]; tensor x_109_cast_fp16 = mul(x = var_4413_cast_fp16, y = var_4419_cast_fp16)[name = string("x_109_cast_fp16")]; tensor hidden_states_107_strides_0 = const()[name = string("hidden_states_107_strides_0"), val = tensor([1, 1])]; string hidden_states_107_pad_type_0 = const()[name = string("hidden_states_107_pad_type_0"), val = string("valid")]; tensor hidden_states_107_pad_0 = const()[name = string("hidden_states_107_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_107_dilations_0 = const()[name = string("hidden_states_107_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_107_groups_0 = const()[name = string("hidden_states_107_groups_0"), val = int32(1)]; tensor hidden_states_107_cast_fp16 = conv(dilations = hidden_states_107_dilations_0, groups = hidden_states_107_groups_0, pad = hidden_states_107_pad_0, pad_type = hidden_states_107_pad_type_0, strides = hidden_states_107_strides_0, weight = layers_10_mlp_down_proj_weight_cast_fp16, x = x_109_cast_fp16)[name = string("hidden_states_107_cast_fp16")]; tensor hidden_states_109_cast_fp16 = add(x = hidden_states_105_cast_fp16, y = hidden_states_107_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; fp16 const_110_promoted_to_fp16 = const()[name = string("const_110_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4437_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_110_promoted_to_fp16)[name = string("op_4437_cast_fp16")]; int32 var_4435 = const()[name = string("op_4435"), val = int32(1)]; bool doubled_89_interleave_0 = const()[name = string("doubled_89_interleave_0"), val = bool(false)]; tensor doubled_89_cast_fp16 = concat(axis = var_4435, interleave = doubled_89_interleave_0, values = (hidden_states_109_cast_fp16, var_4437_cast_fp16))[name = string("doubled_89_cast_fp16")]; tensor out_45_axes_0 = const()[name = string("out_45_axes_0"), val = tensor([1])]; tensor out_45_gamma_0_to_fp16 = const()[name = string("out_45_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415358144)))]; fp16 var_4447_to_fp16 = const()[name = string("op_4447_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_45_cast_fp16 = layer_norm(axes = out_45_axes_0, epsilon = var_4447_to_fp16, gamma = out_45_gamma_0_to_fp16, x = doubled_89_cast_fp16)[name = string("out_45_cast_fp16")]; tensor var_4458_split_sizes_0 = const()[name = string("op_4458_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4458_axis_0 = const()[name = string("op_4458_axis_0"), val = int32(1)]; tensor var_4458_cast_fp16_0, tensor var_4458_cast_fp16_1 = split(axis = var_4458_axis_0, split_sizes = var_4458_split_sizes_0, x = out_45_cast_fp16)[name = string("op_4458_cast_fp16")]; tensor query_states_67_strides_0 = const()[name = string("query_states_67_strides_0"), val = tensor([1, 1])]; string query_states_67_pad_type_0 = const()[name = string("query_states_67_pad_type_0"), val = string("valid")]; tensor query_states_67_pad_0 = const()[name = string("query_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_67_dilations_0 = const()[name = string("query_states_67_dilations_0"), val = tensor([1, 1])]; int32 query_states_67_groups_0 = const()[name = string("query_states_67_groups_0"), val = int32(1)]; tensor query_states_67_cast_fp16 = conv(dilations = query_states_67_dilations_0, groups = query_states_67_groups_0, pad = query_states_67_pad_0, pad_type = query_states_67_pad_type_0, strides = query_states_67_strides_0, weight = layers_11_self_attn_q_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("query_states_67_cast_fp16")]; tensor key_states_111_strides_0 = const()[name = string("key_states_111_strides_0"), val = tensor([1, 1])]; string key_states_111_pad_type_0 = const()[name = string("key_states_111_pad_type_0"), val = string("valid")]; tensor key_states_111_pad_0 = const()[name = string("key_states_111_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_111_dilations_0 = const()[name = string("key_states_111_dilations_0"), val = tensor([1, 1])]; int32 key_states_111_groups_0 = const()[name = string("key_states_111_groups_0"), val = int32(1)]; tensor key_states_111_cast_fp16 = conv(dilations = key_states_111_dilations_0, groups = key_states_111_groups_0, pad = key_states_111_pad_0, pad_type = key_states_111_pad_type_0, strides = key_states_111_strides_0, weight = layers_11_self_attn_k_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("key_states_111_cast_fp16")]; tensor value_states_67_strides_0 = const()[name = string("value_states_67_strides_0"), val = tensor([1, 1])]; string value_states_67_pad_type_0 = const()[name = string("value_states_67_pad_type_0"), val = string("valid")]; tensor value_states_67_pad_0 = const()[name = string("value_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_67_dilations_0 = const()[name = string("value_states_67_dilations_0"), val = tensor([1, 1])]; int32 value_states_67_groups_0 = const()[name = string("value_states_67_groups_0"), val = int32(1)]; tensor value_states_67_cast_fp16 = conv(dilations = value_states_67_dilations_0, groups = value_states_67_groups_0, pad = value_states_67_pad_0, pad_type = value_states_67_pad_type_0, strides = value_states_67_strides_0, weight = layers_11_self_attn_v_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("value_states_67_cast_fp16")]; tensor concat_132x = const()[name = string("concat_132x"), val = tensor([1, 16, 128, -1])]; tensor x_111_cast_fp16 = reshape(shape = concat_132x, x = query_states_67_cast_fp16)[name = string("x_111_cast_fp16")]; tensor concat_133x = const()[name = string("concat_133x"), val = tensor([1, 2, 128, -1])]; tensor var_4515_cast_fp16 = reshape(shape = concat_133x, x = key_states_111_cast_fp16)[name = string("op_4515_cast_fp16")]; tensor concat_134x = const()[name = string("concat_134x"), val = tensor([1, 2, 128, -1])]; tensor var_4522_cast_fp16 = reshape(shape = concat_134x, x = value_states_67_cast_fp16)[name = string("op_4522_cast_fp16")]; tensor var_4526_cast_fp16 = mul(x = x_111_cast_fp16, y = var_869_cast_fp16)[name = string("op_4526_cast_fp16")]; tensor var_4527_split_sizes_0 = const()[name = string("op_4527_split_sizes_0"), val = tensor([64, 64])]; int32 var_4527_axis_0 = const()[name = string("op_4527_axis_0"), val = int32(-2)]; tensor var_4527_cast_fp16_0, tensor var_4527_cast_fp16_1 = split(axis = var_4527_axis_0, split_sizes = var_4527_split_sizes_0, x = x_111_cast_fp16)[name = string("op_4527_cast_fp16")]; fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4529_cast_fp16 = mul(x = var_4527_cast_fp16_1, y = const_112_promoted_to_fp16)[name = string("op_4529_cast_fp16")]; int32 var_4531 = const()[name = string("op_4531"), val = int32(-2)]; bool var_4532_interleave_0 = const()[name = string("op_4532_interleave_0"), val = bool(false)]; tensor var_4532_cast_fp16 = concat(axis = var_4531, interleave = var_4532_interleave_0, values = (var_4529_cast_fp16, var_4527_cast_fp16_0))[name = string("op_4532_cast_fp16")]; tensor var_4533_cast_fp16 = mul(x = var_4532_cast_fp16, y = var_878_cast_fp16)[name = string("op_4533_cast_fp16")]; tensor query_states_69_cast_fp16 = add(x = var_4526_cast_fp16, y = var_4533_cast_fp16)[name = string("query_states_69_cast_fp16")]; tensor var_4539_cast_fp16 = mul(x = var_4515_cast_fp16, y = var_869_cast_fp16)[name = string("op_4539_cast_fp16")]; tensor var_4540_split_sizes_0 = const()[name = string("op_4540_split_sizes_0"), val = tensor([64, 64])]; int32 var_4540_axis_0 = const()[name = string("op_4540_axis_0"), val = int32(-2)]; tensor var_4540_cast_fp16_0, tensor var_4540_cast_fp16_1 = split(axis = var_4540_axis_0, split_sizes = var_4540_split_sizes_0, x = var_4515_cast_fp16)[name = string("op_4540_cast_fp16")]; fp16 const_113_promoted_to_fp16 = const()[name = string("const_113_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4542_cast_fp16 = mul(x = var_4540_cast_fp16_1, y = const_113_promoted_to_fp16)[name = string("op_4542_cast_fp16")]; int32 var_4544 = const()[name = string("op_4544"), val = int32(-2)]; bool var_4545_interleave_0 = const()[name = string("op_4545_interleave_0"), val = bool(false)]; tensor var_4545_cast_fp16 = concat(axis = var_4544, interleave = var_4545_interleave_0, values = (var_4542_cast_fp16, var_4540_cast_fp16_0))[name = string("op_4545_cast_fp16")]; tensor var_4546_cast_fp16 = mul(x = var_4545_cast_fp16, y = var_878_cast_fp16)[name = string("op_4546_cast_fp16")]; tensor key_states_115_cast_fp16 = add(x = var_4539_cast_fp16, y = var_4546_cast_fp16)[name = string("key_states_115_cast_fp16")]; tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; int32 concat_137_axis_0 = const()[name = string("concat_137_axis_0"), val = int32(0)]; bool concat_137_interleave_0 = const()[name = string("concat_137_interleave_0"), val = bool(false)]; tensor concat_137 = concat(axis = concat_137_axis_0, interleave = concat_137_interleave_0, values = (expand_dims_132, expand_dims_133, position_id, expand_dims_135))[name = string("concat_137")]; tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; tensor concat_138_values1_0 = const()[name = string("concat_138_values1_0"), val = tensor([0])]; tensor concat_138_values3_0 = const()[name = string("concat_138_values3_0"), val = tensor([0])]; int32 concat_138_axis_0 = const()[name = string("concat_138_axis_0"), val = int32(0)]; bool concat_138_interleave_0 = const()[name = string("concat_138_interleave_0"), val = bool(false)]; tensor concat_138 = concat(axis = concat_138_axis_0, interleave = concat_138_interleave_0, values = (expand_dims_136, concat_138_values1_0, cache_position_end, concat_138_values3_0))[name = string("concat_138")]; tensor key_states_117_perm_0 = const()[name = string("key_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_12_stride_0 = const()[name = string("key_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_117_cast_fp16 = transpose(perm = key_states_117_perm_0, x = key_states_115_cast_fp16)[name = string("transpose_394")]; tensor key_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = key_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = key_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_12_squeeze_mask_0, stride = key_cache_internal_tensor_assign_12_stride_0, update = key_states_117_cast_fp16, x = coreml_update_state_244)[name = string("key_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_12_cast_fp16, input = key_cache)[name = string("coreml_update_state_246_write_state")]; tensor coreml_update_state_246 = read_state(input = key_cache)[name = string("coreml_update_state_246")]; tensor value_states_69_perm_0 = const()[name = string("value_states_69_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_12_stride_0 = const()[name = string("value_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_69_cast_fp16 = transpose(perm = value_states_69_perm_0, x = var_4522_cast_fp16)[name = string("transpose_393")]; tensor value_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = value_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = value_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_12_squeeze_mask_0, stride = value_cache_internal_tensor_assign_12_stride_0, update = value_states_69_cast_fp16, x = coreml_update_state_245)[name = string("value_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_12_cast_fp16, input = value_cache)[name = string("coreml_update_state_247_write_state")]; tensor coreml_update_state_247 = read_state(input = value_cache)[name = string("coreml_update_state_247")]; tensor var_4616_begin_0 = const()[name = string("op_4616_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4616_end_0 = const()[name = string("op_4616_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4616_end_mask_0 = const()[name = string("op_4616_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4616_cast_fp16 = slice_by_index(begin = var_4616_begin_0, end = var_4616_end_0, end_mask = var_4616_end_mask_0, x = coreml_update_state_246)[name = string("op_4616_cast_fp16")]; tensor tile_22 = const()[name = string("tile_22"), val = tensor([1, 1])]; int32 var_4619_axis_0 = const()[name = string("op_4619_axis_0"), val = int32(1)]; tensor var_4619_cast_fp16_0, tensor var_4619_cast_fp16_1 = split(axis = var_4619_axis_0, split_sizes = tile_22, x = var_4616_cast_fp16)[name = string("op_4619_cast_fp16")]; tensor var_4626_begin_0 = const()[name = string("op_4626_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4626_end_0 = const()[name = string("op_4626_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4626_end_mask_0 = const()[name = string("op_4626_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4626_cast_fp16 = slice_by_index(begin = var_4626_begin_0, end = var_4626_end_0, end_mask = var_4626_end_mask_0, x = coreml_update_state_247)[name = string("op_4626_cast_fp16")]; tensor tile_23 = const()[name = string("tile_23"), val = tensor([1, 1])]; int32 var_4629_axis_0 = const()[name = string("op_4629_axis_0"), val = int32(1)]; tensor var_4629_cast_fp16_0, tensor var_4629_cast_fp16_1 = split(axis = var_4629_axis_0, split_sizes = tile_23, x = var_4626_cast_fp16)[name = string("op_4629_cast_fp16")]; tensor var_4632_split_sizes_0 = const()[name = string("op_4632_split_sizes_0"), val = tensor([8, 8])]; int32 var_4632_axis_0 = const()[name = string("op_4632_axis_0"), val = int32(1)]; tensor var_4632_0, tensor var_4632_1 = split(axis = var_4632_axis_0, split_sizes = var_4632_split_sizes_0, x = query_states_69_cast_fp16)[name = string("op_4632")]; bool attn_weights_177_transpose_x_0 = const()[name = string("attn_weights_177_transpose_x_0"), val = bool(false)]; bool attn_weights_177_transpose_y_0 = const()[name = string("attn_weights_177_transpose_y_0"), val = bool(false)]; tensor attn_weights_177_cast_fp16 = matmul(transpose_x = attn_weights_177_transpose_x_0, transpose_y = attn_weights_177_transpose_y_0, x = var_4619_cast_fp16_0, y = var_4632_0)[name = string("attn_weights_177_cast_fp16")]; fp16 var_4635_to_fp16 = const()[name = string("op_4635_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_179_cast_fp16 = mul(x = attn_weights_177_cast_fp16, y = var_4635_to_fp16)[name = string("attn_weights_179_cast_fp16")]; tensor attn_weights_181_cast_fp16 = add(x = attn_weights_179_cast_fp16, y = attn_mask_1)[name = string("attn_weights_181_cast_fp16")]; int32 var_4639 = const()[name = string("op_4639"), val = int32(-2)]; tensor attn_weights_183_cast_fp16 = softmax(axis = var_4639, x = attn_weights_181_cast_fp16)[name = string("attn_weights_183_cast_fp16")]; bool var_4645_transpose_x_1 = const()[name = string("op_4645_transpose_x_1"), val = bool(true)]; bool var_4645_transpose_y_1 = const()[name = string("op_4645_transpose_y_1"), val = bool(false)]; tensor var_4645_cast_fp16 = matmul(transpose_x = var_4645_transpose_x_1, transpose_y = var_4645_transpose_y_1, x = attn_weights_183_cast_fp16, y = var_4629_cast_fp16_0)[name = string("op_4645_cast_fp16")]; bool attn_weights_185_transpose_x_0 = const()[name = string("attn_weights_185_transpose_x_0"), val = bool(false)]; bool attn_weights_185_transpose_y_0 = const()[name = string("attn_weights_185_transpose_y_0"), val = bool(false)]; tensor attn_weights_185_cast_fp16 = matmul(transpose_x = attn_weights_185_transpose_x_0, transpose_y = attn_weights_185_transpose_y_0, x = var_4619_cast_fp16_1, y = var_4632_1)[name = string("attn_weights_185_cast_fp16")]; fp16 var_4647_to_fp16 = const()[name = string("op_4647_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_187_cast_fp16 = mul(x = attn_weights_185_cast_fp16, y = var_4647_to_fp16)[name = string("attn_weights_187_cast_fp16")]; tensor attn_weights_189_cast_fp16 = add(x = attn_weights_187_cast_fp16, y = attn_mask_1)[name = string("attn_weights_189_cast_fp16")]; int32 var_4651 = const()[name = string("op_4651"), val = int32(-2)]; tensor attn_weights_191_cast_fp16 = softmax(axis = var_4651, x = attn_weights_189_cast_fp16)[name = string("attn_weights_191_cast_fp16")]; bool attn_output_89_transpose_x_1 = const()[name = string("attn_output_89_transpose_x_1"), val = bool(true)]; bool attn_output_89_transpose_y_1 = const()[name = string("attn_output_89_transpose_y_1"), val = bool(false)]; tensor attn_output_89_cast_fp16 = matmul(transpose_x = attn_output_89_transpose_x_1, transpose_y = attn_output_89_transpose_y_1, x = attn_weights_191_cast_fp16, y = var_4629_cast_fp16_1)[name = string("attn_output_89_cast_fp16")]; int32 var_4659 = const()[name = string("op_4659"), val = int32(1)]; bool attn_output_91_interleave_0 = const()[name = string("attn_output_91_interleave_0"), val = bool(false)]; tensor attn_output_91_cast_fp16 = concat(axis = var_4659, interleave = attn_output_91_interleave_0, values = (var_4645_cast_fp16, attn_output_89_cast_fp16))[name = string("attn_output_91_cast_fp16")]; tensor var_4663_perm_0 = const()[name = string("op_4663_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_143x = const()[name = string("concat_143x"), val = tensor([1, 2048, 1, -1])]; tensor var_4663_cast_fp16 = transpose(perm = var_4663_perm_0, x = attn_output_91_cast_fp16)[name = string("transpose_392")]; tensor attn_output_95_cast_fp16 = reshape(shape = concat_143x, x = var_4663_cast_fp16)[name = string("attn_output_95_cast_fp16")]; tensor hidden_states_113_strides_0 = const()[name = string("hidden_states_113_strides_0"), val = tensor([1, 1])]; string hidden_states_113_pad_type_0 = const()[name = string("hidden_states_113_pad_type_0"), val = string("valid")]; tensor hidden_states_113_pad_0 = const()[name = string("hidden_states_113_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_113_dilations_0 = const()[name = string("hidden_states_113_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_113_groups_0 = const()[name = string("hidden_states_113_groups_0"), val = int32(1)]; tensor hidden_states_113_cast_fp16 = conv(dilations = hidden_states_113_dilations_0, groups = hidden_states_113_groups_0, pad = hidden_states_113_pad_0, pad_type = hidden_states_113_pad_type_0, strides = hidden_states_113_strides_0, weight = layers_11_self_attn_o_proj_weight_cast_fp16, x = attn_output_95_cast_fp16)[name = string("hidden_states_113_cast_fp16")]; tensor hidden_states_115_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = hidden_states_113_cast_fp16)[name = string("hidden_states_115_cast_fp16")]; fp16 const_118_promoted_to_fp16 = const()[name = string("const_118_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4696_cast_fp16 = mul(x = hidden_states_115_cast_fp16, y = const_118_promoted_to_fp16)[name = string("op_4696_cast_fp16")]; int32 var_4694 = const()[name = string("op_4694"), val = int32(1)]; bool doubled_93_interleave_0 = const()[name = string("doubled_93_interleave_0"), val = bool(false)]; tensor doubled_93_cast_fp16 = concat(axis = var_4694, interleave = doubled_93_interleave_0, values = (hidden_states_115_cast_fp16, var_4696_cast_fp16))[name = string("doubled_93_cast_fp16")]; tensor out_47_axes_0 = const()[name = string("out_47_axes_0"), val = tensor([1])]; tensor out_47_gamma_0_to_fp16 = const()[name = string("out_47_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415366400)))]; fp16 var_4706_to_fp16 = const()[name = string("op_4706_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_47_cast_fp16 = layer_norm(axes = out_47_axes_0, epsilon = var_4706_to_fp16, gamma = out_47_gamma_0_to_fp16, x = doubled_93_cast_fp16)[name = string("out_47_cast_fp16")]; tensor var_4717_split_sizes_0 = const()[name = string("op_4717_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4717_axis_0 = const()[name = string("op_4717_axis_0"), val = int32(1)]; tensor var_4717_cast_fp16_0, tensor var_4717_cast_fp16_1 = split(axis = var_4717_axis_0, split_sizes = var_4717_split_sizes_0, x = out_47_cast_fp16)[name = string("op_4717_cast_fp16")]; tensor input_23_strides_0 = const()[name = string("input_23_strides_0"), val = tensor([1, 1])]; string input_23_pad_type_0 = const()[name = string("input_23_pad_type_0"), val = string("valid")]; tensor input_23_pad_0 = const()[name = string("input_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_23_dilations_0 = const()[name = string("input_23_dilations_0"), val = tensor([1, 1])]; int32 input_23_groups_0 = const()[name = string("input_23_groups_0"), val = int32(1)]; tensor input_23_cast_fp16 = conv(dilations = input_23_dilations_0, groups = input_23_groups_0, pad = input_23_pad_0, pad_type = input_23_pad_type_0, strides = input_23_strides_0, weight = layers_11_mlp_gate_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("input_23_cast_fp16")]; tensor var_4734_cast_fp16 = silu(x = input_23_cast_fp16)[name = string("op_4734_cast_fp16")]; tensor var_4740_strides_0 = const()[name = string("op_4740_strides_0"), val = tensor([1, 1])]; string var_4740_pad_type_0 = const()[name = string("op_4740_pad_type_0"), val = string("valid")]; tensor var_4740_pad_0 = const()[name = string("op_4740_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4740_dilations_0 = const()[name = string("op_4740_dilations_0"), val = tensor([1, 1])]; int32 var_4740_groups_0 = const()[name = string("op_4740_groups_0"), val = int32(1)]; tensor var_4740_cast_fp16 = conv(dilations = var_4740_dilations_0, groups = var_4740_groups_0, pad = var_4740_pad_0, pad_type = var_4740_pad_type_0, strides = var_4740_strides_0, weight = layers_11_mlp_up_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("op_4740_cast_fp16")]; tensor x_119_cast_fp16 = mul(x = var_4734_cast_fp16, y = var_4740_cast_fp16)[name = string("x_119_cast_fp16")]; tensor hidden_states_117_strides_0 = const()[name = string("hidden_states_117_strides_0"), val = tensor([1, 1])]; string hidden_states_117_pad_type_0 = const()[name = string("hidden_states_117_pad_type_0"), val = string("valid")]; tensor hidden_states_117_pad_0 = const()[name = string("hidden_states_117_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_117_dilations_0 = const()[name = string("hidden_states_117_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_117_groups_0 = const()[name = string("hidden_states_117_groups_0"), val = int32(1)]; tensor hidden_states_117_cast_fp16 = conv(dilations = hidden_states_117_dilations_0, groups = hidden_states_117_groups_0, pad = hidden_states_117_pad_0, pad_type = hidden_states_117_pad_type_0, strides = hidden_states_117_strides_0, weight = layers_11_mlp_down_proj_weight_cast_fp16, x = x_119_cast_fp16)[name = string("hidden_states_117_cast_fp16")]; tensor hidden_states_119_cast_fp16 = add(x = hidden_states_115_cast_fp16, y = hidden_states_117_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; fp16 const_120_promoted_to_fp16 = const()[name = string("const_120_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4758_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_120_promoted_to_fp16)[name = string("op_4758_cast_fp16")]; int32 var_4756 = const()[name = string("op_4756"), val = int32(1)]; bool doubled_97_interleave_0 = const()[name = string("doubled_97_interleave_0"), val = bool(false)]; tensor doubled_97_cast_fp16 = concat(axis = var_4756, interleave = doubled_97_interleave_0, values = (hidden_states_119_cast_fp16, var_4758_cast_fp16))[name = string("doubled_97_cast_fp16")]; tensor out_49_axes_0 = const()[name = string("out_49_axes_0"), val = tensor([1])]; tensor out_49_gamma_0_to_fp16 = const()[name = string("out_49_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415374656)))]; fp16 var_4768_to_fp16 = const()[name = string("op_4768_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_49_cast_fp16 = layer_norm(axes = out_49_axes_0, epsilon = var_4768_to_fp16, gamma = out_49_gamma_0_to_fp16, x = doubled_97_cast_fp16)[name = string("out_49_cast_fp16")]; tensor var_4779_split_sizes_0 = const()[name = string("op_4779_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4779_axis_0 = const()[name = string("op_4779_axis_0"), val = int32(1)]; tensor var_4779_cast_fp16_0, tensor var_4779_cast_fp16_1 = split(axis = var_4779_axis_0, split_sizes = var_4779_split_sizes_0, x = out_49_cast_fp16)[name = string("op_4779_cast_fp16")]; tensor query_states_73_strides_0 = const()[name = string("query_states_73_strides_0"), val = tensor([1, 1])]; string query_states_73_pad_type_0 = const()[name = string("query_states_73_pad_type_0"), val = string("valid")]; tensor query_states_73_pad_0 = const()[name = string("query_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_73_dilations_0 = const()[name = string("query_states_73_dilations_0"), val = tensor([1, 1])]; int32 query_states_73_groups_0 = const()[name = string("query_states_73_groups_0"), val = int32(1)]; tensor query_states_73_cast_fp16 = conv(dilations = query_states_73_dilations_0, groups = query_states_73_groups_0, pad = query_states_73_pad_0, pad_type = query_states_73_pad_type_0, strides = query_states_73_strides_0, weight = layers_12_self_attn_q_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("query_states_73_cast_fp16")]; tensor key_states_121_strides_0 = const()[name = string("key_states_121_strides_0"), val = tensor([1, 1])]; string key_states_121_pad_type_0 = const()[name = string("key_states_121_pad_type_0"), val = string("valid")]; tensor key_states_121_pad_0 = const()[name = string("key_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_121_dilations_0 = const()[name = string("key_states_121_dilations_0"), val = tensor([1, 1])]; int32 key_states_121_groups_0 = const()[name = string("key_states_121_groups_0"), val = int32(1)]; tensor key_states_121_cast_fp16 = conv(dilations = key_states_121_dilations_0, groups = key_states_121_groups_0, pad = key_states_121_pad_0, pad_type = key_states_121_pad_type_0, strides = key_states_121_strides_0, weight = layers_12_self_attn_k_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("key_states_121_cast_fp16")]; tensor value_states_73_strides_0 = const()[name = string("value_states_73_strides_0"), val = tensor([1, 1])]; string value_states_73_pad_type_0 = const()[name = string("value_states_73_pad_type_0"), val = string("valid")]; tensor value_states_73_pad_0 = const()[name = string("value_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_73_dilations_0 = const()[name = string("value_states_73_dilations_0"), val = tensor([1, 1])]; int32 value_states_73_groups_0 = const()[name = string("value_states_73_groups_0"), val = int32(1)]; tensor value_states_73_cast_fp16 = conv(dilations = value_states_73_dilations_0, groups = value_states_73_groups_0, pad = value_states_73_pad_0, pad_type = value_states_73_pad_type_0, strides = value_states_73_strides_0, weight = layers_12_self_attn_v_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("value_states_73_cast_fp16")]; tensor concat_144x = const()[name = string("concat_144x"), val = tensor([1, 16, 128, -1])]; tensor x_121_cast_fp16 = reshape(shape = concat_144x, x = query_states_73_cast_fp16)[name = string("x_121_cast_fp16")]; tensor concat_145x = const()[name = string("concat_145x"), val = tensor([1, 2, 128, -1])]; tensor var_4836_cast_fp16 = reshape(shape = concat_145x, x = key_states_121_cast_fp16)[name = string("op_4836_cast_fp16")]; tensor concat_146x = const()[name = string("concat_146x"), val = tensor([1, 2, 128, -1])]; tensor var_4843_cast_fp16 = reshape(shape = concat_146x, x = value_states_73_cast_fp16)[name = string("op_4843_cast_fp16")]; tensor var_4847_cast_fp16 = mul(x = x_121_cast_fp16, y = var_869_cast_fp16)[name = string("op_4847_cast_fp16")]; tensor var_4848_split_sizes_0 = const()[name = string("op_4848_split_sizes_0"), val = tensor([64, 64])]; int32 var_4848_axis_0 = const()[name = string("op_4848_axis_0"), val = int32(-2)]; tensor var_4848_cast_fp16_0, tensor var_4848_cast_fp16_1 = split(axis = var_4848_axis_0, split_sizes = var_4848_split_sizes_0, x = x_121_cast_fp16)[name = string("op_4848_cast_fp16")]; fp16 const_122_promoted_to_fp16 = const()[name = string("const_122_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4850_cast_fp16 = mul(x = var_4848_cast_fp16_1, y = const_122_promoted_to_fp16)[name = string("op_4850_cast_fp16")]; int32 var_4852 = const()[name = string("op_4852"), val = int32(-2)]; bool var_4853_interleave_0 = const()[name = string("op_4853_interleave_0"), val = bool(false)]; tensor var_4853_cast_fp16 = concat(axis = var_4852, interleave = var_4853_interleave_0, values = (var_4850_cast_fp16, var_4848_cast_fp16_0))[name = string("op_4853_cast_fp16")]; tensor var_4854_cast_fp16 = mul(x = var_4853_cast_fp16, y = var_878_cast_fp16)[name = string("op_4854_cast_fp16")]; tensor query_states_75_cast_fp16 = add(x = var_4847_cast_fp16, y = var_4854_cast_fp16)[name = string("query_states_75_cast_fp16")]; tensor var_4860_cast_fp16 = mul(x = var_4836_cast_fp16, y = var_869_cast_fp16)[name = string("op_4860_cast_fp16")]; tensor var_4861_split_sizes_0 = const()[name = string("op_4861_split_sizes_0"), val = tensor([64, 64])]; int32 var_4861_axis_0 = const()[name = string("op_4861_axis_0"), val = int32(-2)]; tensor var_4861_cast_fp16_0, tensor var_4861_cast_fp16_1 = split(axis = var_4861_axis_0, split_sizes = var_4861_split_sizes_0, x = var_4836_cast_fp16)[name = string("op_4861_cast_fp16")]; fp16 const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4863_cast_fp16 = mul(x = var_4861_cast_fp16_1, y = const_123_promoted_to_fp16)[name = string("op_4863_cast_fp16")]; int32 var_4865 = const()[name = string("op_4865"), val = int32(-2)]; bool var_4866_interleave_0 = const()[name = string("op_4866_interleave_0"), val = bool(false)]; tensor var_4866_cast_fp16 = concat(axis = var_4865, interleave = var_4866_interleave_0, values = (var_4863_cast_fp16, var_4861_cast_fp16_0))[name = string("op_4866_cast_fp16")]; tensor var_4867_cast_fp16 = mul(x = var_4866_cast_fp16, y = var_878_cast_fp16)[name = string("op_4867_cast_fp16")]; tensor key_states_125_cast_fp16 = add(x = var_4860_cast_fp16, y = var_4867_cast_fp16)[name = string("key_states_125_cast_fp16")]; tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; int32 concat_149_axis_0 = const()[name = string("concat_149_axis_0"), val = int32(0)]; bool concat_149_interleave_0 = const()[name = string("concat_149_interleave_0"), val = bool(false)]; tensor concat_149 = concat(axis = concat_149_axis_0, interleave = concat_149_interleave_0, values = (expand_dims_144, expand_dims_145, position_id, expand_dims_147))[name = string("concat_149")]; tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; tensor concat_150_values1_0 = const()[name = string("concat_150_values1_0"), val = tensor([0])]; tensor concat_150_values3_0 = const()[name = string("concat_150_values3_0"), val = tensor([0])]; int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_148, concat_150_values1_0, cache_position_end, concat_150_values3_0))[name = string("concat_150")]; tensor key_states_127_perm_0 = const()[name = string("key_states_127_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_13_stride_0 = const()[name = string("key_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_127_cast_fp16 = transpose(perm = key_states_127_perm_0, x = key_states_125_cast_fp16)[name = string("transpose_391")]; tensor key_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = key_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = key_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_13_squeeze_mask_0, stride = key_cache_internal_tensor_assign_13_stride_0, update = key_states_127_cast_fp16, x = coreml_update_state_246)[name = string("key_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_13_cast_fp16, input = key_cache)[name = string("coreml_update_state_248_write_state")]; tensor coreml_update_state_248 = read_state(input = key_cache)[name = string("coreml_update_state_248")]; tensor value_states_75_perm_0 = const()[name = string("value_states_75_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_13_stride_0 = const()[name = string("value_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_75_cast_fp16 = transpose(perm = value_states_75_perm_0, x = var_4843_cast_fp16)[name = string("transpose_390")]; tensor value_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = value_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = value_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_13_squeeze_mask_0, stride = value_cache_internal_tensor_assign_13_stride_0, update = value_states_75_cast_fp16, x = coreml_update_state_247)[name = string("value_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_13_cast_fp16, input = value_cache)[name = string("coreml_update_state_249_write_state")]; tensor coreml_update_state_249 = read_state(input = value_cache)[name = string("coreml_update_state_249")]; tensor var_4937_begin_0 = const()[name = string("op_4937_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4937_end_0 = const()[name = string("op_4937_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4937_end_mask_0 = const()[name = string("op_4937_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4937_cast_fp16 = slice_by_index(begin = var_4937_begin_0, end = var_4937_end_0, end_mask = var_4937_end_mask_0, x = coreml_update_state_248)[name = string("op_4937_cast_fp16")]; tensor tile_24 = const()[name = string("tile_24"), val = tensor([1, 1])]; int32 var_4940_axis_0 = const()[name = string("op_4940_axis_0"), val = int32(1)]; tensor var_4940_cast_fp16_0, tensor var_4940_cast_fp16_1 = split(axis = var_4940_axis_0, split_sizes = tile_24, x = var_4937_cast_fp16)[name = string("op_4940_cast_fp16")]; tensor var_4947_begin_0 = const()[name = string("op_4947_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4947_end_0 = const()[name = string("op_4947_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4947_end_mask_0 = const()[name = string("op_4947_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4947_cast_fp16 = slice_by_index(begin = var_4947_begin_0, end = var_4947_end_0, end_mask = var_4947_end_mask_0, x = coreml_update_state_249)[name = string("op_4947_cast_fp16")]; tensor tile_25 = const()[name = string("tile_25"), val = tensor([1, 1])]; int32 var_4950_axis_0 = const()[name = string("op_4950_axis_0"), val = int32(1)]; tensor var_4950_cast_fp16_0, tensor var_4950_cast_fp16_1 = split(axis = var_4950_axis_0, split_sizes = tile_25, x = var_4947_cast_fp16)[name = string("op_4950_cast_fp16")]; tensor var_4953_split_sizes_0 = const()[name = string("op_4953_split_sizes_0"), val = tensor([8, 8])]; int32 var_4953_axis_0 = const()[name = string("op_4953_axis_0"), val = int32(1)]; tensor var_4953_0, tensor var_4953_1 = split(axis = var_4953_axis_0, split_sizes = var_4953_split_sizes_0, x = query_states_75_cast_fp16)[name = string("op_4953")]; bool attn_weights_193_transpose_x_0 = const()[name = string("attn_weights_193_transpose_x_0"), val = bool(false)]; bool attn_weights_193_transpose_y_0 = const()[name = string("attn_weights_193_transpose_y_0"), val = bool(false)]; tensor attn_weights_193_cast_fp16 = matmul(transpose_x = attn_weights_193_transpose_x_0, transpose_y = attn_weights_193_transpose_y_0, x = var_4940_cast_fp16_0, y = var_4953_0)[name = string("attn_weights_193_cast_fp16")]; fp16 var_4956_to_fp16 = const()[name = string("op_4956_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_195_cast_fp16 = mul(x = attn_weights_193_cast_fp16, y = var_4956_to_fp16)[name = string("attn_weights_195_cast_fp16")]; tensor attn_weights_197_cast_fp16 = add(x = attn_weights_195_cast_fp16, y = attn_mask_1)[name = string("attn_weights_197_cast_fp16")]; int32 var_4960 = const()[name = string("op_4960"), val = int32(-2)]; tensor attn_weights_199_cast_fp16 = softmax(axis = var_4960, x = attn_weights_197_cast_fp16)[name = string("attn_weights_199_cast_fp16")]; bool var_4966_transpose_x_1 = const()[name = string("op_4966_transpose_x_1"), val = bool(true)]; bool var_4966_transpose_y_1 = const()[name = string("op_4966_transpose_y_1"), val = bool(false)]; tensor var_4966_cast_fp16 = matmul(transpose_x = var_4966_transpose_x_1, transpose_y = var_4966_transpose_y_1, x = attn_weights_199_cast_fp16, y = var_4950_cast_fp16_0)[name = string("op_4966_cast_fp16")]; bool attn_weights_201_transpose_x_0 = const()[name = string("attn_weights_201_transpose_x_0"), val = bool(false)]; bool attn_weights_201_transpose_y_0 = const()[name = string("attn_weights_201_transpose_y_0"), val = bool(false)]; tensor attn_weights_201_cast_fp16 = matmul(transpose_x = attn_weights_201_transpose_x_0, transpose_y = attn_weights_201_transpose_y_0, x = var_4940_cast_fp16_1, y = var_4953_1)[name = string("attn_weights_201_cast_fp16")]; fp16 var_4968_to_fp16 = const()[name = string("op_4968_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_203_cast_fp16 = mul(x = attn_weights_201_cast_fp16, y = var_4968_to_fp16)[name = string("attn_weights_203_cast_fp16")]; tensor attn_weights_205_cast_fp16 = add(x = attn_weights_203_cast_fp16, y = attn_mask_1)[name = string("attn_weights_205_cast_fp16")]; int32 var_4972 = const()[name = string("op_4972"), val = int32(-2)]; tensor attn_weights_207_cast_fp16 = softmax(axis = var_4972, x = attn_weights_205_cast_fp16)[name = string("attn_weights_207_cast_fp16")]; bool attn_output_97_transpose_x_1 = const()[name = string("attn_output_97_transpose_x_1"), val = bool(true)]; bool attn_output_97_transpose_y_1 = const()[name = string("attn_output_97_transpose_y_1"), val = bool(false)]; tensor attn_output_97_cast_fp16 = matmul(transpose_x = attn_output_97_transpose_x_1, transpose_y = attn_output_97_transpose_y_1, x = attn_weights_207_cast_fp16, y = var_4950_cast_fp16_1)[name = string("attn_output_97_cast_fp16")]; int32 var_4980 = const()[name = string("op_4980"), val = int32(1)]; bool attn_output_99_interleave_0 = const()[name = string("attn_output_99_interleave_0"), val = bool(false)]; tensor attn_output_99_cast_fp16 = concat(axis = var_4980, interleave = attn_output_99_interleave_0, values = (var_4966_cast_fp16, attn_output_97_cast_fp16))[name = string("attn_output_99_cast_fp16")]; tensor var_4984_perm_0 = const()[name = string("op_4984_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_155x = const()[name = string("concat_155x"), val = tensor([1, 2048, 1, -1])]; tensor var_4984_cast_fp16 = transpose(perm = var_4984_perm_0, x = attn_output_99_cast_fp16)[name = string("transpose_389")]; tensor attn_output_103_cast_fp16 = reshape(shape = concat_155x, x = var_4984_cast_fp16)[name = string("attn_output_103_cast_fp16")]; tensor hidden_states_123_strides_0 = const()[name = string("hidden_states_123_strides_0"), val = tensor([1, 1])]; string hidden_states_123_pad_type_0 = const()[name = string("hidden_states_123_pad_type_0"), val = string("valid")]; tensor hidden_states_123_pad_0 = const()[name = string("hidden_states_123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_123_dilations_0 = const()[name = string("hidden_states_123_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_123_groups_0 = const()[name = string("hidden_states_123_groups_0"), val = int32(1)]; tensor hidden_states_123_cast_fp16 = conv(dilations = hidden_states_123_dilations_0, groups = hidden_states_123_groups_0, pad = hidden_states_123_pad_0, pad_type = hidden_states_123_pad_type_0, strides = hidden_states_123_strides_0, weight = layers_12_self_attn_o_proj_weight_cast_fp16, x = attn_output_103_cast_fp16)[name = string("hidden_states_123_cast_fp16")]; tensor hidden_states_125_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = hidden_states_123_cast_fp16)[name = string("hidden_states_125_cast_fp16")]; fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5017_cast_fp16 = mul(x = hidden_states_125_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_5017_cast_fp16")]; int32 var_5015 = const()[name = string("op_5015"), val = int32(1)]; bool doubled_101_interleave_0 = const()[name = string("doubled_101_interleave_0"), val = bool(false)]; tensor doubled_101_cast_fp16 = concat(axis = var_5015, interleave = doubled_101_interleave_0, values = (hidden_states_125_cast_fp16, var_5017_cast_fp16))[name = string("doubled_101_cast_fp16")]; tensor out_51_axes_0 = const()[name = string("out_51_axes_0"), val = tensor([1])]; tensor out_51_gamma_0_to_fp16 = const()[name = string("out_51_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415382912)))]; fp16 var_5027_to_fp16 = const()[name = string("op_5027_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_51_cast_fp16 = layer_norm(axes = out_51_axes_0, epsilon = var_5027_to_fp16, gamma = out_51_gamma_0_to_fp16, x = doubled_101_cast_fp16)[name = string("out_51_cast_fp16")]; tensor var_5038_split_sizes_0 = const()[name = string("op_5038_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5038_axis_0 = const()[name = string("op_5038_axis_0"), val = int32(1)]; tensor var_5038_cast_fp16_0, tensor var_5038_cast_fp16_1 = split(axis = var_5038_axis_0, split_sizes = var_5038_split_sizes_0, x = out_51_cast_fp16)[name = string("op_5038_cast_fp16")]; tensor input_25_strides_0 = const()[name = string("input_25_strides_0"), val = tensor([1, 1])]; string input_25_pad_type_0 = const()[name = string("input_25_pad_type_0"), val = string("valid")]; tensor input_25_pad_0 = const()[name = string("input_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_25_dilations_0 = const()[name = string("input_25_dilations_0"), val = tensor([1, 1])]; int32 input_25_groups_0 = const()[name = string("input_25_groups_0"), val = int32(1)]; tensor input_25_cast_fp16 = conv(dilations = input_25_dilations_0, groups = input_25_groups_0, pad = input_25_pad_0, pad_type = input_25_pad_type_0, strides = input_25_strides_0, weight = layers_12_mlp_gate_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("input_25_cast_fp16")]; tensor var_5055_cast_fp16 = silu(x = input_25_cast_fp16)[name = string("op_5055_cast_fp16")]; tensor var_5061_strides_0 = const()[name = string("op_5061_strides_0"), val = tensor([1, 1])]; string var_5061_pad_type_0 = const()[name = string("op_5061_pad_type_0"), val = string("valid")]; tensor var_5061_pad_0 = const()[name = string("op_5061_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5061_dilations_0 = const()[name = string("op_5061_dilations_0"), val = tensor([1, 1])]; int32 var_5061_groups_0 = const()[name = string("op_5061_groups_0"), val = int32(1)]; tensor var_5061_cast_fp16 = conv(dilations = var_5061_dilations_0, groups = var_5061_groups_0, pad = var_5061_pad_0, pad_type = var_5061_pad_type_0, strides = var_5061_strides_0, weight = layers_12_mlp_up_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("op_5061_cast_fp16")]; tensor x_129_cast_fp16 = mul(x = var_5055_cast_fp16, y = var_5061_cast_fp16)[name = string("x_129_cast_fp16")]; tensor hidden_states_127_strides_0 = const()[name = string("hidden_states_127_strides_0"), val = tensor([1, 1])]; string hidden_states_127_pad_type_0 = const()[name = string("hidden_states_127_pad_type_0"), val = string("valid")]; tensor hidden_states_127_pad_0 = const()[name = string("hidden_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_127_dilations_0 = const()[name = string("hidden_states_127_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_127_groups_0 = const()[name = string("hidden_states_127_groups_0"), val = int32(1)]; tensor hidden_states_127_cast_fp16 = conv(dilations = hidden_states_127_dilations_0, groups = hidden_states_127_groups_0, pad = hidden_states_127_pad_0, pad_type = hidden_states_127_pad_type_0, strides = hidden_states_127_strides_0, weight = layers_12_mlp_down_proj_weight_cast_fp16, x = x_129_cast_fp16)[name = string("hidden_states_127_cast_fp16")]; tensor hidden_states_129_cast_fp16 = add(x = hidden_states_125_cast_fp16, y = hidden_states_127_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5079_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_130_promoted_to_fp16)[name = string("op_5079_cast_fp16")]; int32 var_5077 = const()[name = string("op_5077"), val = int32(1)]; bool doubled_105_interleave_0 = const()[name = string("doubled_105_interleave_0"), val = bool(false)]; tensor doubled_105_cast_fp16 = concat(axis = var_5077, interleave = doubled_105_interleave_0, values = (hidden_states_129_cast_fp16, var_5079_cast_fp16))[name = string("doubled_105_cast_fp16")]; tensor out_53_axes_0 = const()[name = string("out_53_axes_0"), val = tensor([1])]; tensor out_53_gamma_0_to_fp16 = const()[name = string("out_53_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415391168)))]; fp16 var_5089_to_fp16 = const()[name = string("op_5089_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_53_cast_fp16 = layer_norm(axes = out_53_axes_0, epsilon = var_5089_to_fp16, gamma = out_53_gamma_0_to_fp16, x = doubled_105_cast_fp16)[name = string("out_53_cast_fp16")]; tensor var_5100_split_sizes_0 = const()[name = string("op_5100_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5100_axis_0 = const()[name = string("op_5100_axis_0"), val = int32(1)]; tensor var_5100_cast_fp16_0, tensor var_5100_cast_fp16_1 = split(axis = var_5100_axis_0, split_sizes = var_5100_split_sizes_0, x = out_53_cast_fp16)[name = string("op_5100_cast_fp16")]; tensor query_states_79_strides_0 = const()[name = string("query_states_79_strides_0"), val = tensor([1, 1])]; string query_states_79_pad_type_0 = const()[name = string("query_states_79_pad_type_0"), val = string("valid")]; tensor query_states_79_pad_0 = const()[name = string("query_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_79_dilations_0 = const()[name = string("query_states_79_dilations_0"), val = tensor([1, 1])]; int32 query_states_79_groups_0 = const()[name = string("query_states_79_groups_0"), val = int32(1)]; tensor query_states_79_cast_fp16 = conv(dilations = query_states_79_dilations_0, groups = query_states_79_groups_0, pad = query_states_79_pad_0, pad_type = query_states_79_pad_type_0, strides = query_states_79_strides_0, weight = layers_13_self_attn_q_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("query_states_79_cast_fp16")]; tensor key_states_131_strides_0 = const()[name = string("key_states_131_strides_0"), val = tensor([1, 1])]; string key_states_131_pad_type_0 = const()[name = string("key_states_131_pad_type_0"), val = string("valid")]; tensor key_states_131_pad_0 = const()[name = string("key_states_131_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_131_dilations_0 = const()[name = string("key_states_131_dilations_0"), val = tensor([1, 1])]; int32 key_states_131_groups_0 = const()[name = string("key_states_131_groups_0"), val = int32(1)]; tensor key_states_131_cast_fp16 = conv(dilations = key_states_131_dilations_0, groups = key_states_131_groups_0, pad = key_states_131_pad_0, pad_type = key_states_131_pad_type_0, strides = key_states_131_strides_0, weight = layers_13_self_attn_k_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("key_states_131_cast_fp16")]; tensor value_states_79_strides_0 = const()[name = string("value_states_79_strides_0"), val = tensor([1, 1])]; string value_states_79_pad_type_0 = const()[name = string("value_states_79_pad_type_0"), val = string("valid")]; tensor value_states_79_pad_0 = const()[name = string("value_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_79_dilations_0 = const()[name = string("value_states_79_dilations_0"), val = tensor([1, 1])]; int32 value_states_79_groups_0 = const()[name = string("value_states_79_groups_0"), val = int32(1)]; tensor value_states_79_cast_fp16 = conv(dilations = value_states_79_dilations_0, groups = value_states_79_groups_0, pad = value_states_79_pad_0, pad_type = value_states_79_pad_type_0, strides = value_states_79_strides_0, weight = layers_13_self_attn_v_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("value_states_79_cast_fp16")]; tensor concat_156x = const()[name = string("concat_156x"), val = tensor([1, 16, 128, -1])]; tensor x_131_cast_fp16 = reshape(shape = concat_156x, x = query_states_79_cast_fp16)[name = string("x_131_cast_fp16")]; tensor concat_157x = const()[name = string("concat_157x"), val = tensor([1, 2, 128, -1])]; tensor var_5157_cast_fp16 = reshape(shape = concat_157x, x = key_states_131_cast_fp16)[name = string("op_5157_cast_fp16")]; tensor concat_158x = const()[name = string("concat_158x"), val = tensor([1, 2, 128, -1])]; tensor var_5164_cast_fp16 = reshape(shape = concat_158x, x = value_states_79_cast_fp16)[name = string("op_5164_cast_fp16")]; tensor var_5168_cast_fp16 = mul(x = x_131_cast_fp16, y = var_869_cast_fp16)[name = string("op_5168_cast_fp16")]; tensor var_5169_split_sizes_0 = const()[name = string("op_5169_split_sizes_0"), val = tensor([64, 64])]; int32 var_5169_axis_0 = const()[name = string("op_5169_axis_0"), val = int32(-2)]; tensor var_5169_cast_fp16_0, tensor var_5169_cast_fp16_1 = split(axis = var_5169_axis_0, split_sizes = var_5169_split_sizes_0, x = x_131_cast_fp16)[name = string("op_5169_cast_fp16")]; fp16 const_132_promoted_to_fp16 = const()[name = string("const_132_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5171_cast_fp16 = mul(x = var_5169_cast_fp16_1, y = const_132_promoted_to_fp16)[name = string("op_5171_cast_fp16")]; int32 var_5173 = const()[name = string("op_5173"), val = int32(-2)]; bool var_5174_interleave_0 = const()[name = string("op_5174_interleave_0"), val = bool(false)]; tensor var_5174_cast_fp16 = concat(axis = var_5173, interleave = var_5174_interleave_0, values = (var_5171_cast_fp16, var_5169_cast_fp16_0))[name = string("op_5174_cast_fp16")]; tensor var_5175_cast_fp16 = mul(x = var_5174_cast_fp16, y = var_878_cast_fp16)[name = string("op_5175_cast_fp16")]; tensor query_states_81_cast_fp16 = add(x = var_5168_cast_fp16, y = var_5175_cast_fp16)[name = string("query_states_81_cast_fp16")]; tensor var_5181_cast_fp16 = mul(x = var_5157_cast_fp16, y = var_869_cast_fp16)[name = string("op_5181_cast_fp16")]; tensor var_5182_split_sizes_0 = const()[name = string("op_5182_split_sizes_0"), val = tensor([64, 64])]; int32 var_5182_axis_0 = const()[name = string("op_5182_axis_0"), val = int32(-2)]; tensor var_5182_cast_fp16_0, tensor var_5182_cast_fp16_1 = split(axis = var_5182_axis_0, split_sizes = var_5182_split_sizes_0, x = var_5157_cast_fp16)[name = string("op_5182_cast_fp16")]; fp16 const_133_promoted_to_fp16 = const()[name = string("const_133_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5184_cast_fp16 = mul(x = var_5182_cast_fp16_1, y = const_133_promoted_to_fp16)[name = string("op_5184_cast_fp16")]; int32 var_5186 = const()[name = string("op_5186"), val = int32(-2)]; bool var_5187_interleave_0 = const()[name = string("op_5187_interleave_0"), val = bool(false)]; tensor var_5187_cast_fp16 = concat(axis = var_5186, interleave = var_5187_interleave_0, values = (var_5184_cast_fp16, var_5182_cast_fp16_0))[name = string("op_5187_cast_fp16")]; tensor var_5188_cast_fp16 = mul(x = var_5187_cast_fp16, y = var_878_cast_fp16)[name = string("op_5188_cast_fp16")]; tensor key_states_135_cast_fp16 = add(x = var_5181_cast_fp16, y = var_5188_cast_fp16)[name = string("key_states_135_cast_fp16")]; tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; int32 concat_161_axis_0 = const()[name = string("concat_161_axis_0"), val = int32(0)]; bool concat_161_interleave_0 = const()[name = string("concat_161_interleave_0"), val = bool(false)]; tensor concat_161 = concat(axis = concat_161_axis_0, interleave = concat_161_interleave_0, values = (expand_dims_156, expand_dims_157, position_id, expand_dims_159))[name = string("concat_161")]; tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; tensor concat_162_values1_0 = const()[name = string("concat_162_values1_0"), val = tensor([0])]; tensor concat_162_values3_0 = const()[name = string("concat_162_values3_0"), val = tensor([0])]; int32 concat_162_axis_0 = const()[name = string("concat_162_axis_0"), val = int32(0)]; bool concat_162_interleave_0 = const()[name = string("concat_162_interleave_0"), val = bool(false)]; tensor concat_162 = concat(axis = concat_162_axis_0, interleave = concat_162_interleave_0, values = (expand_dims_160, concat_162_values1_0, cache_position_end, concat_162_values3_0))[name = string("concat_162")]; tensor key_states_137_perm_0 = const()[name = string("key_states_137_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_14_stride_0 = const()[name = string("key_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_137_cast_fp16 = transpose(perm = key_states_137_perm_0, x = key_states_135_cast_fp16)[name = string("transpose_388")]; tensor key_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = key_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = key_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_14_squeeze_mask_0, stride = key_cache_internal_tensor_assign_14_stride_0, update = key_states_137_cast_fp16, x = coreml_update_state_248)[name = string("key_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_14_cast_fp16, input = key_cache)[name = string("coreml_update_state_250_write_state")]; tensor coreml_update_state_250 = read_state(input = key_cache)[name = string("coreml_update_state_250")]; tensor value_states_81_perm_0 = const()[name = string("value_states_81_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_14_stride_0 = const()[name = string("value_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_81_cast_fp16 = transpose(perm = value_states_81_perm_0, x = var_5164_cast_fp16)[name = string("transpose_387")]; tensor value_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = value_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = value_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_14_squeeze_mask_0, stride = value_cache_internal_tensor_assign_14_stride_0, update = value_states_81_cast_fp16, x = coreml_update_state_249)[name = string("value_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_14_cast_fp16, input = value_cache)[name = string("coreml_update_state_251_write_state")]; tensor coreml_update_state_251 = read_state(input = value_cache)[name = string("coreml_update_state_251")]; tensor var_5258_begin_0 = const()[name = string("op_5258_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5258_end_0 = const()[name = string("op_5258_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5258_end_mask_0 = const()[name = string("op_5258_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5258_cast_fp16 = slice_by_index(begin = var_5258_begin_0, end = var_5258_end_0, end_mask = var_5258_end_mask_0, x = coreml_update_state_250)[name = string("op_5258_cast_fp16")]; tensor tile_26 = const()[name = string("tile_26"), val = tensor([1, 1])]; int32 var_5261_axis_0 = const()[name = string("op_5261_axis_0"), val = int32(1)]; tensor var_5261_cast_fp16_0, tensor var_5261_cast_fp16_1 = split(axis = var_5261_axis_0, split_sizes = tile_26, x = var_5258_cast_fp16)[name = string("op_5261_cast_fp16")]; tensor var_5268_begin_0 = const()[name = string("op_5268_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5268_end_0 = const()[name = string("op_5268_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5268_end_mask_0 = const()[name = string("op_5268_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5268_cast_fp16 = slice_by_index(begin = var_5268_begin_0, end = var_5268_end_0, end_mask = var_5268_end_mask_0, x = coreml_update_state_251)[name = string("op_5268_cast_fp16")]; tensor tile_27 = const()[name = string("tile_27"), val = tensor([1, 1])]; int32 var_5271_axis_0 = const()[name = string("op_5271_axis_0"), val = int32(1)]; tensor var_5271_cast_fp16_0, tensor var_5271_cast_fp16_1 = split(axis = var_5271_axis_0, split_sizes = tile_27, x = var_5268_cast_fp16)[name = string("op_5271_cast_fp16")]; tensor var_5274_split_sizes_0 = const()[name = string("op_5274_split_sizes_0"), val = tensor([8, 8])]; int32 var_5274_axis_0 = const()[name = string("op_5274_axis_0"), val = int32(1)]; tensor var_5274_0, tensor var_5274_1 = split(axis = var_5274_axis_0, split_sizes = var_5274_split_sizes_0, x = query_states_81_cast_fp16)[name = string("op_5274")]; bool attn_weights_209_transpose_x_0 = const()[name = string("attn_weights_209_transpose_x_0"), val = bool(false)]; bool attn_weights_209_transpose_y_0 = const()[name = string("attn_weights_209_transpose_y_0"), val = bool(false)]; tensor attn_weights_209_cast_fp16 = matmul(transpose_x = attn_weights_209_transpose_x_0, transpose_y = attn_weights_209_transpose_y_0, x = var_5261_cast_fp16_0, y = var_5274_0)[name = string("attn_weights_209_cast_fp16")]; fp16 var_5277_to_fp16 = const()[name = string("op_5277_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_211_cast_fp16 = mul(x = attn_weights_209_cast_fp16, y = var_5277_to_fp16)[name = string("attn_weights_211_cast_fp16")]; tensor attn_weights_213_cast_fp16 = add(x = attn_weights_211_cast_fp16, y = attn_mask_1)[name = string("attn_weights_213_cast_fp16")]; int32 var_5281 = const()[name = string("op_5281"), val = int32(-2)]; tensor attn_weights_215_cast_fp16 = softmax(axis = var_5281, x = attn_weights_213_cast_fp16)[name = string("attn_weights_215_cast_fp16")]; bool var_5287_transpose_x_1 = const()[name = string("op_5287_transpose_x_1"), val = bool(true)]; bool var_5287_transpose_y_1 = const()[name = string("op_5287_transpose_y_1"), val = bool(false)]; tensor var_5287_cast_fp16 = matmul(transpose_x = var_5287_transpose_x_1, transpose_y = var_5287_transpose_y_1, x = attn_weights_215_cast_fp16, y = var_5271_cast_fp16_0)[name = string("op_5287_cast_fp16")]; bool attn_weights_217_transpose_x_0 = const()[name = string("attn_weights_217_transpose_x_0"), val = bool(false)]; bool attn_weights_217_transpose_y_0 = const()[name = string("attn_weights_217_transpose_y_0"), val = bool(false)]; tensor attn_weights_217_cast_fp16 = matmul(transpose_x = attn_weights_217_transpose_x_0, transpose_y = attn_weights_217_transpose_y_0, x = var_5261_cast_fp16_1, y = var_5274_1)[name = string("attn_weights_217_cast_fp16")]; fp16 var_5289_to_fp16 = const()[name = string("op_5289_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_219_cast_fp16 = mul(x = attn_weights_217_cast_fp16, y = var_5289_to_fp16)[name = string("attn_weights_219_cast_fp16")]; tensor attn_weights_221_cast_fp16 = add(x = attn_weights_219_cast_fp16, y = attn_mask_1)[name = string("attn_weights_221_cast_fp16")]; int32 var_5293 = const()[name = string("op_5293"), val = int32(-2)]; tensor attn_weights_223_cast_fp16 = softmax(axis = var_5293, x = attn_weights_221_cast_fp16)[name = string("attn_weights_223_cast_fp16")]; bool attn_output_105_transpose_x_1 = const()[name = string("attn_output_105_transpose_x_1"), val = bool(true)]; bool attn_output_105_transpose_y_1 = const()[name = string("attn_output_105_transpose_y_1"), val = bool(false)]; tensor attn_output_105_cast_fp16 = matmul(transpose_x = attn_output_105_transpose_x_1, transpose_y = attn_output_105_transpose_y_1, x = attn_weights_223_cast_fp16, y = var_5271_cast_fp16_1)[name = string("attn_output_105_cast_fp16")]; int32 var_5301 = const()[name = string("op_5301"), val = int32(1)]; bool attn_output_107_interleave_0 = const()[name = string("attn_output_107_interleave_0"), val = bool(false)]; tensor attn_output_107_cast_fp16 = concat(axis = var_5301, interleave = attn_output_107_interleave_0, values = (var_5287_cast_fp16, attn_output_105_cast_fp16))[name = string("attn_output_107_cast_fp16")]; tensor var_5305_perm_0 = const()[name = string("op_5305_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_167x = const()[name = string("concat_167x"), val = tensor([1, 2048, 1, -1])]; tensor var_5305_cast_fp16 = transpose(perm = var_5305_perm_0, x = attn_output_107_cast_fp16)[name = string("transpose_386")]; tensor attn_output_111_cast_fp16 = reshape(shape = concat_167x, x = var_5305_cast_fp16)[name = string("attn_output_111_cast_fp16")]; tensor hidden_states_133_strides_0 = const()[name = string("hidden_states_133_strides_0"), val = tensor([1, 1])]; string hidden_states_133_pad_type_0 = const()[name = string("hidden_states_133_pad_type_0"), val = string("valid")]; tensor hidden_states_133_pad_0 = const()[name = string("hidden_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_133_dilations_0 = const()[name = string("hidden_states_133_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_133_groups_0 = const()[name = string("hidden_states_133_groups_0"), val = int32(1)]; tensor hidden_states_133_cast_fp16 = conv(dilations = hidden_states_133_dilations_0, groups = hidden_states_133_groups_0, pad = hidden_states_133_pad_0, pad_type = hidden_states_133_pad_type_0, strides = hidden_states_133_strides_0, weight = layers_13_self_attn_o_proj_weight_cast_fp16, x = attn_output_111_cast_fp16)[name = string("hidden_states_133_cast_fp16")]; tensor hidden_states_135_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = hidden_states_133_cast_fp16)[name = string("hidden_states_135_cast_fp16")]; fp16 const_138_promoted_to_fp16 = const()[name = string("const_138_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5338_cast_fp16 = mul(x = hidden_states_135_cast_fp16, y = const_138_promoted_to_fp16)[name = string("op_5338_cast_fp16")]; int32 var_5336 = const()[name = string("op_5336"), val = int32(1)]; bool doubled_109_interleave_0 = const()[name = string("doubled_109_interleave_0"), val = bool(false)]; tensor doubled_109_cast_fp16 = concat(axis = var_5336, interleave = doubled_109_interleave_0, values = (hidden_states_135_cast_fp16, var_5338_cast_fp16))[name = string("doubled_109_cast_fp16")]; tensor out_55_axes_0 = const()[name = string("out_55_axes_0"), val = tensor([1])]; tensor out_55_gamma_0_to_fp16 = const()[name = string("out_55_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415399424)))]; fp16 var_5348_to_fp16 = const()[name = string("op_5348_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_55_cast_fp16 = layer_norm(axes = out_55_axes_0, epsilon = var_5348_to_fp16, gamma = out_55_gamma_0_to_fp16, x = doubled_109_cast_fp16)[name = string("out_55_cast_fp16")]; tensor var_5359_split_sizes_0 = const()[name = string("op_5359_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5359_axis_0 = const()[name = string("op_5359_axis_0"), val = int32(1)]; tensor var_5359_cast_fp16_0, tensor var_5359_cast_fp16_1 = split(axis = var_5359_axis_0, split_sizes = var_5359_split_sizes_0, x = out_55_cast_fp16)[name = string("op_5359_cast_fp16")]; tensor input_27_strides_0 = const()[name = string("input_27_strides_0"), val = tensor([1, 1])]; string input_27_pad_type_0 = const()[name = string("input_27_pad_type_0"), val = string("valid")]; tensor input_27_pad_0 = const()[name = string("input_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_27_dilations_0 = const()[name = string("input_27_dilations_0"), val = tensor([1, 1])]; int32 input_27_groups_0 = const()[name = string("input_27_groups_0"), val = int32(1)]; tensor input_27_cast_fp16 = conv(dilations = input_27_dilations_0, groups = input_27_groups_0, pad = input_27_pad_0, pad_type = input_27_pad_type_0, strides = input_27_strides_0, weight = layers_13_mlp_gate_proj_weight_cast_fp16, x = var_5359_cast_fp16_0)[name = string("input_27_cast_fp16")]; tensor var_5376_cast_fp16 = silu(x = input_27_cast_fp16)[name = string("op_5376_cast_fp16")]; tensor layers_13_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_13_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415407680)))]; tensor var_5382_strides_0 = const()[name = string("op_5382_strides_0"), val = tensor([1, 1])]; string var_5382_pad_type_0 = const()[name = string("op_5382_pad_type_0"), val = string("valid")]; tensor var_5382_pad_0 = const()[name = string("op_5382_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5382_dilations_0 = const()[name = string("op_5382_dilations_0"), val = tensor([1, 1])]; int32 var_5382_groups_0 = const()[name = string("op_5382_groups_0"), val = int32(1)]; tensor var_5382_cast_fp16 = conv(dilations = var_5382_dilations_0, groups = var_5382_groups_0, pad = var_5382_pad_0, pad_type = var_5382_pad_type_0, strides = var_5382_strides_0, weight = layers_13_mlp_up_proj_weight_to_fp16, x = var_5359_cast_fp16_0)[name = string("op_5382_cast_fp16")]; tensor x_139_cast_fp16 = mul(x = var_5376_cast_fp16, y = var_5382_cast_fp16)[name = string("x_139_cast_fp16")]; tensor hidden_states_137_strides_0 = const()[name = string("hidden_states_137_strides_0"), val = tensor([1, 1])]; string hidden_states_137_pad_type_0 = const()[name = string("hidden_states_137_pad_type_0"), val = string("valid")]; tensor hidden_states_137_pad_0 = const()[name = string("hidden_states_137_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_137_dilations_0 = const()[name = string("hidden_states_137_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_137_groups_0 = const()[name = string("hidden_states_137_groups_0"), val = int32(1)]; tensor hidden_states_137_cast_fp16 = conv(dilations = hidden_states_137_dilations_0, groups = hidden_states_137_groups_0, pad = hidden_states_137_pad_0, pad_type = hidden_states_137_pad_type_0, strides = hidden_states_137_strides_0, weight = layers_13_mlp_down_proj_weight_cast_fp16, x = x_139_cast_fp16)[name = string("hidden_states_137_cast_fp16")]; tensor hidden_states_139_cast_fp16 = add(x = hidden_states_135_cast_fp16, y = hidden_states_137_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; fp16 const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5400_cast_fp16 = mul(x = hidden_states_139_cast_fp16, y = const_140_promoted_to_fp16)[name = string("op_5400_cast_fp16")]; int32 var_5398 = const()[name = string("op_5398"), val = int32(1)]; bool doubled_113_interleave_0 = const()[name = string("doubled_113_interleave_0"), val = bool(false)]; tensor doubled_113_cast_fp16 = concat(axis = var_5398, interleave = doubled_113_interleave_0, values = (hidden_states_139_cast_fp16, var_5400_cast_fp16))[name = string("doubled_113_cast_fp16")]; tensor out_57_axes_0 = const()[name = string("out_57_axes_0"), val = tensor([1])]; tensor out_57_gamma_0_to_fp16 = const()[name = string("out_57_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440573568)))]; fp16 var_5410_to_fp16 = const()[name = string("op_5410_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_57_cast_fp16 = layer_norm(axes = out_57_axes_0, epsilon = var_5410_to_fp16, gamma = out_57_gamma_0_to_fp16, x = doubled_113_cast_fp16)[name = string("out_57_cast_fp16")]; tensor var_5421_split_sizes_0 = const()[name = string("op_5421_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5421_axis_0 = const()[name = string("op_5421_axis_0"), val = int32(1)]; tensor var_5421_cast_fp16_0, tensor var_5421_cast_fp16_1 = split(axis = var_5421_axis_0, split_sizes = var_5421_split_sizes_0, x = out_57_cast_fp16)[name = string("op_5421_cast_fp16")]; tensor query_states_85_strides_0 = const()[name = string("query_states_85_strides_0"), val = tensor([1, 1])]; string query_states_85_pad_type_0 = const()[name = string("query_states_85_pad_type_0"), val = string("valid")]; tensor query_states_85_pad_0 = const()[name = string("query_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_85_dilations_0 = const()[name = string("query_states_85_dilations_0"), val = tensor([1, 1])]; int32 query_states_85_groups_0 = const()[name = string("query_states_85_groups_0"), val = int32(1)]; tensor query_states_85_cast_fp16 = conv(dilations = query_states_85_dilations_0, groups = query_states_85_groups_0, pad = query_states_85_pad_0, pad_type = query_states_85_pad_type_0, strides = query_states_85_strides_0, weight = layers_14_self_attn_q_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("query_states_85_cast_fp16")]; tensor layers_14_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_14_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440581824)))]; tensor key_states_141_strides_0 = const()[name = string("key_states_141_strides_0"), val = tensor([1, 1])]; string key_states_141_pad_type_0 = const()[name = string("key_states_141_pad_type_0"), val = string("valid")]; tensor key_states_141_pad_0 = const()[name = string("key_states_141_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_141_dilations_0 = const()[name = string("key_states_141_dilations_0"), val = tensor([1, 1])]; int32 key_states_141_groups_0 = const()[name = string("key_states_141_groups_0"), val = int32(1)]; tensor key_states_141_cast_fp16 = conv(dilations = key_states_141_dilations_0, groups = key_states_141_groups_0, pad = key_states_141_pad_0, pad_type = key_states_141_pad_type_0, strides = key_states_141_strides_0, weight = layers_14_self_attn_k_proj_weight_to_fp16, x = var_5421_cast_fp16_0)[name = string("key_states_141_cast_fp16")]; tensor value_states_85_strides_0 = const()[name = string("value_states_85_strides_0"), val = tensor([1, 1])]; string value_states_85_pad_type_0 = const()[name = string("value_states_85_pad_type_0"), val = string("valid")]; tensor value_states_85_pad_0 = const()[name = string("value_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_85_dilations_0 = const()[name = string("value_states_85_dilations_0"), val = tensor([1, 1])]; int32 value_states_85_groups_0 = const()[name = string("value_states_85_groups_0"), val = int32(1)]; tensor value_states_85_cast_fp16 = conv(dilations = value_states_85_dilations_0, groups = value_states_85_groups_0, pad = value_states_85_pad_0, pad_type = value_states_85_pad_type_0, strides = value_states_85_strides_0, weight = layers_14_self_attn_v_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("value_states_85_cast_fp16")]; tensor concat_168x = const()[name = string("concat_168x"), val = tensor([1, 16, 128, -1])]; tensor x_141_cast_fp16 = reshape(shape = concat_168x, x = query_states_85_cast_fp16)[name = string("x_141_cast_fp16")]; tensor concat_169x = const()[name = string("concat_169x"), val = tensor([1, 2, 128, -1])]; tensor var_5478_cast_fp16 = reshape(shape = concat_169x, x = key_states_141_cast_fp16)[name = string("op_5478_cast_fp16")]; tensor concat_170x = const()[name = string("concat_170x"), val = tensor([1, 2, 128, -1])]; tensor var_5485_cast_fp16 = reshape(shape = concat_170x, x = value_states_85_cast_fp16)[name = string("op_5485_cast_fp16")]; tensor var_5489_cast_fp16 = mul(x = x_141_cast_fp16, y = var_869_cast_fp16)[name = string("op_5489_cast_fp16")]; tensor var_5490_split_sizes_0 = const()[name = string("op_5490_split_sizes_0"), val = tensor([64, 64])]; int32 var_5490_axis_0 = const()[name = string("op_5490_axis_0"), val = int32(-2)]; tensor var_5490_cast_fp16_0, tensor var_5490_cast_fp16_1 = split(axis = var_5490_axis_0, split_sizes = var_5490_split_sizes_0, x = x_141_cast_fp16)[name = string("op_5490_cast_fp16")]; fp16 const_142_promoted_to_fp16 = const()[name = string("const_142_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5492_cast_fp16 = mul(x = var_5490_cast_fp16_1, y = const_142_promoted_to_fp16)[name = string("op_5492_cast_fp16")]; int32 var_5494 = const()[name = string("op_5494"), val = int32(-2)]; bool var_5495_interleave_0 = const()[name = string("op_5495_interleave_0"), val = bool(false)]; tensor var_5495_cast_fp16 = concat(axis = var_5494, interleave = var_5495_interleave_0, values = (var_5492_cast_fp16, var_5490_cast_fp16_0))[name = string("op_5495_cast_fp16")]; tensor var_5496_cast_fp16 = mul(x = var_5495_cast_fp16, y = var_878_cast_fp16)[name = string("op_5496_cast_fp16")]; tensor query_states_87_cast_fp16 = add(x = var_5489_cast_fp16, y = var_5496_cast_fp16)[name = string("query_states_87_cast_fp16")]; tensor var_5502_cast_fp16 = mul(x = var_5478_cast_fp16, y = var_869_cast_fp16)[name = string("op_5502_cast_fp16")]; tensor var_5503_split_sizes_0 = const()[name = string("op_5503_split_sizes_0"), val = tensor([64, 64])]; int32 var_5503_axis_0 = const()[name = string("op_5503_axis_0"), val = int32(-2)]; tensor var_5503_cast_fp16_0, tensor var_5503_cast_fp16_1 = split(axis = var_5503_axis_0, split_sizes = var_5503_split_sizes_0, x = var_5478_cast_fp16)[name = string("op_5503_cast_fp16")]; fp16 const_143_promoted_to_fp16 = const()[name = string("const_143_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5505_cast_fp16 = mul(x = var_5503_cast_fp16_1, y = const_143_promoted_to_fp16)[name = string("op_5505_cast_fp16")]; int32 var_5507 = const()[name = string("op_5507"), val = int32(-2)]; bool var_5508_interleave_0 = const()[name = string("op_5508_interleave_0"), val = bool(false)]; tensor var_5508_cast_fp16 = concat(axis = var_5507, interleave = var_5508_interleave_0, values = (var_5505_cast_fp16, var_5503_cast_fp16_0))[name = string("op_5508_cast_fp16")]; tensor var_5509_cast_fp16 = mul(x = var_5508_cast_fp16, y = var_878_cast_fp16)[name = string("op_5509_cast_fp16")]; tensor key_states_145_cast_fp16 = add(x = var_5502_cast_fp16, y = var_5509_cast_fp16)[name = string("key_states_145_cast_fp16")]; tensor expand_dims_168 = const()[name = string("expand_dims_168"), val = tensor([14])]; tensor expand_dims_169 = const()[name = string("expand_dims_169"), val = tensor([0])]; tensor expand_dims_171 = const()[name = string("expand_dims_171"), val = tensor([0])]; int32 concat_173_axis_0 = const()[name = string("concat_173_axis_0"), val = int32(0)]; bool concat_173_interleave_0 = const()[name = string("concat_173_interleave_0"), val = bool(false)]; tensor concat_173 = concat(axis = concat_173_axis_0, interleave = concat_173_interleave_0, values = (expand_dims_168, expand_dims_169, position_id, expand_dims_171))[name = string("concat_173")]; tensor expand_dims_172 = const()[name = string("expand_dims_172"), val = tensor([15])]; tensor concat_174_values1_0 = const()[name = string("concat_174_values1_0"), val = tensor([0])]; tensor concat_174_values3_0 = const()[name = string("concat_174_values3_0"), val = tensor([0])]; int32 concat_174_axis_0 = const()[name = string("concat_174_axis_0"), val = int32(0)]; bool concat_174_interleave_0 = const()[name = string("concat_174_interleave_0"), val = bool(false)]; tensor concat_174 = concat(axis = concat_174_axis_0, interleave = concat_174_interleave_0, values = (expand_dims_172, concat_174_values1_0, cache_position_end, concat_174_values3_0))[name = string("concat_174")]; tensor key_states_147_perm_0 = const()[name = string("key_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_15_stride_0 = const()[name = string("key_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_147_cast_fp16 = transpose(perm = key_states_147_perm_0, x = key_states_145_cast_fp16)[name = string("transpose_385")]; tensor key_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = key_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = key_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_15_squeeze_mask_0, stride = key_cache_internal_tensor_assign_15_stride_0, update = key_states_147_cast_fp16, x = coreml_update_state_250)[name = string("key_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_15_cast_fp16, input = key_cache)[name = string("coreml_update_state_252_write_state")]; tensor coreml_update_state_252 = read_state(input = key_cache)[name = string("coreml_update_state_252")]; tensor value_states_87_perm_0 = const()[name = string("value_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_15_stride_0 = const()[name = string("value_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_87_cast_fp16 = transpose(perm = value_states_87_perm_0, x = var_5485_cast_fp16)[name = string("transpose_384")]; tensor value_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = value_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = value_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_15_squeeze_mask_0, stride = value_cache_internal_tensor_assign_15_stride_0, update = value_states_87_cast_fp16, x = coreml_update_state_251)[name = string("value_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_15_cast_fp16, input = value_cache)[name = string("coreml_update_state_253_write_state")]; tensor coreml_update_state_253 = read_state(input = value_cache)[name = string("coreml_update_state_253")]; tensor var_5579_begin_0 = const()[name = string("op_5579_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5579_end_0 = const()[name = string("op_5579_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5579_end_mask_0 = const()[name = string("op_5579_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5579_cast_fp16 = slice_by_index(begin = var_5579_begin_0, end = var_5579_end_0, end_mask = var_5579_end_mask_0, x = coreml_update_state_252)[name = string("op_5579_cast_fp16")]; tensor tile_28 = const()[name = string("tile_28"), val = tensor([1, 1])]; int32 var_5582_axis_0 = const()[name = string("op_5582_axis_0"), val = int32(1)]; tensor var_5582_cast_fp16_0, tensor var_5582_cast_fp16_1 = split(axis = var_5582_axis_0, split_sizes = tile_28, x = var_5579_cast_fp16)[name = string("op_5582_cast_fp16")]; tensor var_5589_begin_0 = const()[name = string("op_5589_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5589_end_0 = const()[name = string("op_5589_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5589_end_mask_0 = const()[name = string("op_5589_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5589_cast_fp16 = slice_by_index(begin = var_5589_begin_0, end = var_5589_end_0, end_mask = var_5589_end_mask_0, x = coreml_update_state_253)[name = string("op_5589_cast_fp16")]; tensor tile_29 = const()[name = string("tile_29"), val = tensor([1, 1])]; int32 var_5592_axis_0 = const()[name = string("op_5592_axis_0"), val = int32(1)]; tensor var_5592_cast_fp16_0, tensor var_5592_cast_fp16_1 = split(axis = var_5592_axis_0, split_sizes = tile_29, x = var_5589_cast_fp16)[name = string("op_5592_cast_fp16")]; tensor var_5595_split_sizes_0 = const()[name = string("op_5595_split_sizes_0"), val = tensor([8, 8])]; int32 var_5595_axis_0 = const()[name = string("op_5595_axis_0"), val = int32(1)]; tensor var_5595_0, tensor var_5595_1 = split(axis = var_5595_axis_0, split_sizes = var_5595_split_sizes_0, x = query_states_87_cast_fp16)[name = string("op_5595")]; bool attn_weights_225_transpose_x_0 = const()[name = string("attn_weights_225_transpose_x_0"), val = bool(false)]; bool attn_weights_225_transpose_y_0 = const()[name = string("attn_weights_225_transpose_y_0"), val = bool(false)]; tensor attn_weights_225_cast_fp16 = matmul(transpose_x = attn_weights_225_transpose_x_0, transpose_y = attn_weights_225_transpose_y_0, x = var_5582_cast_fp16_0, y = var_5595_0)[name = string("attn_weights_225_cast_fp16")]; fp16 var_5598_to_fp16 = const()[name = string("op_5598_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_227_cast_fp16 = mul(x = attn_weights_225_cast_fp16, y = var_5598_to_fp16)[name = string("attn_weights_227_cast_fp16")]; tensor attn_weights_229_cast_fp16 = add(x = attn_weights_227_cast_fp16, y = attn_mask_1)[name = string("attn_weights_229_cast_fp16")]; int32 var_5602 = const()[name = string("op_5602"), val = int32(-2)]; tensor attn_weights_231_cast_fp16 = softmax(axis = var_5602, x = attn_weights_229_cast_fp16)[name = string("attn_weights_231_cast_fp16")]; bool var_5608_transpose_x_1 = const()[name = string("op_5608_transpose_x_1"), val = bool(true)]; bool var_5608_transpose_y_1 = const()[name = string("op_5608_transpose_y_1"), val = bool(false)]; tensor var_5608_cast_fp16 = matmul(transpose_x = var_5608_transpose_x_1, transpose_y = var_5608_transpose_y_1, x = attn_weights_231_cast_fp16, y = var_5592_cast_fp16_0)[name = string("op_5608_cast_fp16")]; bool attn_weights_233_transpose_x_0 = const()[name = string("attn_weights_233_transpose_x_0"), val = bool(false)]; bool attn_weights_233_transpose_y_0 = const()[name = string("attn_weights_233_transpose_y_0"), val = bool(false)]; tensor attn_weights_233_cast_fp16 = matmul(transpose_x = attn_weights_233_transpose_x_0, transpose_y = attn_weights_233_transpose_y_0, x = var_5582_cast_fp16_1, y = var_5595_1)[name = string("attn_weights_233_cast_fp16")]; fp16 var_5610_to_fp16 = const()[name = string("op_5610_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_235_cast_fp16 = mul(x = attn_weights_233_cast_fp16, y = var_5610_to_fp16)[name = string("attn_weights_235_cast_fp16")]; tensor attn_weights_237_cast_fp16 = add(x = attn_weights_235_cast_fp16, y = attn_mask_1)[name = string("attn_weights_237_cast_fp16")]; int32 var_5614 = const()[name = string("op_5614"), val = int32(-2)]; tensor attn_weights_239_cast_fp16 = softmax(axis = var_5614, x = attn_weights_237_cast_fp16)[name = string("attn_weights_239_cast_fp16")]; bool attn_output_113_transpose_x_1 = const()[name = string("attn_output_113_transpose_x_1"), val = bool(true)]; bool attn_output_113_transpose_y_1 = const()[name = string("attn_output_113_transpose_y_1"), val = bool(false)]; tensor attn_output_113_cast_fp16 = matmul(transpose_x = attn_output_113_transpose_x_1, transpose_y = attn_output_113_transpose_y_1, x = attn_weights_239_cast_fp16, y = var_5592_cast_fp16_1)[name = string("attn_output_113_cast_fp16")]; int32 var_5622 = const()[name = string("op_5622"), val = int32(1)]; bool attn_output_115_interleave_0 = const()[name = string("attn_output_115_interleave_0"), val = bool(false)]; tensor attn_output_115_cast_fp16 = concat(axis = var_5622, interleave = attn_output_115_interleave_0, values = (var_5608_cast_fp16, attn_output_113_cast_fp16))[name = string("attn_output_115_cast_fp16")]; tensor var_5626_perm_0 = const()[name = string("op_5626_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_179x = const()[name = string("concat_179x"), val = tensor([1, 2048, 1, -1])]; tensor var_5626_cast_fp16 = transpose(perm = var_5626_perm_0, x = attn_output_115_cast_fp16)[name = string("transpose_383")]; tensor attn_output_119_cast_fp16 = reshape(shape = concat_179x, x = var_5626_cast_fp16)[name = string("attn_output_119_cast_fp16")]; tensor hidden_states_143_strides_0 = const()[name = string("hidden_states_143_strides_0"), val = tensor([1, 1])]; string hidden_states_143_pad_type_0 = const()[name = string("hidden_states_143_pad_type_0"), val = string("valid")]; tensor hidden_states_143_pad_0 = const()[name = string("hidden_states_143_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_143_dilations_0 = const()[name = string("hidden_states_143_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_143_groups_0 = const()[name = string("hidden_states_143_groups_0"), val = int32(1)]; tensor hidden_states_143_cast_fp16 = conv(dilations = hidden_states_143_dilations_0, groups = hidden_states_143_groups_0, pad = hidden_states_143_pad_0, pad_type = hidden_states_143_pad_type_0, strides = hidden_states_143_strides_0, weight = layers_14_self_attn_o_proj_weight_cast_fp16, x = attn_output_119_cast_fp16)[name = string("hidden_states_143_cast_fp16")]; tensor hidden_states_145_cast_fp16 = add(x = hidden_states_139_cast_fp16, y = hidden_states_143_cast_fp16)[name = string("hidden_states_145_cast_fp16")]; fp16 const_148_promoted_to_fp16 = const()[name = string("const_148_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5659_cast_fp16 = mul(x = hidden_states_145_cast_fp16, y = const_148_promoted_to_fp16)[name = string("op_5659_cast_fp16")]; int32 var_5657 = const()[name = string("op_5657"), val = int32(1)]; bool doubled_117_interleave_0 = const()[name = string("doubled_117_interleave_0"), val = bool(false)]; tensor doubled_117_cast_fp16 = concat(axis = var_5657, interleave = doubled_117_interleave_0, values = (hidden_states_145_cast_fp16, var_5659_cast_fp16))[name = string("doubled_117_cast_fp16")]; tensor out_59_axes_0 = const()[name = string("out_59_axes_0"), val = tensor([1])]; tensor out_59_gamma_0_to_fp16 = const()[name = string("out_59_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441630464)))]; fp16 var_5669_to_fp16 = const()[name = string("op_5669_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_59_cast_fp16 = layer_norm(axes = out_59_axes_0, epsilon = var_5669_to_fp16, gamma = out_59_gamma_0_to_fp16, x = doubled_117_cast_fp16)[name = string("out_59_cast_fp16")]; tensor var_5680_split_sizes_0 = const()[name = string("op_5680_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5680_axis_0 = const()[name = string("op_5680_axis_0"), val = int32(1)]; tensor var_5680_cast_fp16_0, tensor var_5680_cast_fp16_1 = split(axis = var_5680_axis_0, split_sizes = var_5680_split_sizes_0, x = out_59_cast_fp16)[name = string("op_5680_cast_fp16")]; tensor input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor([1, 1])]; string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; tensor input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor([1, 1])]; int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; tensor input_29_cast_fp16 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_14_mlp_gate_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("input_29_cast_fp16")]; tensor var_5697_cast_fp16 = silu(x = input_29_cast_fp16)[name = string("op_5697_cast_fp16")]; tensor var_5703_strides_0 = const()[name = string("op_5703_strides_0"), val = tensor([1, 1])]; string var_5703_pad_type_0 = const()[name = string("op_5703_pad_type_0"), val = string("valid")]; tensor var_5703_pad_0 = const()[name = string("op_5703_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5703_dilations_0 = const()[name = string("op_5703_dilations_0"), val = tensor([1, 1])]; int32 var_5703_groups_0 = const()[name = string("op_5703_groups_0"), val = int32(1)]; tensor var_5703_cast_fp16 = conv(dilations = var_5703_dilations_0, groups = var_5703_groups_0, pad = var_5703_pad_0, pad_type = var_5703_pad_type_0, strides = var_5703_strides_0, weight = layers_14_mlp_up_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("op_5703_cast_fp16")]; tensor x_149_cast_fp16 = mul(x = var_5697_cast_fp16, y = var_5703_cast_fp16)[name = string("x_149_cast_fp16")]; tensor hidden_states_147_strides_0 = const()[name = string("hidden_states_147_strides_0"), val = tensor([1, 1])]; string hidden_states_147_pad_type_0 = const()[name = string("hidden_states_147_pad_type_0"), val = string("valid")]; tensor hidden_states_147_pad_0 = const()[name = string("hidden_states_147_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_147_dilations_0 = const()[name = string("hidden_states_147_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_147_groups_0 = const()[name = string("hidden_states_147_groups_0"), val = int32(1)]; tensor hidden_states_147_cast_fp16 = conv(dilations = hidden_states_147_dilations_0, groups = hidden_states_147_groups_0, pad = hidden_states_147_pad_0, pad_type = hidden_states_147_pad_type_0, strides = hidden_states_147_strides_0, weight = layers_14_mlp_down_proj_weight_cast_fp16, x = x_149_cast_fp16)[name = string("hidden_states_147_cast_fp16")]; tensor hidden_states_149_cast_fp16 = add(x = hidden_states_145_cast_fp16, y = hidden_states_147_cast_fp16)[name = string("hidden_states_149_cast_fp16")]; fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5721_cast_fp16 = mul(x = hidden_states_149_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_5721_cast_fp16")]; int32 var_5719 = const()[name = string("op_5719"), val = int32(1)]; bool doubled_121_interleave_0 = const()[name = string("doubled_121_interleave_0"), val = bool(false)]; tensor doubled_121_cast_fp16 = concat(axis = var_5719, interleave = doubled_121_interleave_0, values = (hidden_states_149_cast_fp16, var_5721_cast_fp16))[name = string("doubled_121_cast_fp16")]; tensor out_61_axes_0 = const()[name = string("out_61_axes_0"), val = tensor([1])]; tensor out_61_gamma_0_to_fp16 = const()[name = string("out_61_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441638720)))]; fp16 var_5731_to_fp16 = const()[name = string("op_5731_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_61_cast_fp16 = layer_norm(axes = out_61_axes_0, epsilon = var_5731_to_fp16, gamma = out_61_gamma_0_to_fp16, x = doubled_121_cast_fp16)[name = string("out_61_cast_fp16")]; tensor var_5742_split_sizes_0 = const()[name = string("op_5742_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5742_axis_0 = const()[name = string("op_5742_axis_0"), val = int32(1)]; tensor var_5742_cast_fp16_0, tensor var_5742_cast_fp16_1 = split(axis = var_5742_axis_0, split_sizes = var_5742_split_sizes_0, x = out_61_cast_fp16)[name = string("op_5742_cast_fp16")]; tensor query_states_91_strides_0 = const()[name = string("query_states_91_strides_0"), val = tensor([1, 1])]; string query_states_91_pad_type_0 = const()[name = string("query_states_91_pad_type_0"), val = string("valid")]; tensor query_states_91_pad_0 = const()[name = string("query_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_91_dilations_0 = const()[name = string("query_states_91_dilations_0"), val = tensor([1, 1])]; int32 query_states_91_groups_0 = const()[name = string("query_states_91_groups_0"), val = int32(1)]; tensor query_states_91_cast_fp16 = conv(dilations = query_states_91_dilations_0, groups = query_states_91_groups_0, pad = query_states_91_pad_0, pad_type = query_states_91_pad_type_0, strides = query_states_91_strides_0, weight = layers_15_self_attn_q_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("query_states_91_cast_fp16")]; tensor key_states_151_strides_0 = const()[name = string("key_states_151_strides_0"), val = tensor([1, 1])]; string key_states_151_pad_type_0 = const()[name = string("key_states_151_pad_type_0"), val = string("valid")]; tensor key_states_151_pad_0 = const()[name = string("key_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_151_dilations_0 = const()[name = string("key_states_151_dilations_0"), val = tensor([1, 1])]; int32 key_states_151_groups_0 = const()[name = string("key_states_151_groups_0"), val = int32(1)]; tensor key_states_151_cast_fp16 = conv(dilations = key_states_151_dilations_0, groups = key_states_151_groups_0, pad = key_states_151_pad_0, pad_type = key_states_151_pad_type_0, strides = key_states_151_strides_0, weight = layers_15_self_attn_k_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("key_states_151_cast_fp16")]; tensor value_states_91_strides_0 = const()[name = string("value_states_91_strides_0"), val = tensor([1, 1])]; string value_states_91_pad_type_0 = const()[name = string("value_states_91_pad_type_0"), val = string("valid")]; tensor value_states_91_pad_0 = const()[name = string("value_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_91_dilations_0 = const()[name = string("value_states_91_dilations_0"), val = tensor([1, 1])]; int32 value_states_91_groups_0 = const()[name = string("value_states_91_groups_0"), val = int32(1)]; tensor value_states_91_cast_fp16 = conv(dilations = value_states_91_dilations_0, groups = value_states_91_groups_0, pad = value_states_91_pad_0, pad_type = value_states_91_pad_type_0, strides = value_states_91_strides_0, weight = layers_15_self_attn_v_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("value_states_91_cast_fp16")]; tensor concat_180x = const()[name = string("concat_180x"), val = tensor([1, 16, 128, -1])]; tensor x_151_cast_fp16 = reshape(shape = concat_180x, x = query_states_91_cast_fp16)[name = string("x_151_cast_fp16")]; tensor concat_181x = const()[name = string("concat_181x"), val = tensor([1, 2, 128, -1])]; tensor var_5799_cast_fp16 = reshape(shape = concat_181x, x = key_states_151_cast_fp16)[name = string("op_5799_cast_fp16")]; tensor concat_182x = const()[name = string("concat_182x"), val = tensor([1, 2, 128, -1])]; tensor var_5806_cast_fp16 = reshape(shape = concat_182x, x = value_states_91_cast_fp16)[name = string("op_5806_cast_fp16")]; tensor var_5810_cast_fp16 = mul(x = x_151_cast_fp16, y = var_869_cast_fp16)[name = string("op_5810_cast_fp16")]; tensor var_5811_split_sizes_0 = const()[name = string("op_5811_split_sizes_0"), val = tensor([64, 64])]; int32 var_5811_axis_0 = const()[name = string("op_5811_axis_0"), val = int32(-2)]; tensor var_5811_cast_fp16_0, tensor var_5811_cast_fp16_1 = split(axis = var_5811_axis_0, split_sizes = var_5811_split_sizes_0, x = x_151_cast_fp16)[name = string("op_5811_cast_fp16")]; fp16 const_152_promoted_to_fp16 = const()[name = string("const_152_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5813_cast_fp16 = mul(x = var_5811_cast_fp16_1, y = const_152_promoted_to_fp16)[name = string("op_5813_cast_fp16")]; int32 var_5815 = const()[name = string("op_5815"), val = int32(-2)]; bool var_5816_interleave_0 = const()[name = string("op_5816_interleave_0"), val = bool(false)]; tensor var_5816_cast_fp16 = concat(axis = var_5815, interleave = var_5816_interleave_0, values = (var_5813_cast_fp16, var_5811_cast_fp16_0))[name = string("op_5816_cast_fp16")]; tensor var_5817_cast_fp16 = mul(x = var_5816_cast_fp16, y = var_878_cast_fp16)[name = string("op_5817_cast_fp16")]; tensor query_states_93_cast_fp16 = add(x = var_5810_cast_fp16, y = var_5817_cast_fp16)[name = string("query_states_93_cast_fp16")]; tensor var_5823_cast_fp16 = mul(x = var_5799_cast_fp16, y = var_869_cast_fp16)[name = string("op_5823_cast_fp16")]; tensor var_5824_split_sizes_0 = const()[name = string("op_5824_split_sizes_0"), val = tensor([64, 64])]; int32 var_5824_axis_0 = const()[name = string("op_5824_axis_0"), val = int32(-2)]; tensor var_5824_cast_fp16_0, tensor var_5824_cast_fp16_1 = split(axis = var_5824_axis_0, split_sizes = var_5824_split_sizes_0, x = var_5799_cast_fp16)[name = string("op_5824_cast_fp16")]; fp16 const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5826_cast_fp16 = mul(x = var_5824_cast_fp16_1, y = const_153_promoted_to_fp16)[name = string("op_5826_cast_fp16")]; int32 var_5828 = const()[name = string("op_5828"), val = int32(-2)]; bool var_5829_interleave_0 = const()[name = string("op_5829_interleave_0"), val = bool(false)]; tensor var_5829_cast_fp16 = concat(axis = var_5828, interleave = var_5829_interleave_0, values = (var_5826_cast_fp16, var_5824_cast_fp16_0))[name = string("op_5829_cast_fp16")]; tensor var_5830_cast_fp16 = mul(x = var_5829_cast_fp16, y = var_878_cast_fp16)[name = string("op_5830_cast_fp16")]; tensor key_states_155_cast_fp16 = add(x = var_5823_cast_fp16, y = var_5830_cast_fp16)[name = string("key_states_155_cast_fp16")]; tensor expand_dims_180 = const()[name = string("expand_dims_180"), val = tensor([15])]; tensor expand_dims_181 = const()[name = string("expand_dims_181"), val = tensor([0])]; tensor expand_dims_183 = const()[name = string("expand_dims_183"), val = tensor([0])]; int32 concat_185_axis_0 = const()[name = string("concat_185_axis_0"), val = int32(0)]; bool concat_185_interleave_0 = const()[name = string("concat_185_interleave_0"), val = bool(false)]; tensor concat_185 = concat(axis = concat_185_axis_0, interleave = concat_185_interleave_0, values = (expand_dims_180, expand_dims_181, position_id, expand_dims_183))[name = string("concat_185")]; tensor expand_dims_184 = const()[name = string("expand_dims_184"), val = tensor([16])]; tensor concat_186_values1_0 = const()[name = string("concat_186_values1_0"), val = tensor([0])]; tensor concat_186_values3_0 = const()[name = string("concat_186_values3_0"), val = tensor([0])]; int32 concat_186_axis_0 = const()[name = string("concat_186_axis_0"), val = int32(0)]; bool concat_186_interleave_0 = const()[name = string("concat_186_interleave_0"), val = bool(false)]; tensor concat_186 = concat(axis = concat_186_axis_0, interleave = concat_186_interleave_0, values = (expand_dims_184, concat_186_values1_0, cache_position_end, concat_186_values3_0))[name = string("concat_186")]; tensor key_states_157_perm_0 = const()[name = string("key_states_157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_16_stride_0 = const()[name = string("key_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_157_cast_fp16 = transpose(perm = key_states_157_perm_0, x = key_states_155_cast_fp16)[name = string("transpose_382")]; tensor key_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = key_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = key_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_16_squeeze_mask_0, stride = key_cache_internal_tensor_assign_16_stride_0, update = key_states_157_cast_fp16, x = coreml_update_state_252)[name = string("key_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_16_cast_fp16, input = key_cache)[name = string("coreml_update_state_254_write_state")]; tensor coreml_update_state_254 = read_state(input = key_cache)[name = string("coreml_update_state_254")]; tensor value_states_93_perm_0 = const()[name = string("value_states_93_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_16_stride_0 = const()[name = string("value_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_93_cast_fp16 = transpose(perm = value_states_93_perm_0, x = var_5806_cast_fp16)[name = string("transpose_381")]; tensor value_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = value_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = value_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_16_squeeze_mask_0, stride = value_cache_internal_tensor_assign_16_stride_0, update = value_states_93_cast_fp16, x = coreml_update_state_253)[name = string("value_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_16_cast_fp16, input = value_cache)[name = string("coreml_update_state_255_write_state")]; tensor coreml_update_state_255 = read_state(input = value_cache)[name = string("coreml_update_state_255")]; tensor var_5900_begin_0 = const()[name = string("op_5900_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5900_end_0 = const()[name = string("op_5900_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5900_end_mask_0 = const()[name = string("op_5900_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5900_cast_fp16 = slice_by_index(begin = var_5900_begin_0, end = var_5900_end_0, end_mask = var_5900_end_mask_0, x = coreml_update_state_254)[name = string("op_5900_cast_fp16")]; tensor tile_30 = const()[name = string("tile_30"), val = tensor([1, 1])]; int32 var_5903_axis_0 = const()[name = string("op_5903_axis_0"), val = int32(1)]; tensor var_5903_cast_fp16_0, tensor var_5903_cast_fp16_1 = split(axis = var_5903_axis_0, split_sizes = tile_30, x = var_5900_cast_fp16)[name = string("op_5903_cast_fp16")]; tensor var_5910_begin_0 = const()[name = string("op_5910_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5910_end_0 = const()[name = string("op_5910_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5910_end_mask_0 = const()[name = string("op_5910_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5910_cast_fp16 = slice_by_index(begin = var_5910_begin_0, end = var_5910_end_0, end_mask = var_5910_end_mask_0, x = coreml_update_state_255)[name = string("op_5910_cast_fp16")]; tensor tile_31 = const()[name = string("tile_31"), val = tensor([1, 1])]; int32 var_5913_axis_0 = const()[name = string("op_5913_axis_0"), val = int32(1)]; tensor var_5913_cast_fp16_0, tensor var_5913_cast_fp16_1 = split(axis = var_5913_axis_0, split_sizes = tile_31, x = var_5910_cast_fp16)[name = string("op_5913_cast_fp16")]; tensor var_5916_split_sizes_0 = const()[name = string("op_5916_split_sizes_0"), val = tensor([8, 8])]; int32 var_5916_axis_0 = const()[name = string("op_5916_axis_0"), val = int32(1)]; tensor var_5916_0, tensor var_5916_1 = split(axis = var_5916_axis_0, split_sizes = var_5916_split_sizes_0, x = query_states_93_cast_fp16)[name = string("op_5916")]; bool attn_weights_241_transpose_x_0 = const()[name = string("attn_weights_241_transpose_x_0"), val = bool(false)]; bool attn_weights_241_transpose_y_0 = const()[name = string("attn_weights_241_transpose_y_0"), val = bool(false)]; tensor attn_weights_241_cast_fp16 = matmul(transpose_x = attn_weights_241_transpose_x_0, transpose_y = attn_weights_241_transpose_y_0, x = var_5903_cast_fp16_0, y = var_5916_0)[name = string("attn_weights_241_cast_fp16")]; fp16 var_5919_to_fp16 = const()[name = string("op_5919_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_243_cast_fp16 = mul(x = attn_weights_241_cast_fp16, y = var_5919_to_fp16)[name = string("attn_weights_243_cast_fp16")]; tensor attn_weights_245_cast_fp16 = add(x = attn_weights_243_cast_fp16, y = attn_mask_1)[name = string("attn_weights_245_cast_fp16")]; int32 var_5923 = const()[name = string("op_5923"), val = int32(-2)]; tensor attn_weights_247_cast_fp16 = softmax(axis = var_5923, x = attn_weights_245_cast_fp16)[name = string("attn_weights_247_cast_fp16")]; bool var_5929_transpose_x_1 = const()[name = string("op_5929_transpose_x_1"), val = bool(true)]; bool var_5929_transpose_y_1 = const()[name = string("op_5929_transpose_y_1"), val = bool(false)]; tensor var_5929_cast_fp16 = matmul(transpose_x = var_5929_transpose_x_1, transpose_y = var_5929_transpose_y_1, x = attn_weights_247_cast_fp16, y = var_5913_cast_fp16_0)[name = string("op_5929_cast_fp16")]; bool attn_weights_249_transpose_x_0 = const()[name = string("attn_weights_249_transpose_x_0"), val = bool(false)]; bool attn_weights_249_transpose_y_0 = const()[name = string("attn_weights_249_transpose_y_0"), val = bool(false)]; tensor attn_weights_249_cast_fp16 = matmul(transpose_x = attn_weights_249_transpose_x_0, transpose_y = attn_weights_249_transpose_y_0, x = var_5903_cast_fp16_1, y = var_5916_1)[name = string("attn_weights_249_cast_fp16")]; fp16 var_5931_to_fp16 = const()[name = string("op_5931_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_251_cast_fp16 = mul(x = attn_weights_249_cast_fp16, y = var_5931_to_fp16)[name = string("attn_weights_251_cast_fp16")]; tensor attn_weights_253_cast_fp16 = add(x = attn_weights_251_cast_fp16, y = attn_mask_1)[name = string("attn_weights_253_cast_fp16")]; int32 var_5935 = const()[name = string("op_5935"), val = int32(-2)]; tensor attn_weights_255_cast_fp16 = softmax(axis = var_5935, x = attn_weights_253_cast_fp16)[name = string("attn_weights_255_cast_fp16")]; bool attn_output_121_transpose_x_1 = const()[name = string("attn_output_121_transpose_x_1"), val = bool(true)]; bool attn_output_121_transpose_y_1 = const()[name = string("attn_output_121_transpose_y_1"), val = bool(false)]; tensor attn_output_121_cast_fp16 = matmul(transpose_x = attn_output_121_transpose_x_1, transpose_y = attn_output_121_transpose_y_1, x = attn_weights_255_cast_fp16, y = var_5913_cast_fp16_1)[name = string("attn_output_121_cast_fp16")]; int32 var_5943 = const()[name = string("op_5943"), val = int32(1)]; bool attn_output_123_interleave_0 = const()[name = string("attn_output_123_interleave_0"), val = bool(false)]; tensor attn_output_123_cast_fp16 = concat(axis = var_5943, interleave = attn_output_123_interleave_0, values = (var_5929_cast_fp16, attn_output_121_cast_fp16))[name = string("attn_output_123_cast_fp16")]; tensor var_5947_perm_0 = const()[name = string("op_5947_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_191x = const()[name = string("concat_191x"), val = tensor([1, 2048, 1, -1])]; tensor var_5947_cast_fp16 = transpose(perm = var_5947_perm_0, x = attn_output_123_cast_fp16)[name = string("transpose_380")]; tensor attn_output_127_cast_fp16 = reshape(shape = concat_191x, x = var_5947_cast_fp16)[name = string("attn_output_127_cast_fp16")]; tensor hidden_states_153_strides_0 = const()[name = string("hidden_states_153_strides_0"), val = tensor([1, 1])]; string hidden_states_153_pad_type_0 = const()[name = string("hidden_states_153_pad_type_0"), val = string("valid")]; tensor hidden_states_153_pad_0 = const()[name = string("hidden_states_153_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_153_dilations_0 = const()[name = string("hidden_states_153_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_153_groups_0 = const()[name = string("hidden_states_153_groups_0"), val = int32(1)]; tensor hidden_states_153_cast_fp16 = conv(dilations = hidden_states_153_dilations_0, groups = hidden_states_153_groups_0, pad = hidden_states_153_pad_0, pad_type = hidden_states_153_pad_type_0, strides = hidden_states_153_strides_0, weight = layers_15_self_attn_o_proj_weight_cast_fp16, x = attn_output_127_cast_fp16)[name = string("hidden_states_153_cast_fp16")]; tensor hidden_states_155_cast_fp16 = add(x = hidden_states_149_cast_fp16, y = hidden_states_153_cast_fp16)[name = string("hidden_states_155_cast_fp16")]; fp16 const_158_promoted_to_fp16 = const()[name = string("const_158_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5980_cast_fp16 = mul(x = hidden_states_155_cast_fp16, y = const_158_promoted_to_fp16)[name = string("op_5980_cast_fp16")]; int32 var_5978 = const()[name = string("op_5978"), val = int32(1)]; bool doubled_125_interleave_0 = const()[name = string("doubled_125_interleave_0"), val = bool(false)]; tensor doubled_125_cast_fp16 = concat(axis = var_5978, interleave = doubled_125_interleave_0, values = (hidden_states_155_cast_fp16, var_5980_cast_fp16))[name = string("doubled_125_cast_fp16")]; tensor out_63_axes_0 = const()[name = string("out_63_axes_0"), val = tensor([1])]; tensor out_63_gamma_0_to_fp16 = const()[name = string("out_63_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441646976)))]; fp16 var_5990_to_fp16 = const()[name = string("op_5990_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_63_cast_fp16 = layer_norm(axes = out_63_axes_0, epsilon = var_5990_to_fp16, gamma = out_63_gamma_0_to_fp16, x = doubled_125_cast_fp16)[name = string("out_63_cast_fp16")]; tensor var_6001_split_sizes_0 = const()[name = string("op_6001_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6001_axis_0 = const()[name = string("op_6001_axis_0"), val = int32(1)]; tensor var_6001_cast_fp16_0, tensor var_6001_cast_fp16_1 = split(axis = var_6001_axis_0, split_sizes = var_6001_split_sizes_0, x = out_63_cast_fp16)[name = string("op_6001_cast_fp16")]; tensor input_31_strides_0 = const()[name = string("input_31_strides_0"), val = tensor([1, 1])]; string input_31_pad_type_0 = const()[name = string("input_31_pad_type_0"), val = string("valid")]; tensor input_31_pad_0 = const()[name = string("input_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_31_dilations_0 = const()[name = string("input_31_dilations_0"), val = tensor([1, 1])]; int32 input_31_groups_0 = const()[name = string("input_31_groups_0"), val = int32(1)]; tensor input_31_cast_fp16 = conv(dilations = input_31_dilations_0, groups = input_31_groups_0, pad = input_31_pad_0, pad_type = input_31_pad_type_0, strides = input_31_strides_0, weight = layers_15_mlp_gate_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("input_31_cast_fp16")]; tensor var_6018_cast_fp16 = silu(x = input_31_cast_fp16)[name = string("op_6018_cast_fp16")]; tensor var_6024_strides_0 = const()[name = string("op_6024_strides_0"), val = tensor([1, 1])]; string var_6024_pad_type_0 = const()[name = string("op_6024_pad_type_0"), val = string("valid")]; tensor var_6024_pad_0 = const()[name = string("op_6024_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6024_dilations_0 = const()[name = string("op_6024_dilations_0"), val = tensor([1, 1])]; int32 var_6024_groups_0 = const()[name = string("op_6024_groups_0"), val = int32(1)]; tensor var_6024_cast_fp16 = conv(dilations = var_6024_dilations_0, groups = var_6024_groups_0, pad = var_6024_pad_0, pad_type = var_6024_pad_type_0, strides = var_6024_strides_0, weight = layers_15_mlp_up_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("op_6024_cast_fp16")]; tensor x_159_cast_fp16 = mul(x = var_6018_cast_fp16, y = var_6024_cast_fp16)[name = string("x_159_cast_fp16")]; tensor hidden_states_157_strides_0 = const()[name = string("hidden_states_157_strides_0"), val = tensor([1, 1])]; string hidden_states_157_pad_type_0 = const()[name = string("hidden_states_157_pad_type_0"), val = string("valid")]; tensor hidden_states_157_pad_0 = const()[name = string("hidden_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_157_dilations_0 = const()[name = string("hidden_states_157_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_157_groups_0 = const()[name = string("hidden_states_157_groups_0"), val = int32(1)]; tensor hidden_states_157_cast_fp16 = conv(dilations = hidden_states_157_dilations_0, groups = hidden_states_157_groups_0, pad = hidden_states_157_pad_0, pad_type = hidden_states_157_pad_type_0, strides = hidden_states_157_strides_0, weight = layers_15_mlp_down_proj_weight_cast_fp16, x = x_159_cast_fp16)[name = string("hidden_states_157_cast_fp16")]; tensor hidden_states_159_cast_fp16 = add(x = hidden_states_155_cast_fp16, y = hidden_states_157_cast_fp16)[name = string("hidden_states_159_cast_fp16")]; fp16 const_160_promoted_to_fp16 = const()[name = string("const_160_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6042_cast_fp16 = mul(x = hidden_states_159_cast_fp16, y = const_160_promoted_to_fp16)[name = string("op_6042_cast_fp16")]; int32 var_6040 = const()[name = string("op_6040"), val = int32(1)]; bool doubled_129_interleave_0 = const()[name = string("doubled_129_interleave_0"), val = bool(false)]; tensor doubled_129_cast_fp16 = concat(axis = var_6040, interleave = doubled_129_interleave_0, values = (hidden_states_159_cast_fp16, var_6042_cast_fp16))[name = string("doubled_129_cast_fp16")]; tensor out_65_axes_0 = const()[name = string("out_65_axes_0"), val = tensor([1])]; tensor out_65_gamma_0_to_fp16 = const()[name = string("out_65_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441655232)))]; fp16 var_6052_to_fp16 = const()[name = string("op_6052_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_65_cast_fp16 = layer_norm(axes = out_65_axes_0, epsilon = var_6052_to_fp16, gamma = out_65_gamma_0_to_fp16, x = doubled_129_cast_fp16)[name = string("out_65_cast_fp16")]; tensor var_6063_split_sizes_0 = const()[name = string("op_6063_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6063_axis_0 = const()[name = string("op_6063_axis_0"), val = int32(1)]; tensor var_6063_cast_fp16_0, tensor var_6063_cast_fp16_1 = split(axis = var_6063_axis_0, split_sizes = var_6063_split_sizes_0, x = out_65_cast_fp16)[name = string("op_6063_cast_fp16")]; tensor query_states_97_strides_0 = const()[name = string("query_states_97_strides_0"), val = tensor([1, 1])]; string query_states_97_pad_type_0 = const()[name = string("query_states_97_pad_type_0"), val = string("valid")]; tensor query_states_97_pad_0 = const()[name = string("query_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_97_dilations_0 = const()[name = string("query_states_97_dilations_0"), val = tensor([1, 1])]; int32 query_states_97_groups_0 = const()[name = string("query_states_97_groups_0"), val = int32(1)]; tensor query_states_97_cast_fp16 = conv(dilations = query_states_97_dilations_0, groups = query_states_97_groups_0, pad = query_states_97_pad_0, pad_type = query_states_97_pad_type_0, strides = query_states_97_strides_0, weight = layers_16_self_attn_q_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("query_states_97_cast_fp16")]; tensor key_states_161_strides_0 = const()[name = string("key_states_161_strides_0"), val = tensor([1, 1])]; string key_states_161_pad_type_0 = const()[name = string("key_states_161_pad_type_0"), val = string("valid")]; tensor key_states_161_pad_0 = const()[name = string("key_states_161_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_161_dilations_0 = const()[name = string("key_states_161_dilations_0"), val = tensor([1, 1])]; int32 key_states_161_groups_0 = const()[name = string("key_states_161_groups_0"), val = int32(1)]; tensor key_states_161_cast_fp16 = conv(dilations = key_states_161_dilations_0, groups = key_states_161_groups_0, pad = key_states_161_pad_0, pad_type = key_states_161_pad_type_0, strides = key_states_161_strides_0, weight = layers_16_self_attn_k_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("key_states_161_cast_fp16")]; tensor value_states_97_strides_0 = const()[name = string("value_states_97_strides_0"), val = tensor([1, 1])]; string value_states_97_pad_type_0 = const()[name = string("value_states_97_pad_type_0"), val = string("valid")]; tensor value_states_97_pad_0 = const()[name = string("value_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_97_dilations_0 = const()[name = string("value_states_97_dilations_0"), val = tensor([1, 1])]; int32 value_states_97_groups_0 = const()[name = string("value_states_97_groups_0"), val = int32(1)]; tensor value_states_97_cast_fp16 = conv(dilations = value_states_97_dilations_0, groups = value_states_97_groups_0, pad = value_states_97_pad_0, pad_type = value_states_97_pad_type_0, strides = value_states_97_strides_0, weight = layers_16_self_attn_v_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("value_states_97_cast_fp16")]; tensor concat_192x = const()[name = string("concat_192x"), val = tensor([1, 16, 128, -1])]; tensor x_161_cast_fp16 = reshape(shape = concat_192x, x = query_states_97_cast_fp16)[name = string("x_161_cast_fp16")]; tensor concat_193x = const()[name = string("concat_193x"), val = tensor([1, 2, 128, -1])]; tensor var_6120_cast_fp16 = reshape(shape = concat_193x, x = key_states_161_cast_fp16)[name = string("op_6120_cast_fp16")]; tensor concat_194x = const()[name = string("concat_194x"), val = tensor([1, 2, 128, -1])]; tensor var_6127_cast_fp16 = reshape(shape = concat_194x, x = value_states_97_cast_fp16)[name = string("op_6127_cast_fp16")]; tensor var_6131_cast_fp16 = mul(x = x_161_cast_fp16, y = var_869_cast_fp16)[name = string("op_6131_cast_fp16")]; tensor var_6132_split_sizes_0 = const()[name = string("op_6132_split_sizes_0"), val = tensor([64, 64])]; int32 var_6132_axis_0 = const()[name = string("op_6132_axis_0"), val = int32(-2)]; tensor var_6132_cast_fp16_0, tensor var_6132_cast_fp16_1 = split(axis = var_6132_axis_0, split_sizes = var_6132_split_sizes_0, x = x_161_cast_fp16)[name = string("op_6132_cast_fp16")]; fp16 const_162_promoted_to_fp16 = const()[name = string("const_162_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6134_cast_fp16 = mul(x = var_6132_cast_fp16_1, y = const_162_promoted_to_fp16)[name = string("op_6134_cast_fp16")]; int32 var_6136 = const()[name = string("op_6136"), val = int32(-2)]; bool var_6137_interleave_0 = const()[name = string("op_6137_interleave_0"), val = bool(false)]; tensor var_6137_cast_fp16 = concat(axis = var_6136, interleave = var_6137_interleave_0, values = (var_6134_cast_fp16, var_6132_cast_fp16_0))[name = string("op_6137_cast_fp16")]; tensor var_6138_cast_fp16 = mul(x = var_6137_cast_fp16, y = var_878_cast_fp16)[name = string("op_6138_cast_fp16")]; tensor query_states_99_cast_fp16 = add(x = var_6131_cast_fp16, y = var_6138_cast_fp16)[name = string("query_states_99_cast_fp16")]; tensor var_6144_cast_fp16 = mul(x = var_6120_cast_fp16, y = var_869_cast_fp16)[name = string("op_6144_cast_fp16")]; tensor var_6145_split_sizes_0 = const()[name = string("op_6145_split_sizes_0"), val = tensor([64, 64])]; int32 var_6145_axis_0 = const()[name = string("op_6145_axis_0"), val = int32(-2)]; tensor var_6145_cast_fp16_0, tensor var_6145_cast_fp16_1 = split(axis = var_6145_axis_0, split_sizes = var_6145_split_sizes_0, x = var_6120_cast_fp16)[name = string("op_6145_cast_fp16")]; fp16 const_163_promoted_to_fp16 = const()[name = string("const_163_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6147_cast_fp16 = mul(x = var_6145_cast_fp16_1, y = const_163_promoted_to_fp16)[name = string("op_6147_cast_fp16")]; int32 var_6149 = const()[name = string("op_6149"), val = int32(-2)]; bool var_6150_interleave_0 = const()[name = string("op_6150_interleave_0"), val = bool(false)]; tensor var_6150_cast_fp16 = concat(axis = var_6149, interleave = var_6150_interleave_0, values = (var_6147_cast_fp16, var_6145_cast_fp16_0))[name = string("op_6150_cast_fp16")]; tensor var_6151_cast_fp16 = mul(x = var_6150_cast_fp16, y = var_878_cast_fp16)[name = string("op_6151_cast_fp16")]; tensor key_states_165_cast_fp16 = add(x = var_6144_cast_fp16, y = var_6151_cast_fp16)[name = string("key_states_165_cast_fp16")]; tensor expand_dims_192 = const()[name = string("expand_dims_192"), val = tensor([16])]; tensor expand_dims_193 = const()[name = string("expand_dims_193"), val = tensor([0])]; tensor expand_dims_195 = const()[name = string("expand_dims_195"), val = tensor([0])]; int32 concat_197_axis_0 = const()[name = string("concat_197_axis_0"), val = int32(0)]; bool concat_197_interleave_0 = const()[name = string("concat_197_interleave_0"), val = bool(false)]; tensor concat_197 = concat(axis = concat_197_axis_0, interleave = concat_197_interleave_0, values = (expand_dims_192, expand_dims_193, position_id, expand_dims_195))[name = string("concat_197")]; tensor expand_dims_196 = const()[name = string("expand_dims_196"), val = tensor([17])]; tensor concat_198_values1_0 = const()[name = string("concat_198_values1_0"), val = tensor([0])]; tensor concat_198_values3_0 = const()[name = string("concat_198_values3_0"), val = tensor([0])]; int32 concat_198_axis_0 = const()[name = string("concat_198_axis_0"), val = int32(0)]; bool concat_198_interleave_0 = const()[name = string("concat_198_interleave_0"), val = bool(false)]; tensor concat_198 = concat(axis = concat_198_axis_0, interleave = concat_198_interleave_0, values = (expand_dims_196, concat_198_values1_0, cache_position_end, concat_198_values3_0))[name = string("concat_198")]; tensor key_states_167_perm_0 = const()[name = string("key_states_167_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_17_stride_0 = const()[name = string("key_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_167_cast_fp16 = transpose(perm = key_states_167_perm_0, x = key_states_165_cast_fp16)[name = string("transpose_379")]; tensor key_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = key_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = key_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_17_squeeze_mask_0, stride = key_cache_internal_tensor_assign_17_stride_0, update = key_states_167_cast_fp16, x = coreml_update_state_254)[name = string("key_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_17_cast_fp16, input = key_cache)[name = string("coreml_update_state_256_write_state")]; tensor coreml_update_state_256 = read_state(input = key_cache)[name = string("coreml_update_state_256")]; tensor value_states_99_perm_0 = const()[name = string("value_states_99_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_17_stride_0 = const()[name = string("value_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_99_cast_fp16 = transpose(perm = value_states_99_perm_0, x = var_6127_cast_fp16)[name = string("transpose_378")]; tensor value_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = value_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = value_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_17_squeeze_mask_0, stride = value_cache_internal_tensor_assign_17_stride_0, update = value_states_99_cast_fp16, x = coreml_update_state_255)[name = string("value_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_17_cast_fp16, input = value_cache)[name = string("coreml_update_state_257_write_state")]; tensor coreml_update_state_257 = read_state(input = value_cache)[name = string("coreml_update_state_257")]; tensor var_6221_begin_0 = const()[name = string("op_6221_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6221_end_0 = const()[name = string("op_6221_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6221_end_mask_0 = const()[name = string("op_6221_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6221_cast_fp16 = slice_by_index(begin = var_6221_begin_0, end = var_6221_end_0, end_mask = var_6221_end_mask_0, x = coreml_update_state_256)[name = string("op_6221_cast_fp16")]; tensor tile_32 = const()[name = string("tile_32"), val = tensor([1, 1])]; int32 var_6224_axis_0 = const()[name = string("op_6224_axis_0"), val = int32(1)]; tensor var_6224_cast_fp16_0, tensor var_6224_cast_fp16_1 = split(axis = var_6224_axis_0, split_sizes = tile_32, x = var_6221_cast_fp16)[name = string("op_6224_cast_fp16")]; tensor var_6231_begin_0 = const()[name = string("op_6231_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6231_end_0 = const()[name = string("op_6231_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6231_end_mask_0 = const()[name = string("op_6231_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6231_cast_fp16 = slice_by_index(begin = var_6231_begin_0, end = var_6231_end_0, end_mask = var_6231_end_mask_0, x = coreml_update_state_257)[name = string("op_6231_cast_fp16")]; tensor tile_33 = const()[name = string("tile_33"), val = tensor([1, 1])]; int32 var_6234_axis_0 = const()[name = string("op_6234_axis_0"), val = int32(1)]; tensor var_6234_cast_fp16_0, tensor var_6234_cast_fp16_1 = split(axis = var_6234_axis_0, split_sizes = tile_33, x = var_6231_cast_fp16)[name = string("op_6234_cast_fp16")]; tensor var_6237_split_sizes_0 = const()[name = string("op_6237_split_sizes_0"), val = tensor([8, 8])]; int32 var_6237_axis_0 = const()[name = string("op_6237_axis_0"), val = int32(1)]; tensor var_6237_0, tensor var_6237_1 = split(axis = var_6237_axis_0, split_sizes = var_6237_split_sizes_0, x = query_states_99_cast_fp16)[name = string("op_6237")]; bool attn_weights_257_transpose_x_0 = const()[name = string("attn_weights_257_transpose_x_0"), val = bool(false)]; bool attn_weights_257_transpose_y_0 = const()[name = string("attn_weights_257_transpose_y_0"), val = bool(false)]; tensor attn_weights_257_cast_fp16 = matmul(transpose_x = attn_weights_257_transpose_x_0, transpose_y = attn_weights_257_transpose_y_0, x = var_6224_cast_fp16_0, y = var_6237_0)[name = string("attn_weights_257_cast_fp16")]; fp16 var_6240_to_fp16 = const()[name = string("op_6240_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_259_cast_fp16 = mul(x = attn_weights_257_cast_fp16, y = var_6240_to_fp16)[name = string("attn_weights_259_cast_fp16")]; tensor attn_weights_261_cast_fp16 = add(x = attn_weights_259_cast_fp16, y = attn_mask_1)[name = string("attn_weights_261_cast_fp16")]; int32 var_6244 = const()[name = string("op_6244"), val = int32(-2)]; tensor attn_weights_263_cast_fp16 = softmax(axis = var_6244, x = attn_weights_261_cast_fp16)[name = string("attn_weights_263_cast_fp16")]; bool var_6250_transpose_x_1 = const()[name = string("op_6250_transpose_x_1"), val = bool(true)]; bool var_6250_transpose_y_1 = const()[name = string("op_6250_transpose_y_1"), val = bool(false)]; tensor var_6250_cast_fp16 = matmul(transpose_x = var_6250_transpose_x_1, transpose_y = var_6250_transpose_y_1, x = attn_weights_263_cast_fp16, y = var_6234_cast_fp16_0)[name = string("op_6250_cast_fp16")]; bool attn_weights_265_transpose_x_0 = const()[name = string("attn_weights_265_transpose_x_0"), val = bool(false)]; bool attn_weights_265_transpose_y_0 = const()[name = string("attn_weights_265_transpose_y_0"), val = bool(false)]; tensor attn_weights_265_cast_fp16 = matmul(transpose_x = attn_weights_265_transpose_x_0, transpose_y = attn_weights_265_transpose_y_0, x = var_6224_cast_fp16_1, y = var_6237_1)[name = string("attn_weights_265_cast_fp16")]; fp16 var_6252_to_fp16 = const()[name = string("op_6252_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_267_cast_fp16 = mul(x = attn_weights_265_cast_fp16, y = var_6252_to_fp16)[name = string("attn_weights_267_cast_fp16")]; tensor attn_weights_269_cast_fp16 = add(x = attn_weights_267_cast_fp16, y = attn_mask_1)[name = string("attn_weights_269_cast_fp16")]; int32 var_6256 = const()[name = string("op_6256"), val = int32(-2)]; tensor attn_weights_271_cast_fp16 = softmax(axis = var_6256, x = attn_weights_269_cast_fp16)[name = string("attn_weights_271_cast_fp16")]; bool attn_output_129_transpose_x_1 = const()[name = string("attn_output_129_transpose_x_1"), val = bool(true)]; bool attn_output_129_transpose_y_1 = const()[name = string("attn_output_129_transpose_y_1"), val = bool(false)]; tensor attn_output_129_cast_fp16 = matmul(transpose_x = attn_output_129_transpose_x_1, transpose_y = attn_output_129_transpose_y_1, x = attn_weights_271_cast_fp16, y = var_6234_cast_fp16_1)[name = string("attn_output_129_cast_fp16")]; int32 var_6264 = const()[name = string("op_6264"), val = int32(1)]; bool attn_output_131_interleave_0 = const()[name = string("attn_output_131_interleave_0"), val = bool(false)]; tensor attn_output_131_cast_fp16 = concat(axis = var_6264, interleave = attn_output_131_interleave_0, values = (var_6250_cast_fp16, attn_output_129_cast_fp16))[name = string("attn_output_131_cast_fp16")]; tensor var_6268_perm_0 = const()[name = string("op_6268_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_203x = const()[name = string("concat_203x"), val = tensor([1, 2048, 1, -1])]; tensor var_6268_cast_fp16 = transpose(perm = var_6268_perm_0, x = attn_output_131_cast_fp16)[name = string("transpose_377")]; tensor attn_output_135_cast_fp16 = reshape(shape = concat_203x, x = var_6268_cast_fp16)[name = string("attn_output_135_cast_fp16")]; tensor hidden_states_163_strides_0 = const()[name = string("hidden_states_163_strides_0"), val = tensor([1, 1])]; string hidden_states_163_pad_type_0 = const()[name = string("hidden_states_163_pad_type_0"), val = string("valid")]; tensor hidden_states_163_pad_0 = const()[name = string("hidden_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_163_dilations_0 = const()[name = string("hidden_states_163_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_163_groups_0 = const()[name = string("hidden_states_163_groups_0"), val = int32(1)]; tensor hidden_states_163_cast_fp16 = conv(dilations = hidden_states_163_dilations_0, groups = hidden_states_163_groups_0, pad = hidden_states_163_pad_0, pad_type = hidden_states_163_pad_type_0, strides = hidden_states_163_strides_0, weight = layers_16_self_attn_o_proj_weight_cast_fp16, x = attn_output_135_cast_fp16)[name = string("hidden_states_163_cast_fp16")]; tensor hidden_states_165_cast_fp16 = add(x = hidden_states_159_cast_fp16, y = hidden_states_163_cast_fp16)[name = string("hidden_states_165_cast_fp16")]; fp16 const_168_promoted_to_fp16 = const()[name = string("const_168_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6301_cast_fp16 = mul(x = hidden_states_165_cast_fp16, y = const_168_promoted_to_fp16)[name = string("op_6301_cast_fp16")]; int32 var_6299 = const()[name = string("op_6299"), val = int32(1)]; bool doubled_133_interleave_0 = const()[name = string("doubled_133_interleave_0"), val = bool(false)]; tensor doubled_133_cast_fp16 = concat(axis = var_6299, interleave = doubled_133_interleave_0, values = (hidden_states_165_cast_fp16, var_6301_cast_fp16))[name = string("doubled_133_cast_fp16")]; tensor out_67_axes_0 = const()[name = string("out_67_axes_0"), val = tensor([1])]; tensor out_67_gamma_0_to_fp16 = const()[name = string("out_67_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441663488)))]; fp16 var_6311_to_fp16 = const()[name = string("op_6311_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_67_cast_fp16 = layer_norm(axes = out_67_axes_0, epsilon = var_6311_to_fp16, gamma = out_67_gamma_0_to_fp16, x = doubled_133_cast_fp16)[name = string("out_67_cast_fp16")]; tensor var_6322_split_sizes_0 = const()[name = string("op_6322_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6322_axis_0 = const()[name = string("op_6322_axis_0"), val = int32(1)]; tensor var_6322_cast_fp16_0, tensor var_6322_cast_fp16_1 = split(axis = var_6322_axis_0, split_sizes = var_6322_split_sizes_0, x = out_67_cast_fp16)[name = string("op_6322_cast_fp16")]; tensor layers_16_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441671744)))]; tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; tensor input_33_cast_fp16 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = layers_16_mlp_gate_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("input_33_cast_fp16")]; tensor var_6339_cast_fp16 = silu(x = input_33_cast_fp16)[name = string("op_6339_cast_fp16")]; tensor layers_16_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1466837632)))]; tensor var_6345_strides_0 = const()[name = string("op_6345_strides_0"), val = tensor([1, 1])]; string var_6345_pad_type_0 = const()[name = string("op_6345_pad_type_0"), val = string("valid")]; tensor var_6345_pad_0 = const()[name = string("op_6345_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6345_dilations_0 = const()[name = string("op_6345_dilations_0"), val = tensor([1, 1])]; int32 var_6345_groups_0 = const()[name = string("op_6345_groups_0"), val = int32(1)]; tensor var_6345_cast_fp16 = conv(dilations = var_6345_dilations_0, groups = var_6345_groups_0, pad = var_6345_pad_0, pad_type = var_6345_pad_type_0, strides = var_6345_strides_0, weight = layers_16_mlp_up_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("op_6345_cast_fp16")]; tensor x_169_cast_fp16 = mul(x = var_6339_cast_fp16, y = var_6345_cast_fp16)[name = string("x_169_cast_fp16")]; tensor hidden_states_167_strides_0 = const()[name = string("hidden_states_167_strides_0"), val = tensor([1, 1])]; string hidden_states_167_pad_type_0 = const()[name = string("hidden_states_167_pad_type_0"), val = string("valid")]; tensor hidden_states_167_pad_0 = const()[name = string("hidden_states_167_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_167_dilations_0 = const()[name = string("hidden_states_167_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_167_groups_0 = const()[name = string("hidden_states_167_groups_0"), val = int32(1)]; tensor hidden_states_167_cast_fp16 = conv(dilations = hidden_states_167_dilations_0, groups = hidden_states_167_groups_0, pad = hidden_states_167_pad_0, pad_type = hidden_states_167_pad_type_0, strides = hidden_states_167_strides_0, weight = layers_16_mlp_down_proj_weight_cast_fp16, x = x_169_cast_fp16)[name = string("hidden_states_167_cast_fp16")]; tensor hidden_states_169_cast_fp16 = add(x = hidden_states_165_cast_fp16, y = hidden_states_167_cast_fp16)[name = string("hidden_states_169_cast_fp16")]; fp16 const_170_promoted_to_fp16 = const()[name = string("const_170_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6363_cast_fp16 = mul(x = hidden_states_169_cast_fp16, y = const_170_promoted_to_fp16)[name = string("op_6363_cast_fp16")]; int32 var_6361 = const()[name = string("op_6361"), val = int32(1)]; bool doubled_137_interleave_0 = const()[name = string("doubled_137_interleave_0"), val = bool(false)]; tensor doubled_137_cast_fp16 = concat(axis = var_6361, interleave = doubled_137_interleave_0, values = (hidden_states_169_cast_fp16, var_6363_cast_fp16))[name = string("doubled_137_cast_fp16")]; tensor out_69_axes_0 = const()[name = string("out_69_axes_0"), val = tensor([1])]; tensor out_69_gamma_0_to_fp16 = const()[name = string("out_69_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492003520)))]; fp16 var_6373_to_fp16 = const()[name = string("op_6373_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_69_cast_fp16 = layer_norm(axes = out_69_axes_0, epsilon = var_6373_to_fp16, gamma = out_69_gamma_0_to_fp16, x = doubled_137_cast_fp16)[name = string("out_69_cast_fp16")]; tensor var_6384_split_sizes_0 = const()[name = string("op_6384_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6384_axis_0 = const()[name = string("op_6384_axis_0"), val = int32(1)]; tensor var_6384_cast_fp16_0, tensor var_6384_cast_fp16_1 = split(axis = var_6384_axis_0, split_sizes = var_6384_split_sizes_0, x = out_69_cast_fp16)[name = string("op_6384_cast_fp16")]; tensor query_states_103_strides_0 = const()[name = string("query_states_103_strides_0"), val = tensor([1, 1])]; string query_states_103_pad_type_0 = const()[name = string("query_states_103_pad_type_0"), val = string("valid")]; tensor query_states_103_pad_0 = const()[name = string("query_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_103_dilations_0 = const()[name = string("query_states_103_dilations_0"), val = tensor([1, 1])]; int32 query_states_103_groups_0 = const()[name = string("query_states_103_groups_0"), val = int32(1)]; tensor query_states_103_cast_fp16 = conv(dilations = query_states_103_dilations_0, groups = query_states_103_groups_0, pad = query_states_103_pad_0, pad_type = query_states_103_pad_type_0, strides = query_states_103_strides_0, weight = layers_17_self_attn_q_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("query_states_103_cast_fp16")]; tensor key_states_171_strides_0 = const()[name = string("key_states_171_strides_0"), val = tensor([1, 1])]; string key_states_171_pad_type_0 = const()[name = string("key_states_171_pad_type_0"), val = string("valid")]; tensor key_states_171_pad_0 = const()[name = string("key_states_171_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_171_dilations_0 = const()[name = string("key_states_171_dilations_0"), val = tensor([1, 1])]; int32 key_states_171_groups_0 = const()[name = string("key_states_171_groups_0"), val = int32(1)]; tensor key_states_171_cast_fp16 = conv(dilations = key_states_171_dilations_0, groups = key_states_171_groups_0, pad = key_states_171_pad_0, pad_type = key_states_171_pad_type_0, strides = key_states_171_strides_0, weight = layers_17_self_attn_k_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("key_states_171_cast_fp16")]; tensor value_states_103_strides_0 = const()[name = string("value_states_103_strides_0"), val = tensor([1, 1])]; string value_states_103_pad_type_0 = const()[name = string("value_states_103_pad_type_0"), val = string("valid")]; tensor value_states_103_pad_0 = const()[name = string("value_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_103_dilations_0 = const()[name = string("value_states_103_dilations_0"), val = tensor([1, 1])]; int32 value_states_103_groups_0 = const()[name = string("value_states_103_groups_0"), val = int32(1)]; tensor value_states_103_cast_fp16 = conv(dilations = value_states_103_dilations_0, groups = value_states_103_groups_0, pad = value_states_103_pad_0, pad_type = value_states_103_pad_type_0, strides = value_states_103_strides_0, weight = layers_17_self_attn_v_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("value_states_103_cast_fp16")]; tensor concat_204x = const()[name = string("concat_204x"), val = tensor([1, 16, 128, -1])]; tensor x_171_cast_fp16 = reshape(shape = concat_204x, x = query_states_103_cast_fp16)[name = string("x_171_cast_fp16")]; tensor concat_205x = const()[name = string("concat_205x"), val = tensor([1, 2, 128, -1])]; tensor var_6441_cast_fp16 = reshape(shape = concat_205x, x = key_states_171_cast_fp16)[name = string("op_6441_cast_fp16")]; tensor concat_206x = const()[name = string("concat_206x"), val = tensor([1, 2, 128, -1])]; tensor var_6448_cast_fp16 = reshape(shape = concat_206x, x = value_states_103_cast_fp16)[name = string("op_6448_cast_fp16")]; tensor var_6452_cast_fp16 = mul(x = x_171_cast_fp16, y = var_869_cast_fp16)[name = string("op_6452_cast_fp16")]; tensor var_6453_split_sizes_0 = const()[name = string("op_6453_split_sizes_0"), val = tensor([64, 64])]; int32 var_6453_axis_0 = const()[name = string("op_6453_axis_0"), val = int32(-2)]; tensor var_6453_cast_fp16_0, tensor var_6453_cast_fp16_1 = split(axis = var_6453_axis_0, split_sizes = var_6453_split_sizes_0, x = x_171_cast_fp16)[name = string("op_6453_cast_fp16")]; fp16 const_172_promoted_to_fp16 = const()[name = string("const_172_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6455_cast_fp16 = mul(x = var_6453_cast_fp16_1, y = const_172_promoted_to_fp16)[name = string("op_6455_cast_fp16")]; int32 var_6457 = const()[name = string("op_6457"), val = int32(-2)]; bool var_6458_interleave_0 = const()[name = string("op_6458_interleave_0"), val = bool(false)]; tensor var_6458_cast_fp16 = concat(axis = var_6457, interleave = var_6458_interleave_0, values = (var_6455_cast_fp16, var_6453_cast_fp16_0))[name = string("op_6458_cast_fp16")]; tensor var_6459_cast_fp16 = mul(x = var_6458_cast_fp16, y = var_878_cast_fp16)[name = string("op_6459_cast_fp16")]; tensor query_states_105_cast_fp16 = add(x = var_6452_cast_fp16, y = var_6459_cast_fp16)[name = string("query_states_105_cast_fp16")]; tensor var_6465_cast_fp16 = mul(x = var_6441_cast_fp16, y = var_869_cast_fp16)[name = string("op_6465_cast_fp16")]; tensor var_6466_split_sizes_0 = const()[name = string("op_6466_split_sizes_0"), val = tensor([64, 64])]; int32 var_6466_axis_0 = const()[name = string("op_6466_axis_0"), val = int32(-2)]; tensor var_6466_cast_fp16_0, tensor var_6466_cast_fp16_1 = split(axis = var_6466_axis_0, split_sizes = var_6466_split_sizes_0, x = var_6441_cast_fp16)[name = string("op_6466_cast_fp16")]; fp16 const_173_promoted_to_fp16 = const()[name = string("const_173_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6468_cast_fp16 = mul(x = var_6466_cast_fp16_1, y = const_173_promoted_to_fp16)[name = string("op_6468_cast_fp16")]; int32 var_6470 = const()[name = string("op_6470"), val = int32(-2)]; bool var_6471_interleave_0 = const()[name = string("op_6471_interleave_0"), val = bool(false)]; tensor var_6471_cast_fp16 = concat(axis = var_6470, interleave = var_6471_interleave_0, values = (var_6468_cast_fp16, var_6466_cast_fp16_0))[name = string("op_6471_cast_fp16")]; tensor var_6472_cast_fp16 = mul(x = var_6471_cast_fp16, y = var_878_cast_fp16)[name = string("op_6472_cast_fp16")]; tensor key_states_175_cast_fp16 = add(x = var_6465_cast_fp16, y = var_6472_cast_fp16)[name = string("key_states_175_cast_fp16")]; tensor expand_dims_204 = const()[name = string("expand_dims_204"), val = tensor([17])]; tensor expand_dims_205 = const()[name = string("expand_dims_205"), val = tensor([0])]; tensor expand_dims_207 = const()[name = string("expand_dims_207"), val = tensor([0])]; int32 concat_209_axis_0 = const()[name = string("concat_209_axis_0"), val = int32(0)]; bool concat_209_interleave_0 = const()[name = string("concat_209_interleave_0"), val = bool(false)]; tensor concat_209 = concat(axis = concat_209_axis_0, interleave = concat_209_interleave_0, values = (expand_dims_204, expand_dims_205, position_id, expand_dims_207))[name = string("concat_209")]; tensor expand_dims_208 = const()[name = string("expand_dims_208"), val = tensor([18])]; tensor concat_210_values1_0 = const()[name = string("concat_210_values1_0"), val = tensor([0])]; tensor concat_210_values3_0 = const()[name = string("concat_210_values3_0"), val = tensor([0])]; int32 concat_210_axis_0 = const()[name = string("concat_210_axis_0"), val = int32(0)]; bool concat_210_interleave_0 = const()[name = string("concat_210_interleave_0"), val = bool(false)]; tensor concat_210 = concat(axis = concat_210_axis_0, interleave = concat_210_interleave_0, values = (expand_dims_208, concat_210_values1_0, cache_position_end, concat_210_values3_0))[name = string("concat_210")]; tensor key_states_177_perm_0 = const()[name = string("key_states_177_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_18_stride_0 = const()[name = string("key_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_177_cast_fp16 = transpose(perm = key_states_177_perm_0, x = key_states_175_cast_fp16)[name = string("transpose_376")]; tensor key_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = key_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = key_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_18_squeeze_mask_0, stride = key_cache_internal_tensor_assign_18_stride_0, update = key_states_177_cast_fp16, x = coreml_update_state_256)[name = string("key_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_18_cast_fp16, input = key_cache)[name = string("coreml_update_state_258_write_state")]; tensor coreml_update_state_258 = read_state(input = key_cache)[name = string("coreml_update_state_258")]; tensor value_states_105_perm_0 = const()[name = string("value_states_105_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_18_stride_0 = const()[name = string("value_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_105_cast_fp16 = transpose(perm = value_states_105_perm_0, x = var_6448_cast_fp16)[name = string("transpose_375")]; tensor value_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = value_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = value_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_18_squeeze_mask_0, stride = value_cache_internal_tensor_assign_18_stride_0, update = value_states_105_cast_fp16, x = coreml_update_state_257)[name = string("value_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_18_cast_fp16, input = value_cache)[name = string("coreml_update_state_259_write_state")]; tensor coreml_update_state_259 = read_state(input = value_cache)[name = string("coreml_update_state_259")]; tensor var_6542_begin_0 = const()[name = string("op_6542_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6542_end_0 = const()[name = string("op_6542_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6542_end_mask_0 = const()[name = string("op_6542_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6542_cast_fp16 = slice_by_index(begin = var_6542_begin_0, end = var_6542_end_0, end_mask = var_6542_end_mask_0, x = coreml_update_state_258)[name = string("op_6542_cast_fp16")]; tensor tile_34 = const()[name = string("tile_34"), val = tensor([1, 1])]; int32 var_6545_axis_0 = const()[name = string("op_6545_axis_0"), val = int32(1)]; tensor var_6545_cast_fp16_0, tensor var_6545_cast_fp16_1 = split(axis = var_6545_axis_0, split_sizes = tile_34, x = var_6542_cast_fp16)[name = string("op_6545_cast_fp16")]; tensor var_6552_begin_0 = const()[name = string("op_6552_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6552_end_0 = const()[name = string("op_6552_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6552_end_mask_0 = const()[name = string("op_6552_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6552_cast_fp16 = slice_by_index(begin = var_6552_begin_0, end = var_6552_end_0, end_mask = var_6552_end_mask_0, x = coreml_update_state_259)[name = string("op_6552_cast_fp16")]; tensor tile_35 = const()[name = string("tile_35"), val = tensor([1, 1])]; int32 var_6555_axis_0 = const()[name = string("op_6555_axis_0"), val = int32(1)]; tensor var_6555_cast_fp16_0, tensor var_6555_cast_fp16_1 = split(axis = var_6555_axis_0, split_sizes = tile_35, x = var_6552_cast_fp16)[name = string("op_6555_cast_fp16")]; tensor var_6558_split_sizes_0 = const()[name = string("op_6558_split_sizes_0"), val = tensor([8, 8])]; int32 var_6558_axis_0 = const()[name = string("op_6558_axis_0"), val = int32(1)]; tensor var_6558_0, tensor var_6558_1 = split(axis = var_6558_axis_0, split_sizes = var_6558_split_sizes_0, x = query_states_105_cast_fp16)[name = string("op_6558")]; bool attn_weights_273_transpose_x_0 = const()[name = string("attn_weights_273_transpose_x_0"), val = bool(false)]; bool attn_weights_273_transpose_y_0 = const()[name = string("attn_weights_273_transpose_y_0"), val = bool(false)]; tensor attn_weights_273_cast_fp16 = matmul(transpose_x = attn_weights_273_transpose_x_0, transpose_y = attn_weights_273_transpose_y_0, x = var_6545_cast_fp16_0, y = var_6558_0)[name = string("attn_weights_273_cast_fp16")]; fp16 var_6561_to_fp16 = const()[name = string("op_6561_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_275_cast_fp16 = mul(x = attn_weights_273_cast_fp16, y = var_6561_to_fp16)[name = string("attn_weights_275_cast_fp16")]; tensor attn_weights_277_cast_fp16 = add(x = attn_weights_275_cast_fp16, y = attn_mask_1)[name = string("attn_weights_277_cast_fp16")]; int32 var_6565 = const()[name = string("op_6565"), val = int32(-2)]; tensor attn_weights_279_cast_fp16 = softmax(axis = var_6565, x = attn_weights_277_cast_fp16)[name = string("attn_weights_279_cast_fp16")]; bool var_6571_transpose_x_1 = const()[name = string("op_6571_transpose_x_1"), val = bool(true)]; bool var_6571_transpose_y_1 = const()[name = string("op_6571_transpose_y_1"), val = bool(false)]; tensor var_6571_cast_fp16 = matmul(transpose_x = var_6571_transpose_x_1, transpose_y = var_6571_transpose_y_1, x = attn_weights_279_cast_fp16, y = var_6555_cast_fp16_0)[name = string("op_6571_cast_fp16")]; bool attn_weights_281_transpose_x_0 = const()[name = string("attn_weights_281_transpose_x_0"), val = bool(false)]; bool attn_weights_281_transpose_y_0 = const()[name = string("attn_weights_281_transpose_y_0"), val = bool(false)]; tensor attn_weights_281_cast_fp16 = matmul(transpose_x = attn_weights_281_transpose_x_0, transpose_y = attn_weights_281_transpose_y_0, x = var_6545_cast_fp16_1, y = var_6558_1)[name = string("attn_weights_281_cast_fp16")]; fp16 var_6573_to_fp16 = const()[name = string("op_6573_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_283_cast_fp16 = mul(x = attn_weights_281_cast_fp16, y = var_6573_to_fp16)[name = string("attn_weights_283_cast_fp16")]; tensor attn_weights_285_cast_fp16 = add(x = attn_weights_283_cast_fp16, y = attn_mask_1)[name = string("attn_weights_285_cast_fp16")]; int32 var_6577 = const()[name = string("op_6577"), val = int32(-2)]; tensor attn_weights_287_cast_fp16 = softmax(axis = var_6577, x = attn_weights_285_cast_fp16)[name = string("attn_weights_287_cast_fp16")]; bool attn_output_137_transpose_x_1 = const()[name = string("attn_output_137_transpose_x_1"), val = bool(true)]; bool attn_output_137_transpose_y_1 = const()[name = string("attn_output_137_transpose_y_1"), val = bool(false)]; tensor attn_output_137_cast_fp16 = matmul(transpose_x = attn_output_137_transpose_x_1, transpose_y = attn_output_137_transpose_y_1, x = attn_weights_287_cast_fp16, y = var_6555_cast_fp16_1)[name = string("attn_output_137_cast_fp16")]; int32 var_6585 = const()[name = string("op_6585"), val = int32(1)]; bool attn_output_139_interleave_0 = const()[name = string("attn_output_139_interleave_0"), val = bool(false)]; tensor attn_output_139_cast_fp16 = concat(axis = var_6585, interleave = attn_output_139_interleave_0, values = (var_6571_cast_fp16, attn_output_137_cast_fp16))[name = string("attn_output_139_cast_fp16")]; tensor var_6589_perm_0 = const()[name = string("op_6589_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_215x = const()[name = string("concat_215x"), val = tensor([1, 2048, 1, -1])]; tensor var_6589_cast_fp16 = transpose(perm = var_6589_perm_0, x = attn_output_139_cast_fp16)[name = string("transpose_374")]; tensor attn_output_143_cast_fp16 = reshape(shape = concat_215x, x = var_6589_cast_fp16)[name = string("attn_output_143_cast_fp16")]; tensor hidden_states_173_strides_0 = const()[name = string("hidden_states_173_strides_0"), val = tensor([1, 1])]; string hidden_states_173_pad_type_0 = const()[name = string("hidden_states_173_pad_type_0"), val = string("valid")]; tensor hidden_states_173_pad_0 = const()[name = string("hidden_states_173_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_173_dilations_0 = const()[name = string("hidden_states_173_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_173_groups_0 = const()[name = string("hidden_states_173_groups_0"), val = int32(1)]; tensor hidden_states_173_cast_fp16 = conv(dilations = hidden_states_173_dilations_0, groups = hidden_states_173_groups_0, pad = hidden_states_173_pad_0, pad_type = hidden_states_173_pad_type_0, strides = hidden_states_173_strides_0, weight = layers_17_self_attn_o_proj_weight_cast_fp16, x = attn_output_143_cast_fp16)[name = string("hidden_states_173_cast_fp16")]; tensor hidden_states_175_cast_fp16 = add(x = hidden_states_169_cast_fp16, y = hidden_states_173_cast_fp16)[name = string("hidden_states_175_cast_fp16")]; fp16 const_178_promoted_to_fp16 = const()[name = string("const_178_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6622_cast_fp16 = mul(x = hidden_states_175_cast_fp16, y = const_178_promoted_to_fp16)[name = string("op_6622_cast_fp16")]; int32 var_6620 = const()[name = string("op_6620"), val = int32(1)]; bool doubled_141_interleave_0 = const()[name = string("doubled_141_interleave_0"), val = bool(false)]; tensor doubled_141_cast_fp16 = concat(axis = var_6620, interleave = doubled_141_interleave_0, values = (hidden_states_175_cast_fp16, var_6622_cast_fp16))[name = string("doubled_141_cast_fp16")]; tensor out_71_axes_0 = const()[name = string("out_71_axes_0"), val = tensor([1])]; tensor out_71_gamma_0_to_fp16 = const()[name = string("out_71_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492011776)))]; fp16 var_6632_to_fp16 = const()[name = string("op_6632_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_71_cast_fp16 = layer_norm(axes = out_71_axes_0, epsilon = var_6632_to_fp16, gamma = out_71_gamma_0_to_fp16, x = doubled_141_cast_fp16)[name = string("out_71_cast_fp16")]; tensor var_6643_split_sizes_0 = const()[name = string("op_6643_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6643_axis_0 = const()[name = string("op_6643_axis_0"), val = int32(1)]; tensor var_6643_cast_fp16_0, tensor var_6643_cast_fp16_1 = split(axis = var_6643_axis_0, split_sizes = var_6643_split_sizes_0, x = out_71_cast_fp16)[name = string("op_6643_cast_fp16")]; tensor input_35_strides_0 = const()[name = string("input_35_strides_0"), val = tensor([1, 1])]; string input_35_pad_type_0 = const()[name = string("input_35_pad_type_0"), val = string("valid")]; tensor input_35_pad_0 = const()[name = string("input_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_35_dilations_0 = const()[name = string("input_35_dilations_0"), val = tensor([1, 1])]; int32 input_35_groups_0 = const()[name = string("input_35_groups_0"), val = int32(1)]; tensor input_35_cast_fp16 = conv(dilations = input_35_dilations_0, groups = input_35_groups_0, pad = input_35_pad_0, pad_type = input_35_pad_type_0, strides = input_35_strides_0, weight = layers_17_mlp_gate_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("input_35_cast_fp16")]; tensor var_6660_cast_fp16 = silu(x = input_35_cast_fp16)[name = string("op_6660_cast_fp16")]; tensor var_6666_strides_0 = const()[name = string("op_6666_strides_0"), val = tensor([1, 1])]; string var_6666_pad_type_0 = const()[name = string("op_6666_pad_type_0"), val = string("valid")]; tensor var_6666_pad_0 = const()[name = string("op_6666_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6666_dilations_0 = const()[name = string("op_6666_dilations_0"), val = tensor([1, 1])]; int32 var_6666_groups_0 = const()[name = string("op_6666_groups_0"), val = int32(1)]; tensor var_6666_cast_fp16 = conv(dilations = var_6666_dilations_0, groups = var_6666_groups_0, pad = var_6666_pad_0, pad_type = var_6666_pad_type_0, strides = var_6666_strides_0, weight = layers_17_mlp_up_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("op_6666_cast_fp16")]; tensor x_179_cast_fp16 = mul(x = var_6660_cast_fp16, y = var_6666_cast_fp16)[name = string("x_179_cast_fp16")]; tensor hidden_states_177_strides_0 = const()[name = string("hidden_states_177_strides_0"), val = tensor([1, 1])]; string hidden_states_177_pad_type_0 = const()[name = string("hidden_states_177_pad_type_0"), val = string("valid")]; tensor hidden_states_177_pad_0 = const()[name = string("hidden_states_177_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_177_dilations_0 = const()[name = string("hidden_states_177_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_177_groups_0 = const()[name = string("hidden_states_177_groups_0"), val = int32(1)]; tensor hidden_states_177_cast_fp16 = conv(dilations = hidden_states_177_dilations_0, groups = hidden_states_177_groups_0, pad = hidden_states_177_pad_0, pad_type = hidden_states_177_pad_type_0, strides = hidden_states_177_strides_0, weight = layers_17_mlp_down_proj_weight_cast_fp16, x = x_179_cast_fp16)[name = string("hidden_states_177_cast_fp16")]; tensor hidden_states_179_cast_fp16 = add(x = hidden_states_175_cast_fp16, y = hidden_states_177_cast_fp16)[name = string("hidden_states_179_cast_fp16")]; fp16 const_180_promoted_to_fp16 = const()[name = string("const_180_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6684_cast_fp16 = mul(x = hidden_states_179_cast_fp16, y = const_180_promoted_to_fp16)[name = string("op_6684_cast_fp16")]; int32 var_6682 = const()[name = string("op_6682"), val = int32(1)]; bool doubled_145_interleave_0 = const()[name = string("doubled_145_interleave_0"), val = bool(false)]; tensor doubled_145_cast_fp16 = concat(axis = var_6682, interleave = doubled_145_interleave_0, values = (hidden_states_179_cast_fp16, var_6684_cast_fp16))[name = string("doubled_145_cast_fp16")]; tensor out_73_axes_0 = const()[name = string("out_73_axes_0"), val = tensor([1])]; tensor out_73_gamma_0_to_fp16 = const()[name = string("out_73_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492020032)))]; fp16 var_6694_to_fp16 = const()[name = string("op_6694_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_73_cast_fp16 = layer_norm(axes = out_73_axes_0, epsilon = var_6694_to_fp16, gamma = out_73_gamma_0_to_fp16, x = doubled_145_cast_fp16)[name = string("out_73_cast_fp16")]; tensor var_6705_split_sizes_0 = const()[name = string("op_6705_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6705_axis_0 = const()[name = string("op_6705_axis_0"), val = int32(1)]; tensor var_6705_cast_fp16_0, tensor var_6705_cast_fp16_1 = split(axis = var_6705_axis_0, split_sizes = var_6705_split_sizes_0, x = out_73_cast_fp16)[name = string("op_6705_cast_fp16")]; tensor query_states_109_strides_0 = const()[name = string("query_states_109_strides_0"), val = tensor([1, 1])]; string query_states_109_pad_type_0 = const()[name = string("query_states_109_pad_type_0"), val = string("valid")]; tensor query_states_109_pad_0 = const()[name = string("query_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_109_dilations_0 = const()[name = string("query_states_109_dilations_0"), val = tensor([1, 1])]; int32 query_states_109_groups_0 = const()[name = string("query_states_109_groups_0"), val = int32(1)]; tensor query_states_109_cast_fp16 = conv(dilations = query_states_109_dilations_0, groups = query_states_109_groups_0, pad = query_states_109_pad_0, pad_type = query_states_109_pad_type_0, strides = query_states_109_strides_0, weight = layers_18_self_attn_q_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("query_states_109_cast_fp16")]; tensor key_states_181_strides_0 = const()[name = string("key_states_181_strides_0"), val = tensor([1, 1])]; string key_states_181_pad_type_0 = const()[name = string("key_states_181_pad_type_0"), val = string("valid")]; tensor key_states_181_pad_0 = const()[name = string("key_states_181_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_181_dilations_0 = const()[name = string("key_states_181_dilations_0"), val = tensor([1, 1])]; int32 key_states_181_groups_0 = const()[name = string("key_states_181_groups_0"), val = int32(1)]; tensor key_states_181_cast_fp16 = conv(dilations = key_states_181_dilations_0, groups = key_states_181_groups_0, pad = key_states_181_pad_0, pad_type = key_states_181_pad_type_0, strides = key_states_181_strides_0, weight = layers_18_self_attn_k_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("key_states_181_cast_fp16")]; tensor value_states_109_strides_0 = const()[name = string("value_states_109_strides_0"), val = tensor([1, 1])]; string value_states_109_pad_type_0 = const()[name = string("value_states_109_pad_type_0"), val = string("valid")]; tensor value_states_109_pad_0 = const()[name = string("value_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_109_dilations_0 = const()[name = string("value_states_109_dilations_0"), val = tensor([1, 1])]; int32 value_states_109_groups_0 = const()[name = string("value_states_109_groups_0"), val = int32(1)]; tensor value_states_109_cast_fp16 = conv(dilations = value_states_109_dilations_0, groups = value_states_109_groups_0, pad = value_states_109_pad_0, pad_type = value_states_109_pad_type_0, strides = value_states_109_strides_0, weight = layers_18_self_attn_v_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("value_states_109_cast_fp16")]; tensor concat_216x = const()[name = string("concat_216x"), val = tensor([1, 16, 128, -1])]; tensor x_181_cast_fp16 = reshape(shape = concat_216x, x = query_states_109_cast_fp16)[name = string("x_181_cast_fp16")]; tensor concat_217x = const()[name = string("concat_217x"), val = tensor([1, 2, 128, -1])]; tensor var_6762_cast_fp16 = reshape(shape = concat_217x, x = key_states_181_cast_fp16)[name = string("op_6762_cast_fp16")]; tensor concat_218x = const()[name = string("concat_218x"), val = tensor([1, 2, 128, -1])]; tensor var_6769_cast_fp16 = reshape(shape = concat_218x, x = value_states_109_cast_fp16)[name = string("op_6769_cast_fp16")]; tensor var_6773_cast_fp16 = mul(x = x_181_cast_fp16, y = var_869_cast_fp16)[name = string("op_6773_cast_fp16")]; tensor var_6774_split_sizes_0 = const()[name = string("op_6774_split_sizes_0"), val = tensor([64, 64])]; int32 var_6774_axis_0 = const()[name = string("op_6774_axis_0"), val = int32(-2)]; tensor var_6774_cast_fp16_0, tensor var_6774_cast_fp16_1 = split(axis = var_6774_axis_0, split_sizes = var_6774_split_sizes_0, x = x_181_cast_fp16)[name = string("op_6774_cast_fp16")]; fp16 const_182_promoted_to_fp16 = const()[name = string("const_182_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6776_cast_fp16 = mul(x = var_6774_cast_fp16_1, y = const_182_promoted_to_fp16)[name = string("op_6776_cast_fp16")]; int32 var_6778 = const()[name = string("op_6778"), val = int32(-2)]; bool var_6779_interleave_0 = const()[name = string("op_6779_interleave_0"), val = bool(false)]; tensor var_6779_cast_fp16 = concat(axis = var_6778, interleave = var_6779_interleave_0, values = (var_6776_cast_fp16, var_6774_cast_fp16_0))[name = string("op_6779_cast_fp16")]; tensor var_6780_cast_fp16 = mul(x = var_6779_cast_fp16, y = var_878_cast_fp16)[name = string("op_6780_cast_fp16")]; tensor query_states_111_cast_fp16 = add(x = var_6773_cast_fp16, y = var_6780_cast_fp16)[name = string("query_states_111_cast_fp16")]; tensor var_6786_cast_fp16 = mul(x = var_6762_cast_fp16, y = var_869_cast_fp16)[name = string("op_6786_cast_fp16")]; tensor var_6787_split_sizes_0 = const()[name = string("op_6787_split_sizes_0"), val = tensor([64, 64])]; int32 var_6787_axis_0 = const()[name = string("op_6787_axis_0"), val = int32(-2)]; tensor var_6787_cast_fp16_0, tensor var_6787_cast_fp16_1 = split(axis = var_6787_axis_0, split_sizes = var_6787_split_sizes_0, x = var_6762_cast_fp16)[name = string("op_6787_cast_fp16")]; fp16 const_183_promoted_to_fp16 = const()[name = string("const_183_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6789_cast_fp16 = mul(x = var_6787_cast_fp16_1, y = const_183_promoted_to_fp16)[name = string("op_6789_cast_fp16")]; int32 var_6791 = const()[name = string("op_6791"), val = int32(-2)]; bool var_6792_interleave_0 = const()[name = string("op_6792_interleave_0"), val = bool(false)]; tensor var_6792_cast_fp16 = concat(axis = var_6791, interleave = var_6792_interleave_0, values = (var_6789_cast_fp16, var_6787_cast_fp16_0))[name = string("op_6792_cast_fp16")]; tensor var_6793_cast_fp16 = mul(x = var_6792_cast_fp16, y = var_878_cast_fp16)[name = string("op_6793_cast_fp16")]; tensor key_states_185_cast_fp16 = add(x = var_6786_cast_fp16, y = var_6793_cast_fp16)[name = string("key_states_185_cast_fp16")]; tensor expand_dims_216 = const()[name = string("expand_dims_216"), val = tensor([18])]; tensor expand_dims_217 = const()[name = string("expand_dims_217"), val = tensor([0])]; tensor expand_dims_219 = const()[name = string("expand_dims_219"), val = tensor([0])]; int32 concat_221_axis_0 = const()[name = string("concat_221_axis_0"), val = int32(0)]; bool concat_221_interleave_0 = const()[name = string("concat_221_interleave_0"), val = bool(false)]; tensor concat_221 = concat(axis = concat_221_axis_0, interleave = concat_221_interleave_0, values = (expand_dims_216, expand_dims_217, position_id, expand_dims_219))[name = string("concat_221")]; tensor expand_dims_220 = const()[name = string("expand_dims_220"), val = tensor([19])]; tensor concat_222_values1_0 = const()[name = string("concat_222_values1_0"), val = tensor([0])]; tensor concat_222_values3_0 = const()[name = string("concat_222_values3_0"), val = tensor([0])]; int32 concat_222_axis_0 = const()[name = string("concat_222_axis_0"), val = int32(0)]; bool concat_222_interleave_0 = const()[name = string("concat_222_interleave_0"), val = bool(false)]; tensor concat_222 = concat(axis = concat_222_axis_0, interleave = concat_222_interleave_0, values = (expand_dims_220, concat_222_values1_0, cache_position_end, concat_222_values3_0))[name = string("concat_222")]; tensor key_states_187_perm_0 = const()[name = string("key_states_187_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_19_stride_0 = const()[name = string("key_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_187_cast_fp16 = transpose(perm = key_states_187_perm_0, x = key_states_185_cast_fp16)[name = string("transpose_373")]; tensor key_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = key_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = key_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_19_squeeze_mask_0, stride = key_cache_internal_tensor_assign_19_stride_0, update = key_states_187_cast_fp16, x = coreml_update_state_258)[name = string("key_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_19_cast_fp16, input = key_cache)[name = string("coreml_update_state_260_write_state")]; tensor coreml_update_state_260 = read_state(input = key_cache)[name = string("coreml_update_state_260")]; tensor value_states_111_perm_0 = const()[name = string("value_states_111_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_19_stride_0 = const()[name = string("value_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_111_cast_fp16 = transpose(perm = value_states_111_perm_0, x = var_6769_cast_fp16)[name = string("transpose_372")]; tensor value_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = value_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = value_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_19_squeeze_mask_0, stride = value_cache_internal_tensor_assign_19_stride_0, update = value_states_111_cast_fp16, x = coreml_update_state_259)[name = string("value_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_19_cast_fp16, input = value_cache)[name = string("coreml_update_state_261_write_state")]; tensor coreml_update_state_261 = read_state(input = value_cache)[name = string("coreml_update_state_261")]; tensor var_6863_begin_0 = const()[name = string("op_6863_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6863_end_0 = const()[name = string("op_6863_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6863_end_mask_0 = const()[name = string("op_6863_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6863_cast_fp16 = slice_by_index(begin = var_6863_begin_0, end = var_6863_end_0, end_mask = var_6863_end_mask_0, x = coreml_update_state_260)[name = string("op_6863_cast_fp16")]; tensor tile_36 = const()[name = string("tile_36"), val = tensor([1, 1])]; int32 var_6866_axis_0 = const()[name = string("op_6866_axis_0"), val = int32(1)]; tensor var_6866_cast_fp16_0, tensor var_6866_cast_fp16_1 = split(axis = var_6866_axis_0, split_sizes = tile_36, x = var_6863_cast_fp16)[name = string("op_6866_cast_fp16")]; tensor var_6873_begin_0 = const()[name = string("op_6873_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6873_end_0 = const()[name = string("op_6873_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6873_end_mask_0 = const()[name = string("op_6873_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6873_cast_fp16 = slice_by_index(begin = var_6873_begin_0, end = var_6873_end_0, end_mask = var_6873_end_mask_0, x = coreml_update_state_261)[name = string("op_6873_cast_fp16")]; tensor tile_37 = const()[name = string("tile_37"), val = tensor([1, 1])]; int32 var_6876_axis_0 = const()[name = string("op_6876_axis_0"), val = int32(1)]; tensor var_6876_cast_fp16_0, tensor var_6876_cast_fp16_1 = split(axis = var_6876_axis_0, split_sizes = tile_37, x = var_6873_cast_fp16)[name = string("op_6876_cast_fp16")]; tensor var_6879_split_sizes_0 = const()[name = string("op_6879_split_sizes_0"), val = tensor([8, 8])]; int32 var_6879_axis_0 = const()[name = string("op_6879_axis_0"), val = int32(1)]; tensor var_6879_0, tensor var_6879_1 = split(axis = var_6879_axis_0, split_sizes = var_6879_split_sizes_0, x = query_states_111_cast_fp16)[name = string("op_6879")]; bool attn_weights_289_transpose_x_0 = const()[name = string("attn_weights_289_transpose_x_0"), val = bool(false)]; bool attn_weights_289_transpose_y_0 = const()[name = string("attn_weights_289_transpose_y_0"), val = bool(false)]; tensor attn_weights_289_cast_fp16 = matmul(transpose_x = attn_weights_289_transpose_x_0, transpose_y = attn_weights_289_transpose_y_0, x = var_6866_cast_fp16_0, y = var_6879_0)[name = string("attn_weights_289_cast_fp16")]; fp16 var_6882_to_fp16 = const()[name = string("op_6882_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_291_cast_fp16 = mul(x = attn_weights_289_cast_fp16, y = var_6882_to_fp16)[name = string("attn_weights_291_cast_fp16")]; tensor attn_weights_293_cast_fp16 = add(x = attn_weights_291_cast_fp16, y = attn_mask_1)[name = string("attn_weights_293_cast_fp16")]; int32 var_6886 = const()[name = string("op_6886"), val = int32(-2)]; tensor attn_weights_295_cast_fp16 = softmax(axis = var_6886, x = attn_weights_293_cast_fp16)[name = string("attn_weights_295_cast_fp16")]; bool var_6892_transpose_x_1 = const()[name = string("op_6892_transpose_x_1"), val = bool(true)]; bool var_6892_transpose_y_1 = const()[name = string("op_6892_transpose_y_1"), val = bool(false)]; tensor var_6892_cast_fp16 = matmul(transpose_x = var_6892_transpose_x_1, transpose_y = var_6892_transpose_y_1, x = attn_weights_295_cast_fp16, y = var_6876_cast_fp16_0)[name = string("op_6892_cast_fp16")]; bool attn_weights_297_transpose_x_0 = const()[name = string("attn_weights_297_transpose_x_0"), val = bool(false)]; bool attn_weights_297_transpose_y_0 = const()[name = string("attn_weights_297_transpose_y_0"), val = bool(false)]; tensor attn_weights_297_cast_fp16 = matmul(transpose_x = attn_weights_297_transpose_x_0, transpose_y = attn_weights_297_transpose_y_0, x = var_6866_cast_fp16_1, y = var_6879_1)[name = string("attn_weights_297_cast_fp16")]; fp16 var_6894_to_fp16 = const()[name = string("op_6894_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_299_cast_fp16 = mul(x = attn_weights_297_cast_fp16, y = var_6894_to_fp16)[name = string("attn_weights_299_cast_fp16")]; tensor attn_weights_301_cast_fp16 = add(x = attn_weights_299_cast_fp16, y = attn_mask_1)[name = string("attn_weights_301_cast_fp16")]; int32 var_6898 = const()[name = string("op_6898"), val = int32(-2)]; tensor attn_weights_303_cast_fp16 = softmax(axis = var_6898, x = attn_weights_301_cast_fp16)[name = string("attn_weights_303_cast_fp16")]; bool attn_output_145_transpose_x_1 = const()[name = string("attn_output_145_transpose_x_1"), val = bool(true)]; bool attn_output_145_transpose_y_1 = const()[name = string("attn_output_145_transpose_y_1"), val = bool(false)]; tensor attn_output_145_cast_fp16 = matmul(transpose_x = attn_output_145_transpose_x_1, transpose_y = attn_output_145_transpose_y_1, x = attn_weights_303_cast_fp16, y = var_6876_cast_fp16_1)[name = string("attn_output_145_cast_fp16")]; int32 var_6906 = const()[name = string("op_6906"), val = int32(1)]; bool attn_output_147_interleave_0 = const()[name = string("attn_output_147_interleave_0"), val = bool(false)]; tensor attn_output_147_cast_fp16 = concat(axis = var_6906, interleave = attn_output_147_interleave_0, values = (var_6892_cast_fp16, attn_output_145_cast_fp16))[name = string("attn_output_147_cast_fp16")]; tensor var_6910_perm_0 = const()[name = string("op_6910_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_227x = const()[name = string("concat_227x"), val = tensor([1, 2048, 1, -1])]; tensor var_6910_cast_fp16 = transpose(perm = var_6910_perm_0, x = attn_output_147_cast_fp16)[name = string("transpose_371")]; tensor attn_output_151_cast_fp16 = reshape(shape = concat_227x, x = var_6910_cast_fp16)[name = string("attn_output_151_cast_fp16")]; tensor hidden_states_183_strides_0 = const()[name = string("hidden_states_183_strides_0"), val = tensor([1, 1])]; string hidden_states_183_pad_type_0 = const()[name = string("hidden_states_183_pad_type_0"), val = string("valid")]; tensor hidden_states_183_pad_0 = const()[name = string("hidden_states_183_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_183_dilations_0 = const()[name = string("hidden_states_183_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_183_groups_0 = const()[name = string("hidden_states_183_groups_0"), val = int32(1)]; tensor hidden_states_183_cast_fp16 = conv(dilations = hidden_states_183_dilations_0, groups = hidden_states_183_groups_0, pad = hidden_states_183_pad_0, pad_type = hidden_states_183_pad_type_0, strides = hidden_states_183_strides_0, weight = layers_18_self_attn_o_proj_weight_cast_fp16, x = attn_output_151_cast_fp16)[name = string("hidden_states_183_cast_fp16")]; tensor hidden_states_185_cast_fp16 = add(x = hidden_states_179_cast_fp16, y = hidden_states_183_cast_fp16)[name = string("hidden_states_185_cast_fp16")]; fp16 const_188_promoted_to_fp16 = const()[name = string("const_188_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6943_cast_fp16 = mul(x = hidden_states_185_cast_fp16, y = const_188_promoted_to_fp16)[name = string("op_6943_cast_fp16")]; int32 var_6941 = const()[name = string("op_6941"), val = int32(1)]; bool doubled_149_interleave_0 = const()[name = string("doubled_149_interleave_0"), val = bool(false)]; tensor doubled_149_cast_fp16 = concat(axis = var_6941, interleave = doubled_149_interleave_0, values = (hidden_states_185_cast_fp16, var_6943_cast_fp16))[name = string("doubled_149_cast_fp16")]; tensor out_75_axes_0 = const()[name = string("out_75_axes_0"), val = tensor([1])]; tensor out_75_gamma_0_to_fp16 = const()[name = string("out_75_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492028288)))]; fp16 var_6953_to_fp16 = const()[name = string("op_6953_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_75_cast_fp16 = layer_norm(axes = out_75_axes_0, epsilon = var_6953_to_fp16, gamma = out_75_gamma_0_to_fp16, x = doubled_149_cast_fp16)[name = string("out_75_cast_fp16")]; tensor var_6964_split_sizes_0 = const()[name = string("op_6964_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6964_axis_0 = const()[name = string("op_6964_axis_0"), val = int32(1)]; tensor var_6964_cast_fp16_0, tensor var_6964_cast_fp16_1 = split(axis = var_6964_axis_0, split_sizes = var_6964_split_sizes_0, x = out_75_cast_fp16)[name = string("op_6964_cast_fp16")]; tensor input_37_strides_0 = const()[name = string("input_37_strides_0"), val = tensor([1, 1])]; string input_37_pad_type_0 = const()[name = string("input_37_pad_type_0"), val = string("valid")]; tensor input_37_pad_0 = const()[name = string("input_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_37_dilations_0 = const()[name = string("input_37_dilations_0"), val = tensor([1, 1])]; int32 input_37_groups_0 = const()[name = string("input_37_groups_0"), val = int32(1)]; tensor input_37_cast_fp16 = conv(dilations = input_37_dilations_0, groups = input_37_groups_0, pad = input_37_pad_0, pad_type = input_37_pad_type_0, strides = input_37_strides_0, weight = layers_18_mlp_gate_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("input_37_cast_fp16")]; tensor var_6981_cast_fp16 = silu(x = input_37_cast_fp16)[name = string("op_6981_cast_fp16")]; tensor var_6987_strides_0 = const()[name = string("op_6987_strides_0"), val = tensor([1, 1])]; string var_6987_pad_type_0 = const()[name = string("op_6987_pad_type_0"), val = string("valid")]; tensor var_6987_pad_0 = const()[name = string("op_6987_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6987_dilations_0 = const()[name = string("op_6987_dilations_0"), val = tensor([1, 1])]; int32 var_6987_groups_0 = const()[name = string("op_6987_groups_0"), val = int32(1)]; tensor var_6987_cast_fp16 = conv(dilations = var_6987_dilations_0, groups = var_6987_groups_0, pad = var_6987_pad_0, pad_type = var_6987_pad_type_0, strides = var_6987_strides_0, weight = layers_18_mlp_up_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("op_6987_cast_fp16")]; tensor x_189_cast_fp16 = mul(x = var_6981_cast_fp16, y = var_6987_cast_fp16)[name = string("x_189_cast_fp16")]; tensor hidden_states_187_strides_0 = const()[name = string("hidden_states_187_strides_0"), val = tensor([1, 1])]; string hidden_states_187_pad_type_0 = const()[name = string("hidden_states_187_pad_type_0"), val = string("valid")]; tensor hidden_states_187_pad_0 = const()[name = string("hidden_states_187_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_187_dilations_0 = const()[name = string("hidden_states_187_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_187_groups_0 = const()[name = string("hidden_states_187_groups_0"), val = int32(1)]; tensor hidden_states_187_cast_fp16 = conv(dilations = hidden_states_187_dilations_0, groups = hidden_states_187_groups_0, pad = hidden_states_187_pad_0, pad_type = hidden_states_187_pad_type_0, strides = hidden_states_187_strides_0, weight = layers_18_mlp_down_proj_weight_cast_fp16, x = x_189_cast_fp16)[name = string("hidden_states_187_cast_fp16")]; tensor hidden_states_189_cast_fp16 = add(x = hidden_states_185_cast_fp16, y = hidden_states_187_cast_fp16)[name = string("hidden_states_189_cast_fp16")]; fp16 const_190_promoted_to_fp16 = const()[name = string("const_190_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7005_cast_fp16 = mul(x = hidden_states_189_cast_fp16, y = const_190_promoted_to_fp16)[name = string("op_7005_cast_fp16")]; int32 var_7003 = const()[name = string("op_7003"), val = int32(1)]; bool doubled_153_interleave_0 = const()[name = string("doubled_153_interleave_0"), val = bool(false)]; tensor doubled_153_cast_fp16 = concat(axis = var_7003, interleave = doubled_153_interleave_0, values = (hidden_states_189_cast_fp16, var_7005_cast_fp16))[name = string("doubled_153_cast_fp16")]; tensor out_77_axes_0 = const()[name = string("out_77_axes_0"), val = tensor([1])]; tensor out_77_gamma_0_to_fp16 = const()[name = string("out_77_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492036544)))]; fp16 var_7015_to_fp16 = const()[name = string("op_7015_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_77_cast_fp16 = layer_norm(axes = out_77_axes_0, epsilon = var_7015_to_fp16, gamma = out_77_gamma_0_to_fp16, x = doubled_153_cast_fp16)[name = string("out_77_cast_fp16")]; tensor var_7026_split_sizes_0 = const()[name = string("op_7026_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7026_axis_0 = const()[name = string("op_7026_axis_0"), val = int32(1)]; tensor var_7026_cast_fp16_0, tensor var_7026_cast_fp16_1 = split(axis = var_7026_axis_0, split_sizes = var_7026_split_sizes_0, x = out_77_cast_fp16)[name = string("op_7026_cast_fp16")]; tensor query_states_115_strides_0 = const()[name = string("query_states_115_strides_0"), val = tensor([1, 1])]; string query_states_115_pad_type_0 = const()[name = string("query_states_115_pad_type_0"), val = string("valid")]; tensor query_states_115_pad_0 = const()[name = string("query_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_115_dilations_0 = const()[name = string("query_states_115_dilations_0"), val = tensor([1, 1])]; int32 query_states_115_groups_0 = const()[name = string("query_states_115_groups_0"), val = int32(1)]; tensor query_states_115_cast_fp16 = conv(dilations = query_states_115_dilations_0, groups = query_states_115_groups_0, pad = query_states_115_pad_0, pad_type = query_states_115_pad_type_0, strides = query_states_115_strides_0, weight = layers_19_self_attn_q_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("query_states_115_cast_fp16")]; tensor key_states_191_strides_0 = const()[name = string("key_states_191_strides_0"), val = tensor([1, 1])]; string key_states_191_pad_type_0 = const()[name = string("key_states_191_pad_type_0"), val = string("valid")]; tensor key_states_191_pad_0 = const()[name = string("key_states_191_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_191_dilations_0 = const()[name = string("key_states_191_dilations_0"), val = tensor([1, 1])]; int32 key_states_191_groups_0 = const()[name = string("key_states_191_groups_0"), val = int32(1)]; tensor key_states_191_cast_fp16 = conv(dilations = key_states_191_dilations_0, groups = key_states_191_groups_0, pad = key_states_191_pad_0, pad_type = key_states_191_pad_type_0, strides = key_states_191_strides_0, weight = layers_19_self_attn_k_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("key_states_191_cast_fp16")]; tensor layers_19_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492044800)))]; tensor value_states_115_strides_0 = const()[name = string("value_states_115_strides_0"), val = tensor([1, 1])]; string value_states_115_pad_type_0 = const()[name = string("value_states_115_pad_type_0"), val = string("valid")]; tensor value_states_115_pad_0 = const()[name = string("value_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_115_dilations_0 = const()[name = string("value_states_115_dilations_0"), val = tensor([1, 1])]; int32 value_states_115_groups_0 = const()[name = string("value_states_115_groups_0"), val = int32(1)]; tensor value_states_115_cast_fp16 = conv(dilations = value_states_115_dilations_0, groups = value_states_115_groups_0, pad = value_states_115_pad_0, pad_type = value_states_115_pad_type_0, strides = value_states_115_strides_0, weight = layers_19_self_attn_v_proj_weight_to_fp16, x = var_7026_cast_fp16_0)[name = string("value_states_115_cast_fp16")]; tensor concat_228x = const()[name = string("concat_228x"), val = tensor([1, 16, 128, -1])]; tensor x_191_cast_fp16 = reshape(shape = concat_228x, x = query_states_115_cast_fp16)[name = string("x_191_cast_fp16")]; tensor concat_229x = const()[name = string("concat_229x"), val = tensor([1, 2, 128, -1])]; tensor var_7083_cast_fp16 = reshape(shape = concat_229x, x = key_states_191_cast_fp16)[name = string("op_7083_cast_fp16")]; tensor concat_230x = const()[name = string("concat_230x"), val = tensor([1, 2, 128, -1])]; tensor var_7090_cast_fp16 = reshape(shape = concat_230x, x = value_states_115_cast_fp16)[name = string("op_7090_cast_fp16")]; tensor var_7094_cast_fp16 = mul(x = x_191_cast_fp16, y = var_869_cast_fp16)[name = string("op_7094_cast_fp16")]; tensor var_7095_split_sizes_0 = const()[name = string("op_7095_split_sizes_0"), val = tensor([64, 64])]; int32 var_7095_axis_0 = const()[name = string("op_7095_axis_0"), val = int32(-2)]; tensor var_7095_cast_fp16_0, tensor var_7095_cast_fp16_1 = split(axis = var_7095_axis_0, split_sizes = var_7095_split_sizes_0, x = x_191_cast_fp16)[name = string("op_7095_cast_fp16")]; fp16 const_192_promoted_to_fp16 = const()[name = string("const_192_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7097_cast_fp16 = mul(x = var_7095_cast_fp16_1, y = const_192_promoted_to_fp16)[name = string("op_7097_cast_fp16")]; int32 var_7099 = const()[name = string("op_7099"), val = int32(-2)]; bool var_7100_interleave_0 = const()[name = string("op_7100_interleave_0"), val = bool(false)]; tensor var_7100_cast_fp16 = concat(axis = var_7099, interleave = var_7100_interleave_0, values = (var_7097_cast_fp16, var_7095_cast_fp16_0))[name = string("op_7100_cast_fp16")]; tensor var_7101_cast_fp16 = mul(x = var_7100_cast_fp16, y = var_878_cast_fp16)[name = string("op_7101_cast_fp16")]; tensor query_states_117_cast_fp16 = add(x = var_7094_cast_fp16, y = var_7101_cast_fp16)[name = string("query_states_117_cast_fp16")]; tensor var_7107_cast_fp16 = mul(x = var_7083_cast_fp16, y = var_869_cast_fp16)[name = string("op_7107_cast_fp16")]; tensor var_7108_split_sizes_0 = const()[name = string("op_7108_split_sizes_0"), val = tensor([64, 64])]; int32 var_7108_axis_0 = const()[name = string("op_7108_axis_0"), val = int32(-2)]; tensor var_7108_cast_fp16_0, tensor var_7108_cast_fp16_1 = split(axis = var_7108_axis_0, split_sizes = var_7108_split_sizes_0, x = var_7083_cast_fp16)[name = string("op_7108_cast_fp16")]; fp16 const_193_promoted_to_fp16 = const()[name = string("const_193_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7110_cast_fp16 = mul(x = var_7108_cast_fp16_1, y = const_193_promoted_to_fp16)[name = string("op_7110_cast_fp16")]; int32 var_7112 = const()[name = string("op_7112"), val = int32(-2)]; bool var_7113_interleave_0 = const()[name = string("op_7113_interleave_0"), val = bool(false)]; tensor var_7113_cast_fp16 = concat(axis = var_7112, interleave = var_7113_interleave_0, values = (var_7110_cast_fp16, var_7108_cast_fp16_0))[name = string("op_7113_cast_fp16")]; tensor var_7114_cast_fp16 = mul(x = var_7113_cast_fp16, y = var_878_cast_fp16)[name = string("op_7114_cast_fp16")]; tensor key_states_195_cast_fp16 = add(x = var_7107_cast_fp16, y = var_7114_cast_fp16)[name = string("key_states_195_cast_fp16")]; tensor expand_dims_228 = const()[name = string("expand_dims_228"), val = tensor([19])]; tensor expand_dims_229 = const()[name = string("expand_dims_229"), val = tensor([0])]; tensor expand_dims_231 = const()[name = string("expand_dims_231"), val = tensor([0])]; int32 concat_233_axis_0 = const()[name = string("concat_233_axis_0"), val = int32(0)]; bool concat_233_interleave_0 = const()[name = string("concat_233_interleave_0"), val = bool(false)]; tensor concat_233 = concat(axis = concat_233_axis_0, interleave = concat_233_interleave_0, values = (expand_dims_228, expand_dims_229, position_id, expand_dims_231))[name = string("concat_233")]; tensor expand_dims_232 = const()[name = string("expand_dims_232"), val = tensor([20])]; tensor concat_234_values1_0 = const()[name = string("concat_234_values1_0"), val = tensor([0])]; tensor concat_234_values3_0 = const()[name = string("concat_234_values3_0"), val = tensor([0])]; int32 concat_234_axis_0 = const()[name = string("concat_234_axis_0"), val = int32(0)]; bool concat_234_interleave_0 = const()[name = string("concat_234_interleave_0"), val = bool(false)]; tensor concat_234 = concat(axis = concat_234_axis_0, interleave = concat_234_interleave_0, values = (expand_dims_232, concat_234_values1_0, cache_position_end, concat_234_values3_0))[name = string("concat_234")]; tensor key_states_197_perm_0 = const()[name = string("key_states_197_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_20_stride_0 = const()[name = string("key_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_197_cast_fp16 = transpose(perm = key_states_197_perm_0, x = key_states_195_cast_fp16)[name = string("transpose_370")]; tensor key_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = key_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = key_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_20_squeeze_mask_0, stride = key_cache_internal_tensor_assign_20_stride_0, update = key_states_197_cast_fp16, x = coreml_update_state_260)[name = string("key_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_20_cast_fp16, input = key_cache)[name = string("coreml_update_state_262_write_state")]; tensor coreml_update_state_262 = read_state(input = key_cache)[name = string("coreml_update_state_262")]; tensor value_states_117_perm_0 = const()[name = string("value_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_20_stride_0 = const()[name = string("value_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_117_cast_fp16 = transpose(perm = value_states_117_perm_0, x = var_7090_cast_fp16)[name = string("transpose_369")]; tensor value_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = value_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = value_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_20_squeeze_mask_0, stride = value_cache_internal_tensor_assign_20_stride_0, update = value_states_117_cast_fp16, x = coreml_update_state_261)[name = string("value_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_20_cast_fp16, input = value_cache)[name = string("coreml_update_state_263_write_state")]; tensor coreml_update_state_263 = read_state(input = value_cache)[name = string("coreml_update_state_263")]; tensor var_7184_begin_0 = const()[name = string("op_7184_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7184_end_0 = const()[name = string("op_7184_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7184_end_mask_0 = const()[name = string("op_7184_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7184_cast_fp16 = slice_by_index(begin = var_7184_begin_0, end = var_7184_end_0, end_mask = var_7184_end_mask_0, x = coreml_update_state_262)[name = string("op_7184_cast_fp16")]; tensor tile_38 = const()[name = string("tile_38"), val = tensor([1, 1])]; int32 var_7187_axis_0 = const()[name = string("op_7187_axis_0"), val = int32(1)]; tensor var_7187_cast_fp16_0, tensor var_7187_cast_fp16_1 = split(axis = var_7187_axis_0, split_sizes = tile_38, x = var_7184_cast_fp16)[name = string("op_7187_cast_fp16")]; tensor var_7194_begin_0 = const()[name = string("op_7194_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7194_end_0 = const()[name = string("op_7194_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7194_end_mask_0 = const()[name = string("op_7194_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7194_cast_fp16 = slice_by_index(begin = var_7194_begin_0, end = var_7194_end_0, end_mask = var_7194_end_mask_0, x = coreml_update_state_263)[name = string("op_7194_cast_fp16")]; tensor tile_39 = const()[name = string("tile_39"), val = tensor([1, 1])]; int32 var_7197_axis_0 = const()[name = string("op_7197_axis_0"), val = int32(1)]; tensor var_7197_cast_fp16_0, tensor var_7197_cast_fp16_1 = split(axis = var_7197_axis_0, split_sizes = tile_39, x = var_7194_cast_fp16)[name = string("op_7197_cast_fp16")]; tensor var_7200_split_sizes_0 = const()[name = string("op_7200_split_sizes_0"), val = tensor([8, 8])]; int32 var_7200_axis_0 = const()[name = string("op_7200_axis_0"), val = int32(1)]; tensor var_7200_0, tensor var_7200_1 = split(axis = var_7200_axis_0, split_sizes = var_7200_split_sizes_0, x = query_states_117_cast_fp16)[name = string("op_7200")]; bool attn_weights_305_transpose_x_0 = const()[name = string("attn_weights_305_transpose_x_0"), val = bool(false)]; bool attn_weights_305_transpose_y_0 = const()[name = string("attn_weights_305_transpose_y_0"), val = bool(false)]; tensor attn_weights_305_cast_fp16 = matmul(transpose_x = attn_weights_305_transpose_x_0, transpose_y = attn_weights_305_transpose_y_0, x = var_7187_cast_fp16_0, y = var_7200_0)[name = string("attn_weights_305_cast_fp16")]; fp16 var_7203_to_fp16 = const()[name = string("op_7203_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_307_cast_fp16 = mul(x = attn_weights_305_cast_fp16, y = var_7203_to_fp16)[name = string("attn_weights_307_cast_fp16")]; tensor attn_weights_309_cast_fp16 = add(x = attn_weights_307_cast_fp16, y = attn_mask_1)[name = string("attn_weights_309_cast_fp16")]; int32 var_7207 = const()[name = string("op_7207"), val = int32(-2)]; tensor attn_weights_311_cast_fp16 = softmax(axis = var_7207, x = attn_weights_309_cast_fp16)[name = string("attn_weights_311_cast_fp16")]; bool var_7213_transpose_x_1 = const()[name = string("op_7213_transpose_x_1"), val = bool(true)]; bool var_7213_transpose_y_1 = const()[name = string("op_7213_transpose_y_1"), val = bool(false)]; tensor var_7213_cast_fp16 = matmul(transpose_x = var_7213_transpose_x_1, transpose_y = var_7213_transpose_y_1, x = attn_weights_311_cast_fp16, y = var_7197_cast_fp16_0)[name = string("op_7213_cast_fp16")]; bool attn_weights_313_transpose_x_0 = const()[name = string("attn_weights_313_transpose_x_0"), val = bool(false)]; bool attn_weights_313_transpose_y_0 = const()[name = string("attn_weights_313_transpose_y_0"), val = bool(false)]; tensor attn_weights_313_cast_fp16 = matmul(transpose_x = attn_weights_313_transpose_x_0, transpose_y = attn_weights_313_transpose_y_0, x = var_7187_cast_fp16_1, y = var_7200_1)[name = string("attn_weights_313_cast_fp16")]; fp16 var_7215_to_fp16 = const()[name = string("op_7215_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_315_cast_fp16 = mul(x = attn_weights_313_cast_fp16, y = var_7215_to_fp16)[name = string("attn_weights_315_cast_fp16")]; tensor attn_weights_317_cast_fp16 = add(x = attn_weights_315_cast_fp16, y = attn_mask_1)[name = string("attn_weights_317_cast_fp16")]; int32 var_7219 = const()[name = string("op_7219"), val = int32(-2)]; tensor attn_weights_319_cast_fp16 = softmax(axis = var_7219, x = attn_weights_317_cast_fp16)[name = string("attn_weights_319_cast_fp16")]; bool attn_output_153_transpose_x_1 = const()[name = string("attn_output_153_transpose_x_1"), val = bool(true)]; bool attn_output_153_transpose_y_1 = const()[name = string("attn_output_153_transpose_y_1"), val = bool(false)]; tensor attn_output_153_cast_fp16 = matmul(transpose_x = attn_output_153_transpose_x_1, transpose_y = attn_output_153_transpose_y_1, x = attn_weights_319_cast_fp16, y = var_7197_cast_fp16_1)[name = string("attn_output_153_cast_fp16")]; int32 var_7227 = const()[name = string("op_7227"), val = int32(1)]; bool attn_output_155_interleave_0 = const()[name = string("attn_output_155_interleave_0"), val = bool(false)]; tensor attn_output_155_cast_fp16 = concat(axis = var_7227, interleave = attn_output_155_interleave_0, values = (var_7213_cast_fp16, attn_output_153_cast_fp16))[name = string("attn_output_155_cast_fp16")]; tensor var_7231_perm_0 = const()[name = string("op_7231_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_239x = const()[name = string("concat_239x"), val = tensor([1, 2048, 1, -1])]; tensor var_7231_cast_fp16 = transpose(perm = var_7231_perm_0, x = attn_output_155_cast_fp16)[name = string("transpose_368")]; tensor attn_output_159_cast_fp16 = reshape(shape = concat_239x, x = var_7231_cast_fp16)[name = string("attn_output_159_cast_fp16")]; tensor layers_19_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1493093440)))]; tensor hidden_states_193_strides_0 = const()[name = string("hidden_states_193_strides_0"), val = tensor([1, 1])]; string hidden_states_193_pad_type_0 = const()[name = string("hidden_states_193_pad_type_0"), val = string("valid")]; tensor hidden_states_193_pad_0 = const()[name = string("hidden_states_193_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_193_dilations_0 = const()[name = string("hidden_states_193_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_193_groups_0 = const()[name = string("hidden_states_193_groups_0"), val = int32(1)]; tensor hidden_states_193_cast_fp16 = conv(dilations = hidden_states_193_dilations_0, groups = hidden_states_193_groups_0, pad = hidden_states_193_pad_0, pad_type = hidden_states_193_pad_type_0, strides = hidden_states_193_strides_0, weight = layers_19_self_attn_o_proj_weight_to_fp16, x = attn_output_159_cast_fp16)[name = string("hidden_states_193_cast_fp16")]; tensor hidden_states_195_cast_fp16 = add(x = hidden_states_189_cast_fp16, y = hidden_states_193_cast_fp16)[name = string("hidden_states_195_cast_fp16")]; fp16 const_198_promoted_to_fp16 = const()[name = string("const_198_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7264_cast_fp16 = mul(x = hidden_states_195_cast_fp16, y = const_198_promoted_to_fp16)[name = string("op_7264_cast_fp16")]; int32 var_7262 = const()[name = string("op_7262"), val = int32(1)]; bool doubled_157_interleave_0 = const()[name = string("doubled_157_interleave_0"), val = bool(false)]; tensor doubled_157_cast_fp16 = concat(axis = var_7262, interleave = doubled_157_interleave_0, values = (hidden_states_195_cast_fp16, var_7264_cast_fp16))[name = string("doubled_157_cast_fp16")]; tensor out_79_axes_0 = const()[name = string("out_79_axes_0"), val = tensor([1])]; tensor out_79_gamma_0_to_fp16 = const()[name = string("out_79_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501482112)))]; fp16 var_7274_to_fp16 = const()[name = string("op_7274_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_79_cast_fp16 = layer_norm(axes = out_79_axes_0, epsilon = var_7274_to_fp16, gamma = out_79_gamma_0_to_fp16, x = doubled_157_cast_fp16)[name = string("out_79_cast_fp16")]; tensor var_7285_split_sizes_0 = const()[name = string("op_7285_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7285_axis_0 = const()[name = string("op_7285_axis_0"), val = int32(1)]; tensor var_7285_cast_fp16_0, tensor var_7285_cast_fp16_1 = split(axis = var_7285_axis_0, split_sizes = var_7285_split_sizes_0, x = out_79_cast_fp16)[name = string("op_7285_cast_fp16")]; tensor input_39_strides_0 = const()[name = string("input_39_strides_0"), val = tensor([1, 1])]; string input_39_pad_type_0 = const()[name = string("input_39_pad_type_0"), val = string("valid")]; tensor input_39_pad_0 = const()[name = string("input_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_39_dilations_0 = const()[name = string("input_39_dilations_0"), val = tensor([1, 1])]; int32 input_39_groups_0 = const()[name = string("input_39_groups_0"), val = int32(1)]; tensor input_39_cast_fp16 = conv(dilations = input_39_dilations_0, groups = input_39_groups_0, pad = input_39_pad_0, pad_type = input_39_pad_type_0, strides = input_39_strides_0, weight = layers_19_mlp_gate_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("input_39_cast_fp16")]; tensor var_7302_cast_fp16 = silu(x = input_39_cast_fp16)[name = string("op_7302_cast_fp16")]; tensor var_7308_strides_0 = const()[name = string("op_7308_strides_0"), val = tensor([1, 1])]; string var_7308_pad_type_0 = const()[name = string("op_7308_pad_type_0"), val = string("valid")]; tensor var_7308_pad_0 = const()[name = string("op_7308_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7308_dilations_0 = const()[name = string("op_7308_dilations_0"), val = tensor([1, 1])]; int32 var_7308_groups_0 = const()[name = string("op_7308_groups_0"), val = int32(1)]; tensor var_7308_cast_fp16 = conv(dilations = var_7308_dilations_0, groups = var_7308_groups_0, pad = var_7308_pad_0, pad_type = var_7308_pad_type_0, strides = var_7308_strides_0, weight = layers_19_mlp_up_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("op_7308_cast_fp16")]; tensor x_199_cast_fp16 = mul(x = var_7302_cast_fp16, y = var_7308_cast_fp16)[name = string("x_199_cast_fp16")]; tensor hidden_states_197_strides_0 = const()[name = string("hidden_states_197_strides_0"), val = tensor([1, 1])]; string hidden_states_197_pad_type_0 = const()[name = string("hidden_states_197_pad_type_0"), val = string("valid")]; tensor hidden_states_197_pad_0 = const()[name = string("hidden_states_197_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_197_dilations_0 = const()[name = string("hidden_states_197_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_197_groups_0 = const()[name = string("hidden_states_197_groups_0"), val = int32(1)]; tensor hidden_states_197_cast_fp16 = conv(dilations = hidden_states_197_dilations_0, groups = hidden_states_197_groups_0, pad = hidden_states_197_pad_0, pad_type = hidden_states_197_pad_type_0, strides = hidden_states_197_strides_0, weight = layers_19_mlp_down_proj_weight_cast_fp16, x = x_199_cast_fp16)[name = string("hidden_states_197_cast_fp16")]; tensor hidden_states_199_cast_fp16 = add(x = hidden_states_195_cast_fp16, y = hidden_states_197_cast_fp16)[name = string("hidden_states_199_cast_fp16")]; fp16 const_200_promoted_to_fp16 = const()[name = string("const_200_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7326_cast_fp16 = mul(x = hidden_states_199_cast_fp16, y = const_200_promoted_to_fp16)[name = string("op_7326_cast_fp16")]; int32 var_7324 = const()[name = string("op_7324"), val = int32(1)]; bool doubled_161_interleave_0 = const()[name = string("doubled_161_interleave_0"), val = bool(false)]; tensor doubled_161_cast_fp16 = concat(axis = var_7324, interleave = doubled_161_interleave_0, values = (hidden_states_199_cast_fp16, var_7326_cast_fp16))[name = string("doubled_161_cast_fp16")]; tensor out_81_axes_0 = const()[name = string("out_81_axes_0"), val = tensor([1])]; tensor out_81_gamma_0_to_fp16 = const()[name = string("out_81_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501490368)))]; fp16 var_7336_to_fp16 = const()[name = string("op_7336_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_81_cast_fp16 = layer_norm(axes = out_81_axes_0, epsilon = var_7336_to_fp16, gamma = out_81_gamma_0_to_fp16, x = doubled_161_cast_fp16)[name = string("out_81_cast_fp16")]; tensor var_7347_split_sizes_0 = const()[name = string("op_7347_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7347_axis_0 = const()[name = string("op_7347_axis_0"), val = int32(1)]; tensor var_7347_cast_fp16_0, tensor var_7347_cast_fp16_1 = split(axis = var_7347_axis_0, split_sizes = var_7347_split_sizes_0, x = out_81_cast_fp16)[name = string("op_7347_cast_fp16")]; tensor query_states_121_strides_0 = const()[name = string("query_states_121_strides_0"), val = tensor([1, 1])]; string query_states_121_pad_type_0 = const()[name = string("query_states_121_pad_type_0"), val = string("valid")]; tensor query_states_121_pad_0 = const()[name = string("query_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_121_dilations_0 = const()[name = string("query_states_121_dilations_0"), val = tensor([1, 1])]; int32 query_states_121_groups_0 = const()[name = string("query_states_121_groups_0"), val = int32(1)]; tensor query_states_121_cast_fp16 = conv(dilations = query_states_121_dilations_0, groups = query_states_121_groups_0, pad = query_states_121_pad_0, pad_type = query_states_121_pad_type_0, strides = query_states_121_strides_0, weight = layers_20_self_attn_q_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("query_states_121_cast_fp16")]; tensor key_states_201_strides_0 = const()[name = string("key_states_201_strides_0"), val = tensor([1, 1])]; string key_states_201_pad_type_0 = const()[name = string("key_states_201_pad_type_0"), val = string("valid")]; tensor key_states_201_pad_0 = const()[name = string("key_states_201_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_201_dilations_0 = const()[name = string("key_states_201_dilations_0"), val = tensor([1, 1])]; int32 key_states_201_groups_0 = const()[name = string("key_states_201_groups_0"), val = int32(1)]; tensor key_states_201_cast_fp16 = conv(dilations = key_states_201_dilations_0, groups = key_states_201_groups_0, pad = key_states_201_pad_0, pad_type = key_states_201_pad_type_0, strides = key_states_201_strides_0, weight = layers_20_self_attn_k_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("key_states_201_cast_fp16")]; tensor layers_20_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_20_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501498624)))]; tensor value_states_121_strides_0 = const()[name = string("value_states_121_strides_0"), val = tensor([1, 1])]; string value_states_121_pad_type_0 = const()[name = string("value_states_121_pad_type_0"), val = string("valid")]; tensor value_states_121_pad_0 = const()[name = string("value_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_121_dilations_0 = const()[name = string("value_states_121_dilations_0"), val = tensor([1, 1])]; int32 value_states_121_groups_0 = const()[name = string("value_states_121_groups_0"), val = int32(1)]; tensor value_states_121_cast_fp16 = conv(dilations = value_states_121_dilations_0, groups = value_states_121_groups_0, pad = value_states_121_pad_0, pad_type = value_states_121_pad_type_0, strides = value_states_121_strides_0, weight = layers_20_self_attn_v_proj_weight_to_fp16, x = var_7347_cast_fp16_0)[name = string("value_states_121_cast_fp16")]; tensor concat_240x = const()[name = string("concat_240x"), val = tensor([1, 16, 128, -1])]; tensor x_201_cast_fp16 = reshape(shape = concat_240x, x = query_states_121_cast_fp16)[name = string("x_201_cast_fp16")]; tensor concat_241x = const()[name = string("concat_241x"), val = tensor([1, 2, 128, -1])]; tensor var_7404_cast_fp16 = reshape(shape = concat_241x, x = key_states_201_cast_fp16)[name = string("op_7404_cast_fp16")]; tensor concat_242x = const()[name = string("concat_242x"), val = tensor([1, 2, 128, -1])]; tensor var_7411_cast_fp16 = reshape(shape = concat_242x, x = value_states_121_cast_fp16)[name = string("op_7411_cast_fp16")]; tensor var_7415_cast_fp16 = mul(x = x_201_cast_fp16, y = var_869_cast_fp16)[name = string("op_7415_cast_fp16")]; tensor var_7416_split_sizes_0 = const()[name = string("op_7416_split_sizes_0"), val = tensor([64, 64])]; int32 var_7416_axis_0 = const()[name = string("op_7416_axis_0"), val = int32(-2)]; tensor var_7416_cast_fp16_0, tensor var_7416_cast_fp16_1 = split(axis = var_7416_axis_0, split_sizes = var_7416_split_sizes_0, x = x_201_cast_fp16)[name = string("op_7416_cast_fp16")]; fp16 const_202_promoted_to_fp16 = const()[name = string("const_202_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7418_cast_fp16 = mul(x = var_7416_cast_fp16_1, y = const_202_promoted_to_fp16)[name = string("op_7418_cast_fp16")]; int32 var_7420 = const()[name = string("op_7420"), val = int32(-2)]; bool var_7421_interleave_0 = const()[name = string("op_7421_interleave_0"), val = bool(false)]; tensor var_7421_cast_fp16 = concat(axis = var_7420, interleave = var_7421_interleave_0, values = (var_7418_cast_fp16, var_7416_cast_fp16_0))[name = string("op_7421_cast_fp16")]; tensor var_7422_cast_fp16 = mul(x = var_7421_cast_fp16, y = var_878_cast_fp16)[name = string("op_7422_cast_fp16")]; tensor query_states_123_cast_fp16 = add(x = var_7415_cast_fp16, y = var_7422_cast_fp16)[name = string("query_states_123_cast_fp16")]; tensor var_7428_cast_fp16 = mul(x = var_7404_cast_fp16, y = var_869_cast_fp16)[name = string("op_7428_cast_fp16")]; tensor var_7429_split_sizes_0 = const()[name = string("op_7429_split_sizes_0"), val = tensor([64, 64])]; int32 var_7429_axis_0 = const()[name = string("op_7429_axis_0"), val = int32(-2)]; tensor var_7429_cast_fp16_0, tensor var_7429_cast_fp16_1 = split(axis = var_7429_axis_0, split_sizes = var_7429_split_sizes_0, x = var_7404_cast_fp16)[name = string("op_7429_cast_fp16")]; fp16 const_203_promoted_to_fp16 = const()[name = string("const_203_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7431_cast_fp16 = mul(x = var_7429_cast_fp16_1, y = const_203_promoted_to_fp16)[name = string("op_7431_cast_fp16")]; int32 var_7433 = const()[name = string("op_7433"), val = int32(-2)]; bool var_7434_interleave_0 = const()[name = string("op_7434_interleave_0"), val = bool(false)]; tensor var_7434_cast_fp16 = concat(axis = var_7433, interleave = var_7434_interleave_0, values = (var_7431_cast_fp16, var_7429_cast_fp16_0))[name = string("op_7434_cast_fp16")]; tensor var_7435_cast_fp16 = mul(x = var_7434_cast_fp16, y = var_878_cast_fp16)[name = string("op_7435_cast_fp16")]; tensor key_states_205_cast_fp16 = add(x = var_7428_cast_fp16, y = var_7435_cast_fp16)[name = string("key_states_205_cast_fp16")]; tensor expand_dims_240 = const()[name = string("expand_dims_240"), val = tensor([20])]; tensor expand_dims_241 = const()[name = string("expand_dims_241"), val = tensor([0])]; tensor expand_dims_243 = const()[name = string("expand_dims_243"), val = tensor([0])]; int32 concat_245_axis_0 = const()[name = string("concat_245_axis_0"), val = int32(0)]; bool concat_245_interleave_0 = const()[name = string("concat_245_interleave_0"), val = bool(false)]; tensor concat_245 = concat(axis = concat_245_axis_0, interleave = concat_245_interleave_0, values = (expand_dims_240, expand_dims_241, position_id, expand_dims_243))[name = string("concat_245")]; tensor expand_dims_244 = const()[name = string("expand_dims_244"), val = tensor([21])]; tensor concat_246_values1_0 = const()[name = string("concat_246_values1_0"), val = tensor([0])]; tensor concat_246_values3_0 = const()[name = string("concat_246_values3_0"), val = tensor([0])]; int32 concat_246_axis_0 = const()[name = string("concat_246_axis_0"), val = int32(0)]; bool concat_246_interleave_0 = const()[name = string("concat_246_interleave_0"), val = bool(false)]; tensor concat_246 = concat(axis = concat_246_axis_0, interleave = concat_246_interleave_0, values = (expand_dims_244, concat_246_values1_0, cache_position_end, concat_246_values3_0))[name = string("concat_246")]; tensor key_states_207_perm_0 = const()[name = string("key_states_207_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_21_stride_0 = const()[name = string("key_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_207_cast_fp16 = transpose(perm = key_states_207_perm_0, x = key_states_205_cast_fp16)[name = string("transpose_367")]; tensor key_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = key_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = key_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_21_squeeze_mask_0, stride = key_cache_internal_tensor_assign_21_stride_0, update = key_states_207_cast_fp16, x = coreml_update_state_262)[name = string("key_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_21_cast_fp16, input = key_cache)[name = string("coreml_update_state_264_write_state")]; tensor coreml_update_state_264 = read_state(input = key_cache)[name = string("coreml_update_state_264")]; tensor value_states_123_perm_0 = const()[name = string("value_states_123_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_21_stride_0 = const()[name = string("value_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_123_cast_fp16 = transpose(perm = value_states_123_perm_0, x = var_7411_cast_fp16)[name = string("transpose_366")]; tensor value_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = value_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = value_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_21_squeeze_mask_0, stride = value_cache_internal_tensor_assign_21_stride_0, update = value_states_123_cast_fp16, x = coreml_update_state_263)[name = string("value_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_21_cast_fp16, input = value_cache)[name = string("coreml_update_state_265_write_state")]; tensor coreml_update_state_265 = read_state(input = value_cache)[name = string("coreml_update_state_265")]; tensor var_7505_begin_0 = const()[name = string("op_7505_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7505_end_0 = const()[name = string("op_7505_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7505_end_mask_0 = const()[name = string("op_7505_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7505_cast_fp16 = slice_by_index(begin = var_7505_begin_0, end = var_7505_end_0, end_mask = var_7505_end_mask_0, x = coreml_update_state_264)[name = string("op_7505_cast_fp16")]; tensor tile_40 = const()[name = string("tile_40"), val = tensor([1, 1])]; int32 var_7508_axis_0 = const()[name = string("op_7508_axis_0"), val = int32(1)]; tensor var_7508_cast_fp16_0, tensor var_7508_cast_fp16_1 = split(axis = var_7508_axis_0, split_sizes = tile_40, x = var_7505_cast_fp16)[name = string("op_7508_cast_fp16")]; tensor var_7515_begin_0 = const()[name = string("op_7515_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7515_end_0 = const()[name = string("op_7515_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7515_end_mask_0 = const()[name = string("op_7515_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7515_cast_fp16 = slice_by_index(begin = var_7515_begin_0, end = var_7515_end_0, end_mask = var_7515_end_mask_0, x = coreml_update_state_265)[name = string("op_7515_cast_fp16")]; tensor tile_41 = const()[name = string("tile_41"), val = tensor([1, 1])]; int32 var_7518_axis_0 = const()[name = string("op_7518_axis_0"), val = int32(1)]; tensor var_7518_cast_fp16_0, tensor var_7518_cast_fp16_1 = split(axis = var_7518_axis_0, split_sizes = tile_41, x = var_7515_cast_fp16)[name = string("op_7518_cast_fp16")]; tensor var_7521_split_sizes_0 = const()[name = string("op_7521_split_sizes_0"), val = tensor([8, 8])]; int32 var_7521_axis_0 = const()[name = string("op_7521_axis_0"), val = int32(1)]; tensor var_7521_0, tensor var_7521_1 = split(axis = var_7521_axis_0, split_sizes = var_7521_split_sizes_0, x = query_states_123_cast_fp16)[name = string("op_7521")]; bool attn_weights_321_transpose_x_0 = const()[name = string("attn_weights_321_transpose_x_0"), val = bool(false)]; bool attn_weights_321_transpose_y_0 = const()[name = string("attn_weights_321_transpose_y_0"), val = bool(false)]; tensor attn_weights_321_cast_fp16 = matmul(transpose_x = attn_weights_321_transpose_x_0, transpose_y = attn_weights_321_transpose_y_0, x = var_7508_cast_fp16_0, y = var_7521_0)[name = string("attn_weights_321_cast_fp16")]; fp16 var_7524_to_fp16 = const()[name = string("op_7524_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_323_cast_fp16 = mul(x = attn_weights_321_cast_fp16, y = var_7524_to_fp16)[name = string("attn_weights_323_cast_fp16")]; tensor attn_weights_325_cast_fp16 = add(x = attn_weights_323_cast_fp16, y = attn_mask_1)[name = string("attn_weights_325_cast_fp16")]; int32 var_7528 = const()[name = string("op_7528"), val = int32(-2)]; tensor attn_weights_327_cast_fp16 = softmax(axis = var_7528, x = attn_weights_325_cast_fp16)[name = string("attn_weights_327_cast_fp16")]; bool var_7534_transpose_x_1 = const()[name = string("op_7534_transpose_x_1"), val = bool(true)]; bool var_7534_transpose_y_1 = const()[name = string("op_7534_transpose_y_1"), val = bool(false)]; tensor var_7534_cast_fp16 = matmul(transpose_x = var_7534_transpose_x_1, transpose_y = var_7534_transpose_y_1, x = attn_weights_327_cast_fp16, y = var_7518_cast_fp16_0)[name = string("op_7534_cast_fp16")]; bool attn_weights_329_transpose_x_0 = const()[name = string("attn_weights_329_transpose_x_0"), val = bool(false)]; bool attn_weights_329_transpose_y_0 = const()[name = string("attn_weights_329_transpose_y_0"), val = bool(false)]; tensor attn_weights_329_cast_fp16 = matmul(transpose_x = attn_weights_329_transpose_x_0, transpose_y = attn_weights_329_transpose_y_0, x = var_7508_cast_fp16_1, y = var_7521_1)[name = string("attn_weights_329_cast_fp16")]; fp16 var_7536_to_fp16 = const()[name = string("op_7536_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_331_cast_fp16 = mul(x = attn_weights_329_cast_fp16, y = var_7536_to_fp16)[name = string("attn_weights_331_cast_fp16")]; tensor attn_weights_333_cast_fp16 = add(x = attn_weights_331_cast_fp16, y = attn_mask_1)[name = string("attn_weights_333_cast_fp16")]; int32 var_7540 = const()[name = string("op_7540"), val = int32(-2)]; tensor attn_weights_335_cast_fp16 = softmax(axis = var_7540, x = attn_weights_333_cast_fp16)[name = string("attn_weights_335_cast_fp16")]; bool attn_output_161_transpose_x_1 = const()[name = string("attn_output_161_transpose_x_1"), val = bool(true)]; bool attn_output_161_transpose_y_1 = const()[name = string("attn_output_161_transpose_y_1"), val = bool(false)]; tensor attn_output_161_cast_fp16 = matmul(transpose_x = attn_output_161_transpose_x_1, transpose_y = attn_output_161_transpose_y_1, x = attn_weights_335_cast_fp16, y = var_7518_cast_fp16_1)[name = string("attn_output_161_cast_fp16")]; int32 var_7548 = const()[name = string("op_7548"), val = int32(1)]; bool attn_output_163_interleave_0 = const()[name = string("attn_output_163_interleave_0"), val = bool(false)]; tensor attn_output_163_cast_fp16 = concat(axis = var_7548, interleave = attn_output_163_interleave_0, values = (var_7534_cast_fp16, attn_output_161_cast_fp16))[name = string("attn_output_163_cast_fp16")]; tensor var_7552_perm_0 = const()[name = string("op_7552_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_251x = const()[name = string("concat_251x"), val = tensor([1, 2048, 1, -1])]; tensor var_7552_cast_fp16 = transpose(perm = var_7552_perm_0, x = attn_output_163_cast_fp16)[name = string("transpose_365")]; tensor attn_output_167_cast_fp16 = reshape(shape = concat_251x, x = var_7552_cast_fp16)[name = string("attn_output_167_cast_fp16")]; tensor hidden_states_203_strides_0 = const()[name = string("hidden_states_203_strides_0"), val = tensor([1, 1])]; string hidden_states_203_pad_type_0 = const()[name = string("hidden_states_203_pad_type_0"), val = string("valid")]; tensor hidden_states_203_pad_0 = const()[name = string("hidden_states_203_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_203_dilations_0 = const()[name = string("hidden_states_203_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_203_groups_0 = const()[name = string("hidden_states_203_groups_0"), val = int32(1)]; tensor hidden_states_203_cast_fp16 = conv(dilations = hidden_states_203_dilations_0, groups = hidden_states_203_groups_0, pad = hidden_states_203_pad_0, pad_type = hidden_states_203_pad_type_0, strides = hidden_states_203_strides_0, weight = layers_20_self_attn_o_proj_weight_cast_fp16, x = attn_output_167_cast_fp16)[name = string("hidden_states_203_cast_fp16")]; tensor hidden_states_205_cast_fp16 = add(x = hidden_states_199_cast_fp16, y = hidden_states_203_cast_fp16)[name = string("hidden_states_205_cast_fp16")]; fp16 const_208_promoted_to_fp16 = const()[name = string("const_208_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7585_cast_fp16 = mul(x = hidden_states_205_cast_fp16, y = const_208_promoted_to_fp16)[name = string("op_7585_cast_fp16")]; int32 var_7583 = const()[name = string("op_7583"), val = int32(1)]; bool doubled_165_interleave_0 = const()[name = string("doubled_165_interleave_0"), val = bool(false)]; tensor doubled_165_cast_fp16 = concat(axis = var_7583, interleave = doubled_165_interleave_0, values = (hidden_states_205_cast_fp16, var_7585_cast_fp16))[name = string("doubled_165_cast_fp16")]; tensor out_83_axes_0 = const()[name = string("out_83_axes_0"), val = tensor([1])]; tensor out_83_gamma_0_to_fp16 = const()[name = string("out_83_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502547264)))]; fp16 var_7595_to_fp16 = const()[name = string("op_7595_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_83_cast_fp16 = layer_norm(axes = out_83_axes_0, epsilon = var_7595_to_fp16, gamma = out_83_gamma_0_to_fp16, x = doubled_165_cast_fp16)[name = string("out_83_cast_fp16")]; tensor var_7606_split_sizes_0 = const()[name = string("op_7606_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7606_axis_0 = const()[name = string("op_7606_axis_0"), val = int32(1)]; tensor var_7606_cast_fp16_0, tensor var_7606_cast_fp16_1 = split(axis = var_7606_axis_0, split_sizes = var_7606_split_sizes_0, x = out_83_cast_fp16)[name = string("op_7606_cast_fp16")]; tensor input_41_strides_0 = const()[name = string("input_41_strides_0"), val = tensor([1, 1])]; string input_41_pad_type_0 = const()[name = string("input_41_pad_type_0"), val = string("valid")]; tensor input_41_pad_0 = const()[name = string("input_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_41_dilations_0 = const()[name = string("input_41_dilations_0"), val = tensor([1, 1])]; int32 input_41_groups_0 = const()[name = string("input_41_groups_0"), val = int32(1)]; tensor input_41_cast_fp16 = conv(dilations = input_41_dilations_0, groups = input_41_groups_0, pad = input_41_pad_0, pad_type = input_41_pad_type_0, strides = input_41_strides_0, weight = layers_20_mlp_gate_proj_weight_cast_fp16, x = var_7606_cast_fp16_0)[name = string("input_41_cast_fp16")]; tensor var_7623_cast_fp16 = silu(x = input_41_cast_fp16)[name = string("op_7623_cast_fp16")]; tensor layers_20_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_20_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502555520)))]; tensor var_7629_strides_0 = const()[name = string("op_7629_strides_0"), val = tensor([1, 1])]; string var_7629_pad_type_0 = const()[name = string("op_7629_pad_type_0"), val = string("valid")]; tensor var_7629_pad_0 = const()[name = string("op_7629_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7629_dilations_0 = const()[name = string("op_7629_dilations_0"), val = tensor([1, 1])]; int32 var_7629_groups_0 = const()[name = string("op_7629_groups_0"), val = int32(1)]; tensor var_7629_cast_fp16 = conv(dilations = var_7629_dilations_0, groups = var_7629_groups_0, pad = var_7629_pad_0, pad_type = var_7629_pad_type_0, strides = var_7629_strides_0, weight = layers_20_mlp_up_proj_weight_to_fp16, x = var_7606_cast_fp16_0)[name = string("op_7629_cast_fp16")]; tensor x_209_cast_fp16 = mul(x = var_7623_cast_fp16, y = var_7629_cast_fp16)[name = string("x_209_cast_fp16")]; tensor hidden_states_207_strides_0 = const()[name = string("hidden_states_207_strides_0"), val = tensor([1, 1])]; string hidden_states_207_pad_type_0 = const()[name = string("hidden_states_207_pad_type_0"), val = string("valid")]; tensor hidden_states_207_pad_0 = const()[name = string("hidden_states_207_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_207_dilations_0 = const()[name = string("hidden_states_207_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_207_groups_0 = const()[name = string("hidden_states_207_groups_0"), val = int32(1)]; tensor hidden_states_207_cast_fp16 = conv(dilations = hidden_states_207_dilations_0, groups = hidden_states_207_groups_0, pad = hidden_states_207_pad_0, pad_type = hidden_states_207_pad_type_0, strides = hidden_states_207_strides_0, weight = layers_20_mlp_down_proj_weight_cast_fp16, x = x_209_cast_fp16)[name = string("hidden_states_207_cast_fp16")]; tensor hidden_states_209_cast_fp16 = add(x = hidden_states_205_cast_fp16, y = hidden_states_207_cast_fp16)[name = string("hidden_states_209_cast_fp16")]; fp16 const_210_promoted_to_fp16 = const()[name = string("const_210_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7647_cast_fp16 = mul(x = hidden_states_209_cast_fp16, y = const_210_promoted_to_fp16)[name = string("op_7647_cast_fp16")]; int32 var_7645 = const()[name = string("op_7645"), val = int32(1)]; bool doubled_169_interleave_0 = const()[name = string("doubled_169_interleave_0"), val = bool(false)]; tensor doubled_169_cast_fp16 = concat(axis = var_7645, interleave = doubled_169_interleave_0, values = (hidden_states_209_cast_fp16, var_7647_cast_fp16))[name = string("doubled_169_cast_fp16")]; tensor out_85_axes_0 = const()[name = string("out_85_axes_0"), val = tensor([1])]; tensor out_85_gamma_0_to_fp16 = const()[name = string("out_85_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527721408)))]; fp16 var_7657_to_fp16 = const()[name = string("op_7657_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_85_cast_fp16 = layer_norm(axes = out_85_axes_0, epsilon = var_7657_to_fp16, gamma = out_85_gamma_0_to_fp16, x = doubled_169_cast_fp16)[name = string("out_85_cast_fp16")]; tensor var_7668_split_sizes_0 = const()[name = string("op_7668_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7668_axis_0 = const()[name = string("op_7668_axis_0"), val = int32(1)]; tensor var_7668_cast_fp16_0, tensor var_7668_cast_fp16_1 = split(axis = var_7668_axis_0, split_sizes = var_7668_split_sizes_0, x = out_85_cast_fp16)[name = string("op_7668_cast_fp16")]; tensor query_states_127_strides_0 = const()[name = string("query_states_127_strides_0"), val = tensor([1, 1])]; string query_states_127_pad_type_0 = const()[name = string("query_states_127_pad_type_0"), val = string("valid")]; tensor query_states_127_pad_0 = const()[name = string("query_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_127_dilations_0 = const()[name = string("query_states_127_dilations_0"), val = tensor([1, 1])]; int32 query_states_127_groups_0 = const()[name = string("query_states_127_groups_0"), val = int32(1)]; tensor query_states_127_cast_fp16 = conv(dilations = query_states_127_dilations_0, groups = query_states_127_groups_0, pad = query_states_127_pad_0, pad_type = query_states_127_pad_type_0, strides = query_states_127_strides_0, weight = layers_21_self_attn_q_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("query_states_127_cast_fp16")]; tensor key_states_211_strides_0 = const()[name = string("key_states_211_strides_0"), val = tensor([1, 1])]; string key_states_211_pad_type_0 = const()[name = string("key_states_211_pad_type_0"), val = string("valid")]; tensor key_states_211_pad_0 = const()[name = string("key_states_211_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_211_dilations_0 = const()[name = string("key_states_211_dilations_0"), val = tensor([1, 1])]; int32 key_states_211_groups_0 = const()[name = string("key_states_211_groups_0"), val = int32(1)]; tensor key_states_211_cast_fp16 = conv(dilations = key_states_211_dilations_0, groups = key_states_211_groups_0, pad = key_states_211_pad_0, pad_type = key_states_211_pad_type_0, strides = key_states_211_strides_0, weight = layers_21_self_attn_k_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("key_states_211_cast_fp16")]; tensor layers_21_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_21_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527729664)))]; tensor value_states_127_strides_0 = const()[name = string("value_states_127_strides_0"), val = tensor([1, 1])]; string value_states_127_pad_type_0 = const()[name = string("value_states_127_pad_type_0"), val = string("valid")]; tensor value_states_127_pad_0 = const()[name = string("value_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_127_dilations_0 = const()[name = string("value_states_127_dilations_0"), val = tensor([1, 1])]; int32 value_states_127_groups_0 = const()[name = string("value_states_127_groups_0"), val = int32(1)]; tensor value_states_127_cast_fp16 = conv(dilations = value_states_127_dilations_0, groups = value_states_127_groups_0, pad = value_states_127_pad_0, pad_type = value_states_127_pad_type_0, strides = value_states_127_strides_0, weight = layers_21_self_attn_v_proj_weight_to_fp16, x = var_7668_cast_fp16_0)[name = string("value_states_127_cast_fp16")]; tensor concat_252x = const()[name = string("concat_252x"), val = tensor([1, 16, 128, -1])]; tensor x_211_cast_fp16 = reshape(shape = concat_252x, x = query_states_127_cast_fp16)[name = string("x_211_cast_fp16")]; tensor concat_253x = const()[name = string("concat_253x"), val = tensor([1, 2, 128, -1])]; tensor var_7725_cast_fp16 = reshape(shape = concat_253x, x = key_states_211_cast_fp16)[name = string("op_7725_cast_fp16")]; tensor concat_254x = const()[name = string("concat_254x"), val = tensor([1, 2, 128, -1])]; tensor var_7732_cast_fp16 = reshape(shape = concat_254x, x = value_states_127_cast_fp16)[name = string("op_7732_cast_fp16")]; tensor var_7736_cast_fp16 = mul(x = x_211_cast_fp16, y = var_869_cast_fp16)[name = string("op_7736_cast_fp16")]; tensor var_7737_split_sizes_0 = const()[name = string("op_7737_split_sizes_0"), val = tensor([64, 64])]; int32 var_7737_axis_0 = const()[name = string("op_7737_axis_0"), val = int32(-2)]; tensor var_7737_cast_fp16_0, tensor var_7737_cast_fp16_1 = split(axis = var_7737_axis_0, split_sizes = var_7737_split_sizes_0, x = x_211_cast_fp16)[name = string("op_7737_cast_fp16")]; fp16 const_212_promoted_to_fp16 = const()[name = string("const_212_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7739_cast_fp16 = mul(x = var_7737_cast_fp16_1, y = const_212_promoted_to_fp16)[name = string("op_7739_cast_fp16")]; int32 var_7741 = const()[name = string("op_7741"), val = int32(-2)]; bool var_7742_interleave_0 = const()[name = string("op_7742_interleave_0"), val = bool(false)]; tensor var_7742_cast_fp16 = concat(axis = var_7741, interleave = var_7742_interleave_0, values = (var_7739_cast_fp16, var_7737_cast_fp16_0))[name = string("op_7742_cast_fp16")]; tensor var_7743_cast_fp16 = mul(x = var_7742_cast_fp16, y = var_878_cast_fp16)[name = string("op_7743_cast_fp16")]; tensor query_states_129_cast_fp16 = add(x = var_7736_cast_fp16, y = var_7743_cast_fp16)[name = string("query_states_129_cast_fp16")]; tensor var_7749_cast_fp16 = mul(x = var_7725_cast_fp16, y = var_869_cast_fp16)[name = string("op_7749_cast_fp16")]; tensor var_7750_split_sizes_0 = const()[name = string("op_7750_split_sizes_0"), val = tensor([64, 64])]; int32 var_7750_axis_0 = const()[name = string("op_7750_axis_0"), val = int32(-2)]; tensor var_7750_cast_fp16_0, tensor var_7750_cast_fp16_1 = split(axis = var_7750_axis_0, split_sizes = var_7750_split_sizes_0, x = var_7725_cast_fp16)[name = string("op_7750_cast_fp16")]; fp16 const_213_promoted_to_fp16 = const()[name = string("const_213_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7752_cast_fp16 = mul(x = var_7750_cast_fp16_1, y = const_213_promoted_to_fp16)[name = string("op_7752_cast_fp16")]; int32 var_7754 = const()[name = string("op_7754"), val = int32(-2)]; bool var_7755_interleave_0 = const()[name = string("op_7755_interleave_0"), val = bool(false)]; tensor var_7755_cast_fp16 = concat(axis = var_7754, interleave = var_7755_interleave_0, values = (var_7752_cast_fp16, var_7750_cast_fp16_0))[name = string("op_7755_cast_fp16")]; tensor var_7756_cast_fp16 = mul(x = var_7755_cast_fp16, y = var_878_cast_fp16)[name = string("op_7756_cast_fp16")]; tensor key_states_215_cast_fp16 = add(x = var_7749_cast_fp16, y = var_7756_cast_fp16)[name = string("key_states_215_cast_fp16")]; tensor expand_dims_252 = const()[name = string("expand_dims_252"), val = tensor([21])]; tensor expand_dims_253 = const()[name = string("expand_dims_253"), val = tensor([0])]; tensor expand_dims_255 = const()[name = string("expand_dims_255"), val = tensor([0])]; int32 concat_257_axis_0 = const()[name = string("concat_257_axis_0"), val = int32(0)]; bool concat_257_interleave_0 = const()[name = string("concat_257_interleave_0"), val = bool(false)]; tensor concat_257 = concat(axis = concat_257_axis_0, interleave = concat_257_interleave_0, values = (expand_dims_252, expand_dims_253, position_id, expand_dims_255))[name = string("concat_257")]; tensor expand_dims_256 = const()[name = string("expand_dims_256"), val = tensor([22])]; tensor concat_258_values1_0 = const()[name = string("concat_258_values1_0"), val = tensor([0])]; tensor concat_258_values3_0 = const()[name = string("concat_258_values3_0"), val = tensor([0])]; int32 concat_258_axis_0 = const()[name = string("concat_258_axis_0"), val = int32(0)]; bool concat_258_interleave_0 = const()[name = string("concat_258_interleave_0"), val = bool(false)]; tensor concat_258 = concat(axis = concat_258_axis_0, interleave = concat_258_interleave_0, values = (expand_dims_256, concat_258_values1_0, cache_position_end, concat_258_values3_0))[name = string("concat_258")]; tensor key_states_217_perm_0 = const()[name = string("key_states_217_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_22_stride_0 = const()[name = string("key_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_217_cast_fp16 = transpose(perm = key_states_217_perm_0, x = key_states_215_cast_fp16)[name = string("transpose_364")]; tensor key_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = key_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = key_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_22_squeeze_mask_0, stride = key_cache_internal_tensor_assign_22_stride_0, update = key_states_217_cast_fp16, x = coreml_update_state_264)[name = string("key_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_22_cast_fp16, input = key_cache)[name = string("coreml_update_state_266_write_state")]; tensor coreml_update_state_266 = read_state(input = key_cache)[name = string("coreml_update_state_266")]; tensor value_states_129_perm_0 = const()[name = string("value_states_129_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_22_stride_0 = const()[name = string("value_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_129_cast_fp16 = transpose(perm = value_states_129_perm_0, x = var_7732_cast_fp16)[name = string("transpose_363")]; tensor value_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = value_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = value_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_22_squeeze_mask_0, stride = value_cache_internal_tensor_assign_22_stride_0, update = value_states_129_cast_fp16, x = coreml_update_state_265)[name = string("value_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_22_cast_fp16, input = value_cache)[name = string("coreml_update_state_267_write_state")]; tensor coreml_update_state_267 = read_state(input = value_cache)[name = string("coreml_update_state_267")]; tensor var_7826_begin_0 = const()[name = string("op_7826_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7826_end_0 = const()[name = string("op_7826_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7826_end_mask_0 = const()[name = string("op_7826_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7826_cast_fp16 = slice_by_index(begin = var_7826_begin_0, end = var_7826_end_0, end_mask = var_7826_end_mask_0, x = coreml_update_state_266)[name = string("op_7826_cast_fp16")]; tensor tile_42 = const()[name = string("tile_42"), val = tensor([1, 1])]; int32 var_7829_axis_0 = const()[name = string("op_7829_axis_0"), val = int32(1)]; tensor var_7829_cast_fp16_0, tensor var_7829_cast_fp16_1 = split(axis = var_7829_axis_0, split_sizes = tile_42, x = var_7826_cast_fp16)[name = string("op_7829_cast_fp16")]; tensor var_7836_begin_0 = const()[name = string("op_7836_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7836_end_0 = const()[name = string("op_7836_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7836_end_mask_0 = const()[name = string("op_7836_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7836_cast_fp16 = slice_by_index(begin = var_7836_begin_0, end = var_7836_end_0, end_mask = var_7836_end_mask_0, x = coreml_update_state_267)[name = string("op_7836_cast_fp16")]; tensor tile_43 = const()[name = string("tile_43"), val = tensor([1, 1])]; int32 var_7839_axis_0 = const()[name = string("op_7839_axis_0"), val = int32(1)]; tensor var_7839_cast_fp16_0, tensor var_7839_cast_fp16_1 = split(axis = var_7839_axis_0, split_sizes = tile_43, x = var_7836_cast_fp16)[name = string("op_7839_cast_fp16")]; tensor var_7842_split_sizes_0 = const()[name = string("op_7842_split_sizes_0"), val = tensor([8, 8])]; int32 var_7842_axis_0 = const()[name = string("op_7842_axis_0"), val = int32(1)]; tensor var_7842_0, tensor var_7842_1 = split(axis = var_7842_axis_0, split_sizes = var_7842_split_sizes_0, x = query_states_129_cast_fp16)[name = string("op_7842")]; bool attn_weights_337_transpose_x_0 = const()[name = string("attn_weights_337_transpose_x_0"), val = bool(false)]; bool attn_weights_337_transpose_y_0 = const()[name = string("attn_weights_337_transpose_y_0"), val = bool(false)]; tensor attn_weights_337_cast_fp16 = matmul(transpose_x = attn_weights_337_transpose_x_0, transpose_y = attn_weights_337_transpose_y_0, x = var_7829_cast_fp16_0, y = var_7842_0)[name = string("attn_weights_337_cast_fp16")]; fp16 var_7845_to_fp16 = const()[name = string("op_7845_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_339_cast_fp16 = mul(x = attn_weights_337_cast_fp16, y = var_7845_to_fp16)[name = string("attn_weights_339_cast_fp16")]; tensor attn_weights_341_cast_fp16 = add(x = attn_weights_339_cast_fp16, y = attn_mask_1)[name = string("attn_weights_341_cast_fp16")]; int32 var_7849 = const()[name = string("op_7849"), val = int32(-2)]; tensor attn_weights_343_cast_fp16 = softmax(axis = var_7849, x = attn_weights_341_cast_fp16)[name = string("attn_weights_343_cast_fp16")]; bool var_7855_transpose_x_1 = const()[name = string("op_7855_transpose_x_1"), val = bool(true)]; bool var_7855_transpose_y_1 = const()[name = string("op_7855_transpose_y_1"), val = bool(false)]; tensor var_7855_cast_fp16 = matmul(transpose_x = var_7855_transpose_x_1, transpose_y = var_7855_transpose_y_1, x = attn_weights_343_cast_fp16, y = var_7839_cast_fp16_0)[name = string("op_7855_cast_fp16")]; bool attn_weights_345_transpose_x_0 = const()[name = string("attn_weights_345_transpose_x_0"), val = bool(false)]; bool attn_weights_345_transpose_y_0 = const()[name = string("attn_weights_345_transpose_y_0"), val = bool(false)]; tensor attn_weights_345_cast_fp16 = matmul(transpose_x = attn_weights_345_transpose_x_0, transpose_y = attn_weights_345_transpose_y_0, x = var_7829_cast_fp16_1, y = var_7842_1)[name = string("attn_weights_345_cast_fp16")]; fp16 var_7857_to_fp16 = const()[name = string("op_7857_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_347_cast_fp16 = mul(x = attn_weights_345_cast_fp16, y = var_7857_to_fp16)[name = string("attn_weights_347_cast_fp16")]; tensor attn_weights_349_cast_fp16 = add(x = attn_weights_347_cast_fp16, y = attn_mask_1)[name = string("attn_weights_349_cast_fp16")]; int32 var_7861 = const()[name = string("op_7861"), val = int32(-2)]; tensor attn_weights_351_cast_fp16 = softmax(axis = var_7861, x = attn_weights_349_cast_fp16)[name = string("attn_weights_351_cast_fp16")]; bool attn_output_169_transpose_x_1 = const()[name = string("attn_output_169_transpose_x_1"), val = bool(true)]; bool attn_output_169_transpose_y_1 = const()[name = string("attn_output_169_transpose_y_1"), val = bool(false)]; tensor attn_output_169_cast_fp16 = matmul(transpose_x = attn_output_169_transpose_x_1, transpose_y = attn_output_169_transpose_y_1, x = attn_weights_351_cast_fp16, y = var_7839_cast_fp16_1)[name = string("attn_output_169_cast_fp16")]; int32 var_7869 = const()[name = string("op_7869"), val = int32(1)]; bool attn_output_171_interleave_0 = const()[name = string("attn_output_171_interleave_0"), val = bool(false)]; tensor attn_output_171_cast_fp16 = concat(axis = var_7869, interleave = attn_output_171_interleave_0, values = (var_7855_cast_fp16, attn_output_169_cast_fp16))[name = string("attn_output_171_cast_fp16")]; tensor var_7873_perm_0 = const()[name = string("op_7873_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_263x = const()[name = string("concat_263x"), val = tensor([1, 2048, 1, -1])]; tensor var_7873_cast_fp16 = transpose(perm = var_7873_perm_0, x = attn_output_171_cast_fp16)[name = string("transpose_362")]; tensor attn_output_175_cast_fp16 = reshape(shape = concat_263x, x = var_7873_cast_fp16)[name = string("attn_output_175_cast_fp16")]; tensor hidden_states_213_strides_0 = const()[name = string("hidden_states_213_strides_0"), val = tensor([1, 1])]; string hidden_states_213_pad_type_0 = const()[name = string("hidden_states_213_pad_type_0"), val = string("valid")]; tensor hidden_states_213_pad_0 = const()[name = string("hidden_states_213_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_213_dilations_0 = const()[name = string("hidden_states_213_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_213_groups_0 = const()[name = string("hidden_states_213_groups_0"), val = int32(1)]; tensor hidden_states_213_cast_fp16 = conv(dilations = hidden_states_213_dilations_0, groups = hidden_states_213_groups_0, pad = hidden_states_213_pad_0, pad_type = hidden_states_213_pad_type_0, strides = hidden_states_213_strides_0, weight = layers_21_self_attn_o_proj_weight_cast_fp16, x = attn_output_175_cast_fp16)[name = string("hidden_states_213_cast_fp16")]; tensor hidden_states_215_cast_fp16 = add(x = hidden_states_209_cast_fp16, y = hidden_states_213_cast_fp16)[name = string("hidden_states_215_cast_fp16")]; fp16 const_218_promoted_to_fp16 = const()[name = string("const_218_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7906_cast_fp16 = mul(x = hidden_states_215_cast_fp16, y = const_218_promoted_to_fp16)[name = string("op_7906_cast_fp16")]; int32 var_7904 = const()[name = string("op_7904"), val = int32(1)]; bool doubled_173_interleave_0 = const()[name = string("doubled_173_interleave_0"), val = bool(false)]; tensor doubled_173_cast_fp16 = concat(axis = var_7904, interleave = doubled_173_interleave_0, values = (hidden_states_215_cast_fp16, var_7906_cast_fp16))[name = string("doubled_173_cast_fp16")]; tensor out_87_axes_0 = const()[name = string("out_87_axes_0"), val = tensor([1])]; tensor out_87_gamma_0_to_fp16 = const()[name = string("out_87_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528778304)))]; fp16 var_7916_to_fp16 = const()[name = string("op_7916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_87_cast_fp16 = layer_norm(axes = out_87_axes_0, epsilon = var_7916_to_fp16, gamma = out_87_gamma_0_to_fp16, x = doubled_173_cast_fp16)[name = string("out_87_cast_fp16")]; tensor var_7927_split_sizes_0 = const()[name = string("op_7927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7927_axis_0 = const()[name = string("op_7927_axis_0"), val = int32(1)]; tensor var_7927_cast_fp16_0, tensor var_7927_cast_fp16_1 = split(axis = var_7927_axis_0, split_sizes = var_7927_split_sizes_0, x = out_87_cast_fp16)[name = string("op_7927_cast_fp16")]; tensor input_43_strides_0 = const()[name = string("input_43_strides_0"), val = tensor([1, 1])]; string input_43_pad_type_0 = const()[name = string("input_43_pad_type_0"), val = string("valid")]; tensor input_43_pad_0 = const()[name = string("input_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_43_dilations_0 = const()[name = string("input_43_dilations_0"), val = tensor([1, 1])]; int32 input_43_groups_0 = const()[name = string("input_43_groups_0"), val = int32(1)]; tensor input_43_cast_fp16 = conv(dilations = input_43_dilations_0, groups = input_43_groups_0, pad = input_43_pad_0, pad_type = input_43_pad_type_0, strides = input_43_strides_0, weight = layers_21_mlp_gate_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("input_43_cast_fp16")]; tensor var_7944_cast_fp16 = silu(x = input_43_cast_fp16)[name = string("op_7944_cast_fp16")]; tensor var_7950_strides_0 = const()[name = string("op_7950_strides_0"), val = tensor([1, 1])]; string var_7950_pad_type_0 = const()[name = string("op_7950_pad_type_0"), val = string("valid")]; tensor var_7950_pad_0 = const()[name = string("op_7950_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7950_dilations_0 = const()[name = string("op_7950_dilations_0"), val = tensor([1, 1])]; int32 var_7950_groups_0 = const()[name = string("op_7950_groups_0"), val = int32(1)]; tensor var_7950_cast_fp16 = conv(dilations = var_7950_dilations_0, groups = var_7950_groups_0, pad = var_7950_pad_0, pad_type = var_7950_pad_type_0, strides = var_7950_strides_0, weight = layers_21_mlp_up_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("op_7950_cast_fp16")]; tensor x_219_cast_fp16 = mul(x = var_7944_cast_fp16, y = var_7950_cast_fp16)[name = string("x_219_cast_fp16")]; tensor hidden_states_217_strides_0 = const()[name = string("hidden_states_217_strides_0"), val = tensor([1, 1])]; string hidden_states_217_pad_type_0 = const()[name = string("hidden_states_217_pad_type_0"), val = string("valid")]; tensor hidden_states_217_pad_0 = const()[name = string("hidden_states_217_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_217_dilations_0 = const()[name = string("hidden_states_217_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_217_groups_0 = const()[name = string("hidden_states_217_groups_0"), val = int32(1)]; tensor hidden_states_217_cast_fp16 = conv(dilations = hidden_states_217_dilations_0, groups = hidden_states_217_groups_0, pad = hidden_states_217_pad_0, pad_type = hidden_states_217_pad_type_0, strides = hidden_states_217_strides_0, weight = layers_21_mlp_down_proj_weight_cast_fp16, x = x_219_cast_fp16)[name = string("hidden_states_217_cast_fp16")]; tensor hidden_states_219_cast_fp16 = add(x = hidden_states_215_cast_fp16, y = hidden_states_217_cast_fp16)[name = string("hidden_states_219_cast_fp16")]; fp16 const_220_promoted_to_fp16 = const()[name = string("const_220_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7968_cast_fp16 = mul(x = hidden_states_219_cast_fp16, y = const_220_promoted_to_fp16)[name = string("op_7968_cast_fp16")]; int32 var_7966 = const()[name = string("op_7966"), val = int32(1)]; bool doubled_177_interleave_0 = const()[name = string("doubled_177_interleave_0"), val = bool(false)]; tensor doubled_177_cast_fp16 = concat(axis = var_7966, interleave = doubled_177_interleave_0, values = (hidden_states_219_cast_fp16, var_7968_cast_fp16))[name = string("doubled_177_cast_fp16")]; tensor out_89_axes_0 = const()[name = string("out_89_axes_0"), val = tensor([1])]; tensor out_89_gamma_0_to_fp16 = const()[name = string("out_89_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528786560)))]; fp16 var_7978_to_fp16 = const()[name = string("op_7978_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_89_cast_fp16 = layer_norm(axes = out_89_axes_0, epsilon = var_7978_to_fp16, gamma = out_89_gamma_0_to_fp16, x = doubled_177_cast_fp16)[name = string("out_89_cast_fp16")]; tensor var_7989_split_sizes_0 = const()[name = string("op_7989_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7989_axis_0 = const()[name = string("op_7989_axis_0"), val = int32(1)]; tensor var_7989_cast_fp16_0, tensor var_7989_cast_fp16_1 = split(axis = var_7989_axis_0, split_sizes = var_7989_split_sizes_0, x = out_89_cast_fp16)[name = string("op_7989_cast_fp16")]; tensor query_states_133_strides_0 = const()[name = string("query_states_133_strides_0"), val = tensor([1, 1])]; string query_states_133_pad_type_0 = const()[name = string("query_states_133_pad_type_0"), val = string("valid")]; tensor query_states_133_pad_0 = const()[name = string("query_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_133_dilations_0 = const()[name = string("query_states_133_dilations_0"), val = tensor([1, 1])]; int32 query_states_133_groups_0 = const()[name = string("query_states_133_groups_0"), val = int32(1)]; tensor query_states_133_cast_fp16 = conv(dilations = query_states_133_dilations_0, groups = query_states_133_groups_0, pad = query_states_133_pad_0, pad_type = query_states_133_pad_type_0, strides = query_states_133_strides_0, weight = layers_22_self_attn_q_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("query_states_133_cast_fp16")]; tensor key_states_221_strides_0 = const()[name = string("key_states_221_strides_0"), val = tensor([1, 1])]; string key_states_221_pad_type_0 = const()[name = string("key_states_221_pad_type_0"), val = string("valid")]; tensor key_states_221_pad_0 = const()[name = string("key_states_221_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_221_dilations_0 = const()[name = string("key_states_221_dilations_0"), val = tensor([1, 1])]; int32 key_states_221_groups_0 = const()[name = string("key_states_221_groups_0"), val = int32(1)]; tensor key_states_221_cast_fp16 = conv(dilations = key_states_221_dilations_0, groups = key_states_221_groups_0, pad = key_states_221_pad_0, pad_type = key_states_221_pad_type_0, strides = key_states_221_strides_0, weight = layers_22_self_attn_k_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("key_states_221_cast_fp16")]; tensor layers_22_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528794816)))]; tensor value_states_133_strides_0 = const()[name = string("value_states_133_strides_0"), val = tensor([1, 1])]; string value_states_133_pad_type_0 = const()[name = string("value_states_133_pad_type_0"), val = string("valid")]; tensor value_states_133_pad_0 = const()[name = string("value_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_133_dilations_0 = const()[name = string("value_states_133_dilations_0"), val = tensor([1, 1])]; int32 value_states_133_groups_0 = const()[name = string("value_states_133_groups_0"), val = int32(1)]; tensor value_states_133_cast_fp16 = conv(dilations = value_states_133_dilations_0, groups = value_states_133_groups_0, pad = value_states_133_pad_0, pad_type = value_states_133_pad_type_0, strides = value_states_133_strides_0, weight = layers_22_self_attn_v_proj_weight_to_fp16, x = var_7989_cast_fp16_0)[name = string("value_states_133_cast_fp16")]; tensor concat_264x = const()[name = string("concat_264x"), val = tensor([1, 16, 128, -1])]; tensor x_221_cast_fp16 = reshape(shape = concat_264x, x = query_states_133_cast_fp16)[name = string("x_221_cast_fp16")]; tensor concat_265x = const()[name = string("concat_265x"), val = tensor([1, 2, 128, -1])]; tensor var_8046_cast_fp16 = reshape(shape = concat_265x, x = key_states_221_cast_fp16)[name = string("op_8046_cast_fp16")]; tensor concat_266x = const()[name = string("concat_266x"), val = tensor([1, 2, 128, -1])]; tensor var_8053_cast_fp16 = reshape(shape = concat_266x, x = value_states_133_cast_fp16)[name = string("op_8053_cast_fp16")]; tensor var_8057_cast_fp16 = mul(x = x_221_cast_fp16, y = var_869_cast_fp16)[name = string("op_8057_cast_fp16")]; tensor var_8058_split_sizes_0 = const()[name = string("op_8058_split_sizes_0"), val = tensor([64, 64])]; int32 var_8058_axis_0 = const()[name = string("op_8058_axis_0"), val = int32(-2)]; tensor var_8058_cast_fp16_0, tensor var_8058_cast_fp16_1 = split(axis = var_8058_axis_0, split_sizes = var_8058_split_sizes_0, x = x_221_cast_fp16)[name = string("op_8058_cast_fp16")]; fp16 const_222_promoted_to_fp16 = const()[name = string("const_222_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8060_cast_fp16 = mul(x = var_8058_cast_fp16_1, y = const_222_promoted_to_fp16)[name = string("op_8060_cast_fp16")]; int32 var_8062 = const()[name = string("op_8062"), val = int32(-2)]; bool var_8063_interleave_0 = const()[name = string("op_8063_interleave_0"), val = bool(false)]; tensor var_8063_cast_fp16 = concat(axis = var_8062, interleave = var_8063_interleave_0, values = (var_8060_cast_fp16, var_8058_cast_fp16_0))[name = string("op_8063_cast_fp16")]; tensor var_8064_cast_fp16 = mul(x = var_8063_cast_fp16, y = var_878_cast_fp16)[name = string("op_8064_cast_fp16")]; tensor query_states_135_cast_fp16 = add(x = var_8057_cast_fp16, y = var_8064_cast_fp16)[name = string("query_states_135_cast_fp16")]; tensor var_8070_cast_fp16 = mul(x = var_8046_cast_fp16, y = var_869_cast_fp16)[name = string("op_8070_cast_fp16")]; tensor var_8071_split_sizes_0 = const()[name = string("op_8071_split_sizes_0"), val = tensor([64, 64])]; int32 var_8071_axis_0 = const()[name = string("op_8071_axis_0"), val = int32(-2)]; tensor var_8071_cast_fp16_0, tensor var_8071_cast_fp16_1 = split(axis = var_8071_axis_0, split_sizes = var_8071_split_sizes_0, x = var_8046_cast_fp16)[name = string("op_8071_cast_fp16")]; fp16 const_223_promoted_to_fp16 = const()[name = string("const_223_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8073_cast_fp16 = mul(x = var_8071_cast_fp16_1, y = const_223_promoted_to_fp16)[name = string("op_8073_cast_fp16")]; int32 var_8075 = const()[name = string("op_8075"), val = int32(-2)]; bool var_8076_interleave_0 = const()[name = string("op_8076_interleave_0"), val = bool(false)]; tensor var_8076_cast_fp16 = concat(axis = var_8075, interleave = var_8076_interleave_0, values = (var_8073_cast_fp16, var_8071_cast_fp16_0))[name = string("op_8076_cast_fp16")]; tensor var_8077_cast_fp16 = mul(x = var_8076_cast_fp16, y = var_878_cast_fp16)[name = string("op_8077_cast_fp16")]; tensor key_states_225_cast_fp16 = add(x = var_8070_cast_fp16, y = var_8077_cast_fp16)[name = string("key_states_225_cast_fp16")]; tensor expand_dims_264 = const()[name = string("expand_dims_264"), val = tensor([22])]; tensor expand_dims_265 = const()[name = string("expand_dims_265"), val = tensor([0])]; tensor expand_dims_267 = const()[name = string("expand_dims_267"), val = tensor([0])]; int32 concat_269_axis_0 = const()[name = string("concat_269_axis_0"), val = int32(0)]; bool concat_269_interleave_0 = const()[name = string("concat_269_interleave_0"), val = bool(false)]; tensor concat_269 = concat(axis = concat_269_axis_0, interleave = concat_269_interleave_0, values = (expand_dims_264, expand_dims_265, position_id, expand_dims_267))[name = string("concat_269")]; tensor expand_dims_268 = const()[name = string("expand_dims_268"), val = tensor([23])]; tensor concat_270_values1_0 = const()[name = string("concat_270_values1_0"), val = tensor([0])]; tensor concat_270_values3_0 = const()[name = string("concat_270_values3_0"), val = tensor([0])]; int32 concat_270_axis_0 = const()[name = string("concat_270_axis_0"), val = int32(0)]; bool concat_270_interleave_0 = const()[name = string("concat_270_interleave_0"), val = bool(false)]; tensor concat_270 = concat(axis = concat_270_axis_0, interleave = concat_270_interleave_0, values = (expand_dims_268, concat_270_values1_0, cache_position_end, concat_270_values3_0))[name = string("concat_270")]; tensor key_states_227_perm_0 = const()[name = string("key_states_227_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_23_stride_0 = const()[name = string("key_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_227_cast_fp16 = transpose(perm = key_states_227_perm_0, x = key_states_225_cast_fp16)[name = string("transpose_361")]; tensor key_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = key_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = key_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_23_squeeze_mask_0, stride = key_cache_internal_tensor_assign_23_stride_0, update = key_states_227_cast_fp16, x = coreml_update_state_266)[name = string("key_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_23_cast_fp16, input = key_cache)[name = string("coreml_update_state_268_write_state")]; tensor coreml_update_state_268 = read_state(input = key_cache)[name = string("coreml_update_state_268")]; tensor value_states_135_perm_0 = const()[name = string("value_states_135_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_23_stride_0 = const()[name = string("value_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_135_cast_fp16 = transpose(perm = value_states_135_perm_0, x = var_8053_cast_fp16)[name = string("transpose_360")]; tensor value_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = value_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = value_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_23_squeeze_mask_0, stride = value_cache_internal_tensor_assign_23_stride_0, update = value_states_135_cast_fp16, x = coreml_update_state_267)[name = string("value_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_23_cast_fp16, input = value_cache)[name = string("coreml_update_state_269_write_state")]; tensor coreml_update_state_269 = read_state(input = value_cache)[name = string("coreml_update_state_269")]; tensor var_8147_begin_0 = const()[name = string("op_8147_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8147_end_0 = const()[name = string("op_8147_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8147_end_mask_0 = const()[name = string("op_8147_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8147_cast_fp16 = slice_by_index(begin = var_8147_begin_0, end = var_8147_end_0, end_mask = var_8147_end_mask_0, x = coreml_update_state_268)[name = string("op_8147_cast_fp16")]; tensor tile_44 = const()[name = string("tile_44"), val = tensor([1, 1])]; int32 var_8150_axis_0 = const()[name = string("op_8150_axis_0"), val = int32(1)]; tensor var_8150_cast_fp16_0, tensor var_8150_cast_fp16_1 = split(axis = var_8150_axis_0, split_sizes = tile_44, x = var_8147_cast_fp16)[name = string("op_8150_cast_fp16")]; tensor var_8157_begin_0 = const()[name = string("op_8157_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8157_end_0 = const()[name = string("op_8157_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8157_end_mask_0 = const()[name = string("op_8157_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8157_cast_fp16 = slice_by_index(begin = var_8157_begin_0, end = var_8157_end_0, end_mask = var_8157_end_mask_0, x = coreml_update_state_269)[name = string("op_8157_cast_fp16")]; tensor tile_45 = const()[name = string("tile_45"), val = tensor([1, 1])]; int32 var_8160_axis_0 = const()[name = string("op_8160_axis_0"), val = int32(1)]; tensor var_8160_cast_fp16_0, tensor var_8160_cast_fp16_1 = split(axis = var_8160_axis_0, split_sizes = tile_45, x = var_8157_cast_fp16)[name = string("op_8160_cast_fp16")]; tensor var_8163_split_sizes_0 = const()[name = string("op_8163_split_sizes_0"), val = tensor([8, 8])]; int32 var_8163_axis_0 = const()[name = string("op_8163_axis_0"), val = int32(1)]; tensor var_8163_0, tensor var_8163_1 = split(axis = var_8163_axis_0, split_sizes = var_8163_split_sizes_0, x = query_states_135_cast_fp16)[name = string("op_8163")]; bool attn_weights_353_transpose_x_0 = const()[name = string("attn_weights_353_transpose_x_0"), val = bool(false)]; bool attn_weights_353_transpose_y_0 = const()[name = string("attn_weights_353_transpose_y_0"), val = bool(false)]; tensor attn_weights_353_cast_fp16 = matmul(transpose_x = attn_weights_353_transpose_x_0, transpose_y = attn_weights_353_transpose_y_0, x = var_8150_cast_fp16_0, y = var_8163_0)[name = string("attn_weights_353_cast_fp16")]; fp16 var_8166_to_fp16 = const()[name = string("op_8166_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_355_cast_fp16 = mul(x = attn_weights_353_cast_fp16, y = var_8166_to_fp16)[name = string("attn_weights_355_cast_fp16")]; tensor attn_weights_357_cast_fp16 = add(x = attn_weights_355_cast_fp16, y = attn_mask_1)[name = string("attn_weights_357_cast_fp16")]; int32 var_8170 = const()[name = string("op_8170"), val = int32(-2)]; tensor attn_weights_359_cast_fp16 = softmax(axis = var_8170, x = attn_weights_357_cast_fp16)[name = string("attn_weights_359_cast_fp16")]; bool var_8176_transpose_x_1 = const()[name = string("op_8176_transpose_x_1"), val = bool(true)]; bool var_8176_transpose_y_1 = const()[name = string("op_8176_transpose_y_1"), val = bool(false)]; tensor var_8176_cast_fp16 = matmul(transpose_x = var_8176_transpose_x_1, transpose_y = var_8176_transpose_y_1, x = attn_weights_359_cast_fp16, y = var_8160_cast_fp16_0)[name = string("op_8176_cast_fp16")]; bool attn_weights_361_transpose_x_0 = const()[name = string("attn_weights_361_transpose_x_0"), val = bool(false)]; bool attn_weights_361_transpose_y_0 = const()[name = string("attn_weights_361_transpose_y_0"), val = bool(false)]; tensor attn_weights_361_cast_fp16 = matmul(transpose_x = attn_weights_361_transpose_x_0, transpose_y = attn_weights_361_transpose_y_0, x = var_8150_cast_fp16_1, y = var_8163_1)[name = string("attn_weights_361_cast_fp16")]; fp16 var_8178_to_fp16 = const()[name = string("op_8178_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_363_cast_fp16 = mul(x = attn_weights_361_cast_fp16, y = var_8178_to_fp16)[name = string("attn_weights_363_cast_fp16")]; tensor attn_weights_365_cast_fp16 = add(x = attn_weights_363_cast_fp16, y = attn_mask_1)[name = string("attn_weights_365_cast_fp16")]; int32 var_8182 = const()[name = string("op_8182"), val = int32(-2)]; tensor attn_weights_367_cast_fp16 = softmax(axis = var_8182, x = attn_weights_365_cast_fp16)[name = string("attn_weights_367_cast_fp16")]; bool attn_output_177_transpose_x_1 = const()[name = string("attn_output_177_transpose_x_1"), val = bool(true)]; bool attn_output_177_transpose_y_1 = const()[name = string("attn_output_177_transpose_y_1"), val = bool(false)]; tensor attn_output_177_cast_fp16 = matmul(transpose_x = attn_output_177_transpose_x_1, transpose_y = attn_output_177_transpose_y_1, x = attn_weights_367_cast_fp16, y = var_8160_cast_fp16_1)[name = string("attn_output_177_cast_fp16")]; int32 var_8190 = const()[name = string("op_8190"), val = int32(1)]; bool attn_output_179_interleave_0 = const()[name = string("attn_output_179_interleave_0"), val = bool(false)]; tensor attn_output_179_cast_fp16 = concat(axis = var_8190, interleave = attn_output_179_interleave_0, values = (var_8176_cast_fp16, attn_output_177_cast_fp16))[name = string("attn_output_179_cast_fp16")]; tensor var_8194_perm_0 = const()[name = string("op_8194_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_275x = const()[name = string("concat_275x"), val = tensor([1, 2048, 1, -1])]; tensor var_8194_cast_fp16 = transpose(perm = var_8194_perm_0, x = attn_output_179_cast_fp16)[name = string("transpose_359")]; tensor attn_output_183_cast_fp16 = reshape(shape = concat_275x, x = var_8194_cast_fp16)[name = string("attn_output_183_cast_fp16")]; tensor layers_22_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1529843456)))]; tensor hidden_states_223_strides_0 = const()[name = string("hidden_states_223_strides_0"), val = tensor([1, 1])]; string hidden_states_223_pad_type_0 = const()[name = string("hidden_states_223_pad_type_0"), val = string("valid")]; tensor hidden_states_223_pad_0 = const()[name = string("hidden_states_223_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_223_dilations_0 = const()[name = string("hidden_states_223_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_223_groups_0 = const()[name = string("hidden_states_223_groups_0"), val = int32(1)]; tensor hidden_states_223_cast_fp16 = conv(dilations = hidden_states_223_dilations_0, groups = hidden_states_223_groups_0, pad = hidden_states_223_pad_0, pad_type = hidden_states_223_pad_type_0, strides = hidden_states_223_strides_0, weight = layers_22_self_attn_o_proj_weight_to_fp16, x = attn_output_183_cast_fp16)[name = string("hidden_states_223_cast_fp16")]; tensor hidden_states_225_cast_fp16 = add(x = hidden_states_219_cast_fp16, y = hidden_states_223_cast_fp16)[name = string("hidden_states_225_cast_fp16")]; fp16 const_228_promoted_to_fp16 = const()[name = string("const_228_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8227_cast_fp16 = mul(x = hidden_states_225_cast_fp16, y = const_228_promoted_to_fp16)[name = string("op_8227_cast_fp16")]; int32 var_8225 = const()[name = string("op_8225"), val = int32(1)]; bool doubled_181_interleave_0 = const()[name = string("doubled_181_interleave_0"), val = bool(false)]; tensor doubled_181_cast_fp16 = concat(axis = var_8225, interleave = doubled_181_interleave_0, values = (hidden_states_225_cast_fp16, var_8227_cast_fp16))[name = string("doubled_181_cast_fp16")]; tensor out_91_axes_0 = const()[name = string("out_91_axes_0"), val = tensor([1])]; tensor out_91_gamma_0_to_fp16 = const()[name = string("out_91_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538232128)))]; fp16 var_8237_to_fp16 = const()[name = string("op_8237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_91_cast_fp16 = layer_norm(axes = out_91_axes_0, epsilon = var_8237_to_fp16, gamma = out_91_gamma_0_to_fp16, x = doubled_181_cast_fp16)[name = string("out_91_cast_fp16")]; tensor var_8248_split_sizes_0 = const()[name = string("op_8248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8248_axis_0 = const()[name = string("op_8248_axis_0"), val = int32(1)]; tensor var_8248_cast_fp16_0, tensor var_8248_cast_fp16_1 = split(axis = var_8248_axis_0, split_sizes = var_8248_split_sizes_0, x = out_91_cast_fp16)[name = string("op_8248_cast_fp16")]; tensor input_45_strides_0 = const()[name = string("input_45_strides_0"), val = tensor([1, 1])]; string input_45_pad_type_0 = const()[name = string("input_45_pad_type_0"), val = string("valid")]; tensor input_45_pad_0 = const()[name = string("input_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_45_dilations_0 = const()[name = string("input_45_dilations_0"), val = tensor([1, 1])]; int32 input_45_groups_0 = const()[name = string("input_45_groups_0"), val = int32(1)]; tensor input_45_cast_fp16 = conv(dilations = input_45_dilations_0, groups = input_45_groups_0, pad = input_45_pad_0, pad_type = input_45_pad_type_0, strides = input_45_strides_0, weight = layers_22_mlp_gate_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("input_45_cast_fp16")]; tensor var_8265_cast_fp16 = silu(x = input_45_cast_fp16)[name = string("op_8265_cast_fp16")]; tensor var_8271_strides_0 = const()[name = string("op_8271_strides_0"), val = tensor([1, 1])]; string var_8271_pad_type_0 = const()[name = string("op_8271_pad_type_0"), val = string("valid")]; tensor var_8271_pad_0 = const()[name = string("op_8271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8271_dilations_0 = const()[name = string("op_8271_dilations_0"), val = tensor([1, 1])]; int32 var_8271_groups_0 = const()[name = string("op_8271_groups_0"), val = int32(1)]; tensor var_8271_cast_fp16 = conv(dilations = var_8271_dilations_0, groups = var_8271_groups_0, pad = var_8271_pad_0, pad_type = var_8271_pad_type_0, strides = var_8271_strides_0, weight = layers_22_mlp_up_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("op_8271_cast_fp16")]; tensor x_229_cast_fp16 = mul(x = var_8265_cast_fp16, y = var_8271_cast_fp16)[name = string("x_229_cast_fp16")]; tensor hidden_states_227_strides_0 = const()[name = string("hidden_states_227_strides_0"), val = tensor([1, 1])]; string hidden_states_227_pad_type_0 = const()[name = string("hidden_states_227_pad_type_0"), val = string("valid")]; tensor hidden_states_227_pad_0 = const()[name = string("hidden_states_227_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_227_dilations_0 = const()[name = string("hidden_states_227_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_227_groups_0 = const()[name = string("hidden_states_227_groups_0"), val = int32(1)]; tensor hidden_states_227_cast_fp16 = conv(dilations = hidden_states_227_dilations_0, groups = hidden_states_227_groups_0, pad = hidden_states_227_pad_0, pad_type = hidden_states_227_pad_type_0, strides = hidden_states_227_strides_0, weight = layers_22_mlp_down_proj_weight_cast_fp16, x = x_229_cast_fp16)[name = string("hidden_states_227_cast_fp16")]; tensor hidden_states_229_cast_fp16 = add(x = hidden_states_225_cast_fp16, y = hidden_states_227_cast_fp16)[name = string("hidden_states_229_cast_fp16")]; fp16 const_230_promoted_to_fp16 = const()[name = string("const_230_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8289_cast_fp16 = mul(x = hidden_states_229_cast_fp16, y = const_230_promoted_to_fp16)[name = string("op_8289_cast_fp16")]; int32 var_8287 = const()[name = string("op_8287"), val = int32(1)]; bool doubled_185_interleave_0 = const()[name = string("doubled_185_interleave_0"), val = bool(false)]; tensor doubled_185_cast_fp16 = concat(axis = var_8287, interleave = doubled_185_interleave_0, values = (hidden_states_229_cast_fp16, var_8289_cast_fp16))[name = string("doubled_185_cast_fp16")]; tensor out_93_axes_0 = const()[name = string("out_93_axes_0"), val = tensor([1])]; tensor out_93_gamma_0_to_fp16 = const()[name = string("out_93_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538240384)))]; fp16 var_8299_to_fp16 = const()[name = string("op_8299_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_93_cast_fp16 = layer_norm(axes = out_93_axes_0, epsilon = var_8299_to_fp16, gamma = out_93_gamma_0_to_fp16, x = doubled_185_cast_fp16)[name = string("out_93_cast_fp16")]; tensor var_8310_split_sizes_0 = const()[name = string("op_8310_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8310_axis_0 = const()[name = string("op_8310_axis_0"), val = int32(1)]; tensor var_8310_cast_fp16_0, tensor var_8310_cast_fp16_1 = split(axis = var_8310_axis_0, split_sizes = var_8310_split_sizes_0, x = out_93_cast_fp16)[name = string("op_8310_cast_fp16")]; tensor query_states_139_strides_0 = const()[name = string("query_states_139_strides_0"), val = tensor([1, 1])]; string query_states_139_pad_type_0 = const()[name = string("query_states_139_pad_type_0"), val = string("valid")]; tensor query_states_139_pad_0 = const()[name = string("query_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_139_dilations_0 = const()[name = string("query_states_139_dilations_0"), val = tensor([1, 1])]; int32 query_states_139_groups_0 = const()[name = string("query_states_139_groups_0"), val = int32(1)]; tensor query_states_139_cast_fp16 = conv(dilations = query_states_139_dilations_0, groups = query_states_139_groups_0, pad = query_states_139_pad_0, pad_type = query_states_139_pad_type_0, strides = query_states_139_strides_0, weight = layers_23_self_attn_q_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("query_states_139_cast_fp16")]; tensor key_states_231_strides_0 = const()[name = string("key_states_231_strides_0"), val = tensor([1, 1])]; string key_states_231_pad_type_0 = const()[name = string("key_states_231_pad_type_0"), val = string("valid")]; tensor key_states_231_pad_0 = const()[name = string("key_states_231_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_231_dilations_0 = const()[name = string("key_states_231_dilations_0"), val = tensor([1, 1])]; int32 key_states_231_groups_0 = const()[name = string("key_states_231_groups_0"), val = int32(1)]; tensor key_states_231_cast_fp16 = conv(dilations = key_states_231_dilations_0, groups = key_states_231_groups_0, pad = key_states_231_pad_0, pad_type = key_states_231_pad_type_0, strides = key_states_231_strides_0, weight = layers_23_self_attn_k_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("key_states_231_cast_fp16")]; tensor layers_23_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_23_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538248640)))]; tensor value_states_139_strides_0 = const()[name = string("value_states_139_strides_0"), val = tensor([1, 1])]; string value_states_139_pad_type_0 = const()[name = string("value_states_139_pad_type_0"), val = string("valid")]; tensor value_states_139_pad_0 = const()[name = string("value_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_139_dilations_0 = const()[name = string("value_states_139_dilations_0"), val = tensor([1, 1])]; int32 value_states_139_groups_0 = const()[name = string("value_states_139_groups_0"), val = int32(1)]; tensor value_states_139_cast_fp16 = conv(dilations = value_states_139_dilations_0, groups = value_states_139_groups_0, pad = value_states_139_pad_0, pad_type = value_states_139_pad_type_0, strides = value_states_139_strides_0, weight = layers_23_self_attn_v_proj_weight_to_fp16, x = var_8310_cast_fp16_0)[name = string("value_states_139_cast_fp16")]; tensor concat_276x = const()[name = string("concat_276x"), val = tensor([1, 16, 128, -1])]; tensor x_231_cast_fp16 = reshape(shape = concat_276x, x = query_states_139_cast_fp16)[name = string("x_231_cast_fp16")]; tensor concat_277x = const()[name = string("concat_277x"), val = tensor([1, 2, 128, -1])]; tensor var_8367_cast_fp16 = reshape(shape = concat_277x, x = key_states_231_cast_fp16)[name = string("op_8367_cast_fp16")]; tensor concat_278x = const()[name = string("concat_278x"), val = tensor([1, 2, 128, -1])]; tensor var_8374_cast_fp16 = reshape(shape = concat_278x, x = value_states_139_cast_fp16)[name = string("op_8374_cast_fp16")]; tensor var_8378_cast_fp16 = mul(x = x_231_cast_fp16, y = var_869_cast_fp16)[name = string("op_8378_cast_fp16")]; tensor var_8379_split_sizes_0 = const()[name = string("op_8379_split_sizes_0"), val = tensor([64, 64])]; int32 var_8379_axis_0 = const()[name = string("op_8379_axis_0"), val = int32(-2)]; tensor var_8379_cast_fp16_0, tensor var_8379_cast_fp16_1 = split(axis = var_8379_axis_0, split_sizes = var_8379_split_sizes_0, x = x_231_cast_fp16)[name = string("op_8379_cast_fp16")]; fp16 const_232_promoted_to_fp16 = const()[name = string("const_232_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8381_cast_fp16 = mul(x = var_8379_cast_fp16_1, y = const_232_promoted_to_fp16)[name = string("op_8381_cast_fp16")]; int32 var_8383 = const()[name = string("op_8383"), val = int32(-2)]; bool var_8384_interleave_0 = const()[name = string("op_8384_interleave_0"), val = bool(false)]; tensor var_8384_cast_fp16 = concat(axis = var_8383, interleave = var_8384_interleave_0, values = (var_8381_cast_fp16, var_8379_cast_fp16_0))[name = string("op_8384_cast_fp16")]; tensor var_8385_cast_fp16 = mul(x = var_8384_cast_fp16, y = var_878_cast_fp16)[name = string("op_8385_cast_fp16")]; tensor query_states_141_cast_fp16 = add(x = var_8378_cast_fp16, y = var_8385_cast_fp16)[name = string("query_states_141_cast_fp16")]; tensor var_8391_cast_fp16 = mul(x = var_8367_cast_fp16, y = var_869_cast_fp16)[name = string("op_8391_cast_fp16")]; tensor var_8392_split_sizes_0 = const()[name = string("op_8392_split_sizes_0"), val = tensor([64, 64])]; int32 var_8392_axis_0 = const()[name = string("op_8392_axis_0"), val = int32(-2)]; tensor var_8392_cast_fp16_0, tensor var_8392_cast_fp16_1 = split(axis = var_8392_axis_0, split_sizes = var_8392_split_sizes_0, x = var_8367_cast_fp16)[name = string("op_8392_cast_fp16")]; fp16 const_233_promoted_to_fp16 = const()[name = string("const_233_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8394_cast_fp16 = mul(x = var_8392_cast_fp16_1, y = const_233_promoted_to_fp16)[name = string("op_8394_cast_fp16")]; int32 var_8396 = const()[name = string("op_8396"), val = int32(-2)]; bool var_8397_interleave_0 = const()[name = string("op_8397_interleave_0"), val = bool(false)]; tensor var_8397_cast_fp16 = concat(axis = var_8396, interleave = var_8397_interleave_0, values = (var_8394_cast_fp16, var_8392_cast_fp16_0))[name = string("op_8397_cast_fp16")]; tensor var_8398_cast_fp16 = mul(x = var_8397_cast_fp16, y = var_878_cast_fp16)[name = string("op_8398_cast_fp16")]; tensor key_states_235_cast_fp16 = add(x = var_8391_cast_fp16, y = var_8398_cast_fp16)[name = string("key_states_235_cast_fp16")]; tensor expand_dims_276 = const()[name = string("expand_dims_276"), val = tensor([23])]; tensor expand_dims_277 = const()[name = string("expand_dims_277"), val = tensor([0])]; tensor expand_dims_279 = const()[name = string("expand_dims_279"), val = tensor([0])]; int32 concat_281_axis_0 = const()[name = string("concat_281_axis_0"), val = int32(0)]; bool concat_281_interleave_0 = const()[name = string("concat_281_interleave_0"), val = bool(false)]; tensor concat_281 = concat(axis = concat_281_axis_0, interleave = concat_281_interleave_0, values = (expand_dims_276, expand_dims_277, position_id, expand_dims_279))[name = string("concat_281")]; tensor expand_dims_280 = const()[name = string("expand_dims_280"), val = tensor([24])]; tensor concat_282_values1_0 = const()[name = string("concat_282_values1_0"), val = tensor([0])]; tensor concat_282_values3_0 = const()[name = string("concat_282_values3_0"), val = tensor([0])]; int32 concat_282_axis_0 = const()[name = string("concat_282_axis_0"), val = int32(0)]; bool concat_282_interleave_0 = const()[name = string("concat_282_interleave_0"), val = bool(false)]; tensor concat_282 = concat(axis = concat_282_axis_0, interleave = concat_282_interleave_0, values = (expand_dims_280, concat_282_values1_0, cache_position_end, concat_282_values3_0))[name = string("concat_282")]; tensor key_states_237_perm_0 = const()[name = string("key_states_237_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_24_stride_0 = const()[name = string("key_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_237_cast_fp16 = transpose(perm = key_states_237_perm_0, x = key_states_235_cast_fp16)[name = string("transpose_358")]; tensor key_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = key_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = key_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_24_squeeze_mask_0, stride = key_cache_internal_tensor_assign_24_stride_0, update = key_states_237_cast_fp16, x = coreml_update_state_268)[name = string("key_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_24_cast_fp16, input = key_cache)[name = string("coreml_update_state_270_write_state")]; tensor coreml_update_state_270 = read_state(input = key_cache)[name = string("coreml_update_state_270")]; tensor value_states_141_perm_0 = const()[name = string("value_states_141_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_24_stride_0 = const()[name = string("value_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_141_cast_fp16 = transpose(perm = value_states_141_perm_0, x = var_8374_cast_fp16)[name = string("transpose_357")]; tensor value_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = value_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = value_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_24_squeeze_mask_0, stride = value_cache_internal_tensor_assign_24_stride_0, update = value_states_141_cast_fp16, x = coreml_update_state_269)[name = string("value_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_24_cast_fp16, input = value_cache)[name = string("coreml_update_state_271_write_state")]; tensor coreml_update_state_271 = read_state(input = value_cache)[name = string("coreml_update_state_271")]; tensor var_8468_begin_0 = const()[name = string("op_8468_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8468_end_0 = const()[name = string("op_8468_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8468_end_mask_0 = const()[name = string("op_8468_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8468_cast_fp16 = slice_by_index(begin = var_8468_begin_0, end = var_8468_end_0, end_mask = var_8468_end_mask_0, x = coreml_update_state_270)[name = string("op_8468_cast_fp16")]; tensor tile_46 = const()[name = string("tile_46"), val = tensor([1, 1])]; int32 var_8471_axis_0 = const()[name = string("op_8471_axis_0"), val = int32(1)]; tensor var_8471_cast_fp16_0, tensor var_8471_cast_fp16_1 = split(axis = var_8471_axis_0, split_sizes = tile_46, x = var_8468_cast_fp16)[name = string("op_8471_cast_fp16")]; tensor var_8478_begin_0 = const()[name = string("op_8478_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8478_end_0 = const()[name = string("op_8478_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8478_end_mask_0 = const()[name = string("op_8478_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8478_cast_fp16 = slice_by_index(begin = var_8478_begin_0, end = var_8478_end_0, end_mask = var_8478_end_mask_0, x = coreml_update_state_271)[name = string("op_8478_cast_fp16")]; tensor tile_47 = const()[name = string("tile_47"), val = tensor([1, 1])]; int32 var_8481_axis_0 = const()[name = string("op_8481_axis_0"), val = int32(1)]; tensor var_8481_cast_fp16_0, tensor var_8481_cast_fp16_1 = split(axis = var_8481_axis_0, split_sizes = tile_47, x = var_8478_cast_fp16)[name = string("op_8481_cast_fp16")]; tensor var_8484_split_sizes_0 = const()[name = string("op_8484_split_sizes_0"), val = tensor([8, 8])]; int32 var_8484_axis_0 = const()[name = string("op_8484_axis_0"), val = int32(1)]; tensor var_8484_0, tensor var_8484_1 = split(axis = var_8484_axis_0, split_sizes = var_8484_split_sizes_0, x = query_states_141_cast_fp16)[name = string("op_8484")]; bool attn_weights_369_transpose_x_0 = const()[name = string("attn_weights_369_transpose_x_0"), val = bool(false)]; bool attn_weights_369_transpose_y_0 = const()[name = string("attn_weights_369_transpose_y_0"), val = bool(false)]; tensor attn_weights_369_cast_fp16 = matmul(transpose_x = attn_weights_369_transpose_x_0, transpose_y = attn_weights_369_transpose_y_0, x = var_8471_cast_fp16_0, y = var_8484_0)[name = string("attn_weights_369_cast_fp16")]; fp16 var_8487_to_fp16 = const()[name = string("op_8487_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_371_cast_fp16 = mul(x = attn_weights_369_cast_fp16, y = var_8487_to_fp16)[name = string("attn_weights_371_cast_fp16")]; tensor attn_weights_373_cast_fp16 = add(x = attn_weights_371_cast_fp16, y = attn_mask_1)[name = string("attn_weights_373_cast_fp16")]; int32 var_8491 = const()[name = string("op_8491"), val = int32(-2)]; tensor attn_weights_375_cast_fp16 = softmax(axis = var_8491, x = attn_weights_373_cast_fp16)[name = string("attn_weights_375_cast_fp16")]; bool var_8497_transpose_x_1 = const()[name = string("op_8497_transpose_x_1"), val = bool(true)]; bool var_8497_transpose_y_1 = const()[name = string("op_8497_transpose_y_1"), val = bool(false)]; tensor var_8497_cast_fp16 = matmul(transpose_x = var_8497_transpose_x_1, transpose_y = var_8497_transpose_y_1, x = attn_weights_375_cast_fp16, y = var_8481_cast_fp16_0)[name = string("op_8497_cast_fp16")]; bool attn_weights_377_transpose_x_0 = const()[name = string("attn_weights_377_transpose_x_0"), val = bool(false)]; bool attn_weights_377_transpose_y_0 = const()[name = string("attn_weights_377_transpose_y_0"), val = bool(false)]; tensor attn_weights_377_cast_fp16 = matmul(transpose_x = attn_weights_377_transpose_x_0, transpose_y = attn_weights_377_transpose_y_0, x = var_8471_cast_fp16_1, y = var_8484_1)[name = string("attn_weights_377_cast_fp16")]; fp16 var_8499_to_fp16 = const()[name = string("op_8499_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_379_cast_fp16 = mul(x = attn_weights_377_cast_fp16, y = var_8499_to_fp16)[name = string("attn_weights_379_cast_fp16")]; tensor attn_weights_381_cast_fp16 = add(x = attn_weights_379_cast_fp16, y = attn_mask_1)[name = string("attn_weights_381_cast_fp16")]; int32 var_8503 = const()[name = string("op_8503"), val = int32(-2)]; tensor attn_weights_383_cast_fp16 = softmax(axis = var_8503, x = attn_weights_381_cast_fp16)[name = string("attn_weights_383_cast_fp16")]; bool attn_output_185_transpose_x_1 = const()[name = string("attn_output_185_transpose_x_1"), val = bool(true)]; bool attn_output_185_transpose_y_1 = const()[name = string("attn_output_185_transpose_y_1"), val = bool(false)]; tensor attn_output_185_cast_fp16 = matmul(transpose_x = attn_output_185_transpose_x_1, transpose_y = attn_output_185_transpose_y_1, x = attn_weights_383_cast_fp16, y = var_8481_cast_fp16_1)[name = string("attn_output_185_cast_fp16")]; int32 var_8511 = const()[name = string("op_8511"), val = int32(1)]; bool attn_output_187_interleave_0 = const()[name = string("attn_output_187_interleave_0"), val = bool(false)]; tensor attn_output_187_cast_fp16 = concat(axis = var_8511, interleave = attn_output_187_interleave_0, values = (var_8497_cast_fp16, attn_output_185_cast_fp16))[name = string("attn_output_187_cast_fp16")]; tensor var_8515_perm_0 = const()[name = string("op_8515_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_287x = const()[name = string("concat_287x"), val = tensor([1, 2048, 1, -1])]; tensor var_8515_cast_fp16 = transpose(perm = var_8515_perm_0, x = attn_output_187_cast_fp16)[name = string("transpose_356")]; tensor attn_output_191_cast_fp16 = reshape(shape = concat_287x, x = var_8515_cast_fp16)[name = string("attn_output_191_cast_fp16")]; tensor hidden_states_233_strides_0 = const()[name = string("hidden_states_233_strides_0"), val = tensor([1, 1])]; string hidden_states_233_pad_type_0 = const()[name = string("hidden_states_233_pad_type_0"), val = string("valid")]; tensor hidden_states_233_pad_0 = const()[name = string("hidden_states_233_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_233_dilations_0 = const()[name = string("hidden_states_233_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_233_groups_0 = const()[name = string("hidden_states_233_groups_0"), val = int32(1)]; tensor hidden_states_233_cast_fp16 = conv(dilations = hidden_states_233_dilations_0, groups = hidden_states_233_groups_0, pad = hidden_states_233_pad_0, pad_type = hidden_states_233_pad_type_0, strides = hidden_states_233_strides_0, weight = layers_23_self_attn_o_proj_weight_cast_fp16, x = attn_output_191_cast_fp16)[name = string("hidden_states_233_cast_fp16")]; tensor hidden_states_235_cast_fp16 = add(x = hidden_states_229_cast_fp16, y = hidden_states_233_cast_fp16)[name = string("hidden_states_235_cast_fp16")]; fp16 const_238_promoted_to_fp16 = const()[name = string("const_238_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8548_cast_fp16 = mul(x = hidden_states_235_cast_fp16, y = const_238_promoted_to_fp16)[name = string("op_8548_cast_fp16")]; int32 var_8546 = const()[name = string("op_8546"), val = int32(1)]; bool doubled_189_interleave_0 = const()[name = string("doubled_189_interleave_0"), val = bool(false)]; tensor doubled_189_cast_fp16 = concat(axis = var_8546, interleave = doubled_189_interleave_0, values = (hidden_states_235_cast_fp16, var_8548_cast_fp16))[name = string("doubled_189_cast_fp16")]; tensor out_95_axes_0 = const()[name = string("out_95_axes_0"), val = tensor([1])]; tensor out_95_gamma_0_to_fp16 = const()[name = string("out_95_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539297280)))]; fp16 var_8558_to_fp16 = const()[name = string("op_8558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_95_cast_fp16 = layer_norm(axes = out_95_axes_0, epsilon = var_8558_to_fp16, gamma = out_95_gamma_0_to_fp16, x = doubled_189_cast_fp16)[name = string("out_95_cast_fp16")]; tensor var_8569_split_sizes_0 = const()[name = string("op_8569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8569_axis_0 = const()[name = string("op_8569_axis_0"), val = int32(1)]; tensor var_8569_cast_fp16_0, tensor var_8569_cast_fp16_1 = split(axis = var_8569_axis_0, split_sizes = var_8569_split_sizes_0, x = out_95_cast_fp16)[name = string("op_8569_cast_fp16")]; tensor input_47_strides_0 = const()[name = string("input_47_strides_0"), val = tensor([1, 1])]; string input_47_pad_type_0 = const()[name = string("input_47_pad_type_0"), val = string("valid")]; tensor input_47_pad_0 = const()[name = string("input_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_47_dilations_0 = const()[name = string("input_47_dilations_0"), val = tensor([1, 1])]; int32 input_47_groups_0 = const()[name = string("input_47_groups_0"), val = int32(1)]; tensor input_47_cast_fp16 = conv(dilations = input_47_dilations_0, groups = input_47_groups_0, pad = input_47_pad_0, pad_type = input_47_pad_type_0, strides = input_47_strides_0, weight = layers_23_mlp_gate_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("input_47_cast_fp16")]; tensor var_8586_cast_fp16 = silu(x = input_47_cast_fp16)[name = string("op_8586_cast_fp16")]; tensor var_8592_strides_0 = const()[name = string("op_8592_strides_0"), val = tensor([1, 1])]; string var_8592_pad_type_0 = const()[name = string("op_8592_pad_type_0"), val = string("valid")]; tensor var_8592_pad_0 = const()[name = string("op_8592_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8592_dilations_0 = const()[name = string("op_8592_dilations_0"), val = tensor([1, 1])]; int32 var_8592_groups_0 = const()[name = string("op_8592_groups_0"), val = int32(1)]; tensor var_8592_cast_fp16 = conv(dilations = var_8592_dilations_0, groups = var_8592_groups_0, pad = var_8592_pad_0, pad_type = var_8592_pad_type_0, strides = var_8592_strides_0, weight = layers_23_mlp_up_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("op_8592_cast_fp16")]; tensor x_239_cast_fp16 = mul(x = var_8586_cast_fp16, y = var_8592_cast_fp16)[name = string("x_239_cast_fp16")]; tensor hidden_states_237_strides_0 = const()[name = string("hidden_states_237_strides_0"), val = tensor([1, 1])]; string hidden_states_237_pad_type_0 = const()[name = string("hidden_states_237_pad_type_0"), val = string("valid")]; tensor hidden_states_237_pad_0 = const()[name = string("hidden_states_237_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_237_dilations_0 = const()[name = string("hidden_states_237_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_237_groups_0 = const()[name = string("hidden_states_237_groups_0"), val = int32(1)]; tensor hidden_states_237_cast_fp16 = conv(dilations = hidden_states_237_dilations_0, groups = hidden_states_237_groups_0, pad = hidden_states_237_pad_0, pad_type = hidden_states_237_pad_type_0, strides = hidden_states_237_strides_0, weight = layers_23_mlp_down_proj_weight_cast_fp16, x = x_239_cast_fp16)[name = string("hidden_states_237_cast_fp16")]; tensor hidden_states_239_cast_fp16 = add(x = hidden_states_235_cast_fp16, y = hidden_states_237_cast_fp16)[name = string("hidden_states_239_cast_fp16")]; fp16 const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8610_cast_fp16 = mul(x = hidden_states_239_cast_fp16, y = const_240_promoted_to_fp16)[name = string("op_8610_cast_fp16")]; int32 var_8608 = const()[name = string("op_8608"), val = int32(1)]; bool doubled_193_interleave_0 = const()[name = string("doubled_193_interleave_0"), val = bool(false)]; tensor doubled_193_cast_fp16 = concat(axis = var_8608, interleave = doubled_193_interleave_0, values = (hidden_states_239_cast_fp16, var_8610_cast_fp16))[name = string("doubled_193_cast_fp16")]; tensor out_97_axes_0 = const()[name = string("out_97_axes_0"), val = tensor([1])]; tensor out_97_gamma_0_to_fp16 = const()[name = string("out_97_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539305536)))]; fp16 var_8620_to_fp16 = const()[name = string("op_8620_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_97_cast_fp16 = layer_norm(axes = out_97_axes_0, epsilon = var_8620_to_fp16, gamma = out_97_gamma_0_to_fp16, x = doubled_193_cast_fp16)[name = string("out_97_cast_fp16")]; tensor var_8631_split_sizes_0 = const()[name = string("op_8631_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8631_axis_0 = const()[name = string("op_8631_axis_0"), val = int32(1)]; tensor var_8631_cast_fp16_0, tensor var_8631_cast_fp16_1 = split(axis = var_8631_axis_0, split_sizes = var_8631_split_sizes_0, x = out_97_cast_fp16)[name = string("op_8631_cast_fp16")]; tensor query_states_145_strides_0 = const()[name = string("query_states_145_strides_0"), val = tensor([1, 1])]; string query_states_145_pad_type_0 = const()[name = string("query_states_145_pad_type_0"), val = string("valid")]; tensor query_states_145_pad_0 = const()[name = string("query_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_145_dilations_0 = const()[name = string("query_states_145_dilations_0"), val = tensor([1, 1])]; int32 query_states_145_groups_0 = const()[name = string("query_states_145_groups_0"), val = int32(1)]; tensor query_states_145_cast_fp16 = conv(dilations = query_states_145_dilations_0, groups = query_states_145_groups_0, pad = query_states_145_pad_0, pad_type = query_states_145_pad_type_0, strides = query_states_145_strides_0, weight = layers_24_self_attn_q_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("query_states_145_cast_fp16")]; tensor key_states_241_strides_0 = const()[name = string("key_states_241_strides_0"), val = tensor([1, 1])]; string key_states_241_pad_type_0 = const()[name = string("key_states_241_pad_type_0"), val = string("valid")]; tensor key_states_241_pad_0 = const()[name = string("key_states_241_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_241_dilations_0 = const()[name = string("key_states_241_dilations_0"), val = tensor([1, 1])]; int32 key_states_241_groups_0 = const()[name = string("key_states_241_groups_0"), val = int32(1)]; tensor key_states_241_cast_fp16 = conv(dilations = key_states_241_dilations_0, groups = key_states_241_groups_0, pad = key_states_241_pad_0, pad_type = key_states_241_pad_type_0, strides = key_states_241_strides_0, weight = layers_24_self_attn_k_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("key_states_241_cast_fp16")]; tensor layers_24_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_24_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539313792)))]; tensor value_states_145_strides_0 = const()[name = string("value_states_145_strides_0"), val = tensor([1, 1])]; string value_states_145_pad_type_0 = const()[name = string("value_states_145_pad_type_0"), val = string("valid")]; tensor value_states_145_pad_0 = const()[name = string("value_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_145_dilations_0 = const()[name = string("value_states_145_dilations_0"), val = tensor([1, 1])]; int32 value_states_145_groups_0 = const()[name = string("value_states_145_groups_0"), val = int32(1)]; tensor value_states_145_cast_fp16 = conv(dilations = value_states_145_dilations_0, groups = value_states_145_groups_0, pad = value_states_145_pad_0, pad_type = value_states_145_pad_type_0, strides = value_states_145_strides_0, weight = layers_24_self_attn_v_proj_weight_to_fp16, x = var_8631_cast_fp16_0)[name = string("value_states_145_cast_fp16")]; tensor concat_288x = const()[name = string("concat_288x"), val = tensor([1, 16, 128, -1])]; tensor x_241_cast_fp16 = reshape(shape = concat_288x, x = query_states_145_cast_fp16)[name = string("x_241_cast_fp16")]; tensor concat_289x = const()[name = string("concat_289x"), val = tensor([1, 2, 128, -1])]; tensor var_8688_cast_fp16 = reshape(shape = concat_289x, x = key_states_241_cast_fp16)[name = string("op_8688_cast_fp16")]; tensor concat_290x = const()[name = string("concat_290x"), val = tensor([1, 2, 128, -1])]; tensor var_8695_cast_fp16 = reshape(shape = concat_290x, x = value_states_145_cast_fp16)[name = string("op_8695_cast_fp16")]; tensor var_8699_cast_fp16 = mul(x = x_241_cast_fp16, y = var_869_cast_fp16)[name = string("op_8699_cast_fp16")]; tensor var_8700_split_sizes_0 = const()[name = string("op_8700_split_sizes_0"), val = tensor([64, 64])]; int32 var_8700_axis_0 = const()[name = string("op_8700_axis_0"), val = int32(-2)]; tensor var_8700_cast_fp16_0, tensor var_8700_cast_fp16_1 = split(axis = var_8700_axis_0, split_sizes = var_8700_split_sizes_0, x = x_241_cast_fp16)[name = string("op_8700_cast_fp16")]; fp16 const_242_promoted_to_fp16 = const()[name = string("const_242_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8702_cast_fp16 = mul(x = var_8700_cast_fp16_1, y = const_242_promoted_to_fp16)[name = string("op_8702_cast_fp16")]; int32 var_8704 = const()[name = string("op_8704"), val = int32(-2)]; bool var_8705_interleave_0 = const()[name = string("op_8705_interleave_0"), val = bool(false)]; tensor var_8705_cast_fp16 = concat(axis = var_8704, interleave = var_8705_interleave_0, values = (var_8702_cast_fp16, var_8700_cast_fp16_0))[name = string("op_8705_cast_fp16")]; tensor var_8706_cast_fp16 = mul(x = var_8705_cast_fp16, y = var_878_cast_fp16)[name = string("op_8706_cast_fp16")]; tensor query_states_147_cast_fp16 = add(x = var_8699_cast_fp16, y = var_8706_cast_fp16)[name = string("query_states_147_cast_fp16")]; tensor var_8712_cast_fp16 = mul(x = var_8688_cast_fp16, y = var_869_cast_fp16)[name = string("op_8712_cast_fp16")]; tensor var_8713_split_sizes_0 = const()[name = string("op_8713_split_sizes_0"), val = tensor([64, 64])]; int32 var_8713_axis_0 = const()[name = string("op_8713_axis_0"), val = int32(-2)]; tensor var_8713_cast_fp16_0, tensor var_8713_cast_fp16_1 = split(axis = var_8713_axis_0, split_sizes = var_8713_split_sizes_0, x = var_8688_cast_fp16)[name = string("op_8713_cast_fp16")]; fp16 const_243_promoted_to_fp16 = const()[name = string("const_243_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8715_cast_fp16 = mul(x = var_8713_cast_fp16_1, y = const_243_promoted_to_fp16)[name = string("op_8715_cast_fp16")]; int32 var_8717 = const()[name = string("op_8717"), val = int32(-2)]; bool var_8718_interleave_0 = const()[name = string("op_8718_interleave_0"), val = bool(false)]; tensor var_8718_cast_fp16 = concat(axis = var_8717, interleave = var_8718_interleave_0, values = (var_8715_cast_fp16, var_8713_cast_fp16_0))[name = string("op_8718_cast_fp16")]; tensor var_8719_cast_fp16 = mul(x = var_8718_cast_fp16, y = var_878_cast_fp16)[name = string("op_8719_cast_fp16")]; tensor key_states_245_cast_fp16 = add(x = var_8712_cast_fp16, y = var_8719_cast_fp16)[name = string("key_states_245_cast_fp16")]; tensor expand_dims_288 = const()[name = string("expand_dims_288"), val = tensor([24])]; tensor expand_dims_289 = const()[name = string("expand_dims_289"), val = tensor([0])]; tensor expand_dims_291 = const()[name = string("expand_dims_291"), val = tensor([0])]; int32 concat_293_axis_0 = const()[name = string("concat_293_axis_0"), val = int32(0)]; bool concat_293_interleave_0 = const()[name = string("concat_293_interleave_0"), val = bool(false)]; tensor concat_293 = concat(axis = concat_293_axis_0, interleave = concat_293_interleave_0, values = (expand_dims_288, expand_dims_289, position_id, expand_dims_291))[name = string("concat_293")]; tensor expand_dims_292 = const()[name = string("expand_dims_292"), val = tensor([25])]; tensor concat_294_values1_0 = const()[name = string("concat_294_values1_0"), val = tensor([0])]; tensor concat_294_values3_0 = const()[name = string("concat_294_values3_0"), val = tensor([0])]; int32 concat_294_axis_0 = const()[name = string("concat_294_axis_0"), val = int32(0)]; bool concat_294_interleave_0 = const()[name = string("concat_294_interleave_0"), val = bool(false)]; tensor concat_294 = concat(axis = concat_294_axis_0, interleave = concat_294_interleave_0, values = (expand_dims_292, concat_294_values1_0, cache_position_end, concat_294_values3_0))[name = string("concat_294")]; tensor key_states_247_perm_0 = const()[name = string("key_states_247_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_25_stride_0 = const()[name = string("key_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_247_cast_fp16 = transpose(perm = key_states_247_perm_0, x = key_states_245_cast_fp16)[name = string("transpose_355")]; tensor key_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = key_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = key_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_25_squeeze_mask_0, stride = key_cache_internal_tensor_assign_25_stride_0, update = key_states_247_cast_fp16, x = coreml_update_state_270)[name = string("key_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_25_cast_fp16, input = key_cache)[name = string("coreml_update_state_272_write_state")]; tensor coreml_update_state_272 = read_state(input = key_cache)[name = string("coreml_update_state_272")]; tensor value_states_147_perm_0 = const()[name = string("value_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_25_stride_0 = const()[name = string("value_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_147_cast_fp16 = transpose(perm = value_states_147_perm_0, x = var_8695_cast_fp16)[name = string("transpose_354")]; tensor value_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = value_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = value_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_25_squeeze_mask_0, stride = value_cache_internal_tensor_assign_25_stride_0, update = value_states_147_cast_fp16, x = coreml_update_state_271)[name = string("value_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_25_cast_fp16, input = value_cache)[name = string("coreml_update_state_273_write_state")]; tensor coreml_update_state_273 = read_state(input = value_cache)[name = string("coreml_update_state_273")]; tensor var_8789_begin_0 = const()[name = string("op_8789_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8789_end_0 = const()[name = string("op_8789_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8789_end_mask_0 = const()[name = string("op_8789_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8789_cast_fp16 = slice_by_index(begin = var_8789_begin_0, end = var_8789_end_0, end_mask = var_8789_end_mask_0, x = coreml_update_state_272)[name = string("op_8789_cast_fp16")]; tensor tile_48 = const()[name = string("tile_48"), val = tensor([1, 1])]; int32 var_8792_axis_0 = const()[name = string("op_8792_axis_0"), val = int32(1)]; tensor var_8792_cast_fp16_0, tensor var_8792_cast_fp16_1 = split(axis = var_8792_axis_0, split_sizes = tile_48, x = var_8789_cast_fp16)[name = string("op_8792_cast_fp16")]; tensor var_8799_begin_0 = const()[name = string("op_8799_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8799_end_0 = const()[name = string("op_8799_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8799_end_mask_0 = const()[name = string("op_8799_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8799_cast_fp16 = slice_by_index(begin = var_8799_begin_0, end = var_8799_end_0, end_mask = var_8799_end_mask_0, x = coreml_update_state_273)[name = string("op_8799_cast_fp16")]; tensor tile_49 = const()[name = string("tile_49"), val = tensor([1, 1])]; int32 var_8802_axis_0 = const()[name = string("op_8802_axis_0"), val = int32(1)]; tensor var_8802_cast_fp16_0, tensor var_8802_cast_fp16_1 = split(axis = var_8802_axis_0, split_sizes = tile_49, x = var_8799_cast_fp16)[name = string("op_8802_cast_fp16")]; tensor var_8805_split_sizes_0 = const()[name = string("op_8805_split_sizes_0"), val = tensor([8, 8])]; int32 var_8805_axis_0 = const()[name = string("op_8805_axis_0"), val = int32(1)]; tensor var_8805_0, tensor var_8805_1 = split(axis = var_8805_axis_0, split_sizes = var_8805_split_sizes_0, x = query_states_147_cast_fp16)[name = string("op_8805")]; bool attn_weights_385_transpose_x_0 = const()[name = string("attn_weights_385_transpose_x_0"), val = bool(false)]; bool attn_weights_385_transpose_y_0 = const()[name = string("attn_weights_385_transpose_y_0"), val = bool(false)]; tensor attn_weights_385_cast_fp16 = matmul(transpose_x = attn_weights_385_transpose_x_0, transpose_y = attn_weights_385_transpose_y_0, x = var_8792_cast_fp16_0, y = var_8805_0)[name = string("attn_weights_385_cast_fp16")]; fp16 var_8808_to_fp16 = const()[name = string("op_8808_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_387_cast_fp16 = mul(x = attn_weights_385_cast_fp16, y = var_8808_to_fp16)[name = string("attn_weights_387_cast_fp16")]; tensor attn_weights_389_cast_fp16 = add(x = attn_weights_387_cast_fp16, y = attn_mask_1)[name = string("attn_weights_389_cast_fp16")]; int32 var_8812 = const()[name = string("op_8812"), val = int32(-2)]; tensor attn_weights_391_cast_fp16 = softmax(axis = var_8812, x = attn_weights_389_cast_fp16)[name = string("attn_weights_391_cast_fp16")]; bool var_8818_transpose_x_1 = const()[name = string("op_8818_transpose_x_1"), val = bool(true)]; bool var_8818_transpose_y_1 = const()[name = string("op_8818_transpose_y_1"), val = bool(false)]; tensor var_8818_cast_fp16 = matmul(transpose_x = var_8818_transpose_x_1, transpose_y = var_8818_transpose_y_1, x = attn_weights_391_cast_fp16, y = var_8802_cast_fp16_0)[name = string("op_8818_cast_fp16")]; bool attn_weights_393_transpose_x_0 = const()[name = string("attn_weights_393_transpose_x_0"), val = bool(false)]; bool attn_weights_393_transpose_y_0 = const()[name = string("attn_weights_393_transpose_y_0"), val = bool(false)]; tensor attn_weights_393_cast_fp16 = matmul(transpose_x = attn_weights_393_transpose_x_0, transpose_y = attn_weights_393_transpose_y_0, x = var_8792_cast_fp16_1, y = var_8805_1)[name = string("attn_weights_393_cast_fp16")]; fp16 var_8820_to_fp16 = const()[name = string("op_8820_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_395_cast_fp16 = mul(x = attn_weights_393_cast_fp16, y = var_8820_to_fp16)[name = string("attn_weights_395_cast_fp16")]; tensor attn_weights_397_cast_fp16 = add(x = attn_weights_395_cast_fp16, y = attn_mask_1)[name = string("attn_weights_397_cast_fp16")]; int32 var_8824 = const()[name = string("op_8824"), val = int32(-2)]; tensor attn_weights_399_cast_fp16 = softmax(axis = var_8824, x = attn_weights_397_cast_fp16)[name = string("attn_weights_399_cast_fp16")]; bool attn_output_193_transpose_x_1 = const()[name = string("attn_output_193_transpose_x_1"), val = bool(true)]; bool attn_output_193_transpose_y_1 = const()[name = string("attn_output_193_transpose_y_1"), val = bool(false)]; tensor attn_output_193_cast_fp16 = matmul(transpose_x = attn_output_193_transpose_x_1, transpose_y = attn_output_193_transpose_y_1, x = attn_weights_399_cast_fp16, y = var_8802_cast_fp16_1)[name = string("attn_output_193_cast_fp16")]; int32 var_8832 = const()[name = string("op_8832"), val = int32(1)]; bool attn_output_195_interleave_0 = const()[name = string("attn_output_195_interleave_0"), val = bool(false)]; tensor attn_output_195_cast_fp16 = concat(axis = var_8832, interleave = attn_output_195_interleave_0, values = (var_8818_cast_fp16, attn_output_193_cast_fp16))[name = string("attn_output_195_cast_fp16")]; tensor var_8836_perm_0 = const()[name = string("op_8836_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_299x = const()[name = string("concat_299x"), val = tensor([1, 2048, 1, -1])]; tensor var_8836_cast_fp16 = transpose(perm = var_8836_perm_0, x = attn_output_195_cast_fp16)[name = string("transpose_353")]; tensor attn_output_199_cast_fp16 = reshape(shape = concat_299x, x = var_8836_cast_fp16)[name = string("attn_output_199_cast_fp16")]; tensor hidden_states_243_strides_0 = const()[name = string("hidden_states_243_strides_0"), val = tensor([1, 1])]; string hidden_states_243_pad_type_0 = const()[name = string("hidden_states_243_pad_type_0"), val = string("valid")]; tensor hidden_states_243_pad_0 = const()[name = string("hidden_states_243_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_243_dilations_0 = const()[name = string("hidden_states_243_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_243_groups_0 = const()[name = string("hidden_states_243_groups_0"), val = int32(1)]; tensor hidden_states_243_cast_fp16 = conv(dilations = hidden_states_243_dilations_0, groups = hidden_states_243_groups_0, pad = hidden_states_243_pad_0, pad_type = hidden_states_243_pad_type_0, strides = hidden_states_243_strides_0, weight = layers_24_self_attn_o_proj_weight_cast_fp16, x = attn_output_199_cast_fp16)[name = string("hidden_states_243_cast_fp16")]; tensor hidden_states_245_cast_fp16 = add(x = hidden_states_239_cast_fp16, y = hidden_states_243_cast_fp16)[name = string("hidden_states_245_cast_fp16")]; fp16 const_248_promoted_to_fp16 = const()[name = string("const_248_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8869_cast_fp16 = mul(x = hidden_states_245_cast_fp16, y = const_248_promoted_to_fp16)[name = string("op_8869_cast_fp16")]; int32 var_8867 = const()[name = string("op_8867"), val = int32(1)]; bool doubled_197_interleave_0 = const()[name = string("doubled_197_interleave_0"), val = bool(false)]; tensor doubled_197_cast_fp16 = concat(axis = var_8867, interleave = doubled_197_interleave_0, values = (hidden_states_245_cast_fp16, var_8869_cast_fp16))[name = string("doubled_197_cast_fp16")]; tensor out_99_axes_0 = const()[name = string("out_99_axes_0"), val = tensor([1])]; tensor out_99_gamma_0_to_fp16 = const()[name = string("out_99_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540362432)))]; fp16 var_8879_to_fp16 = const()[name = string("op_8879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_99_cast_fp16 = layer_norm(axes = out_99_axes_0, epsilon = var_8879_to_fp16, gamma = out_99_gamma_0_to_fp16, x = doubled_197_cast_fp16)[name = string("out_99_cast_fp16")]; tensor var_8890_split_sizes_0 = const()[name = string("op_8890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8890_axis_0 = const()[name = string("op_8890_axis_0"), val = int32(1)]; tensor var_8890_cast_fp16_0, tensor var_8890_cast_fp16_1 = split(axis = var_8890_axis_0, split_sizes = var_8890_split_sizes_0, x = out_99_cast_fp16)[name = string("op_8890_cast_fp16")]; tensor input_49_strides_0 = const()[name = string("input_49_strides_0"), val = tensor([1, 1])]; string input_49_pad_type_0 = const()[name = string("input_49_pad_type_0"), val = string("valid")]; tensor input_49_pad_0 = const()[name = string("input_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_49_dilations_0 = const()[name = string("input_49_dilations_0"), val = tensor([1, 1])]; int32 input_49_groups_0 = const()[name = string("input_49_groups_0"), val = int32(1)]; tensor input_49_cast_fp16 = conv(dilations = input_49_dilations_0, groups = input_49_groups_0, pad = input_49_pad_0, pad_type = input_49_pad_type_0, strides = input_49_strides_0, weight = layers_24_mlp_gate_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("input_49_cast_fp16")]; tensor var_8907_cast_fp16 = silu(x = input_49_cast_fp16)[name = string("op_8907_cast_fp16")]; tensor var_8913_strides_0 = const()[name = string("op_8913_strides_0"), val = tensor([1, 1])]; string var_8913_pad_type_0 = const()[name = string("op_8913_pad_type_0"), val = string("valid")]; tensor var_8913_pad_0 = const()[name = string("op_8913_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8913_dilations_0 = const()[name = string("op_8913_dilations_0"), val = tensor([1, 1])]; int32 var_8913_groups_0 = const()[name = string("op_8913_groups_0"), val = int32(1)]; tensor var_8913_cast_fp16 = conv(dilations = var_8913_dilations_0, groups = var_8913_groups_0, pad = var_8913_pad_0, pad_type = var_8913_pad_type_0, strides = var_8913_strides_0, weight = layers_24_mlp_up_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("op_8913_cast_fp16")]; tensor x_249_cast_fp16 = mul(x = var_8907_cast_fp16, y = var_8913_cast_fp16)[name = string("x_249_cast_fp16")]; tensor hidden_states_247_strides_0 = const()[name = string("hidden_states_247_strides_0"), val = tensor([1, 1])]; string hidden_states_247_pad_type_0 = const()[name = string("hidden_states_247_pad_type_0"), val = string("valid")]; tensor hidden_states_247_pad_0 = const()[name = string("hidden_states_247_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_247_dilations_0 = const()[name = string("hidden_states_247_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_247_groups_0 = const()[name = string("hidden_states_247_groups_0"), val = int32(1)]; tensor hidden_states_247_cast_fp16 = conv(dilations = hidden_states_247_dilations_0, groups = hidden_states_247_groups_0, pad = hidden_states_247_pad_0, pad_type = hidden_states_247_pad_type_0, strides = hidden_states_247_strides_0, weight = layers_24_mlp_down_proj_weight_cast_fp16, x = x_249_cast_fp16)[name = string("hidden_states_247_cast_fp16")]; tensor hidden_states_249_cast_fp16 = add(x = hidden_states_245_cast_fp16, y = hidden_states_247_cast_fp16)[name = string("hidden_states_249_cast_fp16")]; fp16 const_250_promoted_to_fp16 = const()[name = string("const_250_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8931_cast_fp16 = mul(x = hidden_states_249_cast_fp16, y = const_250_promoted_to_fp16)[name = string("op_8931_cast_fp16")]; int32 var_8929 = const()[name = string("op_8929"), val = int32(1)]; bool doubled_201_interleave_0 = const()[name = string("doubled_201_interleave_0"), val = bool(false)]; tensor doubled_201_cast_fp16 = concat(axis = var_8929, interleave = doubled_201_interleave_0, values = (hidden_states_249_cast_fp16, var_8931_cast_fp16))[name = string("doubled_201_cast_fp16")]; tensor out_101_axes_0 = const()[name = string("out_101_axes_0"), val = tensor([1])]; tensor out_101_gamma_0_to_fp16 = const()[name = string("out_101_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540370688)))]; fp16 var_8941_to_fp16 = const()[name = string("op_8941_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_101_cast_fp16 = layer_norm(axes = out_101_axes_0, epsilon = var_8941_to_fp16, gamma = out_101_gamma_0_to_fp16, x = doubled_201_cast_fp16)[name = string("out_101_cast_fp16")]; tensor var_8952_split_sizes_0 = const()[name = string("op_8952_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8952_axis_0 = const()[name = string("op_8952_axis_0"), val = int32(1)]; tensor var_8952_cast_fp16_0, tensor var_8952_cast_fp16_1 = split(axis = var_8952_axis_0, split_sizes = var_8952_split_sizes_0, x = out_101_cast_fp16)[name = string("op_8952_cast_fp16")]; tensor query_states_151_strides_0 = const()[name = string("query_states_151_strides_0"), val = tensor([1, 1])]; string query_states_151_pad_type_0 = const()[name = string("query_states_151_pad_type_0"), val = string("valid")]; tensor query_states_151_pad_0 = const()[name = string("query_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_151_dilations_0 = const()[name = string("query_states_151_dilations_0"), val = tensor([1, 1])]; int32 query_states_151_groups_0 = const()[name = string("query_states_151_groups_0"), val = int32(1)]; tensor query_states_151_cast_fp16 = conv(dilations = query_states_151_dilations_0, groups = query_states_151_groups_0, pad = query_states_151_pad_0, pad_type = query_states_151_pad_type_0, strides = query_states_151_strides_0, weight = layers_25_self_attn_q_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("query_states_151_cast_fp16")]; tensor key_states_251_strides_0 = const()[name = string("key_states_251_strides_0"), val = tensor([1, 1])]; string key_states_251_pad_type_0 = const()[name = string("key_states_251_pad_type_0"), val = string("valid")]; tensor key_states_251_pad_0 = const()[name = string("key_states_251_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_251_dilations_0 = const()[name = string("key_states_251_dilations_0"), val = tensor([1, 1])]; int32 key_states_251_groups_0 = const()[name = string("key_states_251_groups_0"), val = int32(1)]; tensor key_states_251_cast_fp16 = conv(dilations = key_states_251_dilations_0, groups = key_states_251_groups_0, pad = key_states_251_pad_0, pad_type = key_states_251_pad_type_0, strides = key_states_251_strides_0, weight = layers_25_self_attn_k_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("key_states_251_cast_fp16")]; tensor layers_25_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_25_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540378944)))]; tensor value_states_151_strides_0 = const()[name = string("value_states_151_strides_0"), val = tensor([1, 1])]; string value_states_151_pad_type_0 = const()[name = string("value_states_151_pad_type_0"), val = string("valid")]; tensor value_states_151_pad_0 = const()[name = string("value_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_151_dilations_0 = const()[name = string("value_states_151_dilations_0"), val = tensor([1, 1])]; int32 value_states_151_groups_0 = const()[name = string("value_states_151_groups_0"), val = int32(1)]; tensor value_states_151_cast_fp16 = conv(dilations = value_states_151_dilations_0, groups = value_states_151_groups_0, pad = value_states_151_pad_0, pad_type = value_states_151_pad_type_0, strides = value_states_151_strides_0, weight = layers_25_self_attn_v_proj_weight_to_fp16, x = var_8952_cast_fp16_0)[name = string("value_states_151_cast_fp16")]; tensor concat_300x = const()[name = string("concat_300x"), val = tensor([1, 16, 128, -1])]; tensor x_251_cast_fp16 = reshape(shape = concat_300x, x = query_states_151_cast_fp16)[name = string("x_251_cast_fp16")]; tensor concat_301x = const()[name = string("concat_301x"), val = tensor([1, 2, 128, -1])]; tensor var_9009_cast_fp16 = reshape(shape = concat_301x, x = key_states_251_cast_fp16)[name = string("op_9009_cast_fp16")]; tensor concat_302x = const()[name = string("concat_302x"), val = tensor([1, 2, 128, -1])]; tensor var_9016_cast_fp16 = reshape(shape = concat_302x, x = value_states_151_cast_fp16)[name = string("op_9016_cast_fp16")]; tensor var_9020_cast_fp16 = mul(x = x_251_cast_fp16, y = var_869_cast_fp16)[name = string("op_9020_cast_fp16")]; tensor var_9021_split_sizes_0 = const()[name = string("op_9021_split_sizes_0"), val = tensor([64, 64])]; int32 var_9021_axis_0 = const()[name = string("op_9021_axis_0"), val = int32(-2)]; tensor var_9021_cast_fp16_0, tensor var_9021_cast_fp16_1 = split(axis = var_9021_axis_0, split_sizes = var_9021_split_sizes_0, x = x_251_cast_fp16)[name = string("op_9021_cast_fp16")]; fp16 const_252_promoted_to_fp16 = const()[name = string("const_252_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9023_cast_fp16 = mul(x = var_9021_cast_fp16_1, y = const_252_promoted_to_fp16)[name = string("op_9023_cast_fp16")]; int32 var_9025 = const()[name = string("op_9025"), val = int32(-2)]; bool var_9026_interleave_0 = const()[name = string("op_9026_interleave_0"), val = bool(false)]; tensor var_9026_cast_fp16 = concat(axis = var_9025, interleave = var_9026_interleave_0, values = (var_9023_cast_fp16, var_9021_cast_fp16_0))[name = string("op_9026_cast_fp16")]; tensor var_9027_cast_fp16 = mul(x = var_9026_cast_fp16, y = var_878_cast_fp16)[name = string("op_9027_cast_fp16")]; tensor query_states_153_cast_fp16 = add(x = var_9020_cast_fp16, y = var_9027_cast_fp16)[name = string("query_states_153_cast_fp16")]; tensor var_9033_cast_fp16 = mul(x = var_9009_cast_fp16, y = var_869_cast_fp16)[name = string("op_9033_cast_fp16")]; tensor var_9034_split_sizes_0 = const()[name = string("op_9034_split_sizes_0"), val = tensor([64, 64])]; int32 var_9034_axis_0 = const()[name = string("op_9034_axis_0"), val = int32(-2)]; tensor var_9034_cast_fp16_0, tensor var_9034_cast_fp16_1 = split(axis = var_9034_axis_0, split_sizes = var_9034_split_sizes_0, x = var_9009_cast_fp16)[name = string("op_9034_cast_fp16")]; fp16 const_253_promoted_to_fp16 = const()[name = string("const_253_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9036_cast_fp16 = mul(x = var_9034_cast_fp16_1, y = const_253_promoted_to_fp16)[name = string("op_9036_cast_fp16")]; int32 var_9038 = const()[name = string("op_9038"), val = int32(-2)]; bool var_9039_interleave_0 = const()[name = string("op_9039_interleave_0"), val = bool(false)]; tensor var_9039_cast_fp16 = concat(axis = var_9038, interleave = var_9039_interleave_0, values = (var_9036_cast_fp16, var_9034_cast_fp16_0))[name = string("op_9039_cast_fp16")]; tensor var_9040_cast_fp16 = mul(x = var_9039_cast_fp16, y = var_878_cast_fp16)[name = string("op_9040_cast_fp16")]; tensor key_states_255_cast_fp16 = add(x = var_9033_cast_fp16, y = var_9040_cast_fp16)[name = string("key_states_255_cast_fp16")]; tensor expand_dims_300 = const()[name = string("expand_dims_300"), val = tensor([25])]; tensor expand_dims_301 = const()[name = string("expand_dims_301"), val = tensor([0])]; tensor expand_dims_303 = const()[name = string("expand_dims_303"), val = tensor([0])]; int32 concat_305_axis_0 = const()[name = string("concat_305_axis_0"), val = int32(0)]; bool concat_305_interleave_0 = const()[name = string("concat_305_interleave_0"), val = bool(false)]; tensor concat_305 = concat(axis = concat_305_axis_0, interleave = concat_305_interleave_0, values = (expand_dims_300, expand_dims_301, position_id, expand_dims_303))[name = string("concat_305")]; tensor expand_dims_304 = const()[name = string("expand_dims_304"), val = tensor([26])]; tensor concat_306_values1_0 = const()[name = string("concat_306_values1_0"), val = tensor([0])]; tensor concat_306_values3_0 = const()[name = string("concat_306_values3_0"), val = tensor([0])]; int32 concat_306_axis_0 = const()[name = string("concat_306_axis_0"), val = int32(0)]; bool concat_306_interleave_0 = const()[name = string("concat_306_interleave_0"), val = bool(false)]; tensor concat_306 = concat(axis = concat_306_axis_0, interleave = concat_306_interleave_0, values = (expand_dims_304, concat_306_values1_0, cache_position_end, concat_306_values3_0))[name = string("concat_306")]; tensor key_states_257_perm_0 = const()[name = string("key_states_257_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_26_stride_0 = const()[name = string("key_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_257_cast_fp16 = transpose(perm = key_states_257_perm_0, x = key_states_255_cast_fp16)[name = string("transpose_352")]; tensor key_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = key_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = key_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_26_squeeze_mask_0, stride = key_cache_internal_tensor_assign_26_stride_0, update = key_states_257_cast_fp16, x = coreml_update_state_272)[name = string("key_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_26_cast_fp16, input = key_cache)[name = string("coreml_update_state_274_write_state")]; tensor coreml_update_state_274 = read_state(input = key_cache)[name = string("coreml_update_state_274")]; tensor value_states_153_perm_0 = const()[name = string("value_states_153_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_26_stride_0 = const()[name = string("value_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_153_cast_fp16 = transpose(perm = value_states_153_perm_0, x = var_9016_cast_fp16)[name = string("transpose_351")]; tensor value_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = value_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = value_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_26_squeeze_mask_0, stride = value_cache_internal_tensor_assign_26_stride_0, update = value_states_153_cast_fp16, x = coreml_update_state_273)[name = string("value_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_26_cast_fp16, input = value_cache)[name = string("coreml_update_state_275_write_state")]; tensor coreml_update_state_275 = read_state(input = value_cache)[name = string("coreml_update_state_275")]; tensor var_9110_begin_0 = const()[name = string("op_9110_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9110_end_0 = const()[name = string("op_9110_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9110_end_mask_0 = const()[name = string("op_9110_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9110_cast_fp16 = slice_by_index(begin = var_9110_begin_0, end = var_9110_end_0, end_mask = var_9110_end_mask_0, x = coreml_update_state_274)[name = string("op_9110_cast_fp16")]; tensor tile_50 = const()[name = string("tile_50"), val = tensor([1, 1])]; int32 var_9113_axis_0 = const()[name = string("op_9113_axis_0"), val = int32(1)]; tensor var_9113_cast_fp16_0, tensor var_9113_cast_fp16_1 = split(axis = var_9113_axis_0, split_sizes = tile_50, x = var_9110_cast_fp16)[name = string("op_9113_cast_fp16")]; tensor var_9120_begin_0 = const()[name = string("op_9120_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9120_end_0 = const()[name = string("op_9120_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9120_end_mask_0 = const()[name = string("op_9120_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9120_cast_fp16 = slice_by_index(begin = var_9120_begin_0, end = var_9120_end_0, end_mask = var_9120_end_mask_0, x = coreml_update_state_275)[name = string("op_9120_cast_fp16")]; tensor tile_51 = const()[name = string("tile_51"), val = tensor([1, 1])]; int32 var_9123_axis_0 = const()[name = string("op_9123_axis_0"), val = int32(1)]; tensor var_9123_cast_fp16_0, tensor var_9123_cast_fp16_1 = split(axis = var_9123_axis_0, split_sizes = tile_51, x = var_9120_cast_fp16)[name = string("op_9123_cast_fp16")]; tensor var_9126_split_sizes_0 = const()[name = string("op_9126_split_sizes_0"), val = tensor([8, 8])]; int32 var_9126_axis_0 = const()[name = string("op_9126_axis_0"), val = int32(1)]; tensor var_9126_0, tensor var_9126_1 = split(axis = var_9126_axis_0, split_sizes = var_9126_split_sizes_0, x = query_states_153_cast_fp16)[name = string("op_9126")]; bool attn_weights_401_transpose_x_0 = const()[name = string("attn_weights_401_transpose_x_0"), val = bool(false)]; bool attn_weights_401_transpose_y_0 = const()[name = string("attn_weights_401_transpose_y_0"), val = bool(false)]; tensor attn_weights_401_cast_fp16 = matmul(transpose_x = attn_weights_401_transpose_x_0, transpose_y = attn_weights_401_transpose_y_0, x = var_9113_cast_fp16_0, y = var_9126_0)[name = string("attn_weights_401_cast_fp16")]; fp16 var_9129_to_fp16 = const()[name = string("op_9129_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_403_cast_fp16 = mul(x = attn_weights_401_cast_fp16, y = var_9129_to_fp16)[name = string("attn_weights_403_cast_fp16")]; tensor attn_weights_405_cast_fp16 = add(x = attn_weights_403_cast_fp16, y = attn_mask_1)[name = string("attn_weights_405_cast_fp16")]; int32 var_9133 = const()[name = string("op_9133"), val = int32(-2)]; tensor attn_weights_407_cast_fp16 = softmax(axis = var_9133, x = attn_weights_405_cast_fp16)[name = string("attn_weights_407_cast_fp16")]; bool var_9139_transpose_x_1 = const()[name = string("op_9139_transpose_x_1"), val = bool(true)]; bool var_9139_transpose_y_1 = const()[name = string("op_9139_transpose_y_1"), val = bool(false)]; tensor var_9139_cast_fp16 = matmul(transpose_x = var_9139_transpose_x_1, transpose_y = var_9139_transpose_y_1, x = attn_weights_407_cast_fp16, y = var_9123_cast_fp16_0)[name = string("op_9139_cast_fp16")]; bool attn_weights_409_transpose_x_0 = const()[name = string("attn_weights_409_transpose_x_0"), val = bool(false)]; bool attn_weights_409_transpose_y_0 = const()[name = string("attn_weights_409_transpose_y_0"), val = bool(false)]; tensor attn_weights_409_cast_fp16 = matmul(transpose_x = attn_weights_409_transpose_x_0, transpose_y = attn_weights_409_transpose_y_0, x = var_9113_cast_fp16_1, y = var_9126_1)[name = string("attn_weights_409_cast_fp16")]; fp16 var_9141_to_fp16 = const()[name = string("op_9141_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_411_cast_fp16 = mul(x = attn_weights_409_cast_fp16, y = var_9141_to_fp16)[name = string("attn_weights_411_cast_fp16")]; tensor attn_weights_413_cast_fp16 = add(x = attn_weights_411_cast_fp16, y = attn_mask_1)[name = string("attn_weights_413_cast_fp16")]; int32 var_9145 = const()[name = string("op_9145"), val = int32(-2)]; tensor attn_weights_415_cast_fp16 = softmax(axis = var_9145, x = attn_weights_413_cast_fp16)[name = string("attn_weights_415_cast_fp16")]; bool attn_output_201_transpose_x_1 = const()[name = string("attn_output_201_transpose_x_1"), val = bool(true)]; bool attn_output_201_transpose_y_1 = const()[name = string("attn_output_201_transpose_y_1"), val = bool(false)]; tensor attn_output_201_cast_fp16 = matmul(transpose_x = attn_output_201_transpose_x_1, transpose_y = attn_output_201_transpose_y_1, x = attn_weights_415_cast_fp16, y = var_9123_cast_fp16_1)[name = string("attn_output_201_cast_fp16")]; int32 var_9153 = const()[name = string("op_9153"), val = int32(1)]; bool attn_output_203_interleave_0 = const()[name = string("attn_output_203_interleave_0"), val = bool(false)]; tensor attn_output_203_cast_fp16 = concat(axis = var_9153, interleave = attn_output_203_interleave_0, values = (var_9139_cast_fp16, attn_output_201_cast_fp16))[name = string("attn_output_203_cast_fp16")]; tensor var_9157_perm_0 = const()[name = string("op_9157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_311x = const()[name = string("concat_311x"), val = tensor([1, 2048, 1, -1])]; tensor var_9157_cast_fp16 = transpose(perm = var_9157_perm_0, x = attn_output_203_cast_fp16)[name = string("transpose_350")]; tensor attn_output_207_cast_fp16 = reshape(shape = concat_311x, x = var_9157_cast_fp16)[name = string("attn_output_207_cast_fp16")]; tensor hidden_states_253_strides_0 = const()[name = string("hidden_states_253_strides_0"), val = tensor([1, 1])]; string hidden_states_253_pad_type_0 = const()[name = string("hidden_states_253_pad_type_0"), val = string("valid")]; tensor hidden_states_253_pad_0 = const()[name = string("hidden_states_253_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_253_dilations_0 = const()[name = string("hidden_states_253_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_253_groups_0 = const()[name = string("hidden_states_253_groups_0"), val = int32(1)]; tensor hidden_states_253_cast_fp16 = conv(dilations = hidden_states_253_dilations_0, groups = hidden_states_253_groups_0, pad = hidden_states_253_pad_0, pad_type = hidden_states_253_pad_type_0, strides = hidden_states_253_strides_0, weight = layers_25_self_attn_o_proj_weight_cast_fp16, x = attn_output_207_cast_fp16)[name = string("hidden_states_253_cast_fp16")]; tensor hidden_states_255_cast_fp16 = add(x = hidden_states_249_cast_fp16, y = hidden_states_253_cast_fp16)[name = string("hidden_states_255_cast_fp16")]; fp16 const_258_promoted_to_fp16 = const()[name = string("const_258_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9190_cast_fp16 = mul(x = hidden_states_255_cast_fp16, y = const_258_promoted_to_fp16)[name = string("op_9190_cast_fp16")]; int32 var_9188 = const()[name = string("op_9188"), val = int32(1)]; bool doubled_205_interleave_0 = const()[name = string("doubled_205_interleave_0"), val = bool(false)]; tensor doubled_205_cast_fp16 = concat(axis = var_9188, interleave = doubled_205_interleave_0, values = (hidden_states_255_cast_fp16, var_9190_cast_fp16))[name = string("doubled_205_cast_fp16")]; tensor out_103_axes_0 = const()[name = string("out_103_axes_0"), val = tensor([1])]; tensor out_103_gamma_0_to_fp16 = const()[name = string("out_103_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541427584)))]; fp16 var_9200_to_fp16 = const()[name = string("op_9200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_103_cast_fp16 = layer_norm(axes = out_103_axes_0, epsilon = var_9200_to_fp16, gamma = out_103_gamma_0_to_fp16, x = doubled_205_cast_fp16)[name = string("out_103_cast_fp16")]; tensor var_9211_split_sizes_0 = const()[name = string("op_9211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9211_axis_0 = const()[name = string("op_9211_axis_0"), val = int32(1)]; tensor var_9211_cast_fp16_0, tensor var_9211_cast_fp16_1 = split(axis = var_9211_axis_0, split_sizes = var_9211_split_sizes_0, x = out_103_cast_fp16)[name = string("op_9211_cast_fp16")]; tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; tensor input_51_cast_fp16 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = layers_25_mlp_gate_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("input_51_cast_fp16")]; tensor var_9228_cast_fp16 = silu(x = input_51_cast_fp16)[name = string("op_9228_cast_fp16")]; tensor var_9234_strides_0 = const()[name = string("op_9234_strides_0"), val = tensor([1, 1])]; string var_9234_pad_type_0 = const()[name = string("op_9234_pad_type_0"), val = string("valid")]; tensor var_9234_pad_0 = const()[name = string("op_9234_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9234_dilations_0 = const()[name = string("op_9234_dilations_0"), val = tensor([1, 1])]; int32 var_9234_groups_0 = const()[name = string("op_9234_groups_0"), val = int32(1)]; tensor var_9234_cast_fp16 = conv(dilations = var_9234_dilations_0, groups = var_9234_groups_0, pad = var_9234_pad_0, pad_type = var_9234_pad_type_0, strides = var_9234_strides_0, weight = layers_25_mlp_up_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("op_9234_cast_fp16")]; tensor x_259_cast_fp16 = mul(x = var_9228_cast_fp16, y = var_9234_cast_fp16)[name = string("x_259_cast_fp16")]; tensor layers_25_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_25_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541435840)))]; tensor hidden_states_257_strides_0 = const()[name = string("hidden_states_257_strides_0"), val = tensor([1, 1])]; string hidden_states_257_pad_type_0 = const()[name = string("hidden_states_257_pad_type_0"), val = string("valid")]; tensor hidden_states_257_pad_0 = const()[name = string("hidden_states_257_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_257_dilations_0 = const()[name = string("hidden_states_257_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_257_groups_0 = const()[name = string("hidden_states_257_groups_0"), val = int32(1)]; tensor hidden_states_257_cast_fp16 = conv(dilations = hidden_states_257_dilations_0, groups = hidden_states_257_groups_0, pad = hidden_states_257_pad_0, pad_type = hidden_states_257_pad_type_0, strides = hidden_states_257_strides_0, weight = layers_25_mlp_down_proj_weight_to_fp16, x = x_259_cast_fp16)[name = string("hidden_states_257_cast_fp16")]; tensor hidden_states_259_cast_fp16 = add(x = hidden_states_255_cast_fp16, y = hidden_states_257_cast_fp16)[name = string("hidden_states_259_cast_fp16")]; fp16 const_260_promoted_to_fp16 = const()[name = string("const_260_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9252_cast_fp16 = mul(x = hidden_states_259_cast_fp16, y = const_260_promoted_to_fp16)[name = string("op_9252_cast_fp16")]; int32 var_9250 = const()[name = string("op_9250"), val = int32(1)]; bool doubled_209_interleave_0 = const()[name = string("doubled_209_interleave_0"), val = bool(false)]; tensor doubled_209_cast_fp16 = concat(axis = var_9250, interleave = doubled_209_interleave_0, values = (hidden_states_259_cast_fp16, var_9252_cast_fp16))[name = string("doubled_209_cast_fp16")]; tensor out_105_axes_0 = const()[name = string("out_105_axes_0"), val = tensor([1])]; tensor out_105_gamma_0_to_fp16 = const()[name = string("out_105_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566601728)))]; fp16 var_9262_to_fp16 = const()[name = string("op_9262_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_105_cast_fp16 = layer_norm(axes = out_105_axes_0, epsilon = var_9262_to_fp16, gamma = out_105_gamma_0_to_fp16, x = doubled_209_cast_fp16)[name = string("out_105_cast_fp16")]; tensor var_9273_split_sizes_0 = const()[name = string("op_9273_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9273_axis_0 = const()[name = string("op_9273_axis_0"), val = int32(1)]; tensor var_9273_cast_fp16_0, tensor var_9273_cast_fp16_1 = split(axis = var_9273_axis_0, split_sizes = var_9273_split_sizes_0, x = out_105_cast_fp16)[name = string("op_9273_cast_fp16")]; tensor query_states_157_strides_0 = const()[name = string("query_states_157_strides_0"), val = tensor([1, 1])]; string query_states_157_pad_type_0 = const()[name = string("query_states_157_pad_type_0"), val = string("valid")]; tensor query_states_157_pad_0 = const()[name = string("query_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_157_dilations_0 = const()[name = string("query_states_157_dilations_0"), val = tensor([1, 1])]; int32 query_states_157_groups_0 = const()[name = string("query_states_157_groups_0"), val = int32(1)]; tensor query_states_157_cast_fp16 = conv(dilations = query_states_157_dilations_0, groups = query_states_157_groups_0, pad = query_states_157_pad_0, pad_type = query_states_157_pad_type_0, strides = query_states_157_strides_0, weight = layers_26_self_attn_q_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("query_states_157_cast_fp16")]; tensor key_states_261_strides_0 = const()[name = string("key_states_261_strides_0"), val = tensor([1, 1])]; string key_states_261_pad_type_0 = const()[name = string("key_states_261_pad_type_0"), val = string("valid")]; tensor key_states_261_pad_0 = const()[name = string("key_states_261_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_261_dilations_0 = const()[name = string("key_states_261_dilations_0"), val = tensor([1, 1])]; int32 key_states_261_groups_0 = const()[name = string("key_states_261_groups_0"), val = int32(1)]; tensor key_states_261_cast_fp16 = conv(dilations = key_states_261_dilations_0, groups = key_states_261_groups_0, pad = key_states_261_pad_0, pad_type = key_states_261_pad_type_0, strides = key_states_261_strides_0, weight = layers_26_self_attn_k_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("key_states_261_cast_fp16")]; tensor layers_26_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_26_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566609984)))]; tensor value_states_157_strides_0 = const()[name = string("value_states_157_strides_0"), val = tensor([1, 1])]; string value_states_157_pad_type_0 = const()[name = string("value_states_157_pad_type_0"), val = string("valid")]; tensor value_states_157_pad_0 = const()[name = string("value_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_157_dilations_0 = const()[name = string("value_states_157_dilations_0"), val = tensor([1, 1])]; int32 value_states_157_groups_0 = const()[name = string("value_states_157_groups_0"), val = int32(1)]; tensor value_states_157_cast_fp16 = conv(dilations = value_states_157_dilations_0, groups = value_states_157_groups_0, pad = value_states_157_pad_0, pad_type = value_states_157_pad_type_0, strides = value_states_157_strides_0, weight = layers_26_self_attn_v_proj_weight_to_fp16, x = var_9273_cast_fp16_0)[name = string("value_states_157_cast_fp16")]; tensor concat_312x = const()[name = string("concat_312x"), val = tensor([1, 16, 128, -1])]; tensor x_261_cast_fp16 = reshape(shape = concat_312x, x = query_states_157_cast_fp16)[name = string("x_261_cast_fp16")]; tensor concat_313x = const()[name = string("concat_313x"), val = tensor([1, 2, 128, -1])]; tensor var_9330_cast_fp16 = reshape(shape = concat_313x, x = key_states_261_cast_fp16)[name = string("op_9330_cast_fp16")]; tensor concat_314x = const()[name = string("concat_314x"), val = tensor([1, 2, 128, -1])]; tensor var_9337_cast_fp16 = reshape(shape = concat_314x, x = value_states_157_cast_fp16)[name = string("op_9337_cast_fp16")]; tensor var_9341_cast_fp16 = mul(x = x_261_cast_fp16, y = var_869_cast_fp16)[name = string("op_9341_cast_fp16")]; tensor var_9342_split_sizes_0 = const()[name = string("op_9342_split_sizes_0"), val = tensor([64, 64])]; int32 var_9342_axis_0 = const()[name = string("op_9342_axis_0"), val = int32(-2)]; tensor var_9342_cast_fp16_0, tensor var_9342_cast_fp16_1 = split(axis = var_9342_axis_0, split_sizes = var_9342_split_sizes_0, x = x_261_cast_fp16)[name = string("op_9342_cast_fp16")]; fp16 const_262_promoted_to_fp16 = const()[name = string("const_262_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9344_cast_fp16 = mul(x = var_9342_cast_fp16_1, y = const_262_promoted_to_fp16)[name = string("op_9344_cast_fp16")]; int32 var_9346 = const()[name = string("op_9346"), val = int32(-2)]; bool var_9347_interleave_0 = const()[name = string("op_9347_interleave_0"), val = bool(false)]; tensor var_9347_cast_fp16 = concat(axis = var_9346, interleave = var_9347_interleave_0, values = (var_9344_cast_fp16, var_9342_cast_fp16_0))[name = string("op_9347_cast_fp16")]; tensor var_9348_cast_fp16 = mul(x = var_9347_cast_fp16, y = var_878_cast_fp16)[name = string("op_9348_cast_fp16")]; tensor query_states_159_cast_fp16 = add(x = var_9341_cast_fp16, y = var_9348_cast_fp16)[name = string("query_states_159_cast_fp16")]; tensor var_9354_cast_fp16 = mul(x = var_9330_cast_fp16, y = var_869_cast_fp16)[name = string("op_9354_cast_fp16")]; tensor var_9355_split_sizes_0 = const()[name = string("op_9355_split_sizes_0"), val = tensor([64, 64])]; int32 var_9355_axis_0 = const()[name = string("op_9355_axis_0"), val = int32(-2)]; tensor var_9355_cast_fp16_0, tensor var_9355_cast_fp16_1 = split(axis = var_9355_axis_0, split_sizes = var_9355_split_sizes_0, x = var_9330_cast_fp16)[name = string("op_9355_cast_fp16")]; fp16 const_263_promoted_to_fp16 = const()[name = string("const_263_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9357_cast_fp16 = mul(x = var_9355_cast_fp16_1, y = const_263_promoted_to_fp16)[name = string("op_9357_cast_fp16")]; int32 var_9359 = const()[name = string("op_9359"), val = int32(-2)]; bool var_9360_interleave_0 = const()[name = string("op_9360_interleave_0"), val = bool(false)]; tensor var_9360_cast_fp16 = concat(axis = var_9359, interleave = var_9360_interleave_0, values = (var_9357_cast_fp16, var_9355_cast_fp16_0))[name = string("op_9360_cast_fp16")]; tensor var_9361_cast_fp16 = mul(x = var_9360_cast_fp16, y = var_878_cast_fp16)[name = string("op_9361_cast_fp16")]; tensor key_states_265_cast_fp16 = add(x = var_9354_cast_fp16, y = var_9361_cast_fp16)[name = string("key_states_265_cast_fp16")]; tensor expand_dims_312 = const()[name = string("expand_dims_312"), val = tensor([26])]; tensor expand_dims_313 = const()[name = string("expand_dims_313"), val = tensor([0])]; tensor expand_dims_315 = const()[name = string("expand_dims_315"), val = tensor([0])]; int32 concat_317_axis_0 = const()[name = string("concat_317_axis_0"), val = int32(0)]; bool concat_317_interleave_0 = const()[name = string("concat_317_interleave_0"), val = bool(false)]; tensor concat_317 = concat(axis = concat_317_axis_0, interleave = concat_317_interleave_0, values = (expand_dims_312, expand_dims_313, position_id, expand_dims_315))[name = string("concat_317")]; tensor expand_dims_316 = const()[name = string("expand_dims_316"), val = tensor([27])]; tensor concat_318_values1_0 = const()[name = string("concat_318_values1_0"), val = tensor([0])]; tensor concat_318_values3_0 = const()[name = string("concat_318_values3_0"), val = tensor([0])]; int32 concat_318_axis_0 = const()[name = string("concat_318_axis_0"), val = int32(0)]; bool concat_318_interleave_0 = const()[name = string("concat_318_interleave_0"), val = bool(false)]; tensor concat_318 = concat(axis = concat_318_axis_0, interleave = concat_318_interleave_0, values = (expand_dims_316, concat_318_values1_0, cache_position_end, concat_318_values3_0))[name = string("concat_318")]; tensor key_states_267_perm_0 = const()[name = string("key_states_267_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_27_stride_0 = const()[name = string("key_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_267_cast_fp16 = transpose(perm = key_states_267_perm_0, x = key_states_265_cast_fp16)[name = string("transpose_349")]; tensor key_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = key_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = key_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_27_squeeze_mask_0, stride = key_cache_internal_tensor_assign_27_stride_0, update = key_states_267_cast_fp16, x = coreml_update_state_274)[name = string("key_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_27_cast_fp16, input = key_cache)[name = string("coreml_update_state_276_write_state")]; tensor coreml_update_state_276 = read_state(input = key_cache)[name = string("coreml_update_state_276")]; tensor value_states_159_perm_0 = const()[name = string("value_states_159_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_27_stride_0 = const()[name = string("value_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_159_cast_fp16 = transpose(perm = value_states_159_perm_0, x = var_9337_cast_fp16)[name = string("transpose_348")]; tensor value_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = value_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = value_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_27_squeeze_mask_0, stride = value_cache_internal_tensor_assign_27_stride_0, update = value_states_159_cast_fp16, x = coreml_update_state_275)[name = string("value_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_27_cast_fp16, input = value_cache)[name = string("coreml_update_state_277_write_state")]; tensor coreml_update_state_277 = read_state(input = value_cache)[name = string("coreml_update_state_277")]; tensor var_9431_begin_0 = const()[name = string("op_9431_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9431_end_0 = const()[name = string("op_9431_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9431_end_mask_0 = const()[name = string("op_9431_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9431_cast_fp16 = slice_by_index(begin = var_9431_begin_0, end = var_9431_end_0, end_mask = var_9431_end_mask_0, x = coreml_update_state_276)[name = string("op_9431_cast_fp16")]; tensor tile_52 = const()[name = string("tile_52"), val = tensor([1, 1])]; int32 var_9434_axis_0 = const()[name = string("op_9434_axis_0"), val = int32(1)]; tensor var_9434_cast_fp16_0, tensor var_9434_cast_fp16_1 = split(axis = var_9434_axis_0, split_sizes = tile_52, x = var_9431_cast_fp16)[name = string("op_9434_cast_fp16")]; tensor var_9441_begin_0 = const()[name = string("op_9441_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9441_end_0 = const()[name = string("op_9441_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9441_end_mask_0 = const()[name = string("op_9441_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9441_cast_fp16 = slice_by_index(begin = var_9441_begin_0, end = var_9441_end_0, end_mask = var_9441_end_mask_0, x = coreml_update_state_277)[name = string("op_9441_cast_fp16")]; tensor tile_53 = const()[name = string("tile_53"), val = tensor([1, 1])]; int32 var_9444_axis_0 = const()[name = string("op_9444_axis_0"), val = int32(1)]; tensor var_9444_cast_fp16_0, tensor var_9444_cast_fp16_1 = split(axis = var_9444_axis_0, split_sizes = tile_53, x = var_9441_cast_fp16)[name = string("op_9444_cast_fp16")]; tensor var_9447_split_sizes_0 = const()[name = string("op_9447_split_sizes_0"), val = tensor([8, 8])]; int32 var_9447_axis_0 = const()[name = string("op_9447_axis_0"), val = int32(1)]; tensor var_9447_0, tensor var_9447_1 = split(axis = var_9447_axis_0, split_sizes = var_9447_split_sizes_0, x = query_states_159_cast_fp16)[name = string("op_9447")]; bool attn_weights_417_transpose_x_0 = const()[name = string("attn_weights_417_transpose_x_0"), val = bool(false)]; bool attn_weights_417_transpose_y_0 = const()[name = string("attn_weights_417_transpose_y_0"), val = bool(false)]; tensor attn_weights_417_cast_fp16 = matmul(transpose_x = attn_weights_417_transpose_x_0, transpose_y = attn_weights_417_transpose_y_0, x = var_9434_cast_fp16_0, y = var_9447_0)[name = string("attn_weights_417_cast_fp16")]; fp16 var_9450_to_fp16 = const()[name = string("op_9450_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_419_cast_fp16 = mul(x = attn_weights_417_cast_fp16, y = var_9450_to_fp16)[name = string("attn_weights_419_cast_fp16")]; tensor attn_weights_421_cast_fp16 = add(x = attn_weights_419_cast_fp16, y = attn_mask_1)[name = string("attn_weights_421_cast_fp16")]; int32 var_9454 = const()[name = string("op_9454"), val = int32(-2)]; tensor attn_weights_423_cast_fp16 = softmax(axis = var_9454, x = attn_weights_421_cast_fp16)[name = string("attn_weights_423_cast_fp16")]; bool var_9460_transpose_x_1 = const()[name = string("op_9460_transpose_x_1"), val = bool(true)]; bool var_9460_transpose_y_1 = const()[name = string("op_9460_transpose_y_1"), val = bool(false)]; tensor var_9460_cast_fp16 = matmul(transpose_x = var_9460_transpose_x_1, transpose_y = var_9460_transpose_y_1, x = attn_weights_423_cast_fp16, y = var_9444_cast_fp16_0)[name = string("op_9460_cast_fp16")]; bool attn_weights_425_transpose_x_0 = const()[name = string("attn_weights_425_transpose_x_0"), val = bool(false)]; bool attn_weights_425_transpose_y_0 = const()[name = string("attn_weights_425_transpose_y_0"), val = bool(false)]; tensor attn_weights_425_cast_fp16 = matmul(transpose_x = attn_weights_425_transpose_x_0, transpose_y = attn_weights_425_transpose_y_0, x = var_9434_cast_fp16_1, y = var_9447_1)[name = string("attn_weights_425_cast_fp16")]; fp16 var_9462_to_fp16 = const()[name = string("op_9462_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_427_cast_fp16 = mul(x = attn_weights_425_cast_fp16, y = var_9462_to_fp16)[name = string("attn_weights_427_cast_fp16")]; tensor attn_weights_429_cast_fp16 = add(x = attn_weights_427_cast_fp16, y = attn_mask_1)[name = string("attn_weights_429_cast_fp16")]; int32 var_9466 = const()[name = string("op_9466"), val = int32(-2)]; tensor attn_weights_431_cast_fp16 = softmax(axis = var_9466, x = attn_weights_429_cast_fp16)[name = string("attn_weights_431_cast_fp16")]; bool attn_output_209_transpose_x_1 = const()[name = string("attn_output_209_transpose_x_1"), val = bool(true)]; bool attn_output_209_transpose_y_1 = const()[name = string("attn_output_209_transpose_y_1"), val = bool(false)]; tensor attn_output_209_cast_fp16 = matmul(transpose_x = attn_output_209_transpose_x_1, transpose_y = attn_output_209_transpose_y_1, x = attn_weights_431_cast_fp16, y = var_9444_cast_fp16_1)[name = string("attn_output_209_cast_fp16")]; int32 var_9474 = const()[name = string("op_9474"), val = int32(1)]; bool attn_output_211_interleave_0 = const()[name = string("attn_output_211_interleave_0"), val = bool(false)]; tensor attn_output_211_cast_fp16 = concat(axis = var_9474, interleave = attn_output_211_interleave_0, values = (var_9460_cast_fp16, attn_output_209_cast_fp16))[name = string("attn_output_211_cast_fp16")]; tensor var_9478_perm_0 = const()[name = string("op_9478_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_323x = const()[name = string("concat_323x"), val = tensor([1, 2048, 1, -1])]; tensor var_9478_cast_fp16 = transpose(perm = var_9478_perm_0, x = attn_output_211_cast_fp16)[name = string("transpose_347")]; tensor attn_output_215_cast_fp16 = reshape(shape = concat_323x, x = var_9478_cast_fp16)[name = string("attn_output_215_cast_fp16")]; tensor hidden_states_263_strides_0 = const()[name = string("hidden_states_263_strides_0"), val = tensor([1, 1])]; string hidden_states_263_pad_type_0 = const()[name = string("hidden_states_263_pad_type_0"), val = string("valid")]; tensor hidden_states_263_pad_0 = const()[name = string("hidden_states_263_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_263_dilations_0 = const()[name = string("hidden_states_263_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_263_groups_0 = const()[name = string("hidden_states_263_groups_0"), val = int32(1)]; tensor hidden_states_263_cast_fp16 = conv(dilations = hidden_states_263_dilations_0, groups = hidden_states_263_groups_0, pad = hidden_states_263_pad_0, pad_type = hidden_states_263_pad_type_0, strides = hidden_states_263_strides_0, weight = layers_26_self_attn_o_proj_weight_cast_fp16, x = attn_output_215_cast_fp16)[name = string("hidden_states_263_cast_fp16")]; tensor hidden_states_265_cast_fp16 = add(x = hidden_states_259_cast_fp16, y = hidden_states_263_cast_fp16)[name = string("hidden_states_265_cast_fp16")]; fp16 const_268_promoted_to_fp16 = const()[name = string("const_268_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9511_cast_fp16 = mul(x = hidden_states_265_cast_fp16, y = const_268_promoted_to_fp16)[name = string("op_9511_cast_fp16")]; int32 var_9509 = const()[name = string("op_9509"), val = int32(1)]; bool doubled_213_interleave_0 = const()[name = string("doubled_213_interleave_0"), val = bool(false)]; tensor doubled_213_cast_fp16 = concat(axis = var_9509, interleave = doubled_213_interleave_0, values = (hidden_states_265_cast_fp16, var_9511_cast_fp16))[name = string("doubled_213_cast_fp16")]; tensor out_107_axes_0 = const()[name = string("out_107_axes_0"), val = tensor([1])]; tensor out_107_gamma_0_to_fp16 = const()[name = string("out_107_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567658624)))]; fp16 var_9521_to_fp16 = const()[name = string("op_9521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_107_cast_fp16 = layer_norm(axes = out_107_axes_0, epsilon = var_9521_to_fp16, gamma = out_107_gamma_0_to_fp16, x = doubled_213_cast_fp16)[name = string("out_107_cast_fp16")]; tensor var_9532_split_sizes_0 = const()[name = string("op_9532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9532_axis_0 = const()[name = string("op_9532_axis_0"), val = int32(1)]; tensor var_9532_cast_fp16_0, tensor var_9532_cast_fp16_1 = split(axis = var_9532_axis_0, split_sizes = var_9532_split_sizes_0, x = out_107_cast_fp16)[name = string("op_9532_cast_fp16")]; tensor input_53_strides_0 = const()[name = string("input_53_strides_0"), val = tensor([1, 1])]; string input_53_pad_type_0 = const()[name = string("input_53_pad_type_0"), val = string("valid")]; tensor input_53_pad_0 = const()[name = string("input_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_53_dilations_0 = const()[name = string("input_53_dilations_0"), val = tensor([1, 1])]; int32 input_53_groups_0 = const()[name = string("input_53_groups_0"), val = int32(1)]; tensor input_53_cast_fp16 = conv(dilations = input_53_dilations_0, groups = input_53_groups_0, pad = input_53_pad_0, pad_type = input_53_pad_type_0, strides = input_53_strides_0, weight = layers_26_mlp_gate_proj_weight_cast_fp16, x = var_9532_cast_fp16_0)[name = string("input_53_cast_fp16")]; tensor var_9549_cast_fp16 = silu(x = input_53_cast_fp16)[name = string("op_9549_cast_fp16")]; tensor layers_26_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567666880)))]; tensor var_9555_strides_0 = const()[name = string("op_9555_strides_0"), val = tensor([1, 1])]; string var_9555_pad_type_0 = const()[name = string("op_9555_pad_type_0"), val = string("valid")]; tensor var_9555_pad_0 = const()[name = string("op_9555_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9555_dilations_0 = const()[name = string("op_9555_dilations_0"), val = tensor([1, 1])]; int32 var_9555_groups_0 = const()[name = string("op_9555_groups_0"), val = int32(1)]; tensor var_9555_cast_fp16 = conv(dilations = var_9555_dilations_0, groups = var_9555_groups_0, pad = var_9555_pad_0, pad_type = var_9555_pad_type_0, strides = var_9555_strides_0, weight = layers_26_mlp_up_proj_weight_to_fp16, x = var_9532_cast_fp16_0)[name = string("op_9555_cast_fp16")]; tensor x_269_cast_fp16 = mul(x = var_9549_cast_fp16, y = var_9555_cast_fp16)[name = string("x_269_cast_fp16")]; tensor layers_26_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1592832768)))]; tensor hidden_states_267_strides_0 = const()[name = string("hidden_states_267_strides_0"), val = tensor([1, 1])]; string hidden_states_267_pad_type_0 = const()[name = string("hidden_states_267_pad_type_0"), val = string("valid")]; tensor hidden_states_267_pad_0 = const()[name = string("hidden_states_267_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_267_dilations_0 = const()[name = string("hidden_states_267_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_267_groups_0 = const()[name = string("hidden_states_267_groups_0"), val = int32(1)]; tensor hidden_states_267_cast_fp16 = conv(dilations = hidden_states_267_dilations_0, groups = hidden_states_267_groups_0, pad = hidden_states_267_pad_0, pad_type = hidden_states_267_pad_type_0, strides = hidden_states_267_strides_0, weight = layers_26_mlp_down_proj_weight_to_fp16, x = x_269_cast_fp16)[name = string("hidden_states_267_cast_fp16")]; tensor hidden_states_269_cast_fp16 = add(x = hidden_states_265_cast_fp16, y = hidden_states_267_cast_fp16)[name = string("hidden_states_269_cast_fp16")]; fp16 const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9573_cast_fp16 = mul(x = hidden_states_269_cast_fp16, y = const_270_promoted_to_fp16)[name = string("op_9573_cast_fp16")]; int32 var_9571 = const()[name = string("op_9571"), val = int32(1)]; bool doubled_217_interleave_0 = const()[name = string("doubled_217_interleave_0"), val = bool(false)]; tensor doubled_217_cast_fp16 = concat(axis = var_9571, interleave = doubled_217_interleave_0, values = (hidden_states_269_cast_fp16, var_9573_cast_fp16))[name = string("doubled_217_cast_fp16")]; tensor out_109_axes_0 = const()[name = string("out_109_axes_0"), val = tensor([1])]; tensor out_109_gamma_0_to_fp16 = const()[name = string("out_109_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1617998656)))]; fp16 var_9583_to_fp16 = const()[name = string("op_9583_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_109_cast_fp16 = layer_norm(axes = out_109_axes_0, epsilon = var_9583_to_fp16, gamma = out_109_gamma_0_to_fp16, x = doubled_217_cast_fp16)[name = string("out_109_cast_fp16")]; tensor var_9594_split_sizes_0 = const()[name = string("op_9594_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9594_axis_0 = const()[name = string("op_9594_axis_0"), val = int32(1)]; tensor var_9594_cast_fp16_0, tensor var_9594_cast_fp16_1 = split(axis = var_9594_axis_0, split_sizes = var_9594_split_sizes_0, x = out_109_cast_fp16)[name = string("op_9594_cast_fp16")]; tensor layers_27_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1618006912)))]; tensor query_states_163_strides_0 = const()[name = string("query_states_163_strides_0"), val = tensor([1, 1])]; string query_states_163_pad_type_0 = const()[name = string("query_states_163_pad_type_0"), val = string("valid")]; tensor query_states_163_pad_0 = const()[name = string("query_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_163_dilations_0 = const()[name = string("query_states_163_dilations_0"), val = tensor([1, 1])]; int32 query_states_163_groups_0 = const()[name = string("query_states_163_groups_0"), val = int32(1)]; tensor query_states_163_cast_fp16 = conv(dilations = query_states_163_dilations_0, groups = query_states_163_groups_0, pad = query_states_163_pad_0, pad_type = query_states_163_pad_type_0, strides = query_states_163_strides_0, weight = layers_27_self_attn_q_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("query_states_163_cast_fp16")]; tensor layers_27_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1626395584)))]; tensor key_states_271_strides_0 = const()[name = string("key_states_271_strides_0"), val = tensor([1, 1])]; string key_states_271_pad_type_0 = const()[name = string("key_states_271_pad_type_0"), val = string("valid")]; tensor key_states_271_pad_0 = const()[name = string("key_states_271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_271_dilations_0 = const()[name = string("key_states_271_dilations_0"), val = tensor([1, 1])]; int32 key_states_271_groups_0 = const()[name = string("key_states_271_groups_0"), val = int32(1)]; tensor key_states_271_cast_fp16 = conv(dilations = key_states_271_dilations_0, groups = key_states_271_groups_0, pad = key_states_271_pad_0, pad_type = key_states_271_pad_type_0, strides = key_states_271_strides_0, weight = layers_27_self_attn_k_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("key_states_271_cast_fp16")]; tensor layers_27_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1627444224)))]; tensor value_states_163_strides_0 = const()[name = string("value_states_163_strides_0"), val = tensor([1, 1])]; string value_states_163_pad_type_0 = const()[name = string("value_states_163_pad_type_0"), val = string("valid")]; tensor value_states_163_pad_0 = const()[name = string("value_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_163_dilations_0 = const()[name = string("value_states_163_dilations_0"), val = tensor([1, 1])]; int32 value_states_163_groups_0 = const()[name = string("value_states_163_groups_0"), val = int32(1)]; tensor value_states_163_cast_fp16 = conv(dilations = value_states_163_dilations_0, groups = value_states_163_groups_0, pad = value_states_163_pad_0, pad_type = value_states_163_pad_type_0, strides = value_states_163_strides_0, weight = layers_27_self_attn_v_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("value_states_163_cast_fp16")]; tensor concat_324x = const()[name = string("concat_324x"), val = tensor([1, 16, 128, -1])]; tensor x_271_cast_fp16 = reshape(shape = concat_324x, x = query_states_163_cast_fp16)[name = string("x_271_cast_fp16")]; tensor concat_325x = const()[name = string("concat_325x"), val = tensor([1, 2, 128, -1])]; tensor var_9651_cast_fp16 = reshape(shape = concat_325x, x = key_states_271_cast_fp16)[name = string("op_9651_cast_fp16")]; tensor concat_326x = const()[name = string("concat_326x"), val = tensor([1, 2, 128, -1])]; tensor var_9658_cast_fp16 = reshape(shape = concat_326x, x = value_states_163_cast_fp16)[name = string("op_9658_cast_fp16")]; tensor var_9662_cast_fp16 = mul(x = x_271_cast_fp16, y = var_869_cast_fp16)[name = string("op_9662_cast_fp16")]; tensor var_9663_split_sizes_0 = const()[name = string("op_9663_split_sizes_0"), val = tensor([64, 64])]; int32 var_9663_axis_0 = const()[name = string("op_9663_axis_0"), val = int32(-2)]; tensor var_9663_cast_fp16_0, tensor var_9663_cast_fp16_1 = split(axis = var_9663_axis_0, split_sizes = var_9663_split_sizes_0, x = x_271_cast_fp16)[name = string("op_9663_cast_fp16")]; fp16 const_272_promoted_to_fp16 = const()[name = string("const_272_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9665_cast_fp16 = mul(x = var_9663_cast_fp16_1, y = const_272_promoted_to_fp16)[name = string("op_9665_cast_fp16")]; int32 var_9667 = const()[name = string("op_9667"), val = int32(-2)]; bool var_9668_interleave_0 = const()[name = string("op_9668_interleave_0"), val = bool(false)]; tensor var_9668_cast_fp16 = concat(axis = var_9667, interleave = var_9668_interleave_0, values = (var_9665_cast_fp16, var_9663_cast_fp16_0))[name = string("op_9668_cast_fp16")]; tensor var_9669_cast_fp16 = mul(x = var_9668_cast_fp16, y = var_878_cast_fp16)[name = string("op_9669_cast_fp16")]; tensor query_states_165_cast_fp16 = add(x = var_9662_cast_fp16, y = var_9669_cast_fp16)[name = string("query_states_165_cast_fp16")]; tensor var_9675_cast_fp16 = mul(x = var_9651_cast_fp16, y = var_869_cast_fp16)[name = string("op_9675_cast_fp16")]; tensor var_9676_split_sizes_0 = const()[name = string("op_9676_split_sizes_0"), val = tensor([64, 64])]; int32 var_9676_axis_0 = const()[name = string("op_9676_axis_0"), val = int32(-2)]; tensor var_9676_cast_fp16_0, tensor var_9676_cast_fp16_1 = split(axis = var_9676_axis_0, split_sizes = var_9676_split_sizes_0, x = var_9651_cast_fp16)[name = string("op_9676_cast_fp16")]; fp16 const_273_promoted_to_fp16 = const()[name = string("const_273_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9678_cast_fp16 = mul(x = var_9676_cast_fp16_1, y = const_273_promoted_to_fp16)[name = string("op_9678_cast_fp16")]; int32 var_9680 = const()[name = string("op_9680"), val = int32(-2)]; bool var_9681_interleave_0 = const()[name = string("op_9681_interleave_0"), val = bool(false)]; tensor var_9681_cast_fp16 = concat(axis = var_9680, interleave = var_9681_interleave_0, values = (var_9678_cast_fp16, var_9676_cast_fp16_0))[name = string("op_9681_cast_fp16")]; tensor var_9682_cast_fp16 = mul(x = var_9681_cast_fp16, y = var_878_cast_fp16)[name = string("op_9682_cast_fp16")]; tensor key_states_275_cast_fp16 = add(x = var_9675_cast_fp16, y = var_9682_cast_fp16)[name = string("key_states_275_cast_fp16")]; tensor expand_dims_324 = const()[name = string("expand_dims_324"), val = tensor([27])]; tensor expand_dims_325 = const()[name = string("expand_dims_325"), val = tensor([0])]; tensor expand_dims_327 = const()[name = string("expand_dims_327"), val = tensor([0])]; int32 concat_329_axis_0 = const()[name = string("concat_329_axis_0"), val = int32(0)]; bool concat_329_interleave_0 = const()[name = string("concat_329_interleave_0"), val = bool(false)]; tensor concat_329 = concat(axis = concat_329_axis_0, interleave = concat_329_interleave_0, values = (expand_dims_324, expand_dims_325, position_id, expand_dims_327))[name = string("concat_329")]; tensor expand_dims_328 = const()[name = string("expand_dims_328"), val = tensor([28])]; tensor concat_330_values1_0 = const()[name = string("concat_330_values1_0"), val = tensor([0])]; tensor concat_330_values3_0 = const()[name = string("concat_330_values3_0"), val = tensor([0])]; int32 concat_330_axis_0 = const()[name = string("concat_330_axis_0"), val = int32(0)]; bool concat_330_interleave_0 = const()[name = string("concat_330_interleave_0"), val = bool(false)]; tensor concat_330 = concat(axis = concat_330_axis_0, interleave = concat_330_interleave_0, values = (expand_dims_328, concat_330_values1_0, cache_position_end, concat_330_values3_0))[name = string("concat_330")]; tensor key_states_277_perm_0 = const()[name = string("key_states_277_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_28_stride_0 = const()[name = string("key_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_277_cast_fp16 = transpose(perm = key_states_277_perm_0, x = key_states_275_cast_fp16)[name = string("transpose_346")]; tensor key_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = key_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = key_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_28_squeeze_mask_0, stride = key_cache_internal_tensor_assign_28_stride_0, update = key_states_277_cast_fp16, x = coreml_update_state_276)[name = string("key_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_28_cast_fp16, input = key_cache)[name = string("coreml_update_state_278_write_state")]; tensor coreml_update_state_278 = read_state(input = key_cache)[name = string("coreml_update_state_278")]; tensor value_states_165_perm_0 = const()[name = string("value_states_165_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_28_stride_0 = const()[name = string("value_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_165_cast_fp16 = transpose(perm = value_states_165_perm_0, x = var_9658_cast_fp16)[name = string("transpose_345")]; tensor value_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = value_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = value_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_28_squeeze_mask_0, stride = value_cache_internal_tensor_assign_28_stride_0, update = value_states_165_cast_fp16, x = coreml_update_state_277)[name = string("value_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_28_cast_fp16, input = value_cache)[name = string("coreml_update_state_279_write_state")]; tensor coreml_update_state_279 = read_state(input = value_cache)[name = string("coreml_update_state_279")]; tensor var_9752_begin_0 = const()[name = string("op_9752_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9752_end_0 = const()[name = string("op_9752_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9752_end_mask_0 = const()[name = string("op_9752_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9752_cast_fp16 = slice_by_index(begin = var_9752_begin_0, end = var_9752_end_0, end_mask = var_9752_end_mask_0, x = coreml_update_state_278)[name = string("op_9752_cast_fp16")]; tensor tile_54 = const()[name = string("tile_54"), val = tensor([1, 1])]; int32 var_9755_axis_0 = const()[name = string("op_9755_axis_0"), val = int32(1)]; tensor var_9755_cast_fp16_0, tensor var_9755_cast_fp16_1 = split(axis = var_9755_axis_0, split_sizes = tile_54, x = var_9752_cast_fp16)[name = string("op_9755_cast_fp16")]; tensor var_9762_begin_0 = const()[name = string("op_9762_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9762_end_0 = const()[name = string("op_9762_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9762_end_mask_0 = const()[name = string("op_9762_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9762_cast_fp16 = slice_by_index(begin = var_9762_begin_0, end = var_9762_end_0, end_mask = var_9762_end_mask_0, x = coreml_update_state_279)[name = string("op_9762_cast_fp16")]; tensor tile_55 = const()[name = string("tile_55"), val = tensor([1, 1])]; int32 var_9765_axis_0 = const()[name = string("op_9765_axis_0"), val = int32(1)]; tensor var_9765_cast_fp16_0, tensor var_9765_cast_fp16_1 = split(axis = var_9765_axis_0, split_sizes = tile_55, x = var_9762_cast_fp16)[name = string("op_9765_cast_fp16")]; tensor var_9768_split_sizes_0 = const()[name = string("op_9768_split_sizes_0"), val = tensor([8, 8])]; int32 var_9768_axis_0 = const()[name = string("op_9768_axis_0"), val = int32(1)]; tensor var_9768_0, tensor var_9768_1 = split(axis = var_9768_axis_0, split_sizes = var_9768_split_sizes_0, x = query_states_165_cast_fp16)[name = string("op_9768")]; bool attn_weights_433_transpose_x_0 = const()[name = string("attn_weights_433_transpose_x_0"), val = bool(false)]; bool attn_weights_433_transpose_y_0 = const()[name = string("attn_weights_433_transpose_y_0"), val = bool(false)]; tensor attn_weights_433_cast_fp16 = matmul(transpose_x = attn_weights_433_transpose_x_0, transpose_y = attn_weights_433_transpose_y_0, x = var_9755_cast_fp16_0, y = var_9768_0)[name = string("attn_weights_433_cast_fp16")]; fp16 var_9771_to_fp16 = const()[name = string("op_9771_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_435_cast_fp16 = mul(x = attn_weights_433_cast_fp16, y = var_9771_to_fp16)[name = string("attn_weights_435_cast_fp16")]; tensor attn_weights_437_cast_fp16 = add(x = attn_weights_435_cast_fp16, y = attn_mask_1)[name = string("attn_weights_437_cast_fp16")]; int32 var_9775 = const()[name = string("op_9775"), val = int32(-2)]; tensor attn_weights_439_cast_fp16 = softmax(axis = var_9775, x = attn_weights_437_cast_fp16)[name = string("attn_weights_439_cast_fp16")]; bool var_9781_transpose_x_1 = const()[name = string("op_9781_transpose_x_1"), val = bool(true)]; bool var_9781_transpose_y_1 = const()[name = string("op_9781_transpose_y_1"), val = bool(false)]; tensor var_9781_cast_fp16 = matmul(transpose_x = var_9781_transpose_x_1, transpose_y = var_9781_transpose_y_1, x = attn_weights_439_cast_fp16, y = var_9765_cast_fp16_0)[name = string("op_9781_cast_fp16")]; bool attn_weights_441_transpose_x_0 = const()[name = string("attn_weights_441_transpose_x_0"), val = bool(false)]; bool attn_weights_441_transpose_y_0 = const()[name = string("attn_weights_441_transpose_y_0"), val = bool(false)]; tensor attn_weights_441_cast_fp16 = matmul(transpose_x = attn_weights_441_transpose_x_0, transpose_y = attn_weights_441_transpose_y_0, x = var_9755_cast_fp16_1, y = var_9768_1)[name = string("attn_weights_441_cast_fp16")]; fp16 var_9783_to_fp16 = const()[name = string("op_9783_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_443_cast_fp16 = mul(x = attn_weights_441_cast_fp16, y = var_9783_to_fp16)[name = string("attn_weights_443_cast_fp16")]; tensor attn_weights_445_cast_fp16 = add(x = attn_weights_443_cast_fp16, y = attn_mask_1)[name = string("attn_weights_445_cast_fp16")]; int32 var_9787 = const()[name = string("op_9787"), val = int32(-2)]; tensor attn_weights_cast_fp16 = softmax(axis = var_9787, x = attn_weights_445_cast_fp16)[name = string("attn_weights_cast_fp16")]; bool attn_output_217_transpose_x_1 = const()[name = string("attn_output_217_transpose_x_1"), val = bool(true)]; bool attn_output_217_transpose_y_1 = const()[name = string("attn_output_217_transpose_y_1"), val = bool(false)]; tensor attn_output_217_cast_fp16 = matmul(transpose_x = attn_output_217_transpose_x_1, transpose_y = attn_output_217_transpose_y_1, x = attn_weights_cast_fp16, y = var_9765_cast_fp16_1)[name = string("attn_output_217_cast_fp16")]; int32 var_9795 = const()[name = string("op_9795"), val = int32(1)]; bool attn_output_219_interleave_0 = const()[name = string("attn_output_219_interleave_0"), val = bool(false)]; tensor attn_output_219_cast_fp16 = concat(axis = var_9795, interleave = attn_output_219_interleave_0, values = (var_9781_cast_fp16, attn_output_217_cast_fp16))[name = string("attn_output_219_cast_fp16")]; tensor var_9799_perm_0 = const()[name = string("op_9799_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_335x = const()[name = string("concat_335x"), val = tensor([1, 2048, 1, -1])]; tensor var_9799_cast_fp16 = transpose(perm = var_9799_perm_0, x = attn_output_219_cast_fp16)[name = string("transpose_344")]; tensor attn_output_cast_fp16 = reshape(shape = concat_335x, x = var_9799_cast_fp16)[name = string("attn_output_cast_fp16")]; tensor layers_27_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1628492864)))]; tensor hidden_states_273_strides_0 = const()[name = string("hidden_states_273_strides_0"), val = tensor([1, 1])]; string hidden_states_273_pad_type_0 = const()[name = string("hidden_states_273_pad_type_0"), val = string("valid")]; tensor hidden_states_273_pad_0 = const()[name = string("hidden_states_273_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_273_dilations_0 = const()[name = string("hidden_states_273_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_273_groups_0 = const()[name = string("hidden_states_273_groups_0"), val = int32(1)]; tensor hidden_states_273_cast_fp16 = conv(dilations = hidden_states_273_dilations_0, groups = hidden_states_273_groups_0, pad = hidden_states_273_pad_0, pad_type = hidden_states_273_pad_type_0, strides = hidden_states_273_strides_0, weight = layers_27_self_attn_o_proj_weight_to_fp16, x = attn_output_cast_fp16)[name = string("hidden_states_273_cast_fp16")]; tensor hidden_states_275_cast_fp16 = add(x = hidden_states_269_cast_fp16, y = hidden_states_273_cast_fp16)[name = string("hidden_states_275_cast_fp16")]; fp16 const_278_promoted_to_fp16 = const()[name = string("const_278_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9832_cast_fp16 = mul(x = hidden_states_275_cast_fp16, y = const_278_promoted_to_fp16)[name = string("op_9832_cast_fp16")]; int32 var_9830 = const()[name = string("op_9830"), val = int32(1)]; bool doubled_221_interleave_0 = const()[name = string("doubled_221_interleave_0"), val = bool(false)]; tensor doubled_221_cast_fp16 = concat(axis = var_9830, interleave = doubled_221_interleave_0, values = (hidden_states_275_cast_fp16, var_9832_cast_fp16))[name = string("doubled_221_cast_fp16")]; tensor out_111_axes_0 = const()[name = string("out_111_axes_0"), val = tensor([1])]; tensor out_111_gamma_0_to_fp16 = const()[name = string("out_111_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636881536)))]; fp16 var_9842_to_fp16 = const()[name = string("op_9842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_111_cast_fp16 = layer_norm(axes = out_111_axes_0, epsilon = var_9842_to_fp16, gamma = out_111_gamma_0_to_fp16, x = doubled_221_cast_fp16)[name = string("out_111_cast_fp16")]; tensor var_9853_split_sizes_0 = const()[name = string("op_9853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9853_axis_0 = const()[name = string("op_9853_axis_0"), val = int32(1)]; tensor var_9853_cast_fp16_0, tensor var_9853_cast_fp16_1 = split(axis = var_9853_axis_0, split_sizes = var_9853_split_sizes_0, x = out_111_cast_fp16)[name = string("op_9853_cast_fp16")]; tensor layers_27_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636889792)))]; tensor input_strides_0 = const()[name = string("input_strides_0"), val = tensor([1, 1])]; string input_pad_type_0 = const()[name = string("input_pad_type_0"), val = string("valid")]; tensor input_pad_0 = const()[name = string("input_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_dilations_0 = const()[name = string("input_dilations_0"), val = tensor([1, 1])]; int32 input_groups_0 = const()[name = string("input_groups_0"), val = int32(1)]; tensor input_cast_fp16 = conv(dilations = input_dilations_0, groups = input_groups_0, pad = input_pad_0, pad_type = input_pad_type_0, strides = input_strides_0, weight = layers_27_mlp_gate_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("input_cast_fp16")]; tensor var_9870_cast_fp16 = silu(x = input_cast_fp16)[name = string("op_9870_cast_fp16")]; tensor layers_27_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1662055680)))]; tensor var_9876_strides_0 = const()[name = string("op_9876_strides_0"), val = tensor([1, 1])]; string var_9876_pad_type_0 = const()[name = string("op_9876_pad_type_0"), val = string("valid")]; tensor var_9876_pad_0 = const()[name = string("op_9876_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9876_dilations_0 = const()[name = string("op_9876_dilations_0"), val = tensor([1, 1])]; int32 var_9876_groups_0 = const()[name = string("op_9876_groups_0"), val = int32(1)]; tensor var_9876_cast_fp16 = conv(dilations = var_9876_dilations_0, groups = var_9876_groups_0, pad = var_9876_pad_0, pad_type = var_9876_pad_type_0, strides = var_9876_strides_0, weight = layers_27_mlp_up_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("op_9876_cast_fp16")]; tensor x_cast_fp16 = mul(x = var_9870_cast_fp16, y = var_9876_cast_fp16)[name = string("x_cast_fp16")]; tensor layers_27_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1687221568)))]; tensor hidden_states_277_strides_0 = const()[name = string("hidden_states_277_strides_0"), val = tensor([1, 1])]; string hidden_states_277_pad_type_0 = const()[name = string("hidden_states_277_pad_type_0"), val = string("valid")]; tensor hidden_states_277_pad_0 = const()[name = string("hidden_states_277_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_277_dilations_0 = const()[name = string("hidden_states_277_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_277_groups_0 = const()[name = string("hidden_states_277_groups_0"), val = int32(1)]; tensor hidden_states_277_cast_fp16 = conv(dilations = hidden_states_277_dilations_0, groups = hidden_states_277_groups_0, pad = hidden_states_277_pad_0, pad_type = hidden_states_277_pad_type_0, strides = hidden_states_277_strides_0, weight = layers_27_mlp_down_proj_weight_to_fp16, x = x_cast_fp16)[name = string("hidden_states_277_cast_fp16")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_275_cast_fp16, y = hidden_states_277_cast_fp16)[name = string("hidden_states_cast_fp16")]; fp16 const_280_promoted_to_fp16 = const()[name = string("const_280_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9894_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_280_promoted_to_fp16)[name = string("op_9894_cast_fp16")]; int32 var_9892 = const()[name = string("op_9892"), val = int32(1)]; bool doubled_225_interleave_0 = const()[name = string("doubled_225_interleave_0"), val = bool(false)]; tensor doubled_225_cast_fp16 = concat(axis = var_9892, interleave = doubled_225_interleave_0, values = (hidden_states_cast_fp16, var_9894_cast_fp16))[name = string("doubled_225_cast_fp16")]; tensor out_axes_0 = const()[name = string("out_axes_0"), val = tensor([1])]; tensor out_gamma_0_to_fp16 = const()[name = string("out_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1712387456)))]; fp16 var_9904_to_fp16 = const()[name = string("op_9904_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_cast_fp16 = layer_norm(axes = out_axes_0, epsilon = var_9904_to_fp16, gamma = out_gamma_0_to_fp16, x = doubled_225_cast_fp16)[name = string("out_cast_fp16")]; tensor var_9915_split_sizes_0 = const()[name = string("op_9915_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9915_axis_0 = const()[name = string("op_9915_axis_0"), val = int32(1)]; tensor hidden_states, tensor var_9915_cast_fp16_1 = split(axis = var_9915_axis_0, split_sizes = var_9915_split_sizes_0, x = out_cast_fp16)[name = string("op_9915_cast_fp16")]; } -> (hidden_states); func length_64(tensor inputs_embeds, state> key_cache, tensor position_id, tensor position_index_seed, state> value_cache) { tensor layers_1_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524992))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524416))))[name = string("layers_1_self_attn_v_proj_weight_cast_fp16")]; tensor layers_1_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(525312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13120640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13108288))))[name = string("layers_1_mlp_up_proj_weight_cast_fp16")]; tensor layers_2_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13126848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651776))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651200))))[name = string("layers_2_self_attn_v_proj_weight_cast_fp16")]; tensor layers_2_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13652096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26247424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26235072))))[name = string("layers_2_mlp_up_proj_weight_cast_fp16")]; tensor layers_3_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26253632))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778560))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26777984))))[name = string("layers_3_self_attn_v_proj_weight_cast_fp16")]; tensor layers_3_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30977408))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30973248))))[name = string("layers_3_self_attn_o_proj_weight_cast_fp16")]; tensor layers_3_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30979520))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43566656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43562496))))[name = string("layers_3_mlp_down_proj_weight_cast_fp16")]; tensor layers_4_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43568768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093696))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093120))))[name = string("layers_4_self_attn_v_proj_weight_cast_fp16")]; tensor layers_4_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44094016))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48292544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48288384))))[name = string("layers_4_self_attn_o_proj_weight_cast_fp16")]; tensor layers_4_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48294656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60877632))))[name = string("layers_4_mlp_gate_proj_weight_cast_fp16")]; tensor layers_4_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60896192))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73491520))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73479168))))[name = string("layers_4_mlp_up_proj_weight_cast_fp16")]; tensor layers_4_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73497728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86084864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86080704))))[name = string("layers_4_mlp_down_proj_weight_cast_fp16")]; tensor layers_5_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86086976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611328))))[name = string("layers_5_self_attn_v_proj_weight_cast_fp16")]; tensor layers_5_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86612224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90810752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90806592))))[name = string("layers_5_self_attn_o_proj_weight_cast_fp16")]; tensor layers_5_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90812864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103408192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103395840))))[name = string("layers_5_mlp_up_proj_weight_cast_fp16")]; tensor layers_5_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103414400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116001536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115997376))))[name = string("layers_5_mlp_down_proj_weight_cast_fp16")]; tensor layers_6_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116003648))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528576))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528000))))[name = string("layers_6_self_attn_v_proj_weight_cast_fp16")]; tensor layers_6_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528896))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120727424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120723264))))[name = string("layers_6_self_attn_o_proj_weight_cast_fp16")]; tensor layers_6_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120729536))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133324864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133312512))))[name = string("layers_6_mlp_gate_proj_weight_cast_fp16")]; tensor layers_6_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133331072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145926400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145914048))))[name = string("layers_6_mlp_up_proj_weight_cast_fp16")]; tensor layers_6_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145932608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158519744))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158515584))))[name = string("layers_6_mlp_down_proj_weight_cast_fp16")]; tensor layers_7_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158521856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046208))))[name = string("layers_7_self_attn_v_proj_weight_cast_fp16")]; tensor layers_7_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159047104))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163245632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163241472))))[name = string("layers_7_self_attn_o_proj_weight_cast_fp16")]; tensor layers_7_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163247744))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175843072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175830720))))[name = string("layers_7_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175849280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374208))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176373632))))[name = string("layers_8_self_attn_v_proj_weight_cast_fp16")]; tensor layers_8_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374528))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180573056))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180568896))))[name = string("layers_8_self_attn_o_proj_weight_cast_fp16")]; tensor layers_8_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180575168))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193170496))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193158144))))[name = string("layers_8_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193176704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205772032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205759680))))[name = string("layers_8_mlp_up_proj_weight_cast_fp16")]; tensor layers_8_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205778240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218365376))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218361216))))[name = string("layers_8_mlp_down_proj_weight_cast_fp16")]; tensor layers_9_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218367488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218891840))))[name = string("layers_9_self_attn_v_proj_weight_cast_fp16")]; tensor layers_9_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223091264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223087104))))[name = string("layers_9_self_attn_o_proj_weight_cast_fp16")]; tensor layers_9_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223093376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235688704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235676352))))[name = string("layers_9_mlp_gate_proj_weight_cast_fp16")]; tensor layers_9_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235694912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248290240))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248277888))))[name = string("layers_9_mlp_up_proj_weight_cast_fp16")]; tensor layers_9_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248296448))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260883584))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260879424))))[name = string("layers_9_mlp_down_proj_weight_cast_fp16")]; tensor layers_10_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260885696))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410048))))[name = string("layers_10_self_attn_v_proj_weight_cast_fp16")]; tensor layers_10_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410944))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265609472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265605312))))[name = string("layers_10_self_attn_o_proj_weight_cast_fp16")]; tensor layers_10_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265611584))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278206912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278194560))))[name = string("layers_10_mlp_gate_proj_weight_cast_fp16")]; tensor layers_10_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278213120))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290808448))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290796096))))[name = string("layers_10_mlp_up_proj_weight_cast_fp16")]; tensor layers_10_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290814656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303401792))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303397632))))[name = string("layers_10_mlp_down_proj_weight_cast_fp16")]; tensor layers_11_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303403904))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307602432))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307598272))))[name = string("layers_11_self_attn_q_proj_weight_cast_fp16")]; tensor layers_11_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307604544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308128896))))[name = string("layers_11_self_attn_k_proj_weight_cast_fp16")]; tensor layers_11_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129792))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654720))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654144))))[name = string("layers_11_self_attn_v_proj_weight_cast_fp16")]; tensor layers_11_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308655040))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312853568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312849408))))[name = string("layers_11_self_attn_o_proj_weight_cast_fp16")]; tensor layers_11_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312855680))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325451008))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325438656))))[name = string("layers_11_mlp_gate_proj_weight_cast_fp16")]; tensor layers_11_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325457216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338052544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338040192))))[name = string("layers_11_mlp_up_proj_weight_cast_fp16")]; tensor layers_11_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338058752))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350645888))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350641728))))[name = string("layers_11_mlp_down_proj_weight_cast_fp16")]; tensor layers_12_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350648000))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354846528))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354842368))))[name = string("layers_12_self_attn_q_proj_weight_cast_fp16")]; tensor layers_12_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354848640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355372992))))[name = string("layers_12_self_attn_k_proj_weight_cast_fp16")]; tensor layers_12_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373888))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898816))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898240))))[name = string("layers_12_self_attn_v_proj_weight_cast_fp16")]; tensor layers_12_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355899136))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360097664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360093504))))[name = string("layers_12_self_attn_o_proj_weight_cast_fp16")]; tensor layers_12_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360099776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372695104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372682752))))[name = string("layers_12_mlp_gate_proj_weight_cast_fp16")]; tensor layers_12_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372701312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385296640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385284288))))[name = string("layers_12_mlp_up_proj_weight_cast_fp16")]; tensor layers_12_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385302848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397885824))))[name = string("layers_12_mlp_down_proj_weight_cast_fp16")]; tensor layers_13_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397892096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402090624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402086464))))[name = string("layers_13_self_attn_q_proj_weight_cast_fp16")]; tensor layers_13_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402092736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617088))))[name = string("layers_13_self_attn_k_proj_weight_cast_fp16")]; tensor layers_13_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617984))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142336))))[name = string("layers_13_self_attn_v_proj_weight_cast_fp16")]; tensor layers_13_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403143232))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407341760))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407337600))))[name = string("layers_13_self_attn_o_proj_weight_cast_fp16")]; tensor layers_13_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407343872))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419939200))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419926848))))[name = string("layers_13_mlp_gate_proj_weight_cast_fp16")]; tensor layers_13_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419945408))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432532544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432528384))))[name = string("layers_13_mlp_down_proj_weight_cast_fp16")]; tensor layers_14_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432534656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436733184))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436729024))))[name = string("layers_14_self_attn_q_proj_weight_cast_fp16")]; tensor layers_14_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436735296))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437259648))))[name = string("layers_14_self_attn_v_proj_weight_cast_fp16")]; tensor layers_14_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441459072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441454912))))[name = string("layers_14_self_attn_o_proj_weight_cast_fp16")]; tensor layers_14_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441461184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454056512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454044160))))[name = string("layers_14_mlp_gate_proj_weight_cast_fp16")]; tensor layers_14_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454062720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466658048))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466645696))))[name = string("layers_14_mlp_up_proj_weight_cast_fp16")]; tensor layers_14_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466664256))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479251392))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479247232))))[name = string("layers_14_mlp_down_proj_weight_cast_fp16")]; tensor layers_15_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479253504))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483452032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483447872))))[name = string("layers_15_self_attn_q_proj_weight_cast_fp16")]; tensor layers_15_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483454144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483978496))))[name = string("layers_15_self_attn_k_proj_weight_cast_fp16")]; tensor layers_15_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504320))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484503744))))[name = string("layers_15_self_attn_v_proj_weight_cast_fp16")]; tensor layers_15_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488703168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488699008))))[name = string("layers_15_self_attn_o_proj_weight_cast_fp16")]; tensor layers_15_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488705280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501300608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501288256))))[name = string("layers_15_mlp_gate_proj_weight_cast_fp16")]; tensor layers_15_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501306816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513902144))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513889792))))[name = string("layers_15_mlp_up_proj_weight_cast_fp16")]; tensor layers_15_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513908352))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526495488))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526491328))))[name = string("layers_15_mlp_down_proj_weight_cast_fp16")]; tensor layers_16_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526497600))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530696128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530691968))))[name = string("layers_16_self_attn_q_proj_weight_cast_fp16")]; tensor layers_16_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530698240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531222592))))[name = string("layers_16_self_attn_k_proj_weight_cast_fp16")]; tensor layers_16_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531747840))))[name = string("layers_16_self_attn_v_proj_weight_cast_fp16")]; tensor layers_16_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535947264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535943104))))[name = string("layers_16_self_attn_o_proj_weight_cast_fp16")]; tensor layers_16_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535949376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548536512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548532352))))[name = string("layers_16_mlp_down_proj_weight_cast_fp16")]; tensor layers_17_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548538624))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552737152))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552732992))))[name = string("layers_17_self_attn_q_proj_weight_cast_fp16")]; tensor layers_17_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552739264))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553263616))))[name = string("layers_17_self_attn_k_proj_weight_cast_fp16")]; tensor layers_17_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264512))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553788864))))[name = string("layers_17_self_attn_v_proj_weight_cast_fp16")]; tensor layers_17_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789760))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557988288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557984128))))[name = string("layers_17_self_attn_o_proj_weight_cast_fp16")]; tensor layers_17_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557990400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570585728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570573376))))[name = string("layers_17_mlp_gate_proj_weight_cast_fp16")]; tensor layers_17_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570591936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583187264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583174912))))[name = string("layers_17_mlp_up_proj_weight_cast_fp16")]; tensor layers_17_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583193472))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595780608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595776448))))[name = string("layers_17_mlp_down_proj_weight_cast_fp16")]; tensor layers_18_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595782720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599981248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599977088))))[name = string("layers_18_self_attn_q_proj_weight_cast_fp16")]; tensor layers_18_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599983360))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600507712))))[name = string("layers_18_self_attn_k_proj_weight_cast_fp16")]; tensor layers_18_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601032960))))[name = string("layers_18_self_attn_v_proj_weight_cast_fp16")]; tensor layers_18_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605232384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605228224))))[name = string("layers_18_self_attn_o_proj_weight_cast_fp16")]; tensor layers_18_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605234496))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617829824))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617817472))))[name = string("layers_18_mlp_gate_proj_weight_cast_fp16")]; tensor layers_18_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617836032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630431360))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630419008))))[name = string("layers_18_mlp_up_proj_weight_cast_fp16")]; tensor layers_18_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630437568))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643024704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643020544))))[name = string("layers_18_mlp_down_proj_weight_cast_fp16")]; tensor layers_19_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643026816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647225344))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647221184))))[name = string("layers_19_self_attn_q_proj_weight_cast_fp16")]; tensor layers_19_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647227456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647751808))))[name = string("layers_19_self_attn_k_proj_weight_cast_fp16")]; tensor layers_19_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660348032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660335680))))[name = string("layers_19_mlp_gate_proj_weight_cast_fp16")]; tensor layers_19_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660354240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672949568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672937216))))[name = string("layers_19_mlp_up_proj_weight_cast_fp16")]; tensor layers_19_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672955776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685542912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685538752))))[name = string("layers_19_mlp_down_proj_weight_cast_fp16")]; tensor layers_20_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685545024))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689743552))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689739392))))[name = string("layers_20_self_attn_q_proj_weight_cast_fp16")]; tensor layers_20_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689745664))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270592))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270016))))[name = string("layers_20_self_attn_k_proj_weight_cast_fp16")]; tensor layers_20_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694469440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694465280))))[name = string("layers_20_self_attn_o_proj_weight_cast_fp16")]; tensor layers_20_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694471552))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707066880))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707054528))))[name = string("layers_20_mlp_gate_proj_weight_cast_fp16")]; tensor layers_20_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707073088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719660224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719656064))))[name = string("layers_20_mlp_down_proj_weight_cast_fp16")]; tensor layers_21_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719662336))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723860864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723856704))))[name = string("layers_21_self_attn_q_proj_weight_cast_fp16")]; tensor layers_21_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723862976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387328))))[name = string("layers_21_self_attn_k_proj_weight_cast_fp16")]; tensor layers_21_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724388224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728586752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728582592))))[name = string("layers_21_self_attn_o_proj_weight_cast_fp16")]; tensor layers_21_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728588864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741184192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741171840))))[name = string("layers_21_mlp_gate_proj_weight_cast_fp16")]; tensor layers_21_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741190400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753785728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753773376))))[name = string("layers_21_mlp_up_proj_weight_cast_fp16")]; tensor layers_21_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753791936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766379072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766374912))))[name = string("layers_21_mlp_down_proj_weight_cast_fp16")]; tensor layers_22_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766381184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770579712))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770575552))))[name = string("layers_22_self_attn_q_proj_weight_cast_fp16")]; tensor layers_22_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770581824))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106176))))[name = string("layers_22_self_attn_k_proj_weight_cast_fp16")]; tensor layers_22_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771107072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783702400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783690048))))[name = string("layers_22_mlp_gate_proj_weight_cast_fp16")]; tensor layers_22_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783708608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796303936))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796291584))))[name = string("layers_22_mlp_up_proj_weight_cast_fp16")]; tensor layers_22_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796310144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808897280))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808893120))))[name = string("layers_22_mlp_down_proj_weight_cast_fp16")]; tensor layers_23_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808899392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813097920))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813093760))))[name = string("layers_23_self_attn_q_proj_weight_cast_fp16")]; tensor layers_23_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813100032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624960))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624384))))[name = string("layers_23_self_attn_k_proj_weight_cast_fp16")]; tensor layers_23_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813625280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817823808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817819648))))[name = string("layers_23_self_attn_o_proj_weight_cast_fp16")]; tensor layers_23_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817825920))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830421248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830408896))))[name = string("layers_23_mlp_gate_proj_weight_cast_fp16")]; tensor layers_23_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830427456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843022784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843010432))))[name = string("layers_23_mlp_up_proj_weight_cast_fp16")]; tensor layers_23_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843028992))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855616128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855611968))))[name = string("layers_23_mlp_down_proj_weight_cast_fp16")]; tensor layers_24_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855618240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859816768))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859812608))))[name = string("layers_24_self_attn_q_proj_weight_cast_fp16")]; tensor layers_24_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859818880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343232))))[name = string("layers_24_self_attn_k_proj_weight_cast_fp16")]; tensor layers_24_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860344128))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864542656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864538496))))[name = string("layers_24_self_attn_o_proj_weight_cast_fp16")]; tensor layers_24_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864544768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877140096))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877127744))))[name = string("layers_24_mlp_gate_proj_weight_cast_fp16")]; tensor layers_24_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877146304))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889741632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889729280))))[name = string("layers_24_mlp_up_proj_weight_cast_fp16")]; tensor layers_24_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889747840))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902334976))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902330816))))[name = string("layers_24_mlp_down_proj_weight_cast_fp16")]; tensor layers_25_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902337088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906535616))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906531456))))[name = string("layers_25_self_attn_q_proj_weight_cast_fp16")]; tensor layers_25_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906537728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062080))))[name = string("layers_25_self_attn_k_proj_weight_cast_fp16")]; tensor layers_25_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911261504))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911257344))))[name = string("layers_25_self_attn_o_proj_weight_cast_fp16")]; tensor layers_25_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911263616))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923858944))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923846592))))[name = string("layers_25_mlp_gate_proj_weight_cast_fp16")]; tensor layers_25_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923865152))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936460480))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936448128))))[name = string("layers_25_mlp_up_proj_weight_cast_fp16")]; tensor layers_26_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936466688))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940665216))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940661056))))[name = string("layers_26_self_attn_q_proj_weight_cast_fp16")]; tensor layers_26_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940667328))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192256))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941191680))))[name = string("layers_26_self_attn_k_proj_weight_cast_fp16")]; tensor layers_26_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192576))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945391104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945386944))))[name = string("layers_26_self_attn_o_proj_weight_cast_fp16")]; tensor layers_26_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945393216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957988544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957976192))))[name = string("layers_26_mlp_gate_proj_weight_cast_fp16")]; int32 var_765 = const()[name = string("op_765"), val = int32(0)]; tensor var_766 = mul(x = position_index_seed, y = var_765)[name = string("op_766")]; int32 var_768 = const()[name = string("op_768"), val = int32(1)]; tensor ones = add(x = var_766, y = var_768)[name = string("ones")]; int32 var_770 = const()[name = string("op_770"), val = int32(0)]; bool var_772_exclusive_0 = const()[name = string("op_772_exclusive_0"), val = bool(false)]; bool var_772_reverse_0 = const()[name = string("op_772_reverse_0"), val = bool(false)]; tensor var_772 = cumsum(axis = var_770, exclusive = var_772_exclusive_0, reverse = var_772_reverse_0, x = ones)[name = string("op_772")]; int32 var_774 = const()[name = string("op_774"), val = int32(1)]; tensor position_offsets = sub(x = var_772, y = var_774)[name = string("position_offsets")]; tensor position_ids_1 = add(x = position_offsets, y = position_id)[name = string("position_ids_1")]; bool var_784_keep_dims_0 = const()[name = string("op_784_keep_dims_0"), val = bool(false)]; int32 var_784 = reduce_sum(keep_dims = var_784_keep_dims_0, x = ones)[name = string("op_784")]; int32 var_786 = const()[name = string("op_786"), val = int32(1)]; int32 offset = sub(x = var_784, y = var_786)[name = string("offset")]; tensor var_789 = add(x = position_id, y = offset)[name = string("op_789")]; int32 var_791 = const()[name = string("op_791"), val = int32(1)]; tensor cache_position_end = add(x = var_789, y = var_791)[name = string("cache_position_end")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor greater_equal_0 = greater_equal(x = position_ids_1, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(32768)]; tensor add_0 = add(x = position_ids_1, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = position_ids_1, b = add_0, cond = greater_equal_0)[name = string("select_0")]; tensor rope_emb_cos_cached_to_fp16 = const()[name = string("rope_emb_cos_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957994752)))]; int32 cos_1_batch_dims_0 = const()[name = string("cos_1_batch_dims_0"), val = int32(0)]; bool cos_1_validate_indices_0 = const()[name = string("cos_1_validate_indices_0"), val = bool(false)]; int32 greater_equal_10_y_0 = const()[name = string("greater_equal_10_y_0"), val = int32(0)]; tensor greater_equal_10 = greater_equal(x = select_0, y = greater_equal_10_y_0)[name = string("greater_equal_10")]; int32 slice_by_index_10 = const()[name = string("slice_by_index_10"), val = int32(32768)]; tensor add_10 = add(x = select_0, y = slice_by_index_10)[name = string("add_10")]; tensor select_10 = select(a = select_0, b = add_10, cond = greater_equal_10)[name = string("select_10")]; int32 cos_1_cast_fp16_axis_5 = const()[name = string("cos_1_cast_fp16_axis_5"), val = int32(0)]; tensor cos_1_cast_fp16 = gather(axis = cos_1_cast_fp16_axis_5, batch_dims = cos_1_batch_dims_0, indices = select_10, validate_indices = cos_1_validate_indices_0, x = rope_emb_cos_cached_to_fp16)[name = string("cos_1_cast_fp16")]; tensor rope_emb_sin_cached_to_fp16 = const()[name = string("rope_emb_sin_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(966383424)))]; int32 sin_1_batch_dims_0 = const()[name = string("sin_1_batch_dims_0"), val = int32(0)]; bool sin_1_validate_indices_0 = const()[name = string("sin_1_validate_indices_0"), val = bool(false)]; int32 sin_1_cast_fp16_axis_5 = const()[name = string("sin_1_cast_fp16_axis_5"), val = int32(0)]; tensor sin_1_cast_fp16 = gather(axis = sin_1_cast_fp16_axis_5, batch_dims = sin_1_batch_dims_0, indices = select_10, validate_indices = sin_1_validate_indices_0, x = rope_emb_sin_cached_to_fp16)[name = string("sin_1_cast_fp16")]; tensor var_865_perm_0 = const()[name = string("op_865_perm_0"), val = tensor([-1, -2])]; tensor var_867_axes_0 = const()[name = string("op_867_axes_0"), val = tensor([0])]; tensor var_865_cast_fp16 = transpose(perm = var_865_perm_0, x = cos_1_cast_fp16)[name = string("transpose_515")]; tensor var_867_cast_fp16 = expand_dims(axes = var_867_axes_0, x = var_865_cast_fp16)[name = string("op_867_cast_fp16")]; tensor var_869_axes_0 = const()[name = string("op_869_axes_0"), val = tensor([0])]; tensor var_869_cast_fp16 = expand_dims(axes = var_869_axes_0, x = var_867_cast_fp16)[name = string("op_869_cast_fp16")]; tensor var_874_perm_0 = const()[name = string("op_874_perm_0"), val = tensor([-1, -2])]; tensor var_876_axes_0 = const()[name = string("op_876_axes_0"), val = tensor([0])]; tensor var_874_cast_fp16 = transpose(perm = var_874_perm_0, x = sin_1_cast_fp16)[name = string("transpose_514")]; tensor var_876_cast_fp16 = expand_dims(axes = var_876_axes_0, x = var_874_cast_fp16)[name = string("op_876_cast_fp16")]; tensor var_878_axes_0 = const()[name = string("op_878_axes_0"), val = tensor([0])]; tensor var_878_cast_fp16 = expand_dims(axes = var_878_axes_0, x = var_876_cast_fp16)[name = string("op_878_cast_fp16")]; string position_ids_1_to_uint16_dtype_0 = const()[name = string("position_ids_1_to_uint16_dtype_0"), val = string("uint16")]; tensor causal_mask = const()[name = string("causal_mask"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(974772096)))]; int32 mask_axis_0 = const()[name = string("mask_axis_0"), val = int32(1)]; int32 mask_batch_dims_0 = const()[name = string("mask_batch_dims_0"), val = int32(0)]; bool mask_validate_indices_0 = const()[name = string("mask_validate_indices_0"), val = bool(false)]; tensor position_ids_1_to_uint16 = cast(dtype = position_ids_1_to_uint16_dtype_0, x = position_ids_1)[name = string("cast_11")]; tensor mask_cast_uint16 = gather(axis = mask_axis_0, batch_dims = mask_batch_dims_0, indices = position_ids_1_to_uint16, validate_indices = mask_validate_indices_0, x = causal_mask)[name = string("mask_cast_uint16")]; tensor var_895_axes_0 = const()[name = string("op_895_axes_0"), val = tensor([0])]; tensor var_895 = expand_dims(axes = var_895_axes_0, x = mask_cast_uint16)[name = string("op_895")]; tensor attn_mask_1_axes_0 = const()[name = string("attn_mask_1_axes_0"), val = tensor([0])]; tensor attn_mask_1 = expand_dims(axes = attn_mask_1_axes_0, x = var_895)[name = string("attn_mask_1")]; string inputs_embeds_to_fp16_dtype_0 = const()[name = string("inputs_embeds_to_fp16_dtype_0"), val = string("fp16")]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor inputs_embeds_to_fp16 = cast(dtype = inputs_embeds_to_fp16_dtype_0, x = inputs_embeds)[name = string("cast_10")]; tensor var_906_cast_fp16 = mul(x = inputs_embeds_to_fp16, y = const_0_promoted_to_fp16)[name = string("op_906_cast_fp16")]; int32 var_904 = const()[name = string("op_904"), val = int32(1)]; bool doubled_1_interleave_0 = const()[name = string("doubled_1_interleave_0"), val = bool(false)]; tensor doubled_1_cast_fp16 = concat(axis = var_904, interleave = doubled_1_interleave_0, values = (inputs_embeds_to_fp16, var_906_cast_fp16))[name = string("doubled_1_cast_fp16")]; tensor out_1_axes_0 = const()[name = string("out_1_axes_0"), val = tensor([1])]; tensor out_1_gamma_0_to_fp16 = const()[name = string("out_1_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983160768)))]; fp16 var_916_to_fp16 = const()[name = string("op_916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_1_cast_fp16 = layer_norm(axes = out_1_axes_0, epsilon = var_916_to_fp16, gamma = out_1_gamma_0_to_fp16, x = doubled_1_cast_fp16)[name = string("out_1_cast_fp16")]; tensor var_927_split_sizes_0 = const()[name = string("op_927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_927_axis_0 = const()[name = string("op_927_axis_0"), val = int32(1)]; tensor var_927_cast_fp16_0, tensor var_927_cast_fp16_1 = split(axis = var_927_axis_0, split_sizes = var_927_split_sizes_0, x = out_1_cast_fp16)[name = string("op_927_cast_fp16")]; tensor layers_0_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983169024)))]; tensor query_states_1_strides_0 = const()[name = string("query_states_1_strides_0"), val = tensor([1, 1])]; string query_states_1_pad_type_0 = const()[name = string("query_states_1_pad_type_0"), val = string("valid")]; tensor query_states_1_pad_0 = const()[name = string("query_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_1_dilations_0 = const()[name = string("query_states_1_dilations_0"), val = tensor([1, 1])]; int32 query_states_1_groups_0 = const()[name = string("query_states_1_groups_0"), val = int32(1)]; tensor query_states_1_cast_fp16 = conv(dilations = query_states_1_dilations_0, groups = query_states_1_groups_0, pad = query_states_1_pad_0, pad_type = query_states_1_pad_type_0, strides = query_states_1_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("query_states_1_cast_fp16")]; tensor layers_0_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(991557696)))]; tensor key_states_1_strides_0 = const()[name = string("key_states_1_strides_0"), val = tensor([1, 1])]; string key_states_1_pad_type_0 = const()[name = string("key_states_1_pad_type_0"), val = string("valid")]; tensor key_states_1_pad_0 = const()[name = string("key_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_1_dilations_0 = const()[name = string("key_states_1_dilations_0"), val = tensor([1, 1])]; int32 key_states_1_groups_0 = const()[name = string("key_states_1_groups_0"), val = int32(1)]; tensor key_states_1_cast_fp16 = conv(dilations = key_states_1_dilations_0, groups = key_states_1_groups_0, pad = key_states_1_pad_0, pad_type = key_states_1_pad_type_0, strides = key_states_1_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("key_states_1_cast_fp16")]; tensor layers_0_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(992606336)))]; tensor value_states_1_strides_0 = const()[name = string("value_states_1_strides_0"), val = tensor([1, 1])]; string value_states_1_pad_type_0 = const()[name = string("value_states_1_pad_type_0"), val = string("valid")]; tensor value_states_1_pad_0 = const()[name = string("value_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_1_dilations_0 = const()[name = string("value_states_1_dilations_0"), val = tensor([1, 1])]; int32 value_states_1_groups_0 = const()[name = string("value_states_1_groups_0"), val = int32(1)]; tensor value_states_1_cast_fp16 = conv(dilations = value_states_1_dilations_0, groups = value_states_1_groups_0, pad = value_states_1_pad_0, pad_type = value_states_1_pad_type_0, strides = value_states_1_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("value_states_1_cast_fp16")]; tensor concat_0x = const()[name = string("concat_0x"), val = tensor([1, 16, 128, -1])]; tensor x_1_cast_fp16 = reshape(shape = concat_0x, x = query_states_1_cast_fp16)[name = string("x_1_cast_fp16")]; tensor concat_1x = const()[name = string("concat_1x"), val = tensor([1, 2, 128, -1])]; tensor var_984_cast_fp16 = reshape(shape = concat_1x, x = key_states_1_cast_fp16)[name = string("op_984_cast_fp16")]; tensor concat_2x = const()[name = string("concat_2x"), val = tensor([1, 2, 128, -1])]; tensor var_991_cast_fp16 = reshape(shape = concat_2x, x = value_states_1_cast_fp16)[name = string("op_991_cast_fp16")]; tensor var_995_cast_fp16 = mul(x = x_1_cast_fp16, y = var_869_cast_fp16)[name = string("op_995_cast_fp16")]; tensor var_996_split_sizes_0 = const()[name = string("op_996_split_sizes_0"), val = tensor([64, 64])]; int32 var_996_axis_0 = const()[name = string("op_996_axis_0"), val = int32(-2)]; tensor var_996_cast_fp16_0, tensor var_996_cast_fp16_1 = split(axis = var_996_axis_0, split_sizes = var_996_split_sizes_0, x = x_1_cast_fp16)[name = string("op_996_cast_fp16")]; fp16 const_2_promoted_to_fp16 = const()[name = string("const_2_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_998_cast_fp16 = mul(x = var_996_cast_fp16_1, y = const_2_promoted_to_fp16)[name = string("op_998_cast_fp16")]; int32 var_1000 = const()[name = string("op_1000"), val = int32(-2)]; bool var_1001_interleave_0 = const()[name = string("op_1001_interleave_0"), val = bool(false)]; tensor var_1001_cast_fp16 = concat(axis = var_1000, interleave = var_1001_interleave_0, values = (var_998_cast_fp16, var_996_cast_fp16_0))[name = string("op_1001_cast_fp16")]; tensor var_1002_cast_fp16 = mul(x = var_1001_cast_fp16, y = var_878_cast_fp16)[name = string("op_1002_cast_fp16")]; tensor query_states_3_cast_fp16 = add(x = var_995_cast_fp16, y = var_1002_cast_fp16)[name = string("query_states_3_cast_fp16")]; tensor var_1008_cast_fp16 = mul(x = var_984_cast_fp16, y = var_869_cast_fp16)[name = string("op_1008_cast_fp16")]; tensor var_1009_split_sizes_0 = const()[name = string("op_1009_split_sizes_0"), val = tensor([64, 64])]; int32 var_1009_axis_0 = const()[name = string("op_1009_axis_0"), val = int32(-2)]; tensor var_1009_cast_fp16_0, tensor var_1009_cast_fp16_1 = split(axis = var_1009_axis_0, split_sizes = var_1009_split_sizes_0, x = var_984_cast_fp16)[name = string("op_1009_cast_fp16")]; fp16 const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1011_cast_fp16 = mul(x = var_1009_cast_fp16_1, y = const_3_promoted_to_fp16)[name = string("op_1011_cast_fp16")]; int32 var_1013 = const()[name = string("op_1013"), val = int32(-2)]; bool var_1014_interleave_0 = const()[name = string("op_1014_interleave_0"), val = bool(false)]; tensor var_1014_cast_fp16 = concat(axis = var_1013, interleave = var_1014_interleave_0, values = (var_1011_cast_fp16, var_1009_cast_fp16_0))[name = string("op_1014_cast_fp16")]; tensor var_1015_cast_fp16 = mul(x = var_1014_cast_fp16, y = var_878_cast_fp16)[name = string("op_1015_cast_fp16")]; tensor key_states_5_cast_fp16 = add(x = var_1008_cast_fp16, y = var_1015_cast_fp16)[name = string("key_states_5_cast_fp16")]; tensor read_state_0 = read_state(input = key_cache)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; int32 concat_5_axis_0 = const()[name = string("concat_5_axis_0"), val = int32(0)]; bool concat_5_interleave_0 = const()[name = string("concat_5_interleave_0"), val = bool(false)]; tensor concat_5 = concat(axis = concat_5_axis_0, interleave = concat_5_interleave_0, values = (expand_dims_0, expand_dims_1, position_id, expand_dims_3))[name = string("concat_5")]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; tensor concat_6_values1_0 = const()[name = string("concat_6_values1_0"), val = tensor([0])]; tensor concat_6_values3_0 = const()[name = string("concat_6_values3_0"), val = tensor([0])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_4, concat_6_values1_0, cache_position_end, concat_6_values3_0))[name = string("concat_6")]; tensor key_states_7_perm_0 = const()[name = string("key_states_7_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_1_stride_0 = const()[name = string("key_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_7_cast_fp16 = transpose(perm = key_states_7_perm_0, x = key_states_5_cast_fp16)[name = string("transpose_513")]; tensor key_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = key_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = key_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_1_squeeze_mask_0, stride = key_cache_internal_tensor_assign_1_stride_0, update = key_states_7_cast_fp16, x = read_state_0)[name = string("key_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_1_cast_fp16, input = key_cache)[name = string("coreml_update_state_280_write_state")]; tensor coreml_update_state_280 = read_state(input = key_cache)[name = string("coreml_update_state_280")]; tensor read_state_1 = read_state(input = value_cache)[name = string("read_state_1")]; tensor value_states_3_perm_0 = const()[name = string("value_states_3_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_1_stride_0 = const()[name = string("value_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_3_cast_fp16 = transpose(perm = value_states_3_perm_0, x = var_991_cast_fp16)[name = string("transpose_512")]; tensor value_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = value_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = value_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_1_squeeze_mask_0, stride = value_cache_internal_tensor_assign_1_stride_0, update = value_states_3_cast_fp16, x = read_state_1)[name = string("value_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_1_cast_fp16, input = value_cache)[name = string("coreml_update_state_281_write_state")]; tensor coreml_update_state_281 = read_state(input = value_cache)[name = string("coreml_update_state_281")]; tensor var_1085_begin_0 = const()[name = string("op_1085_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1085_end_0 = const()[name = string("op_1085_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1085_end_mask_0 = const()[name = string("op_1085_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1085_cast_fp16 = slice_by_index(begin = var_1085_begin_0, end = var_1085_end_0, end_mask = var_1085_end_mask_0, x = coreml_update_state_280)[name = string("op_1085_cast_fp16")]; tensor tile_0 = const()[name = string("tile_0"), val = tensor([1, 1])]; int32 var_1088_axis_0 = const()[name = string("op_1088_axis_0"), val = int32(1)]; tensor var_1088_cast_fp16_0, tensor var_1088_cast_fp16_1 = split(axis = var_1088_axis_0, split_sizes = tile_0, x = var_1085_cast_fp16)[name = string("op_1088_cast_fp16")]; tensor var_1095_begin_0 = const()[name = string("op_1095_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1095_end_0 = const()[name = string("op_1095_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1095_end_mask_0 = const()[name = string("op_1095_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1095_cast_fp16 = slice_by_index(begin = var_1095_begin_0, end = var_1095_end_0, end_mask = var_1095_end_mask_0, x = coreml_update_state_281)[name = string("op_1095_cast_fp16")]; tensor tile_1 = const()[name = string("tile_1"), val = tensor([1, 1])]; int32 var_1098_axis_0 = const()[name = string("op_1098_axis_0"), val = int32(1)]; tensor var_1098_cast_fp16_0, tensor var_1098_cast_fp16_1 = split(axis = var_1098_axis_0, split_sizes = tile_1, x = var_1095_cast_fp16)[name = string("op_1098_cast_fp16")]; tensor var_1101_split_sizes_0 = const()[name = string("op_1101_split_sizes_0"), val = tensor([8, 8])]; int32 var_1101_axis_0 = const()[name = string("op_1101_axis_0"), val = int32(1)]; tensor var_1101_0, tensor var_1101_1 = split(axis = var_1101_axis_0, split_sizes = var_1101_split_sizes_0, x = query_states_3_cast_fp16)[name = string("op_1101")]; bool attn_weights_1_transpose_x_0 = const()[name = string("attn_weights_1_transpose_x_0"), val = bool(false)]; bool attn_weights_1_transpose_y_0 = const()[name = string("attn_weights_1_transpose_y_0"), val = bool(false)]; tensor attn_weights_1_cast_fp16 = matmul(transpose_x = attn_weights_1_transpose_x_0, transpose_y = attn_weights_1_transpose_y_0, x = var_1088_cast_fp16_0, y = var_1101_0)[name = string("attn_weights_1_cast_fp16")]; fp16 var_1104_to_fp16 = const()[name = string("op_1104_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_3_cast_fp16 = mul(x = attn_weights_1_cast_fp16, y = var_1104_to_fp16)[name = string("attn_weights_3_cast_fp16")]; tensor attn_weights_5_cast_fp16 = add(x = attn_weights_3_cast_fp16, y = attn_mask_1)[name = string("attn_weights_5_cast_fp16")]; int32 var_1108 = const()[name = string("op_1108"), val = int32(-2)]; tensor attn_weights_7_cast_fp16 = softmax(axis = var_1108, x = attn_weights_5_cast_fp16)[name = string("attn_weights_7_cast_fp16")]; bool var_1114_transpose_x_1 = const()[name = string("op_1114_transpose_x_1"), val = bool(true)]; bool var_1114_transpose_y_1 = const()[name = string("op_1114_transpose_y_1"), val = bool(false)]; tensor var_1114_cast_fp16 = matmul(transpose_x = var_1114_transpose_x_1, transpose_y = var_1114_transpose_y_1, x = attn_weights_7_cast_fp16, y = var_1098_cast_fp16_0)[name = string("op_1114_cast_fp16")]; bool attn_weights_9_transpose_x_0 = const()[name = string("attn_weights_9_transpose_x_0"), val = bool(false)]; bool attn_weights_9_transpose_y_0 = const()[name = string("attn_weights_9_transpose_y_0"), val = bool(false)]; tensor attn_weights_9_cast_fp16 = matmul(transpose_x = attn_weights_9_transpose_x_0, transpose_y = attn_weights_9_transpose_y_0, x = var_1088_cast_fp16_1, y = var_1101_1)[name = string("attn_weights_9_cast_fp16")]; fp16 var_1116_to_fp16 = const()[name = string("op_1116_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_11_cast_fp16 = mul(x = attn_weights_9_cast_fp16, y = var_1116_to_fp16)[name = string("attn_weights_11_cast_fp16")]; tensor attn_weights_13_cast_fp16 = add(x = attn_weights_11_cast_fp16, y = attn_mask_1)[name = string("attn_weights_13_cast_fp16")]; int32 var_1120 = const()[name = string("op_1120"), val = int32(-2)]; tensor attn_weights_15_cast_fp16 = softmax(axis = var_1120, x = attn_weights_13_cast_fp16)[name = string("attn_weights_15_cast_fp16")]; bool attn_output_1_transpose_x_1 = const()[name = string("attn_output_1_transpose_x_1"), val = bool(true)]; bool attn_output_1_transpose_y_1 = const()[name = string("attn_output_1_transpose_y_1"), val = bool(false)]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_1, transpose_y = attn_output_1_transpose_y_1, x = attn_weights_15_cast_fp16, y = var_1098_cast_fp16_1)[name = string("attn_output_1_cast_fp16")]; int32 var_1128 = const()[name = string("op_1128"), val = int32(1)]; bool attn_output_3_interleave_0 = const()[name = string("attn_output_3_interleave_0"), val = bool(false)]; tensor attn_output_3_cast_fp16 = concat(axis = var_1128, interleave = attn_output_3_interleave_0, values = (var_1114_cast_fp16, attn_output_1_cast_fp16))[name = string("attn_output_3_cast_fp16")]; tensor var_1132_perm_0 = const()[name = string("op_1132_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_11x = const()[name = string("concat_11x"), val = tensor([1, 2048, 1, -1])]; tensor var_1132_cast_fp16 = transpose(perm = var_1132_perm_0, x = attn_output_3_cast_fp16)[name = string("transpose_511")]; tensor attn_output_7_cast_fp16 = reshape(shape = concat_11x, x = var_1132_cast_fp16)[name = string("attn_output_7_cast_fp16")]; tensor layers_0_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(993654976)))]; tensor hidden_states_3_strides_0 = const()[name = string("hidden_states_3_strides_0"), val = tensor([1, 1])]; string hidden_states_3_pad_type_0 = const()[name = string("hidden_states_3_pad_type_0"), val = string("valid")]; tensor hidden_states_3_pad_0 = const()[name = string("hidden_states_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_3_dilations_0 = const()[name = string("hidden_states_3_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_3_groups_0 = const()[name = string("hidden_states_3_groups_0"), val = int32(1)]; tensor hidden_states_3_cast_fp16 = conv(dilations = hidden_states_3_dilations_0, groups = hidden_states_3_groups_0, pad = hidden_states_3_pad_0, pad_type = hidden_states_3_pad_type_0, strides = hidden_states_3_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16, x = attn_output_7_cast_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor hidden_states_5_cast_fp16 = add(x = inputs_embeds_to_fp16, y = hidden_states_3_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1165_cast_fp16 = mul(x = hidden_states_5_cast_fp16, y = const_8_promoted_to_fp16)[name = string("op_1165_cast_fp16")]; int32 var_1163 = const()[name = string("op_1163"), val = int32(1)]; bool doubled_5_interleave_0 = const()[name = string("doubled_5_interleave_0"), val = bool(false)]; tensor doubled_5_cast_fp16 = concat(axis = var_1163, interleave = doubled_5_interleave_0, values = (hidden_states_5_cast_fp16, var_1165_cast_fp16))[name = string("doubled_5_cast_fp16")]; tensor out_3_axes_0 = const()[name = string("out_3_axes_0"), val = tensor([1])]; tensor out_3_gamma_0_to_fp16 = const()[name = string("out_3_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002043648)))]; fp16 var_1175_to_fp16 = const()[name = string("op_1175_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_3_cast_fp16 = layer_norm(axes = out_3_axes_0, epsilon = var_1175_to_fp16, gamma = out_3_gamma_0_to_fp16, x = doubled_5_cast_fp16)[name = string("out_3_cast_fp16")]; tensor var_1186_split_sizes_0 = const()[name = string("op_1186_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1186_axis_0 = const()[name = string("op_1186_axis_0"), val = int32(1)]; tensor var_1186_cast_fp16_0, tensor var_1186_cast_fp16_1 = split(axis = var_1186_axis_0, split_sizes = var_1186_split_sizes_0, x = out_3_cast_fp16)[name = string("op_1186_cast_fp16")]; tensor layers_0_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002051904)))]; tensor input_1_strides_0 = const()[name = string("input_1_strides_0"), val = tensor([1, 1])]; string input_1_pad_type_0 = const()[name = string("input_1_pad_type_0"), val = string("valid")]; tensor input_1_pad_0 = const()[name = string("input_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_1_dilations_0 = const()[name = string("input_1_dilations_0"), val = tensor([1, 1])]; int32 input_1_groups_0 = const()[name = string("input_1_groups_0"), val = int32(1)]; tensor input_1_cast_fp16 = conv(dilations = input_1_dilations_0, groups = input_1_groups_0, pad = input_1_pad_0, pad_type = input_1_pad_type_0, strides = input_1_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("input_1_cast_fp16")]; tensor var_1203_cast_fp16 = silu(x = input_1_cast_fp16)[name = string("op_1203_cast_fp16")]; tensor layers_0_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1027217792)))]; tensor var_1209_strides_0 = const()[name = string("op_1209_strides_0"), val = tensor([1, 1])]; string var_1209_pad_type_0 = const()[name = string("op_1209_pad_type_0"), val = string("valid")]; tensor var_1209_pad_0 = const()[name = string("op_1209_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1209_dilations_0 = const()[name = string("op_1209_dilations_0"), val = tensor([1, 1])]; int32 var_1209_groups_0 = const()[name = string("op_1209_groups_0"), val = int32(1)]; tensor var_1209_cast_fp16 = conv(dilations = var_1209_dilations_0, groups = var_1209_groups_0, pad = var_1209_pad_0, pad_type = var_1209_pad_type_0, strides = var_1209_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("op_1209_cast_fp16")]; tensor x_9_cast_fp16 = mul(x = var_1203_cast_fp16, y = var_1209_cast_fp16)[name = string("x_9_cast_fp16")]; tensor layers_0_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1052383680)))]; tensor hidden_states_7_strides_0 = const()[name = string("hidden_states_7_strides_0"), val = tensor([1, 1])]; string hidden_states_7_pad_type_0 = const()[name = string("hidden_states_7_pad_type_0"), val = string("valid")]; tensor hidden_states_7_pad_0 = const()[name = string("hidden_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_7_dilations_0 = const()[name = string("hidden_states_7_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_7_groups_0 = const()[name = string("hidden_states_7_groups_0"), val = int32(1)]; tensor hidden_states_7_cast_fp16 = conv(dilations = hidden_states_7_dilations_0, groups = hidden_states_7_groups_0, pad = hidden_states_7_pad_0, pad_type = hidden_states_7_pad_type_0, strides = hidden_states_7_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16, x = x_9_cast_fp16)[name = string("hidden_states_7_cast_fp16")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = hidden_states_7_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; fp16 const_10_promoted_to_fp16 = const()[name = string("const_10_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1227_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_10_promoted_to_fp16)[name = string("op_1227_cast_fp16")]; int32 var_1225 = const()[name = string("op_1225"), val = int32(1)]; bool doubled_9_interleave_0 = const()[name = string("doubled_9_interleave_0"), val = bool(false)]; tensor doubled_9_cast_fp16 = concat(axis = var_1225, interleave = doubled_9_interleave_0, values = (hidden_states_9_cast_fp16, var_1227_cast_fp16))[name = string("doubled_9_cast_fp16")]; tensor out_5_axes_0 = const()[name = string("out_5_axes_0"), val = tensor([1])]; tensor out_5_gamma_0_to_fp16 = const()[name = string("out_5_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077549568)))]; fp16 var_1237_to_fp16 = const()[name = string("op_1237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_5_cast_fp16 = layer_norm(axes = out_5_axes_0, epsilon = var_1237_to_fp16, gamma = out_5_gamma_0_to_fp16, x = doubled_9_cast_fp16)[name = string("out_5_cast_fp16")]; tensor var_1248_split_sizes_0 = const()[name = string("op_1248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1248_axis_0 = const()[name = string("op_1248_axis_0"), val = int32(1)]; tensor var_1248_cast_fp16_0, tensor var_1248_cast_fp16_1 = split(axis = var_1248_axis_0, split_sizes = var_1248_split_sizes_0, x = out_5_cast_fp16)[name = string("op_1248_cast_fp16")]; tensor layers_1_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077557824)))]; tensor query_states_7_strides_0 = const()[name = string("query_states_7_strides_0"), val = tensor([1, 1])]; string query_states_7_pad_type_0 = const()[name = string("query_states_7_pad_type_0"), val = string("valid")]; tensor query_states_7_pad_0 = const()[name = string("query_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_7_dilations_0 = const()[name = string("query_states_7_dilations_0"), val = tensor([1, 1])]; int32 query_states_7_groups_0 = const()[name = string("query_states_7_groups_0"), val = int32(1)]; tensor query_states_7_cast_fp16 = conv(dilations = query_states_7_dilations_0, groups = query_states_7_groups_0, pad = query_states_7_pad_0, pad_type = query_states_7_pad_type_0, strides = query_states_7_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("query_states_7_cast_fp16")]; tensor layers_1_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1085946496)))]; tensor key_states_11_strides_0 = const()[name = string("key_states_11_strides_0"), val = tensor([1, 1])]; string key_states_11_pad_type_0 = const()[name = string("key_states_11_pad_type_0"), val = string("valid")]; tensor key_states_11_pad_0 = const()[name = string("key_states_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_11_dilations_0 = const()[name = string("key_states_11_dilations_0"), val = tensor([1, 1])]; int32 key_states_11_groups_0 = const()[name = string("key_states_11_groups_0"), val = int32(1)]; tensor key_states_11_cast_fp16 = conv(dilations = key_states_11_dilations_0, groups = key_states_11_groups_0, pad = key_states_11_pad_0, pad_type = key_states_11_pad_type_0, strides = key_states_11_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("key_states_11_cast_fp16")]; tensor value_states_7_strides_0 = const()[name = string("value_states_7_strides_0"), val = tensor([1, 1])]; string value_states_7_pad_type_0 = const()[name = string("value_states_7_pad_type_0"), val = string("valid")]; tensor value_states_7_pad_0 = const()[name = string("value_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_7_dilations_0 = const()[name = string("value_states_7_dilations_0"), val = tensor([1, 1])]; int32 value_states_7_groups_0 = const()[name = string("value_states_7_groups_0"), val = int32(1)]; tensor value_states_7_cast_fp16 = conv(dilations = value_states_7_dilations_0, groups = value_states_7_groups_0, pad = value_states_7_pad_0, pad_type = value_states_7_pad_type_0, strides = value_states_7_strides_0, weight = layers_1_self_attn_v_proj_weight_cast_fp16, x = var_1248_cast_fp16_0)[name = string("value_states_7_cast_fp16")]; tensor concat_12x = const()[name = string("concat_12x"), val = tensor([1, 16, 128, -1])]; tensor x_11_cast_fp16 = reshape(shape = concat_12x, x = query_states_7_cast_fp16)[name = string("x_11_cast_fp16")]; tensor concat_13x = const()[name = string("concat_13x"), val = tensor([1, 2, 128, -1])]; tensor var_1305_cast_fp16 = reshape(shape = concat_13x, x = key_states_11_cast_fp16)[name = string("op_1305_cast_fp16")]; tensor concat_14x = const()[name = string("concat_14x"), val = tensor([1, 2, 128, -1])]; tensor var_1312_cast_fp16 = reshape(shape = concat_14x, x = value_states_7_cast_fp16)[name = string("op_1312_cast_fp16")]; tensor var_1316_cast_fp16 = mul(x = x_11_cast_fp16, y = var_869_cast_fp16)[name = string("op_1316_cast_fp16")]; tensor var_1317_split_sizes_0 = const()[name = string("op_1317_split_sizes_0"), val = tensor([64, 64])]; int32 var_1317_axis_0 = const()[name = string("op_1317_axis_0"), val = int32(-2)]; tensor var_1317_cast_fp16_0, tensor var_1317_cast_fp16_1 = split(axis = var_1317_axis_0, split_sizes = var_1317_split_sizes_0, x = x_11_cast_fp16)[name = string("op_1317_cast_fp16")]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1319_cast_fp16 = mul(x = var_1317_cast_fp16_1, y = const_12_promoted_to_fp16)[name = string("op_1319_cast_fp16")]; int32 var_1321 = const()[name = string("op_1321"), val = int32(-2)]; bool var_1322_interleave_0 = const()[name = string("op_1322_interleave_0"), val = bool(false)]; tensor var_1322_cast_fp16 = concat(axis = var_1321, interleave = var_1322_interleave_0, values = (var_1319_cast_fp16, var_1317_cast_fp16_0))[name = string("op_1322_cast_fp16")]; tensor var_1323_cast_fp16 = mul(x = var_1322_cast_fp16, y = var_878_cast_fp16)[name = string("op_1323_cast_fp16")]; tensor query_states_9_cast_fp16 = add(x = var_1316_cast_fp16, y = var_1323_cast_fp16)[name = string("query_states_9_cast_fp16")]; tensor var_1329_cast_fp16 = mul(x = var_1305_cast_fp16, y = var_869_cast_fp16)[name = string("op_1329_cast_fp16")]; tensor var_1330_split_sizes_0 = const()[name = string("op_1330_split_sizes_0"), val = tensor([64, 64])]; int32 var_1330_axis_0 = const()[name = string("op_1330_axis_0"), val = int32(-2)]; tensor var_1330_cast_fp16_0, tensor var_1330_cast_fp16_1 = split(axis = var_1330_axis_0, split_sizes = var_1330_split_sizes_0, x = var_1305_cast_fp16)[name = string("op_1330_cast_fp16")]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1332_cast_fp16 = mul(x = var_1330_cast_fp16_1, y = const_13_promoted_to_fp16)[name = string("op_1332_cast_fp16")]; int32 var_1334 = const()[name = string("op_1334"), val = int32(-2)]; bool var_1335_interleave_0 = const()[name = string("op_1335_interleave_0"), val = bool(false)]; tensor var_1335_cast_fp16 = concat(axis = var_1334, interleave = var_1335_interleave_0, values = (var_1332_cast_fp16, var_1330_cast_fp16_0))[name = string("op_1335_cast_fp16")]; tensor var_1336_cast_fp16 = mul(x = var_1335_cast_fp16, y = var_878_cast_fp16)[name = string("op_1336_cast_fp16")]; tensor key_states_15_cast_fp16 = add(x = var_1329_cast_fp16, y = var_1336_cast_fp16)[name = string("key_states_15_cast_fp16")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; int32 concat_17_axis_0 = const()[name = string("concat_17_axis_0"), val = int32(0)]; bool concat_17_interleave_0 = const()[name = string("concat_17_interleave_0"), val = bool(false)]; tensor concat_17 = concat(axis = concat_17_axis_0, interleave = concat_17_interleave_0, values = (expand_dims_12, expand_dims_13, position_id, expand_dims_15))[name = string("concat_17")]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; tensor concat_18_values1_0 = const()[name = string("concat_18_values1_0"), val = tensor([0])]; tensor concat_18_values3_0 = const()[name = string("concat_18_values3_0"), val = tensor([0])]; int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_16, concat_18_values1_0, cache_position_end, concat_18_values3_0))[name = string("concat_18")]; tensor key_states_17_perm_0 = const()[name = string("key_states_17_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_2_stride_0 = const()[name = string("key_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_17_cast_fp16 = transpose(perm = key_states_17_perm_0, x = key_states_15_cast_fp16)[name = string("transpose_510")]; tensor key_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = key_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = key_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_2_squeeze_mask_0, stride = key_cache_internal_tensor_assign_2_stride_0, update = key_states_17_cast_fp16, x = coreml_update_state_280)[name = string("key_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_2_cast_fp16, input = key_cache)[name = string("coreml_update_state_282_write_state")]; tensor coreml_update_state_282 = read_state(input = key_cache)[name = string("coreml_update_state_282")]; tensor value_states_9_perm_0 = const()[name = string("value_states_9_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_2_stride_0 = const()[name = string("value_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_9_cast_fp16 = transpose(perm = value_states_9_perm_0, x = var_1312_cast_fp16)[name = string("transpose_509")]; tensor value_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = value_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = value_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_2_squeeze_mask_0, stride = value_cache_internal_tensor_assign_2_stride_0, update = value_states_9_cast_fp16, x = coreml_update_state_281)[name = string("value_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_2_cast_fp16, input = value_cache)[name = string("coreml_update_state_283_write_state")]; tensor coreml_update_state_283 = read_state(input = value_cache)[name = string("coreml_update_state_283")]; tensor var_1406_begin_0 = const()[name = string("op_1406_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1406_end_0 = const()[name = string("op_1406_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1406_end_mask_0 = const()[name = string("op_1406_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1406_cast_fp16 = slice_by_index(begin = var_1406_begin_0, end = var_1406_end_0, end_mask = var_1406_end_mask_0, x = coreml_update_state_282)[name = string("op_1406_cast_fp16")]; tensor tile_2 = const()[name = string("tile_2"), val = tensor([1, 1])]; int32 var_1409_axis_0 = const()[name = string("op_1409_axis_0"), val = int32(1)]; tensor var_1409_cast_fp16_0, tensor var_1409_cast_fp16_1 = split(axis = var_1409_axis_0, split_sizes = tile_2, x = var_1406_cast_fp16)[name = string("op_1409_cast_fp16")]; tensor var_1416_begin_0 = const()[name = string("op_1416_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1416_end_0 = const()[name = string("op_1416_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1416_end_mask_0 = const()[name = string("op_1416_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1416_cast_fp16 = slice_by_index(begin = var_1416_begin_0, end = var_1416_end_0, end_mask = var_1416_end_mask_0, x = coreml_update_state_283)[name = string("op_1416_cast_fp16")]; tensor tile_3 = const()[name = string("tile_3"), val = tensor([1, 1])]; int32 var_1419_axis_0 = const()[name = string("op_1419_axis_0"), val = int32(1)]; tensor var_1419_cast_fp16_0, tensor var_1419_cast_fp16_1 = split(axis = var_1419_axis_0, split_sizes = tile_3, x = var_1416_cast_fp16)[name = string("op_1419_cast_fp16")]; tensor var_1422_split_sizes_0 = const()[name = string("op_1422_split_sizes_0"), val = tensor([8, 8])]; int32 var_1422_axis_0 = const()[name = string("op_1422_axis_0"), val = int32(1)]; tensor var_1422_0, tensor var_1422_1 = split(axis = var_1422_axis_0, split_sizes = var_1422_split_sizes_0, x = query_states_9_cast_fp16)[name = string("op_1422")]; bool attn_weights_17_transpose_x_0 = const()[name = string("attn_weights_17_transpose_x_0"), val = bool(false)]; bool attn_weights_17_transpose_y_0 = const()[name = string("attn_weights_17_transpose_y_0"), val = bool(false)]; tensor attn_weights_17_cast_fp16 = matmul(transpose_x = attn_weights_17_transpose_x_0, transpose_y = attn_weights_17_transpose_y_0, x = var_1409_cast_fp16_0, y = var_1422_0)[name = string("attn_weights_17_cast_fp16")]; fp16 var_1425_to_fp16 = const()[name = string("op_1425_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_19_cast_fp16 = mul(x = attn_weights_17_cast_fp16, y = var_1425_to_fp16)[name = string("attn_weights_19_cast_fp16")]; tensor attn_weights_21_cast_fp16 = add(x = attn_weights_19_cast_fp16, y = attn_mask_1)[name = string("attn_weights_21_cast_fp16")]; int32 var_1429 = const()[name = string("op_1429"), val = int32(-2)]; tensor attn_weights_23_cast_fp16 = softmax(axis = var_1429, x = attn_weights_21_cast_fp16)[name = string("attn_weights_23_cast_fp16")]; bool var_1435_transpose_x_1 = const()[name = string("op_1435_transpose_x_1"), val = bool(true)]; bool var_1435_transpose_y_1 = const()[name = string("op_1435_transpose_y_1"), val = bool(false)]; tensor var_1435_cast_fp16 = matmul(transpose_x = var_1435_transpose_x_1, transpose_y = var_1435_transpose_y_1, x = attn_weights_23_cast_fp16, y = var_1419_cast_fp16_0)[name = string("op_1435_cast_fp16")]; bool attn_weights_25_transpose_x_0 = const()[name = string("attn_weights_25_transpose_x_0"), val = bool(false)]; bool attn_weights_25_transpose_y_0 = const()[name = string("attn_weights_25_transpose_y_0"), val = bool(false)]; tensor attn_weights_25_cast_fp16 = matmul(transpose_x = attn_weights_25_transpose_x_0, transpose_y = attn_weights_25_transpose_y_0, x = var_1409_cast_fp16_1, y = var_1422_1)[name = string("attn_weights_25_cast_fp16")]; fp16 var_1437_to_fp16 = const()[name = string("op_1437_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_27_cast_fp16 = mul(x = attn_weights_25_cast_fp16, y = var_1437_to_fp16)[name = string("attn_weights_27_cast_fp16")]; tensor attn_weights_29_cast_fp16 = add(x = attn_weights_27_cast_fp16, y = attn_mask_1)[name = string("attn_weights_29_cast_fp16")]; int32 var_1441 = const()[name = string("op_1441"), val = int32(-2)]; tensor attn_weights_31_cast_fp16 = softmax(axis = var_1441, x = attn_weights_29_cast_fp16)[name = string("attn_weights_31_cast_fp16")]; bool attn_output_9_transpose_x_1 = const()[name = string("attn_output_9_transpose_x_1"), val = bool(true)]; bool attn_output_9_transpose_y_1 = const()[name = string("attn_output_9_transpose_y_1"), val = bool(false)]; tensor attn_output_9_cast_fp16 = matmul(transpose_x = attn_output_9_transpose_x_1, transpose_y = attn_output_9_transpose_y_1, x = attn_weights_31_cast_fp16, y = var_1419_cast_fp16_1)[name = string("attn_output_9_cast_fp16")]; int32 var_1449 = const()[name = string("op_1449"), val = int32(1)]; bool attn_output_11_interleave_0 = const()[name = string("attn_output_11_interleave_0"), val = bool(false)]; tensor attn_output_11_cast_fp16 = concat(axis = var_1449, interleave = attn_output_11_interleave_0, values = (var_1435_cast_fp16, attn_output_9_cast_fp16))[name = string("attn_output_11_cast_fp16")]; tensor var_1453_perm_0 = const()[name = string("op_1453_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_23x = const()[name = string("concat_23x"), val = tensor([1, 2048, 1, -1])]; tensor var_1453_cast_fp16 = transpose(perm = var_1453_perm_0, x = attn_output_11_cast_fp16)[name = string("transpose_508")]; tensor attn_output_15_cast_fp16 = reshape(shape = concat_23x, x = var_1453_cast_fp16)[name = string("attn_output_15_cast_fp16")]; tensor layers_1_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086995136)))]; tensor hidden_states_13_strides_0 = const()[name = string("hidden_states_13_strides_0"), val = tensor([1, 1])]; string hidden_states_13_pad_type_0 = const()[name = string("hidden_states_13_pad_type_0"), val = string("valid")]; tensor hidden_states_13_pad_0 = const()[name = string("hidden_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_13_dilations_0 = const()[name = string("hidden_states_13_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_13_groups_0 = const()[name = string("hidden_states_13_groups_0"), val = int32(1)]; tensor hidden_states_13_cast_fp16 = conv(dilations = hidden_states_13_dilations_0, groups = hidden_states_13_groups_0, pad = hidden_states_13_pad_0, pad_type = hidden_states_13_pad_type_0, strides = hidden_states_13_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16, x = attn_output_15_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; tensor hidden_states_15_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = hidden_states_13_cast_fp16)[name = string("hidden_states_15_cast_fp16")]; fp16 const_18_promoted_to_fp16 = const()[name = string("const_18_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1486_cast_fp16 = mul(x = hidden_states_15_cast_fp16, y = const_18_promoted_to_fp16)[name = string("op_1486_cast_fp16")]; int32 var_1484 = const()[name = string("op_1484"), val = int32(1)]; bool doubled_13_interleave_0 = const()[name = string("doubled_13_interleave_0"), val = bool(false)]; tensor doubled_13_cast_fp16 = concat(axis = var_1484, interleave = doubled_13_interleave_0, values = (hidden_states_15_cast_fp16, var_1486_cast_fp16))[name = string("doubled_13_cast_fp16")]; tensor out_7_axes_0 = const()[name = string("out_7_axes_0"), val = tensor([1])]; tensor out_7_gamma_0_to_fp16 = const()[name = string("out_7_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095383808)))]; fp16 var_1496_to_fp16 = const()[name = string("op_1496_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_7_cast_fp16 = layer_norm(axes = out_7_axes_0, epsilon = var_1496_to_fp16, gamma = out_7_gamma_0_to_fp16, x = doubled_13_cast_fp16)[name = string("out_7_cast_fp16")]; tensor var_1507_split_sizes_0 = const()[name = string("op_1507_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1507_axis_0 = const()[name = string("op_1507_axis_0"), val = int32(1)]; tensor var_1507_cast_fp16_0, tensor var_1507_cast_fp16_1 = split(axis = var_1507_axis_0, split_sizes = var_1507_split_sizes_0, x = out_7_cast_fp16)[name = string("op_1507_cast_fp16")]; tensor layers_1_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095392064)))]; tensor input_3_strides_0 = const()[name = string("input_3_strides_0"), val = tensor([1, 1])]; string input_3_pad_type_0 = const()[name = string("input_3_pad_type_0"), val = string("valid")]; tensor input_3_pad_0 = const()[name = string("input_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_3_dilations_0 = const()[name = string("input_3_dilations_0"), val = tensor([1, 1])]; int32 input_3_groups_0 = const()[name = string("input_3_groups_0"), val = int32(1)]; tensor input_3_cast_fp16 = conv(dilations = input_3_dilations_0, groups = input_3_groups_0, pad = input_3_pad_0, pad_type = input_3_pad_type_0, strides = input_3_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16, x = var_1507_cast_fp16_0)[name = string("input_3_cast_fp16")]; tensor var_1524_cast_fp16 = silu(x = input_3_cast_fp16)[name = string("op_1524_cast_fp16")]; tensor var_1530_strides_0 = const()[name = string("op_1530_strides_0"), val = tensor([1, 1])]; string var_1530_pad_type_0 = const()[name = string("op_1530_pad_type_0"), val = string("valid")]; tensor var_1530_pad_0 = const()[name = string("op_1530_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1530_dilations_0 = const()[name = string("op_1530_dilations_0"), val = tensor([1, 1])]; int32 var_1530_groups_0 = const()[name = string("op_1530_groups_0"), val = int32(1)]; tensor var_1530_cast_fp16 = conv(dilations = var_1530_dilations_0, groups = var_1530_groups_0, pad = var_1530_pad_0, pad_type = var_1530_pad_type_0, strides = var_1530_strides_0, weight = layers_1_mlp_up_proj_weight_cast_fp16, x = var_1507_cast_fp16_0)[name = string("op_1530_cast_fp16")]; tensor x_19_cast_fp16 = mul(x = var_1524_cast_fp16, y = var_1530_cast_fp16)[name = string("x_19_cast_fp16")]; tensor layers_1_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1120557952)))]; tensor hidden_states_17_strides_0 = const()[name = string("hidden_states_17_strides_0"), val = tensor([1, 1])]; string hidden_states_17_pad_type_0 = const()[name = string("hidden_states_17_pad_type_0"), val = string("valid")]; tensor hidden_states_17_pad_0 = const()[name = string("hidden_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_17_dilations_0 = const()[name = string("hidden_states_17_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_17_groups_0 = const()[name = string("hidden_states_17_groups_0"), val = int32(1)]; tensor hidden_states_17_cast_fp16 = conv(dilations = hidden_states_17_dilations_0, groups = hidden_states_17_groups_0, pad = hidden_states_17_pad_0, pad_type = hidden_states_17_pad_type_0, strides = hidden_states_17_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16, x = x_19_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; tensor hidden_states_19_cast_fp16 = add(x = hidden_states_15_cast_fp16, y = hidden_states_17_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1548_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_1548_cast_fp16")]; int32 var_1546 = const()[name = string("op_1546"), val = int32(1)]; bool doubled_17_interleave_0 = const()[name = string("doubled_17_interleave_0"), val = bool(false)]; tensor doubled_17_cast_fp16 = concat(axis = var_1546, interleave = doubled_17_interleave_0, values = (hidden_states_19_cast_fp16, var_1548_cast_fp16))[name = string("doubled_17_cast_fp16")]; tensor out_9_axes_0 = const()[name = string("out_9_axes_0"), val = tensor([1])]; tensor out_9_gamma_0_to_fp16 = const()[name = string("out_9_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145723840)))]; fp16 var_1558_to_fp16 = const()[name = string("op_1558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_9_cast_fp16 = layer_norm(axes = out_9_axes_0, epsilon = var_1558_to_fp16, gamma = out_9_gamma_0_to_fp16, x = doubled_17_cast_fp16)[name = string("out_9_cast_fp16")]; tensor var_1569_split_sizes_0 = const()[name = string("op_1569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1569_axis_0 = const()[name = string("op_1569_axis_0"), val = int32(1)]; tensor var_1569_cast_fp16_0, tensor var_1569_cast_fp16_1 = split(axis = var_1569_axis_0, split_sizes = var_1569_split_sizes_0, x = out_9_cast_fp16)[name = string("op_1569_cast_fp16")]; tensor layers_2_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145732096)))]; tensor query_states_13_strides_0 = const()[name = string("query_states_13_strides_0"), val = tensor([1, 1])]; string query_states_13_pad_type_0 = const()[name = string("query_states_13_pad_type_0"), val = string("valid")]; tensor query_states_13_pad_0 = const()[name = string("query_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_13_dilations_0 = const()[name = string("query_states_13_dilations_0"), val = tensor([1, 1])]; int32 query_states_13_groups_0 = const()[name = string("query_states_13_groups_0"), val = int32(1)]; tensor query_states_13_cast_fp16 = conv(dilations = query_states_13_dilations_0, groups = query_states_13_groups_0, pad = query_states_13_pad_0, pad_type = query_states_13_pad_type_0, strides = query_states_13_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("query_states_13_cast_fp16")]; tensor layers_2_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1154120768)))]; tensor key_states_21_strides_0 = const()[name = string("key_states_21_strides_0"), val = tensor([1, 1])]; string key_states_21_pad_type_0 = const()[name = string("key_states_21_pad_type_0"), val = string("valid")]; tensor key_states_21_pad_0 = const()[name = string("key_states_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_21_dilations_0 = const()[name = string("key_states_21_dilations_0"), val = tensor([1, 1])]; int32 key_states_21_groups_0 = const()[name = string("key_states_21_groups_0"), val = int32(1)]; tensor key_states_21_cast_fp16 = conv(dilations = key_states_21_dilations_0, groups = key_states_21_groups_0, pad = key_states_21_pad_0, pad_type = key_states_21_pad_type_0, strides = key_states_21_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("key_states_21_cast_fp16")]; tensor value_states_13_strides_0 = const()[name = string("value_states_13_strides_0"), val = tensor([1, 1])]; string value_states_13_pad_type_0 = const()[name = string("value_states_13_pad_type_0"), val = string("valid")]; tensor value_states_13_pad_0 = const()[name = string("value_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_13_dilations_0 = const()[name = string("value_states_13_dilations_0"), val = tensor([1, 1])]; int32 value_states_13_groups_0 = const()[name = string("value_states_13_groups_0"), val = int32(1)]; tensor value_states_13_cast_fp16 = conv(dilations = value_states_13_dilations_0, groups = value_states_13_groups_0, pad = value_states_13_pad_0, pad_type = value_states_13_pad_type_0, strides = value_states_13_strides_0, weight = layers_2_self_attn_v_proj_weight_cast_fp16, x = var_1569_cast_fp16_0)[name = string("value_states_13_cast_fp16")]; tensor concat_24x = const()[name = string("concat_24x"), val = tensor([1, 16, 128, -1])]; tensor x_21_cast_fp16 = reshape(shape = concat_24x, x = query_states_13_cast_fp16)[name = string("x_21_cast_fp16")]; tensor concat_25x = const()[name = string("concat_25x"), val = tensor([1, 2, 128, -1])]; tensor var_1626_cast_fp16 = reshape(shape = concat_25x, x = key_states_21_cast_fp16)[name = string("op_1626_cast_fp16")]; tensor concat_26x = const()[name = string("concat_26x"), val = tensor([1, 2, 128, -1])]; tensor var_1633_cast_fp16 = reshape(shape = concat_26x, x = value_states_13_cast_fp16)[name = string("op_1633_cast_fp16")]; tensor var_1637_cast_fp16 = mul(x = x_21_cast_fp16, y = var_869_cast_fp16)[name = string("op_1637_cast_fp16")]; tensor var_1638_split_sizes_0 = const()[name = string("op_1638_split_sizes_0"), val = tensor([64, 64])]; int32 var_1638_axis_0 = const()[name = string("op_1638_axis_0"), val = int32(-2)]; tensor var_1638_cast_fp16_0, tensor var_1638_cast_fp16_1 = split(axis = var_1638_axis_0, split_sizes = var_1638_split_sizes_0, x = x_21_cast_fp16)[name = string("op_1638_cast_fp16")]; fp16 const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1640_cast_fp16 = mul(x = var_1638_cast_fp16_1, y = const_22_promoted_to_fp16)[name = string("op_1640_cast_fp16")]; int32 var_1642 = const()[name = string("op_1642"), val = int32(-2)]; bool var_1643_interleave_0 = const()[name = string("op_1643_interleave_0"), val = bool(false)]; tensor var_1643_cast_fp16 = concat(axis = var_1642, interleave = var_1643_interleave_0, values = (var_1640_cast_fp16, var_1638_cast_fp16_0))[name = string("op_1643_cast_fp16")]; tensor var_1644_cast_fp16 = mul(x = var_1643_cast_fp16, y = var_878_cast_fp16)[name = string("op_1644_cast_fp16")]; tensor query_states_15_cast_fp16 = add(x = var_1637_cast_fp16, y = var_1644_cast_fp16)[name = string("query_states_15_cast_fp16")]; tensor var_1650_cast_fp16 = mul(x = var_1626_cast_fp16, y = var_869_cast_fp16)[name = string("op_1650_cast_fp16")]; tensor var_1651_split_sizes_0 = const()[name = string("op_1651_split_sizes_0"), val = tensor([64, 64])]; int32 var_1651_axis_0 = const()[name = string("op_1651_axis_0"), val = int32(-2)]; tensor var_1651_cast_fp16_0, tensor var_1651_cast_fp16_1 = split(axis = var_1651_axis_0, split_sizes = var_1651_split_sizes_0, x = var_1626_cast_fp16)[name = string("op_1651_cast_fp16")]; fp16 const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1653_cast_fp16 = mul(x = var_1651_cast_fp16_1, y = const_23_promoted_to_fp16)[name = string("op_1653_cast_fp16")]; int32 var_1655 = const()[name = string("op_1655"), val = int32(-2)]; bool var_1656_interleave_0 = const()[name = string("op_1656_interleave_0"), val = bool(false)]; tensor var_1656_cast_fp16 = concat(axis = var_1655, interleave = var_1656_interleave_0, values = (var_1653_cast_fp16, var_1651_cast_fp16_0))[name = string("op_1656_cast_fp16")]; tensor var_1657_cast_fp16 = mul(x = var_1656_cast_fp16, y = var_878_cast_fp16)[name = string("op_1657_cast_fp16")]; tensor key_states_25_cast_fp16 = add(x = var_1650_cast_fp16, y = var_1657_cast_fp16)[name = string("key_states_25_cast_fp16")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; int32 concat_29_axis_0 = const()[name = string("concat_29_axis_0"), val = int32(0)]; bool concat_29_interleave_0 = const()[name = string("concat_29_interleave_0"), val = bool(false)]; tensor concat_29 = concat(axis = concat_29_axis_0, interleave = concat_29_interleave_0, values = (expand_dims_24, expand_dims_25, position_id, expand_dims_27))[name = string("concat_29")]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; tensor concat_30_values1_0 = const()[name = string("concat_30_values1_0"), val = tensor([0])]; tensor concat_30_values3_0 = const()[name = string("concat_30_values3_0"), val = tensor([0])]; int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_28, concat_30_values1_0, cache_position_end, concat_30_values3_0))[name = string("concat_30")]; tensor key_states_27_perm_0 = const()[name = string("key_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_3_stride_0 = const()[name = string("key_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_27_cast_fp16 = transpose(perm = key_states_27_perm_0, x = key_states_25_cast_fp16)[name = string("transpose_507")]; tensor key_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = key_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = key_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_3_squeeze_mask_0, stride = key_cache_internal_tensor_assign_3_stride_0, update = key_states_27_cast_fp16, x = coreml_update_state_282)[name = string("key_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_3_cast_fp16, input = key_cache)[name = string("coreml_update_state_284_write_state")]; tensor coreml_update_state_284 = read_state(input = key_cache)[name = string("coreml_update_state_284")]; tensor value_states_15_perm_0 = const()[name = string("value_states_15_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_3_stride_0 = const()[name = string("value_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_15_cast_fp16 = transpose(perm = value_states_15_perm_0, x = var_1633_cast_fp16)[name = string("transpose_506")]; tensor value_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = value_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = value_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_3_squeeze_mask_0, stride = value_cache_internal_tensor_assign_3_stride_0, update = value_states_15_cast_fp16, x = coreml_update_state_283)[name = string("value_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_3_cast_fp16, input = value_cache)[name = string("coreml_update_state_285_write_state")]; tensor coreml_update_state_285 = read_state(input = value_cache)[name = string("coreml_update_state_285")]; tensor var_1727_begin_0 = const()[name = string("op_1727_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1727_end_0 = const()[name = string("op_1727_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1727_end_mask_0 = const()[name = string("op_1727_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1727_cast_fp16 = slice_by_index(begin = var_1727_begin_0, end = var_1727_end_0, end_mask = var_1727_end_mask_0, x = coreml_update_state_284)[name = string("op_1727_cast_fp16")]; tensor tile_4 = const()[name = string("tile_4"), val = tensor([1, 1])]; int32 var_1730_axis_0 = const()[name = string("op_1730_axis_0"), val = int32(1)]; tensor var_1730_cast_fp16_0, tensor var_1730_cast_fp16_1 = split(axis = var_1730_axis_0, split_sizes = tile_4, x = var_1727_cast_fp16)[name = string("op_1730_cast_fp16")]; tensor var_1737_begin_0 = const()[name = string("op_1737_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1737_end_0 = const()[name = string("op_1737_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1737_end_mask_0 = const()[name = string("op_1737_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1737_cast_fp16 = slice_by_index(begin = var_1737_begin_0, end = var_1737_end_0, end_mask = var_1737_end_mask_0, x = coreml_update_state_285)[name = string("op_1737_cast_fp16")]; tensor tile_5 = const()[name = string("tile_5"), val = tensor([1, 1])]; int32 var_1740_axis_0 = const()[name = string("op_1740_axis_0"), val = int32(1)]; tensor var_1740_cast_fp16_0, tensor var_1740_cast_fp16_1 = split(axis = var_1740_axis_0, split_sizes = tile_5, x = var_1737_cast_fp16)[name = string("op_1740_cast_fp16")]; tensor var_1743_split_sizes_0 = const()[name = string("op_1743_split_sizes_0"), val = tensor([8, 8])]; int32 var_1743_axis_0 = const()[name = string("op_1743_axis_0"), val = int32(1)]; tensor var_1743_0, tensor var_1743_1 = split(axis = var_1743_axis_0, split_sizes = var_1743_split_sizes_0, x = query_states_15_cast_fp16)[name = string("op_1743")]; bool attn_weights_33_transpose_x_0 = const()[name = string("attn_weights_33_transpose_x_0"), val = bool(false)]; bool attn_weights_33_transpose_y_0 = const()[name = string("attn_weights_33_transpose_y_0"), val = bool(false)]; tensor attn_weights_33_cast_fp16 = matmul(transpose_x = attn_weights_33_transpose_x_0, transpose_y = attn_weights_33_transpose_y_0, x = var_1730_cast_fp16_0, y = var_1743_0)[name = string("attn_weights_33_cast_fp16")]; fp16 var_1746_to_fp16 = const()[name = string("op_1746_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_35_cast_fp16 = mul(x = attn_weights_33_cast_fp16, y = var_1746_to_fp16)[name = string("attn_weights_35_cast_fp16")]; tensor attn_weights_37_cast_fp16 = add(x = attn_weights_35_cast_fp16, y = attn_mask_1)[name = string("attn_weights_37_cast_fp16")]; int32 var_1750 = const()[name = string("op_1750"), val = int32(-2)]; tensor attn_weights_39_cast_fp16 = softmax(axis = var_1750, x = attn_weights_37_cast_fp16)[name = string("attn_weights_39_cast_fp16")]; bool var_1756_transpose_x_1 = const()[name = string("op_1756_transpose_x_1"), val = bool(true)]; bool var_1756_transpose_y_1 = const()[name = string("op_1756_transpose_y_1"), val = bool(false)]; tensor var_1756_cast_fp16 = matmul(transpose_x = var_1756_transpose_x_1, transpose_y = var_1756_transpose_y_1, x = attn_weights_39_cast_fp16, y = var_1740_cast_fp16_0)[name = string("op_1756_cast_fp16")]; bool attn_weights_41_transpose_x_0 = const()[name = string("attn_weights_41_transpose_x_0"), val = bool(false)]; bool attn_weights_41_transpose_y_0 = const()[name = string("attn_weights_41_transpose_y_0"), val = bool(false)]; tensor attn_weights_41_cast_fp16 = matmul(transpose_x = attn_weights_41_transpose_x_0, transpose_y = attn_weights_41_transpose_y_0, x = var_1730_cast_fp16_1, y = var_1743_1)[name = string("attn_weights_41_cast_fp16")]; fp16 var_1758_to_fp16 = const()[name = string("op_1758_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_43_cast_fp16 = mul(x = attn_weights_41_cast_fp16, y = var_1758_to_fp16)[name = string("attn_weights_43_cast_fp16")]; tensor attn_weights_45_cast_fp16 = add(x = attn_weights_43_cast_fp16, y = attn_mask_1)[name = string("attn_weights_45_cast_fp16")]; int32 var_1762 = const()[name = string("op_1762"), val = int32(-2)]; tensor attn_weights_47_cast_fp16 = softmax(axis = var_1762, x = attn_weights_45_cast_fp16)[name = string("attn_weights_47_cast_fp16")]; bool attn_output_17_transpose_x_1 = const()[name = string("attn_output_17_transpose_x_1"), val = bool(true)]; bool attn_output_17_transpose_y_1 = const()[name = string("attn_output_17_transpose_y_1"), val = bool(false)]; tensor attn_output_17_cast_fp16 = matmul(transpose_x = attn_output_17_transpose_x_1, transpose_y = attn_output_17_transpose_y_1, x = attn_weights_47_cast_fp16, y = var_1740_cast_fp16_1)[name = string("attn_output_17_cast_fp16")]; int32 var_1770 = const()[name = string("op_1770"), val = int32(1)]; bool attn_output_19_interleave_0 = const()[name = string("attn_output_19_interleave_0"), val = bool(false)]; tensor attn_output_19_cast_fp16 = concat(axis = var_1770, interleave = attn_output_19_interleave_0, values = (var_1756_cast_fp16, attn_output_17_cast_fp16))[name = string("attn_output_19_cast_fp16")]; tensor var_1774_perm_0 = const()[name = string("op_1774_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_35x = const()[name = string("concat_35x"), val = tensor([1, 2048, 1, -1])]; tensor var_1774_cast_fp16 = transpose(perm = var_1774_perm_0, x = attn_output_19_cast_fp16)[name = string("transpose_505")]; tensor attn_output_23_cast_fp16 = reshape(shape = concat_35x, x = var_1774_cast_fp16)[name = string("attn_output_23_cast_fp16")]; tensor layers_2_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1155169408)))]; tensor hidden_states_23_strides_0 = const()[name = string("hidden_states_23_strides_0"), val = tensor([1, 1])]; string hidden_states_23_pad_type_0 = const()[name = string("hidden_states_23_pad_type_0"), val = string("valid")]; tensor hidden_states_23_pad_0 = const()[name = string("hidden_states_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_23_dilations_0 = const()[name = string("hidden_states_23_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_23_groups_0 = const()[name = string("hidden_states_23_groups_0"), val = int32(1)]; tensor hidden_states_23_cast_fp16 = conv(dilations = hidden_states_23_dilations_0, groups = hidden_states_23_groups_0, pad = hidden_states_23_pad_0, pad_type = hidden_states_23_pad_type_0, strides = hidden_states_23_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16, x = attn_output_23_cast_fp16)[name = string("hidden_states_23_cast_fp16")]; tensor hidden_states_25_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = hidden_states_23_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1807_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_1807_cast_fp16")]; int32 var_1805 = const()[name = string("op_1805"), val = int32(1)]; bool doubled_21_interleave_0 = const()[name = string("doubled_21_interleave_0"), val = bool(false)]; tensor doubled_21_cast_fp16 = concat(axis = var_1805, interleave = doubled_21_interleave_0, values = (hidden_states_25_cast_fp16, var_1807_cast_fp16))[name = string("doubled_21_cast_fp16")]; tensor out_11_axes_0 = const()[name = string("out_11_axes_0"), val = tensor([1])]; tensor out_11_gamma_0_to_fp16 = const()[name = string("out_11_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163558080)))]; fp16 var_1817_to_fp16 = const()[name = string("op_1817_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_11_cast_fp16 = layer_norm(axes = out_11_axes_0, epsilon = var_1817_to_fp16, gamma = out_11_gamma_0_to_fp16, x = doubled_21_cast_fp16)[name = string("out_11_cast_fp16")]; tensor var_1828_split_sizes_0 = const()[name = string("op_1828_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1828_axis_0 = const()[name = string("op_1828_axis_0"), val = int32(1)]; tensor var_1828_cast_fp16_0, tensor var_1828_cast_fp16_1 = split(axis = var_1828_axis_0, split_sizes = var_1828_split_sizes_0, x = out_11_cast_fp16)[name = string("op_1828_cast_fp16")]; tensor layers_2_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163566336)))]; tensor input_5_strides_0 = const()[name = string("input_5_strides_0"), val = tensor([1, 1])]; string input_5_pad_type_0 = const()[name = string("input_5_pad_type_0"), val = string("valid")]; tensor input_5_pad_0 = const()[name = string("input_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_5_dilations_0 = const()[name = string("input_5_dilations_0"), val = tensor([1, 1])]; int32 input_5_groups_0 = const()[name = string("input_5_groups_0"), val = int32(1)]; tensor input_5_cast_fp16 = conv(dilations = input_5_dilations_0, groups = input_5_groups_0, pad = input_5_pad_0, pad_type = input_5_pad_type_0, strides = input_5_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16, x = var_1828_cast_fp16_0)[name = string("input_5_cast_fp16")]; tensor var_1845_cast_fp16 = silu(x = input_5_cast_fp16)[name = string("op_1845_cast_fp16")]; tensor var_1851_strides_0 = const()[name = string("op_1851_strides_0"), val = tensor([1, 1])]; string var_1851_pad_type_0 = const()[name = string("op_1851_pad_type_0"), val = string("valid")]; tensor var_1851_pad_0 = const()[name = string("op_1851_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1851_dilations_0 = const()[name = string("op_1851_dilations_0"), val = tensor([1, 1])]; int32 var_1851_groups_0 = const()[name = string("op_1851_groups_0"), val = int32(1)]; tensor var_1851_cast_fp16 = conv(dilations = var_1851_dilations_0, groups = var_1851_groups_0, pad = var_1851_pad_0, pad_type = var_1851_pad_type_0, strides = var_1851_strides_0, weight = layers_2_mlp_up_proj_weight_cast_fp16, x = var_1828_cast_fp16_0)[name = string("op_1851_cast_fp16")]; tensor x_29_cast_fp16 = mul(x = var_1845_cast_fp16, y = var_1851_cast_fp16)[name = string("x_29_cast_fp16")]; tensor layers_2_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1188732224)))]; tensor hidden_states_27_strides_0 = const()[name = string("hidden_states_27_strides_0"), val = tensor([1, 1])]; string hidden_states_27_pad_type_0 = const()[name = string("hidden_states_27_pad_type_0"), val = string("valid")]; tensor hidden_states_27_pad_0 = const()[name = string("hidden_states_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_27_dilations_0 = const()[name = string("hidden_states_27_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_27_groups_0 = const()[name = string("hidden_states_27_groups_0"), val = int32(1)]; tensor hidden_states_27_cast_fp16 = conv(dilations = hidden_states_27_dilations_0, groups = hidden_states_27_groups_0, pad = hidden_states_27_pad_0, pad_type = hidden_states_27_pad_type_0, strides = hidden_states_27_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16, x = x_29_cast_fp16)[name = string("hidden_states_27_cast_fp16")]; tensor hidden_states_29_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = hidden_states_27_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; fp16 const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1869_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_30_promoted_to_fp16)[name = string("op_1869_cast_fp16")]; int32 var_1867 = const()[name = string("op_1867"), val = int32(1)]; bool doubled_25_interleave_0 = const()[name = string("doubled_25_interleave_0"), val = bool(false)]; tensor doubled_25_cast_fp16 = concat(axis = var_1867, interleave = doubled_25_interleave_0, values = (hidden_states_29_cast_fp16, var_1869_cast_fp16))[name = string("doubled_25_cast_fp16")]; tensor out_13_axes_0 = const()[name = string("out_13_axes_0"), val = tensor([1])]; tensor out_13_gamma_0_to_fp16 = const()[name = string("out_13_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213898112)))]; fp16 var_1879_to_fp16 = const()[name = string("op_1879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_13_cast_fp16 = layer_norm(axes = out_13_axes_0, epsilon = var_1879_to_fp16, gamma = out_13_gamma_0_to_fp16, x = doubled_25_cast_fp16)[name = string("out_13_cast_fp16")]; tensor var_1890_split_sizes_0 = const()[name = string("op_1890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1890_axis_0 = const()[name = string("op_1890_axis_0"), val = int32(1)]; tensor var_1890_cast_fp16_0, tensor var_1890_cast_fp16_1 = split(axis = var_1890_axis_0, split_sizes = var_1890_split_sizes_0, x = out_13_cast_fp16)[name = string("op_1890_cast_fp16")]; tensor layers_3_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213906368)))]; tensor query_states_19_strides_0 = const()[name = string("query_states_19_strides_0"), val = tensor([1, 1])]; string query_states_19_pad_type_0 = const()[name = string("query_states_19_pad_type_0"), val = string("valid")]; tensor query_states_19_pad_0 = const()[name = string("query_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_19_dilations_0 = const()[name = string("query_states_19_dilations_0"), val = tensor([1, 1])]; int32 query_states_19_groups_0 = const()[name = string("query_states_19_groups_0"), val = int32(1)]; tensor query_states_19_cast_fp16 = conv(dilations = query_states_19_dilations_0, groups = query_states_19_groups_0, pad = query_states_19_pad_0, pad_type = query_states_19_pad_type_0, strides = query_states_19_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("query_states_19_cast_fp16")]; tensor layers_3_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1222295040)))]; tensor key_states_31_strides_0 = const()[name = string("key_states_31_strides_0"), val = tensor([1, 1])]; string key_states_31_pad_type_0 = const()[name = string("key_states_31_pad_type_0"), val = string("valid")]; tensor key_states_31_pad_0 = const()[name = string("key_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_31_dilations_0 = const()[name = string("key_states_31_dilations_0"), val = tensor([1, 1])]; int32 key_states_31_groups_0 = const()[name = string("key_states_31_groups_0"), val = int32(1)]; tensor key_states_31_cast_fp16 = conv(dilations = key_states_31_dilations_0, groups = key_states_31_groups_0, pad = key_states_31_pad_0, pad_type = key_states_31_pad_type_0, strides = key_states_31_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("key_states_31_cast_fp16")]; tensor value_states_19_strides_0 = const()[name = string("value_states_19_strides_0"), val = tensor([1, 1])]; string value_states_19_pad_type_0 = const()[name = string("value_states_19_pad_type_0"), val = string("valid")]; tensor value_states_19_pad_0 = const()[name = string("value_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_19_dilations_0 = const()[name = string("value_states_19_dilations_0"), val = tensor([1, 1])]; int32 value_states_19_groups_0 = const()[name = string("value_states_19_groups_0"), val = int32(1)]; tensor value_states_19_cast_fp16 = conv(dilations = value_states_19_dilations_0, groups = value_states_19_groups_0, pad = value_states_19_pad_0, pad_type = value_states_19_pad_type_0, strides = value_states_19_strides_0, weight = layers_3_self_attn_v_proj_weight_cast_fp16, x = var_1890_cast_fp16_0)[name = string("value_states_19_cast_fp16")]; tensor concat_36x = const()[name = string("concat_36x"), val = tensor([1, 16, 128, -1])]; tensor x_31_cast_fp16 = reshape(shape = concat_36x, x = query_states_19_cast_fp16)[name = string("x_31_cast_fp16")]; tensor concat_37x = const()[name = string("concat_37x"), val = tensor([1, 2, 128, -1])]; tensor var_1947_cast_fp16 = reshape(shape = concat_37x, x = key_states_31_cast_fp16)[name = string("op_1947_cast_fp16")]; tensor concat_38x = const()[name = string("concat_38x"), val = tensor([1, 2, 128, -1])]; tensor var_1954_cast_fp16 = reshape(shape = concat_38x, x = value_states_19_cast_fp16)[name = string("op_1954_cast_fp16")]; tensor var_1958_cast_fp16 = mul(x = x_31_cast_fp16, y = var_869_cast_fp16)[name = string("op_1958_cast_fp16")]; tensor var_1959_split_sizes_0 = const()[name = string("op_1959_split_sizes_0"), val = tensor([64, 64])]; int32 var_1959_axis_0 = const()[name = string("op_1959_axis_0"), val = int32(-2)]; tensor var_1959_cast_fp16_0, tensor var_1959_cast_fp16_1 = split(axis = var_1959_axis_0, split_sizes = var_1959_split_sizes_0, x = x_31_cast_fp16)[name = string("op_1959_cast_fp16")]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1961_cast_fp16 = mul(x = var_1959_cast_fp16_1, y = const_32_promoted_to_fp16)[name = string("op_1961_cast_fp16")]; int32 var_1963 = const()[name = string("op_1963"), val = int32(-2)]; bool var_1964_interleave_0 = const()[name = string("op_1964_interleave_0"), val = bool(false)]; tensor var_1964_cast_fp16 = concat(axis = var_1963, interleave = var_1964_interleave_0, values = (var_1961_cast_fp16, var_1959_cast_fp16_0))[name = string("op_1964_cast_fp16")]; tensor var_1965_cast_fp16 = mul(x = var_1964_cast_fp16, y = var_878_cast_fp16)[name = string("op_1965_cast_fp16")]; tensor query_states_21_cast_fp16 = add(x = var_1958_cast_fp16, y = var_1965_cast_fp16)[name = string("query_states_21_cast_fp16")]; tensor var_1971_cast_fp16 = mul(x = var_1947_cast_fp16, y = var_869_cast_fp16)[name = string("op_1971_cast_fp16")]; tensor var_1972_split_sizes_0 = const()[name = string("op_1972_split_sizes_0"), val = tensor([64, 64])]; int32 var_1972_axis_0 = const()[name = string("op_1972_axis_0"), val = int32(-2)]; tensor var_1972_cast_fp16_0, tensor var_1972_cast_fp16_1 = split(axis = var_1972_axis_0, split_sizes = var_1972_split_sizes_0, x = var_1947_cast_fp16)[name = string("op_1972_cast_fp16")]; fp16 const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1974_cast_fp16 = mul(x = var_1972_cast_fp16_1, y = const_33_promoted_to_fp16)[name = string("op_1974_cast_fp16")]; int32 var_1976 = const()[name = string("op_1976"), val = int32(-2)]; bool var_1977_interleave_0 = const()[name = string("op_1977_interleave_0"), val = bool(false)]; tensor var_1977_cast_fp16 = concat(axis = var_1976, interleave = var_1977_interleave_0, values = (var_1974_cast_fp16, var_1972_cast_fp16_0))[name = string("op_1977_cast_fp16")]; tensor var_1978_cast_fp16 = mul(x = var_1977_cast_fp16, y = var_878_cast_fp16)[name = string("op_1978_cast_fp16")]; tensor key_states_35_cast_fp16 = add(x = var_1971_cast_fp16, y = var_1978_cast_fp16)[name = string("key_states_35_cast_fp16")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; int32 concat_41_axis_0 = const()[name = string("concat_41_axis_0"), val = int32(0)]; bool concat_41_interleave_0 = const()[name = string("concat_41_interleave_0"), val = bool(false)]; tensor concat_41 = concat(axis = concat_41_axis_0, interleave = concat_41_interleave_0, values = (expand_dims_36, expand_dims_37, position_id, expand_dims_39))[name = string("concat_41")]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; tensor concat_42_values1_0 = const()[name = string("concat_42_values1_0"), val = tensor([0])]; tensor concat_42_values3_0 = const()[name = string("concat_42_values3_0"), val = tensor([0])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_40, concat_42_values1_0, cache_position_end, concat_42_values3_0))[name = string("concat_42")]; tensor key_states_37_perm_0 = const()[name = string("key_states_37_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_4_stride_0 = const()[name = string("key_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_37_cast_fp16 = transpose(perm = key_states_37_perm_0, x = key_states_35_cast_fp16)[name = string("transpose_504")]; tensor key_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = key_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = key_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_4_squeeze_mask_0, stride = key_cache_internal_tensor_assign_4_stride_0, update = key_states_37_cast_fp16, x = coreml_update_state_284)[name = string("key_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_4_cast_fp16, input = key_cache)[name = string("coreml_update_state_286_write_state")]; tensor coreml_update_state_286 = read_state(input = key_cache)[name = string("coreml_update_state_286")]; tensor value_states_21_perm_0 = const()[name = string("value_states_21_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_4_stride_0 = const()[name = string("value_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_21_cast_fp16 = transpose(perm = value_states_21_perm_0, x = var_1954_cast_fp16)[name = string("transpose_503")]; tensor value_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = value_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = value_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_4_squeeze_mask_0, stride = value_cache_internal_tensor_assign_4_stride_0, update = value_states_21_cast_fp16, x = coreml_update_state_285)[name = string("value_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_4_cast_fp16, input = value_cache)[name = string("coreml_update_state_287_write_state")]; tensor coreml_update_state_287 = read_state(input = value_cache)[name = string("coreml_update_state_287")]; tensor var_2048_begin_0 = const()[name = string("op_2048_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2048_end_0 = const()[name = string("op_2048_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2048_end_mask_0 = const()[name = string("op_2048_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2048_cast_fp16 = slice_by_index(begin = var_2048_begin_0, end = var_2048_end_0, end_mask = var_2048_end_mask_0, x = coreml_update_state_286)[name = string("op_2048_cast_fp16")]; tensor tile_6 = const()[name = string("tile_6"), val = tensor([1, 1])]; int32 var_2051_axis_0 = const()[name = string("op_2051_axis_0"), val = int32(1)]; tensor var_2051_cast_fp16_0, tensor var_2051_cast_fp16_1 = split(axis = var_2051_axis_0, split_sizes = tile_6, x = var_2048_cast_fp16)[name = string("op_2051_cast_fp16")]; tensor var_2058_begin_0 = const()[name = string("op_2058_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2058_end_0 = const()[name = string("op_2058_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2058_end_mask_0 = const()[name = string("op_2058_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2058_cast_fp16 = slice_by_index(begin = var_2058_begin_0, end = var_2058_end_0, end_mask = var_2058_end_mask_0, x = coreml_update_state_287)[name = string("op_2058_cast_fp16")]; tensor tile_7 = const()[name = string("tile_7"), val = tensor([1, 1])]; int32 var_2061_axis_0 = const()[name = string("op_2061_axis_0"), val = int32(1)]; tensor var_2061_cast_fp16_0, tensor var_2061_cast_fp16_1 = split(axis = var_2061_axis_0, split_sizes = tile_7, x = var_2058_cast_fp16)[name = string("op_2061_cast_fp16")]; tensor var_2064_split_sizes_0 = const()[name = string("op_2064_split_sizes_0"), val = tensor([8, 8])]; int32 var_2064_axis_0 = const()[name = string("op_2064_axis_0"), val = int32(1)]; tensor var_2064_0, tensor var_2064_1 = split(axis = var_2064_axis_0, split_sizes = var_2064_split_sizes_0, x = query_states_21_cast_fp16)[name = string("op_2064")]; bool attn_weights_49_transpose_x_0 = const()[name = string("attn_weights_49_transpose_x_0"), val = bool(false)]; bool attn_weights_49_transpose_y_0 = const()[name = string("attn_weights_49_transpose_y_0"), val = bool(false)]; tensor attn_weights_49_cast_fp16 = matmul(transpose_x = attn_weights_49_transpose_x_0, transpose_y = attn_weights_49_transpose_y_0, x = var_2051_cast_fp16_0, y = var_2064_0)[name = string("attn_weights_49_cast_fp16")]; fp16 var_2067_to_fp16 = const()[name = string("op_2067_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_51_cast_fp16 = mul(x = attn_weights_49_cast_fp16, y = var_2067_to_fp16)[name = string("attn_weights_51_cast_fp16")]; tensor attn_weights_53_cast_fp16 = add(x = attn_weights_51_cast_fp16, y = attn_mask_1)[name = string("attn_weights_53_cast_fp16")]; int32 var_2071 = const()[name = string("op_2071"), val = int32(-2)]; tensor attn_weights_55_cast_fp16 = softmax(axis = var_2071, x = attn_weights_53_cast_fp16)[name = string("attn_weights_55_cast_fp16")]; bool var_2077_transpose_x_1 = const()[name = string("op_2077_transpose_x_1"), val = bool(true)]; bool var_2077_transpose_y_1 = const()[name = string("op_2077_transpose_y_1"), val = bool(false)]; tensor var_2077_cast_fp16 = matmul(transpose_x = var_2077_transpose_x_1, transpose_y = var_2077_transpose_y_1, x = attn_weights_55_cast_fp16, y = var_2061_cast_fp16_0)[name = string("op_2077_cast_fp16")]; bool attn_weights_57_transpose_x_0 = const()[name = string("attn_weights_57_transpose_x_0"), val = bool(false)]; bool attn_weights_57_transpose_y_0 = const()[name = string("attn_weights_57_transpose_y_0"), val = bool(false)]; tensor attn_weights_57_cast_fp16 = matmul(transpose_x = attn_weights_57_transpose_x_0, transpose_y = attn_weights_57_transpose_y_0, x = var_2051_cast_fp16_1, y = var_2064_1)[name = string("attn_weights_57_cast_fp16")]; fp16 var_2079_to_fp16 = const()[name = string("op_2079_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_59_cast_fp16 = mul(x = attn_weights_57_cast_fp16, y = var_2079_to_fp16)[name = string("attn_weights_59_cast_fp16")]; tensor attn_weights_61_cast_fp16 = add(x = attn_weights_59_cast_fp16, y = attn_mask_1)[name = string("attn_weights_61_cast_fp16")]; int32 var_2083 = const()[name = string("op_2083"), val = int32(-2)]; tensor attn_weights_63_cast_fp16 = softmax(axis = var_2083, x = attn_weights_61_cast_fp16)[name = string("attn_weights_63_cast_fp16")]; bool attn_output_25_transpose_x_1 = const()[name = string("attn_output_25_transpose_x_1"), val = bool(true)]; bool attn_output_25_transpose_y_1 = const()[name = string("attn_output_25_transpose_y_1"), val = bool(false)]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_1, transpose_y = attn_output_25_transpose_y_1, x = attn_weights_63_cast_fp16, y = var_2061_cast_fp16_1)[name = string("attn_output_25_cast_fp16")]; int32 var_2091 = const()[name = string("op_2091"), val = int32(1)]; bool attn_output_27_interleave_0 = const()[name = string("attn_output_27_interleave_0"), val = bool(false)]; tensor attn_output_27_cast_fp16 = concat(axis = var_2091, interleave = attn_output_27_interleave_0, values = (var_2077_cast_fp16, attn_output_25_cast_fp16))[name = string("attn_output_27_cast_fp16")]; tensor var_2095_perm_0 = const()[name = string("op_2095_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_47x = const()[name = string("concat_47x"), val = tensor([1, 2048, 1, -1])]; tensor var_2095_cast_fp16 = transpose(perm = var_2095_perm_0, x = attn_output_27_cast_fp16)[name = string("transpose_502")]; tensor attn_output_31_cast_fp16 = reshape(shape = concat_47x, x = var_2095_cast_fp16)[name = string("attn_output_31_cast_fp16")]; tensor hidden_states_33_strides_0 = const()[name = string("hidden_states_33_strides_0"), val = tensor([1, 1])]; string hidden_states_33_pad_type_0 = const()[name = string("hidden_states_33_pad_type_0"), val = string("valid")]; tensor hidden_states_33_pad_0 = const()[name = string("hidden_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_33_dilations_0 = const()[name = string("hidden_states_33_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_33_groups_0 = const()[name = string("hidden_states_33_groups_0"), val = int32(1)]; tensor hidden_states_33_cast_fp16 = conv(dilations = hidden_states_33_dilations_0, groups = hidden_states_33_groups_0, pad = hidden_states_33_pad_0, pad_type = hidden_states_33_pad_type_0, strides = hidden_states_33_strides_0, weight = layers_3_self_attn_o_proj_weight_cast_fp16, x = attn_output_31_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; tensor hidden_states_35_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = hidden_states_33_cast_fp16)[name = string("hidden_states_35_cast_fp16")]; fp16 const_38_promoted_to_fp16 = const()[name = string("const_38_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2128_cast_fp16 = mul(x = hidden_states_35_cast_fp16, y = const_38_promoted_to_fp16)[name = string("op_2128_cast_fp16")]; int32 var_2126 = const()[name = string("op_2126"), val = int32(1)]; bool doubled_29_interleave_0 = const()[name = string("doubled_29_interleave_0"), val = bool(false)]; tensor doubled_29_cast_fp16 = concat(axis = var_2126, interleave = doubled_29_interleave_0, values = (hidden_states_35_cast_fp16, var_2128_cast_fp16))[name = string("doubled_29_cast_fp16")]; tensor out_15_axes_0 = const()[name = string("out_15_axes_0"), val = tensor([1])]; tensor out_15_gamma_0_to_fp16 = const()[name = string("out_15_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223343680)))]; fp16 var_2138_to_fp16 = const()[name = string("op_2138_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_15_cast_fp16 = layer_norm(axes = out_15_axes_0, epsilon = var_2138_to_fp16, gamma = out_15_gamma_0_to_fp16, x = doubled_29_cast_fp16)[name = string("out_15_cast_fp16")]; tensor var_2149_split_sizes_0 = const()[name = string("op_2149_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2149_axis_0 = const()[name = string("op_2149_axis_0"), val = int32(1)]; tensor var_2149_cast_fp16_0, tensor var_2149_cast_fp16_1 = split(axis = var_2149_axis_0, split_sizes = var_2149_split_sizes_0, x = out_15_cast_fp16)[name = string("op_2149_cast_fp16")]; tensor layers_3_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223351936)))]; tensor input_7_strides_0 = const()[name = string("input_7_strides_0"), val = tensor([1, 1])]; string input_7_pad_type_0 = const()[name = string("input_7_pad_type_0"), val = string("valid")]; tensor input_7_pad_0 = const()[name = string("input_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_7_dilations_0 = const()[name = string("input_7_dilations_0"), val = tensor([1, 1])]; int32 input_7_groups_0 = const()[name = string("input_7_groups_0"), val = int32(1)]; tensor input_7_cast_fp16 = conv(dilations = input_7_dilations_0, groups = input_7_groups_0, pad = input_7_pad_0, pad_type = input_7_pad_type_0, strides = input_7_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("input_7_cast_fp16")]; tensor var_2166_cast_fp16 = silu(x = input_7_cast_fp16)[name = string("op_2166_cast_fp16")]; tensor layers_3_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1248517824)))]; tensor var_2172_strides_0 = const()[name = string("op_2172_strides_0"), val = tensor([1, 1])]; string var_2172_pad_type_0 = const()[name = string("op_2172_pad_type_0"), val = string("valid")]; tensor var_2172_pad_0 = const()[name = string("op_2172_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2172_dilations_0 = const()[name = string("op_2172_dilations_0"), val = tensor([1, 1])]; int32 var_2172_groups_0 = const()[name = string("op_2172_groups_0"), val = int32(1)]; tensor var_2172_cast_fp16 = conv(dilations = var_2172_dilations_0, groups = var_2172_groups_0, pad = var_2172_pad_0, pad_type = var_2172_pad_type_0, strides = var_2172_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("op_2172_cast_fp16")]; tensor x_39_cast_fp16 = mul(x = var_2166_cast_fp16, y = var_2172_cast_fp16)[name = string("x_39_cast_fp16")]; tensor hidden_states_37_strides_0 = const()[name = string("hidden_states_37_strides_0"), val = tensor([1, 1])]; string hidden_states_37_pad_type_0 = const()[name = string("hidden_states_37_pad_type_0"), val = string("valid")]; tensor hidden_states_37_pad_0 = const()[name = string("hidden_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_37_dilations_0 = const()[name = string("hidden_states_37_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_37_groups_0 = const()[name = string("hidden_states_37_groups_0"), val = int32(1)]; tensor hidden_states_37_cast_fp16 = conv(dilations = hidden_states_37_dilations_0, groups = hidden_states_37_groups_0, pad = hidden_states_37_pad_0, pad_type = hidden_states_37_pad_type_0, strides = hidden_states_37_strides_0, weight = layers_3_mlp_down_proj_weight_cast_fp16, x = x_39_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; tensor hidden_states_39_cast_fp16 = add(x = hidden_states_35_cast_fp16, y = hidden_states_37_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; fp16 const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2190_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_40_promoted_to_fp16)[name = string("op_2190_cast_fp16")]; int32 var_2188 = const()[name = string("op_2188"), val = int32(1)]; bool doubled_33_interleave_0 = const()[name = string("doubled_33_interleave_0"), val = bool(false)]; tensor doubled_33_cast_fp16 = concat(axis = var_2188, interleave = doubled_33_interleave_0, values = (hidden_states_39_cast_fp16, var_2190_cast_fp16))[name = string("doubled_33_cast_fp16")]; tensor out_17_axes_0 = const()[name = string("out_17_axes_0"), val = tensor([1])]; tensor out_17_gamma_0_to_fp16 = const()[name = string("out_17_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273683712)))]; fp16 var_2200_to_fp16 = const()[name = string("op_2200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_17_cast_fp16 = layer_norm(axes = out_17_axes_0, epsilon = var_2200_to_fp16, gamma = out_17_gamma_0_to_fp16, x = doubled_33_cast_fp16)[name = string("out_17_cast_fp16")]; tensor var_2211_split_sizes_0 = const()[name = string("op_2211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2211_axis_0 = const()[name = string("op_2211_axis_0"), val = int32(1)]; tensor var_2211_cast_fp16_0, tensor var_2211_cast_fp16_1 = split(axis = var_2211_axis_0, split_sizes = var_2211_split_sizes_0, x = out_17_cast_fp16)[name = string("op_2211_cast_fp16")]; tensor layers_4_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273691968)))]; tensor query_states_25_strides_0 = const()[name = string("query_states_25_strides_0"), val = tensor([1, 1])]; string query_states_25_pad_type_0 = const()[name = string("query_states_25_pad_type_0"), val = string("valid")]; tensor query_states_25_pad_0 = const()[name = string("query_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_25_dilations_0 = const()[name = string("query_states_25_dilations_0"), val = tensor([1, 1])]; int32 query_states_25_groups_0 = const()[name = string("query_states_25_groups_0"), val = int32(1)]; tensor query_states_25_cast_fp16 = conv(dilations = query_states_25_dilations_0, groups = query_states_25_groups_0, pad = query_states_25_pad_0, pad_type = query_states_25_pad_type_0, strides = query_states_25_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("query_states_25_cast_fp16")]; tensor layers_4_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1282080640)))]; tensor key_states_41_strides_0 = const()[name = string("key_states_41_strides_0"), val = tensor([1, 1])]; string key_states_41_pad_type_0 = const()[name = string("key_states_41_pad_type_0"), val = string("valid")]; tensor key_states_41_pad_0 = const()[name = string("key_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_41_dilations_0 = const()[name = string("key_states_41_dilations_0"), val = tensor([1, 1])]; int32 key_states_41_groups_0 = const()[name = string("key_states_41_groups_0"), val = int32(1)]; tensor key_states_41_cast_fp16 = conv(dilations = key_states_41_dilations_0, groups = key_states_41_groups_0, pad = key_states_41_pad_0, pad_type = key_states_41_pad_type_0, strides = key_states_41_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("key_states_41_cast_fp16")]; tensor value_states_25_strides_0 = const()[name = string("value_states_25_strides_0"), val = tensor([1, 1])]; string value_states_25_pad_type_0 = const()[name = string("value_states_25_pad_type_0"), val = string("valid")]; tensor value_states_25_pad_0 = const()[name = string("value_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_25_dilations_0 = const()[name = string("value_states_25_dilations_0"), val = tensor([1, 1])]; int32 value_states_25_groups_0 = const()[name = string("value_states_25_groups_0"), val = int32(1)]; tensor value_states_25_cast_fp16 = conv(dilations = value_states_25_dilations_0, groups = value_states_25_groups_0, pad = value_states_25_pad_0, pad_type = value_states_25_pad_type_0, strides = value_states_25_strides_0, weight = layers_4_self_attn_v_proj_weight_cast_fp16, x = var_2211_cast_fp16_0)[name = string("value_states_25_cast_fp16")]; tensor concat_48x = const()[name = string("concat_48x"), val = tensor([1, 16, 128, -1])]; tensor x_41_cast_fp16 = reshape(shape = concat_48x, x = query_states_25_cast_fp16)[name = string("x_41_cast_fp16")]; tensor concat_49x = const()[name = string("concat_49x"), val = tensor([1, 2, 128, -1])]; tensor var_2268_cast_fp16 = reshape(shape = concat_49x, x = key_states_41_cast_fp16)[name = string("op_2268_cast_fp16")]; tensor concat_50x = const()[name = string("concat_50x"), val = tensor([1, 2, 128, -1])]; tensor var_2275_cast_fp16 = reshape(shape = concat_50x, x = value_states_25_cast_fp16)[name = string("op_2275_cast_fp16")]; tensor var_2279_cast_fp16 = mul(x = x_41_cast_fp16, y = var_869_cast_fp16)[name = string("op_2279_cast_fp16")]; tensor var_2280_split_sizes_0 = const()[name = string("op_2280_split_sizes_0"), val = tensor([64, 64])]; int32 var_2280_axis_0 = const()[name = string("op_2280_axis_0"), val = int32(-2)]; tensor var_2280_cast_fp16_0, tensor var_2280_cast_fp16_1 = split(axis = var_2280_axis_0, split_sizes = var_2280_split_sizes_0, x = x_41_cast_fp16)[name = string("op_2280_cast_fp16")]; fp16 const_42_promoted_to_fp16 = const()[name = string("const_42_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2282_cast_fp16 = mul(x = var_2280_cast_fp16_1, y = const_42_promoted_to_fp16)[name = string("op_2282_cast_fp16")]; int32 var_2284 = const()[name = string("op_2284"), val = int32(-2)]; bool var_2285_interleave_0 = const()[name = string("op_2285_interleave_0"), val = bool(false)]; tensor var_2285_cast_fp16 = concat(axis = var_2284, interleave = var_2285_interleave_0, values = (var_2282_cast_fp16, var_2280_cast_fp16_0))[name = string("op_2285_cast_fp16")]; tensor var_2286_cast_fp16 = mul(x = var_2285_cast_fp16, y = var_878_cast_fp16)[name = string("op_2286_cast_fp16")]; tensor query_states_27_cast_fp16 = add(x = var_2279_cast_fp16, y = var_2286_cast_fp16)[name = string("query_states_27_cast_fp16")]; tensor var_2292_cast_fp16 = mul(x = var_2268_cast_fp16, y = var_869_cast_fp16)[name = string("op_2292_cast_fp16")]; tensor var_2293_split_sizes_0 = const()[name = string("op_2293_split_sizes_0"), val = tensor([64, 64])]; int32 var_2293_axis_0 = const()[name = string("op_2293_axis_0"), val = int32(-2)]; tensor var_2293_cast_fp16_0, tensor var_2293_cast_fp16_1 = split(axis = var_2293_axis_0, split_sizes = var_2293_split_sizes_0, x = var_2268_cast_fp16)[name = string("op_2293_cast_fp16")]; fp16 const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2295_cast_fp16 = mul(x = var_2293_cast_fp16_1, y = const_43_promoted_to_fp16)[name = string("op_2295_cast_fp16")]; int32 var_2297 = const()[name = string("op_2297"), val = int32(-2)]; bool var_2298_interleave_0 = const()[name = string("op_2298_interleave_0"), val = bool(false)]; tensor var_2298_cast_fp16 = concat(axis = var_2297, interleave = var_2298_interleave_0, values = (var_2295_cast_fp16, var_2293_cast_fp16_0))[name = string("op_2298_cast_fp16")]; tensor var_2299_cast_fp16 = mul(x = var_2298_cast_fp16, y = var_878_cast_fp16)[name = string("op_2299_cast_fp16")]; tensor key_states_45_cast_fp16 = add(x = var_2292_cast_fp16, y = var_2299_cast_fp16)[name = string("key_states_45_cast_fp16")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; int32 concat_53_axis_0 = const()[name = string("concat_53_axis_0"), val = int32(0)]; bool concat_53_interleave_0 = const()[name = string("concat_53_interleave_0"), val = bool(false)]; tensor concat_53 = concat(axis = concat_53_axis_0, interleave = concat_53_interleave_0, values = (expand_dims_48, expand_dims_49, position_id, expand_dims_51))[name = string("concat_53")]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; tensor concat_54_values1_0 = const()[name = string("concat_54_values1_0"), val = tensor([0])]; tensor concat_54_values3_0 = const()[name = string("concat_54_values3_0"), val = tensor([0])]; int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_52, concat_54_values1_0, cache_position_end, concat_54_values3_0))[name = string("concat_54")]; tensor key_states_47_perm_0 = const()[name = string("key_states_47_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_5_stride_0 = const()[name = string("key_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_47_cast_fp16 = transpose(perm = key_states_47_perm_0, x = key_states_45_cast_fp16)[name = string("transpose_501")]; tensor key_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = key_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = key_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_5_squeeze_mask_0, stride = key_cache_internal_tensor_assign_5_stride_0, update = key_states_47_cast_fp16, x = coreml_update_state_286)[name = string("key_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_5_cast_fp16, input = key_cache)[name = string("coreml_update_state_288_write_state")]; tensor coreml_update_state_288 = read_state(input = key_cache)[name = string("coreml_update_state_288")]; tensor value_states_27_perm_0 = const()[name = string("value_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_5_stride_0 = const()[name = string("value_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_27_cast_fp16 = transpose(perm = value_states_27_perm_0, x = var_2275_cast_fp16)[name = string("transpose_500")]; tensor value_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = value_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = value_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_5_squeeze_mask_0, stride = value_cache_internal_tensor_assign_5_stride_0, update = value_states_27_cast_fp16, x = coreml_update_state_287)[name = string("value_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_5_cast_fp16, input = value_cache)[name = string("coreml_update_state_289_write_state")]; tensor coreml_update_state_289 = read_state(input = value_cache)[name = string("coreml_update_state_289")]; tensor var_2369_begin_0 = const()[name = string("op_2369_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2369_end_0 = const()[name = string("op_2369_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2369_end_mask_0 = const()[name = string("op_2369_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2369_cast_fp16 = slice_by_index(begin = var_2369_begin_0, end = var_2369_end_0, end_mask = var_2369_end_mask_0, x = coreml_update_state_288)[name = string("op_2369_cast_fp16")]; tensor tile_8 = const()[name = string("tile_8"), val = tensor([1, 1])]; int32 var_2372_axis_0 = const()[name = string("op_2372_axis_0"), val = int32(1)]; tensor var_2372_cast_fp16_0, tensor var_2372_cast_fp16_1 = split(axis = var_2372_axis_0, split_sizes = tile_8, x = var_2369_cast_fp16)[name = string("op_2372_cast_fp16")]; tensor var_2379_begin_0 = const()[name = string("op_2379_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2379_end_0 = const()[name = string("op_2379_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2379_end_mask_0 = const()[name = string("op_2379_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2379_cast_fp16 = slice_by_index(begin = var_2379_begin_0, end = var_2379_end_0, end_mask = var_2379_end_mask_0, x = coreml_update_state_289)[name = string("op_2379_cast_fp16")]; tensor tile_9 = const()[name = string("tile_9"), val = tensor([1, 1])]; int32 var_2382_axis_0 = const()[name = string("op_2382_axis_0"), val = int32(1)]; tensor var_2382_cast_fp16_0, tensor var_2382_cast_fp16_1 = split(axis = var_2382_axis_0, split_sizes = tile_9, x = var_2379_cast_fp16)[name = string("op_2382_cast_fp16")]; tensor var_2385_split_sizes_0 = const()[name = string("op_2385_split_sizes_0"), val = tensor([8, 8])]; int32 var_2385_axis_0 = const()[name = string("op_2385_axis_0"), val = int32(1)]; tensor var_2385_0, tensor var_2385_1 = split(axis = var_2385_axis_0, split_sizes = var_2385_split_sizes_0, x = query_states_27_cast_fp16)[name = string("op_2385")]; bool attn_weights_65_transpose_x_0 = const()[name = string("attn_weights_65_transpose_x_0"), val = bool(false)]; bool attn_weights_65_transpose_y_0 = const()[name = string("attn_weights_65_transpose_y_0"), val = bool(false)]; tensor attn_weights_65_cast_fp16 = matmul(transpose_x = attn_weights_65_transpose_x_0, transpose_y = attn_weights_65_transpose_y_0, x = var_2372_cast_fp16_0, y = var_2385_0)[name = string("attn_weights_65_cast_fp16")]; fp16 var_2388_to_fp16 = const()[name = string("op_2388_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_67_cast_fp16 = mul(x = attn_weights_65_cast_fp16, y = var_2388_to_fp16)[name = string("attn_weights_67_cast_fp16")]; tensor attn_weights_69_cast_fp16 = add(x = attn_weights_67_cast_fp16, y = attn_mask_1)[name = string("attn_weights_69_cast_fp16")]; int32 var_2392 = const()[name = string("op_2392"), val = int32(-2)]; tensor attn_weights_71_cast_fp16 = softmax(axis = var_2392, x = attn_weights_69_cast_fp16)[name = string("attn_weights_71_cast_fp16")]; bool var_2398_transpose_x_1 = const()[name = string("op_2398_transpose_x_1"), val = bool(true)]; bool var_2398_transpose_y_1 = const()[name = string("op_2398_transpose_y_1"), val = bool(false)]; tensor var_2398_cast_fp16 = matmul(transpose_x = var_2398_transpose_x_1, transpose_y = var_2398_transpose_y_1, x = attn_weights_71_cast_fp16, y = var_2382_cast_fp16_0)[name = string("op_2398_cast_fp16")]; bool attn_weights_73_transpose_x_0 = const()[name = string("attn_weights_73_transpose_x_0"), val = bool(false)]; bool attn_weights_73_transpose_y_0 = const()[name = string("attn_weights_73_transpose_y_0"), val = bool(false)]; tensor attn_weights_73_cast_fp16 = matmul(transpose_x = attn_weights_73_transpose_x_0, transpose_y = attn_weights_73_transpose_y_0, x = var_2372_cast_fp16_1, y = var_2385_1)[name = string("attn_weights_73_cast_fp16")]; fp16 var_2400_to_fp16 = const()[name = string("op_2400_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_75_cast_fp16 = mul(x = attn_weights_73_cast_fp16, y = var_2400_to_fp16)[name = string("attn_weights_75_cast_fp16")]; tensor attn_weights_77_cast_fp16 = add(x = attn_weights_75_cast_fp16, y = attn_mask_1)[name = string("attn_weights_77_cast_fp16")]; int32 var_2404 = const()[name = string("op_2404"), val = int32(-2)]; tensor attn_weights_79_cast_fp16 = softmax(axis = var_2404, x = attn_weights_77_cast_fp16)[name = string("attn_weights_79_cast_fp16")]; bool attn_output_33_transpose_x_1 = const()[name = string("attn_output_33_transpose_x_1"), val = bool(true)]; bool attn_output_33_transpose_y_1 = const()[name = string("attn_output_33_transpose_y_1"), val = bool(false)]; tensor attn_output_33_cast_fp16 = matmul(transpose_x = attn_output_33_transpose_x_1, transpose_y = attn_output_33_transpose_y_1, x = attn_weights_79_cast_fp16, y = var_2382_cast_fp16_1)[name = string("attn_output_33_cast_fp16")]; int32 var_2412 = const()[name = string("op_2412"), val = int32(1)]; bool attn_output_35_interleave_0 = const()[name = string("attn_output_35_interleave_0"), val = bool(false)]; tensor attn_output_35_cast_fp16 = concat(axis = var_2412, interleave = attn_output_35_interleave_0, values = (var_2398_cast_fp16, attn_output_33_cast_fp16))[name = string("attn_output_35_cast_fp16")]; tensor var_2416_perm_0 = const()[name = string("op_2416_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_59x = const()[name = string("concat_59x"), val = tensor([1, 2048, 1, -1])]; tensor var_2416_cast_fp16 = transpose(perm = var_2416_perm_0, x = attn_output_35_cast_fp16)[name = string("transpose_499")]; tensor attn_output_39_cast_fp16 = reshape(shape = concat_59x, x = var_2416_cast_fp16)[name = string("attn_output_39_cast_fp16")]; tensor hidden_states_43_strides_0 = const()[name = string("hidden_states_43_strides_0"), val = tensor([1, 1])]; string hidden_states_43_pad_type_0 = const()[name = string("hidden_states_43_pad_type_0"), val = string("valid")]; tensor hidden_states_43_pad_0 = const()[name = string("hidden_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_43_dilations_0 = const()[name = string("hidden_states_43_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_43_groups_0 = const()[name = string("hidden_states_43_groups_0"), val = int32(1)]; tensor hidden_states_43_cast_fp16 = conv(dilations = hidden_states_43_dilations_0, groups = hidden_states_43_groups_0, pad = hidden_states_43_pad_0, pad_type = hidden_states_43_pad_type_0, strides = hidden_states_43_strides_0, weight = layers_4_self_attn_o_proj_weight_cast_fp16, x = attn_output_39_cast_fp16)[name = string("hidden_states_43_cast_fp16")]; tensor hidden_states_45_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = hidden_states_43_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; fp16 const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2449_cast_fp16 = mul(x = hidden_states_45_cast_fp16, y = const_48_promoted_to_fp16)[name = string("op_2449_cast_fp16")]; int32 var_2447 = const()[name = string("op_2447"), val = int32(1)]; bool doubled_37_interleave_0 = const()[name = string("doubled_37_interleave_0"), val = bool(false)]; tensor doubled_37_cast_fp16 = concat(axis = var_2447, interleave = doubled_37_interleave_0, values = (hidden_states_45_cast_fp16, var_2449_cast_fp16))[name = string("doubled_37_cast_fp16")]; tensor out_19_axes_0 = const()[name = string("out_19_axes_0"), val = tensor([1])]; tensor out_19_gamma_0_to_fp16 = const()[name = string("out_19_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283129280)))]; fp16 var_2459_to_fp16 = const()[name = string("op_2459_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_19_cast_fp16 = layer_norm(axes = out_19_axes_0, epsilon = var_2459_to_fp16, gamma = out_19_gamma_0_to_fp16, x = doubled_37_cast_fp16)[name = string("out_19_cast_fp16")]; tensor var_2470_split_sizes_0 = const()[name = string("op_2470_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2470_axis_0 = const()[name = string("op_2470_axis_0"), val = int32(1)]; tensor var_2470_cast_fp16_0, tensor var_2470_cast_fp16_1 = split(axis = var_2470_axis_0, split_sizes = var_2470_split_sizes_0, x = out_19_cast_fp16)[name = string("op_2470_cast_fp16")]; tensor input_9_strides_0 = const()[name = string("input_9_strides_0"), val = tensor([1, 1])]; string input_9_pad_type_0 = const()[name = string("input_9_pad_type_0"), val = string("valid")]; tensor input_9_pad_0 = const()[name = string("input_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_9_dilations_0 = const()[name = string("input_9_dilations_0"), val = tensor([1, 1])]; int32 input_9_groups_0 = const()[name = string("input_9_groups_0"), val = int32(1)]; tensor input_9_cast_fp16 = conv(dilations = input_9_dilations_0, groups = input_9_groups_0, pad = input_9_pad_0, pad_type = input_9_pad_type_0, strides = input_9_strides_0, weight = layers_4_mlp_gate_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("input_9_cast_fp16")]; tensor var_2487_cast_fp16 = silu(x = input_9_cast_fp16)[name = string("op_2487_cast_fp16")]; tensor var_2493_strides_0 = const()[name = string("op_2493_strides_0"), val = tensor([1, 1])]; string var_2493_pad_type_0 = const()[name = string("op_2493_pad_type_0"), val = string("valid")]; tensor var_2493_pad_0 = const()[name = string("op_2493_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2493_dilations_0 = const()[name = string("op_2493_dilations_0"), val = tensor([1, 1])]; int32 var_2493_groups_0 = const()[name = string("op_2493_groups_0"), val = int32(1)]; tensor var_2493_cast_fp16 = conv(dilations = var_2493_dilations_0, groups = var_2493_groups_0, pad = var_2493_pad_0, pad_type = var_2493_pad_type_0, strides = var_2493_strides_0, weight = layers_4_mlp_up_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("op_2493_cast_fp16")]; tensor x_49_cast_fp16 = mul(x = var_2487_cast_fp16, y = var_2493_cast_fp16)[name = string("x_49_cast_fp16")]; tensor hidden_states_47_strides_0 = const()[name = string("hidden_states_47_strides_0"), val = tensor([1, 1])]; string hidden_states_47_pad_type_0 = const()[name = string("hidden_states_47_pad_type_0"), val = string("valid")]; tensor hidden_states_47_pad_0 = const()[name = string("hidden_states_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_47_dilations_0 = const()[name = string("hidden_states_47_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_47_groups_0 = const()[name = string("hidden_states_47_groups_0"), val = int32(1)]; tensor hidden_states_47_cast_fp16 = conv(dilations = hidden_states_47_dilations_0, groups = hidden_states_47_groups_0, pad = hidden_states_47_pad_0, pad_type = hidden_states_47_pad_type_0, strides = hidden_states_47_strides_0, weight = layers_4_mlp_down_proj_weight_cast_fp16, x = x_49_cast_fp16)[name = string("hidden_states_47_cast_fp16")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = hidden_states_47_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; fp16 const_50_promoted_to_fp16 = const()[name = string("const_50_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2511_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_50_promoted_to_fp16)[name = string("op_2511_cast_fp16")]; int32 var_2509 = const()[name = string("op_2509"), val = int32(1)]; bool doubled_41_interleave_0 = const()[name = string("doubled_41_interleave_0"), val = bool(false)]; tensor doubled_41_cast_fp16 = concat(axis = var_2509, interleave = doubled_41_interleave_0, values = (hidden_states_49_cast_fp16, var_2511_cast_fp16))[name = string("doubled_41_cast_fp16")]; tensor out_21_axes_0 = const()[name = string("out_21_axes_0"), val = tensor([1])]; tensor out_21_gamma_0_to_fp16 = const()[name = string("out_21_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283137536)))]; fp16 var_2521_to_fp16 = const()[name = string("op_2521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_21_cast_fp16 = layer_norm(axes = out_21_axes_0, epsilon = var_2521_to_fp16, gamma = out_21_gamma_0_to_fp16, x = doubled_41_cast_fp16)[name = string("out_21_cast_fp16")]; tensor var_2532_split_sizes_0 = const()[name = string("op_2532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2532_axis_0 = const()[name = string("op_2532_axis_0"), val = int32(1)]; tensor var_2532_cast_fp16_0, tensor var_2532_cast_fp16_1 = split(axis = var_2532_axis_0, split_sizes = var_2532_split_sizes_0, x = out_21_cast_fp16)[name = string("op_2532_cast_fp16")]; tensor layers_5_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283145792)))]; tensor query_states_31_strides_0 = const()[name = string("query_states_31_strides_0"), val = tensor([1, 1])]; string query_states_31_pad_type_0 = const()[name = string("query_states_31_pad_type_0"), val = string("valid")]; tensor query_states_31_pad_0 = const()[name = string("query_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_31_dilations_0 = const()[name = string("query_states_31_dilations_0"), val = tensor([1, 1])]; int32 query_states_31_groups_0 = const()[name = string("query_states_31_groups_0"), val = int32(1)]; tensor query_states_31_cast_fp16 = conv(dilations = query_states_31_dilations_0, groups = query_states_31_groups_0, pad = query_states_31_pad_0, pad_type = query_states_31_pad_type_0, strides = query_states_31_strides_0, weight = layers_5_self_attn_q_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("query_states_31_cast_fp16")]; tensor layers_5_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1291534464)))]; tensor key_states_51_strides_0 = const()[name = string("key_states_51_strides_0"), val = tensor([1, 1])]; string key_states_51_pad_type_0 = const()[name = string("key_states_51_pad_type_0"), val = string("valid")]; tensor key_states_51_pad_0 = const()[name = string("key_states_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_51_dilations_0 = const()[name = string("key_states_51_dilations_0"), val = tensor([1, 1])]; int32 key_states_51_groups_0 = const()[name = string("key_states_51_groups_0"), val = int32(1)]; tensor key_states_51_cast_fp16 = conv(dilations = key_states_51_dilations_0, groups = key_states_51_groups_0, pad = key_states_51_pad_0, pad_type = key_states_51_pad_type_0, strides = key_states_51_strides_0, weight = layers_5_self_attn_k_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("key_states_51_cast_fp16")]; tensor value_states_31_strides_0 = const()[name = string("value_states_31_strides_0"), val = tensor([1, 1])]; string value_states_31_pad_type_0 = const()[name = string("value_states_31_pad_type_0"), val = string("valid")]; tensor value_states_31_pad_0 = const()[name = string("value_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_31_dilations_0 = const()[name = string("value_states_31_dilations_0"), val = tensor([1, 1])]; int32 value_states_31_groups_0 = const()[name = string("value_states_31_groups_0"), val = int32(1)]; tensor value_states_31_cast_fp16 = conv(dilations = value_states_31_dilations_0, groups = value_states_31_groups_0, pad = value_states_31_pad_0, pad_type = value_states_31_pad_type_0, strides = value_states_31_strides_0, weight = layers_5_self_attn_v_proj_weight_cast_fp16, x = var_2532_cast_fp16_0)[name = string("value_states_31_cast_fp16")]; tensor concat_60x = const()[name = string("concat_60x"), val = tensor([1, 16, 128, -1])]; tensor x_51_cast_fp16 = reshape(shape = concat_60x, x = query_states_31_cast_fp16)[name = string("x_51_cast_fp16")]; tensor concat_61x = const()[name = string("concat_61x"), val = tensor([1, 2, 128, -1])]; tensor var_2589_cast_fp16 = reshape(shape = concat_61x, x = key_states_51_cast_fp16)[name = string("op_2589_cast_fp16")]; tensor concat_62x = const()[name = string("concat_62x"), val = tensor([1, 2, 128, -1])]; tensor var_2596_cast_fp16 = reshape(shape = concat_62x, x = value_states_31_cast_fp16)[name = string("op_2596_cast_fp16")]; tensor var_2600_cast_fp16 = mul(x = x_51_cast_fp16, y = var_869_cast_fp16)[name = string("op_2600_cast_fp16")]; tensor var_2601_split_sizes_0 = const()[name = string("op_2601_split_sizes_0"), val = tensor([64, 64])]; int32 var_2601_axis_0 = const()[name = string("op_2601_axis_0"), val = int32(-2)]; tensor var_2601_cast_fp16_0, tensor var_2601_cast_fp16_1 = split(axis = var_2601_axis_0, split_sizes = var_2601_split_sizes_0, x = x_51_cast_fp16)[name = string("op_2601_cast_fp16")]; fp16 const_52_promoted_to_fp16 = const()[name = string("const_52_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2603_cast_fp16 = mul(x = var_2601_cast_fp16_1, y = const_52_promoted_to_fp16)[name = string("op_2603_cast_fp16")]; int32 var_2605 = const()[name = string("op_2605"), val = int32(-2)]; bool var_2606_interleave_0 = const()[name = string("op_2606_interleave_0"), val = bool(false)]; tensor var_2606_cast_fp16 = concat(axis = var_2605, interleave = var_2606_interleave_0, values = (var_2603_cast_fp16, var_2601_cast_fp16_0))[name = string("op_2606_cast_fp16")]; tensor var_2607_cast_fp16 = mul(x = var_2606_cast_fp16, y = var_878_cast_fp16)[name = string("op_2607_cast_fp16")]; tensor query_states_33_cast_fp16 = add(x = var_2600_cast_fp16, y = var_2607_cast_fp16)[name = string("query_states_33_cast_fp16")]; tensor var_2613_cast_fp16 = mul(x = var_2589_cast_fp16, y = var_869_cast_fp16)[name = string("op_2613_cast_fp16")]; tensor var_2614_split_sizes_0 = const()[name = string("op_2614_split_sizes_0"), val = tensor([64, 64])]; int32 var_2614_axis_0 = const()[name = string("op_2614_axis_0"), val = int32(-2)]; tensor var_2614_cast_fp16_0, tensor var_2614_cast_fp16_1 = split(axis = var_2614_axis_0, split_sizes = var_2614_split_sizes_0, x = var_2589_cast_fp16)[name = string("op_2614_cast_fp16")]; fp16 const_53_promoted_to_fp16 = const()[name = string("const_53_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2616_cast_fp16 = mul(x = var_2614_cast_fp16_1, y = const_53_promoted_to_fp16)[name = string("op_2616_cast_fp16")]; int32 var_2618 = const()[name = string("op_2618"), val = int32(-2)]; bool var_2619_interleave_0 = const()[name = string("op_2619_interleave_0"), val = bool(false)]; tensor var_2619_cast_fp16 = concat(axis = var_2618, interleave = var_2619_interleave_0, values = (var_2616_cast_fp16, var_2614_cast_fp16_0))[name = string("op_2619_cast_fp16")]; tensor var_2620_cast_fp16 = mul(x = var_2619_cast_fp16, y = var_878_cast_fp16)[name = string("op_2620_cast_fp16")]; tensor key_states_55_cast_fp16 = add(x = var_2613_cast_fp16, y = var_2620_cast_fp16)[name = string("key_states_55_cast_fp16")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; int32 concat_65_axis_0 = const()[name = string("concat_65_axis_0"), val = int32(0)]; bool concat_65_interleave_0 = const()[name = string("concat_65_interleave_0"), val = bool(false)]; tensor concat_65 = concat(axis = concat_65_axis_0, interleave = concat_65_interleave_0, values = (expand_dims_60, expand_dims_61, position_id, expand_dims_63))[name = string("concat_65")]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; tensor concat_66_values1_0 = const()[name = string("concat_66_values1_0"), val = tensor([0])]; tensor concat_66_values3_0 = const()[name = string("concat_66_values3_0"), val = tensor([0])]; int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_64, concat_66_values1_0, cache_position_end, concat_66_values3_0))[name = string("concat_66")]; tensor key_states_57_perm_0 = const()[name = string("key_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_6_stride_0 = const()[name = string("key_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_57_cast_fp16 = transpose(perm = key_states_57_perm_0, x = key_states_55_cast_fp16)[name = string("transpose_498")]; tensor key_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = key_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = key_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_6_squeeze_mask_0, stride = key_cache_internal_tensor_assign_6_stride_0, update = key_states_57_cast_fp16, x = coreml_update_state_288)[name = string("key_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_6_cast_fp16, input = key_cache)[name = string("coreml_update_state_290_write_state")]; tensor coreml_update_state_290 = read_state(input = key_cache)[name = string("coreml_update_state_290")]; tensor value_states_33_perm_0 = const()[name = string("value_states_33_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_6_stride_0 = const()[name = string("value_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_33_cast_fp16 = transpose(perm = value_states_33_perm_0, x = var_2596_cast_fp16)[name = string("transpose_497")]; tensor value_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = value_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = value_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_6_squeeze_mask_0, stride = value_cache_internal_tensor_assign_6_stride_0, update = value_states_33_cast_fp16, x = coreml_update_state_289)[name = string("value_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_6_cast_fp16, input = value_cache)[name = string("coreml_update_state_291_write_state")]; tensor coreml_update_state_291 = read_state(input = value_cache)[name = string("coreml_update_state_291")]; tensor var_2690_begin_0 = const()[name = string("op_2690_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2690_end_0 = const()[name = string("op_2690_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2690_end_mask_0 = const()[name = string("op_2690_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2690_cast_fp16 = slice_by_index(begin = var_2690_begin_0, end = var_2690_end_0, end_mask = var_2690_end_mask_0, x = coreml_update_state_290)[name = string("op_2690_cast_fp16")]; tensor tile_10 = const()[name = string("tile_10"), val = tensor([1, 1])]; int32 var_2693_axis_0 = const()[name = string("op_2693_axis_0"), val = int32(1)]; tensor var_2693_cast_fp16_0, tensor var_2693_cast_fp16_1 = split(axis = var_2693_axis_0, split_sizes = tile_10, x = var_2690_cast_fp16)[name = string("op_2693_cast_fp16")]; tensor var_2700_begin_0 = const()[name = string("op_2700_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2700_end_0 = const()[name = string("op_2700_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2700_end_mask_0 = const()[name = string("op_2700_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2700_cast_fp16 = slice_by_index(begin = var_2700_begin_0, end = var_2700_end_0, end_mask = var_2700_end_mask_0, x = coreml_update_state_291)[name = string("op_2700_cast_fp16")]; tensor tile_11 = const()[name = string("tile_11"), val = tensor([1, 1])]; int32 var_2703_axis_0 = const()[name = string("op_2703_axis_0"), val = int32(1)]; tensor var_2703_cast_fp16_0, tensor var_2703_cast_fp16_1 = split(axis = var_2703_axis_0, split_sizes = tile_11, x = var_2700_cast_fp16)[name = string("op_2703_cast_fp16")]; tensor var_2706_split_sizes_0 = const()[name = string("op_2706_split_sizes_0"), val = tensor([8, 8])]; int32 var_2706_axis_0 = const()[name = string("op_2706_axis_0"), val = int32(1)]; tensor var_2706_0, tensor var_2706_1 = split(axis = var_2706_axis_0, split_sizes = var_2706_split_sizes_0, x = query_states_33_cast_fp16)[name = string("op_2706")]; bool attn_weights_81_transpose_x_0 = const()[name = string("attn_weights_81_transpose_x_0"), val = bool(false)]; bool attn_weights_81_transpose_y_0 = const()[name = string("attn_weights_81_transpose_y_0"), val = bool(false)]; tensor attn_weights_81_cast_fp16 = matmul(transpose_x = attn_weights_81_transpose_x_0, transpose_y = attn_weights_81_transpose_y_0, x = var_2693_cast_fp16_0, y = var_2706_0)[name = string("attn_weights_81_cast_fp16")]; fp16 var_2709_to_fp16 = const()[name = string("op_2709_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_83_cast_fp16 = mul(x = attn_weights_81_cast_fp16, y = var_2709_to_fp16)[name = string("attn_weights_83_cast_fp16")]; tensor attn_weights_85_cast_fp16 = add(x = attn_weights_83_cast_fp16, y = attn_mask_1)[name = string("attn_weights_85_cast_fp16")]; int32 var_2713 = const()[name = string("op_2713"), val = int32(-2)]; tensor attn_weights_87_cast_fp16 = softmax(axis = var_2713, x = attn_weights_85_cast_fp16)[name = string("attn_weights_87_cast_fp16")]; bool var_2719_transpose_x_1 = const()[name = string("op_2719_transpose_x_1"), val = bool(true)]; bool var_2719_transpose_y_1 = const()[name = string("op_2719_transpose_y_1"), val = bool(false)]; tensor var_2719_cast_fp16 = matmul(transpose_x = var_2719_transpose_x_1, transpose_y = var_2719_transpose_y_1, x = attn_weights_87_cast_fp16, y = var_2703_cast_fp16_0)[name = string("op_2719_cast_fp16")]; bool attn_weights_89_transpose_x_0 = const()[name = string("attn_weights_89_transpose_x_0"), val = bool(false)]; bool attn_weights_89_transpose_y_0 = const()[name = string("attn_weights_89_transpose_y_0"), val = bool(false)]; tensor attn_weights_89_cast_fp16 = matmul(transpose_x = attn_weights_89_transpose_x_0, transpose_y = attn_weights_89_transpose_y_0, x = var_2693_cast_fp16_1, y = var_2706_1)[name = string("attn_weights_89_cast_fp16")]; fp16 var_2721_to_fp16 = const()[name = string("op_2721_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_91_cast_fp16 = mul(x = attn_weights_89_cast_fp16, y = var_2721_to_fp16)[name = string("attn_weights_91_cast_fp16")]; tensor attn_weights_93_cast_fp16 = add(x = attn_weights_91_cast_fp16, y = attn_mask_1)[name = string("attn_weights_93_cast_fp16")]; int32 var_2725 = const()[name = string("op_2725"), val = int32(-2)]; tensor attn_weights_95_cast_fp16 = softmax(axis = var_2725, x = attn_weights_93_cast_fp16)[name = string("attn_weights_95_cast_fp16")]; bool attn_output_41_transpose_x_1 = const()[name = string("attn_output_41_transpose_x_1"), val = bool(true)]; bool attn_output_41_transpose_y_1 = const()[name = string("attn_output_41_transpose_y_1"), val = bool(false)]; tensor attn_output_41_cast_fp16 = matmul(transpose_x = attn_output_41_transpose_x_1, transpose_y = attn_output_41_transpose_y_1, x = attn_weights_95_cast_fp16, y = var_2703_cast_fp16_1)[name = string("attn_output_41_cast_fp16")]; int32 var_2733 = const()[name = string("op_2733"), val = int32(1)]; bool attn_output_43_interleave_0 = const()[name = string("attn_output_43_interleave_0"), val = bool(false)]; tensor attn_output_43_cast_fp16 = concat(axis = var_2733, interleave = attn_output_43_interleave_0, values = (var_2719_cast_fp16, attn_output_41_cast_fp16))[name = string("attn_output_43_cast_fp16")]; tensor var_2737_perm_0 = const()[name = string("op_2737_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_71x = const()[name = string("concat_71x"), val = tensor([1, 2048, 1, -1])]; tensor var_2737_cast_fp16 = transpose(perm = var_2737_perm_0, x = attn_output_43_cast_fp16)[name = string("transpose_496")]; tensor attn_output_47_cast_fp16 = reshape(shape = concat_71x, x = var_2737_cast_fp16)[name = string("attn_output_47_cast_fp16")]; tensor hidden_states_53_strides_0 = const()[name = string("hidden_states_53_strides_0"), val = tensor([1, 1])]; string hidden_states_53_pad_type_0 = const()[name = string("hidden_states_53_pad_type_0"), val = string("valid")]; tensor hidden_states_53_pad_0 = const()[name = string("hidden_states_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_53_dilations_0 = const()[name = string("hidden_states_53_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_53_groups_0 = const()[name = string("hidden_states_53_groups_0"), val = int32(1)]; tensor hidden_states_53_cast_fp16 = conv(dilations = hidden_states_53_dilations_0, groups = hidden_states_53_groups_0, pad = hidden_states_53_pad_0, pad_type = hidden_states_53_pad_type_0, strides = hidden_states_53_strides_0, weight = layers_5_self_attn_o_proj_weight_cast_fp16, x = attn_output_47_cast_fp16)[name = string("hidden_states_53_cast_fp16")]; tensor hidden_states_55_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = hidden_states_53_cast_fp16)[name = string("hidden_states_55_cast_fp16")]; fp16 const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2770_cast_fp16 = mul(x = hidden_states_55_cast_fp16, y = const_58_promoted_to_fp16)[name = string("op_2770_cast_fp16")]; int32 var_2768 = const()[name = string("op_2768"), val = int32(1)]; bool doubled_45_interleave_0 = const()[name = string("doubled_45_interleave_0"), val = bool(false)]; tensor doubled_45_cast_fp16 = concat(axis = var_2768, interleave = doubled_45_interleave_0, values = (hidden_states_55_cast_fp16, var_2770_cast_fp16))[name = string("doubled_45_cast_fp16")]; tensor out_23_axes_0 = const()[name = string("out_23_axes_0"), val = tensor([1])]; tensor out_23_gamma_0_to_fp16 = const()[name = string("out_23_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292583104)))]; fp16 var_2780_to_fp16 = const()[name = string("op_2780_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_23_cast_fp16 = layer_norm(axes = out_23_axes_0, epsilon = var_2780_to_fp16, gamma = out_23_gamma_0_to_fp16, x = doubled_45_cast_fp16)[name = string("out_23_cast_fp16")]; tensor var_2791_split_sizes_0 = const()[name = string("op_2791_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2791_axis_0 = const()[name = string("op_2791_axis_0"), val = int32(1)]; tensor var_2791_cast_fp16_0, tensor var_2791_cast_fp16_1 = split(axis = var_2791_axis_0, split_sizes = var_2791_split_sizes_0, x = out_23_cast_fp16)[name = string("op_2791_cast_fp16")]; tensor layers_5_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_5_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292591360)))]; tensor input_11_strides_0 = const()[name = string("input_11_strides_0"), val = tensor([1, 1])]; string input_11_pad_type_0 = const()[name = string("input_11_pad_type_0"), val = string("valid")]; tensor input_11_pad_0 = const()[name = string("input_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_11_dilations_0 = const()[name = string("input_11_dilations_0"), val = tensor([1, 1])]; int32 input_11_groups_0 = const()[name = string("input_11_groups_0"), val = int32(1)]; tensor input_11_cast_fp16 = conv(dilations = input_11_dilations_0, groups = input_11_groups_0, pad = input_11_pad_0, pad_type = input_11_pad_type_0, strides = input_11_strides_0, weight = layers_5_mlp_gate_proj_weight_to_fp16, x = var_2791_cast_fp16_0)[name = string("input_11_cast_fp16")]; tensor var_2808_cast_fp16 = silu(x = input_11_cast_fp16)[name = string("op_2808_cast_fp16")]; tensor var_2814_strides_0 = const()[name = string("op_2814_strides_0"), val = tensor([1, 1])]; string var_2814_pad_type_0 = const()[name = string("op_2814_pad_type_0"), val = string("valid")]; tensor var_2814_pad_0 = const()[name = string("op_2814_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2814_dilations_0 = const()[name = string("op_2814_dilations_0"), val = tensor([1, 1])]; int32 var_2814_groups_0 = const()[name = string("op_2814_groups_0"), val = int32(1)]; tensor var_2814_cast_fp16 = conv(dilations = var_2814_dilations_0, groups = var_2814_groups_0, pad = var_2814_pad_0, pad_type = var_2814_pad_type_0, strides = var_2814_strides_0, weight = layers_5_mlp_up_proj_weight_cast_fp16, x = var_2791_cast_fp16_0)[name = string("op_2814_cast_fp16")]; tensor x_59_cast_fp16 = mul(x = var_2808_cast_fp16, y = var_2814_cast_fp16)[name = string("x_59_cast_fp16")]; tensor hidden_states_57_strides_0 = const()[name = string("hidden_states_57_strides_0"), val = tensor([1, 1])]; string hidden_states_57_pad_type_0 = const()[name = string("hidden_states_57_pad_type_0"), val = string("valid")]; tensor hidden_states_57_pad_0 = const()[name = string("hidden_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_57_dilations_0 = const()[name = string("hidden_states_57_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_57_groups_0 = const()[name = string("hidden_states_57_groups_0"), val = int32(1)]; tensor hidden_states_57_cast_fp16 = conv(dilations = hidden_states_57_dilations_0, groups = hidden_states_57_groups_0, pad = hidden_states_57_pad_0, pad_type = hidden_states_57_pad_type_0, strides = hidden_states_57_strides_0, weight = layers_5_mlp_down_proj_weight_cast_fp16, x = x_59_cast_fp16)[name = string("hidden_states_57_cast_fp16")]; tensor hidden_states_59_cast_fp16 = add(x = hidden_states_55_cast_fp16, y = hidden_states_57_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2832_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_2832_cast_fp16")]; int32 var_2830 = const()[name = string("op_2830"), val = int32(1)]; bool doubled_49_interleave_0 = const()[name = string("doubled_49_interleave_0"), val = bool(false)]; tensor doubled_49_cast_fp16 = concat(axis = var_2830, interleave = doubled_49_interleave_0, values = (hidden_states_59_cast_fp16, var_2832_cast_fp16))[name = string("doubled_49_cast_fp16")]; tensor out_25_axes_0 = const()[name = string("out_25_axes_0"), val = tensor([1])]; tensor out_25_gamma_0_to_fp16 = const()[name = string("out_25_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317757248)))]; fp16 var_2842_to_fp16 = const()[name = string("op_2842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_25_cast_fp16 = layer_norm(axes = out_25_axes_0, epsilon = var_2842_to_fp16, gamma = out_25_gamma_0_to_fp16, x = doubled_49_cast_fp16)[name = string("out_25_cast_fp16")]; tensor var_2853_split_sizes_0 = const()[name = string("op_2853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2853_axis_0 = const()[name = string("op_2853_axis_0"), val = int32(1)]; tensor var_2853_cast_fp16_0, tensor var_2853_cast_fp16_1 = split(axis = var_2853_axis_0, split_sizes = var_2853_split_sizes_0, x = out_25_cast_fp16)[name = string("op_2853_cast_fp16")]; tensor layers_6_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317765504)))]; tensor query_states_37_strides_0 = const()[name = string("query_states_37_strides_0"), val = tensor([1, 1])]; string query_states_37_pad_type_0 = const()[name = string("query_states_37_pad_type_0"), val = string("valid")]; tensor query_states_37_pad_0 = const()[name = string("query_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_37_dilations_0 = const()[name = string("query_states_37_dilations_0"), val = tensor([1, 1])]; int32 query_states_37_groups_0 = const()[name = string("query_states_37_groups_0"), val = int32(1)]; tensor query_states_37_cast_fp16 = conv(dilations = query_states_37_dilations_0, groups = query_states_37_groups_0, pad = query_states_37_pad_0, pad_type = query_states_37_pad_type_0, strides = query_states_37_strides_0, weight = layers_6_self_attn_q_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("query_states_37_cast_fp16")]; tensor layers_6_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1326154176)))]; tensor key_states_61_strides_0 = const()[name = string("key_states_61_strides_0"), val = tensor([1, 1])]; string key_states_61_pad_type_0 = const()[name = string("key_states_61_pad_type_0"), val = string("valid")]; tensor key_states_61_pad_0 = const()[name = string("key_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_61_dilations_0 = const()[name = string("key_states_61_dilations_0"), val = tensor([1, 1])]; int32 key_states_61_groups_0 = const()[name = string("key_states_61_groups_0"), val = int32(1)]; tensor key_states_61_cast_fp16 = conv(dilations = key_states_61_dilations_0, groups = key_states_61_groups_0, pad = key_states_61_pad_0, pad_type = key_states_61_pad_type_0, strides = key_states_61_strides_0, weight = layers_6_self_attn_k_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("key_states_61_cast_fp16")]; tensor value_states_37_strides_0 = const()[name = string("value_states_37_strides_0"), val = tensor([1, 1])]; string value_states_37_pad_type_0 = const()[name = string("value_states_37_pad_type_0"), val = string("valid")]; tensor value_states_37_pad_0 = const()[name = string("value_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_37_dilations_0 = const()[name = string("value_states_37_dilations_0"), val = tensor([1, 1])]; int32 value_states_37_groups_0 = const()[name = string("value_states_37_groups_0"), val = int32(1)]; tensor value_states_37_cast_fp16 = conv(dilations = value_states_37_dilations_0, groups = value_states_37_groups_0, pad = value_states_37_pad_0, pad_type = value_states_37_pad_type_0, strides = value_states_37_strides_0, weight = layers_6_self_attn_v_proj_weight_cast_fp16, x = var_2853_cast_fp16_0)[name = string("value_states_37_cast_fp16")]; tensor concat_72x = const()[name = string("concat_72x"), val = tensor([1, 16, 128, -1])]; tensor x_61_cast_fp16 = reshape(shape = concat_72x, x = query_states_37_cast_fp16)[name = string("x_61_cast_fp16")]; tensor concat_73x = const()[name = string("concat_73x"), val = tensor([1, 2, 128, -1])]; tensor var_2910_cast_fp16 = reshape(shape = concat_73x, x = key_states_61_cast_fp16)[name = string("op_2910_cast_fp16")]; tensor concat_74x = const()[name = string("concat_74x"), val = tensor([1, 2, 128, -1])]; tensor var_2917_cast_fp16 = reshape(shape = concat_74x, x = value_states_37_cast_fp16)[name = string("op_2917_cast_fp16")]; tensor var_2921_cast_fp16 = mul(x = x_61_cast_fp16, y = var_869_cast_fp16)[name = string("op_2921_cast_fp16")]; tensor var_2922_split_sizes_0 = const()[name = string("op_2922_split_sizes_0"), val = tensor([64, 64])]; int32 var_2922_axis_0 = const()[name = string("op_2922_axis_0"), val = int32(-2)]; tensor var_2922_cast_fp16_0, tensor var_2922_cast_fp16_1 = split(axis = var_2922_axis_0, split_sizes = var_2922_split_sizes_0, x = x_61_cast_fp16)[name = string("op_2922_cast_fp16")]; fp16 const_62_promoted_to_fp16 = const()[name = string("const_62_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2924_cast_fp16 = mul(x = var_2922_cast_fp16_1, y = const_62_promoted_to_fp16)[name = string("op_2924_cast_fp16")]; int32 var_2926 = const()[name = string("op_2926"), val = int32(-2)]; bool var_2927_interleave_0 = const()[name = string("op_2927_interleave_0"), val = bool(false)]; tensor var_2927_cast_fp16 = concat(axis = var_2926, interleave = var_2927_interleave_0, values = (var_2924_cast_fp16, var_2922_cast_fp16_0))[name = string("op_2927_cast_fp16")]; tensor var_2928_cast_fp16 = mul(x = var_2927_cast_fp16, y = var_878_cast_fp16)[name = string("op_2928_cast_fp16")]; tensor query_states_39_cast_fp16 = add(x = var_2921_cast_fp16, y = var_2928_cast_fp16)[name = string("query_states_39_cast_fp16")]; tensor var_2934_cast_fp16 = mul(x = var_2910_cast_fp16, y = var_869_cast_fp16)[name = string("op_2934_cast_fp16")]; tensor var_2935_split_sizes_0 = const()[name = string("op_2935_split_sizes_0"), val = tensor([64, 64])]; int32 var_2935_axis_0 = const()[name = string("op_2935_axis_0"), val = int32(-2)]; tensor var_2935_cast_fp16_0, tensor var_2935_cast_fp16_1 = split(axis = var_2935_axis_0, split_sizes = var_2935_split_sizes_0, x = var_2910_cast_fp16)[name = string("op_2935_cast_fp16")]; fp16 const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2937_cast_fp16 = mul(x = var_2935_cast_fp16_1, y = const_63_promoted_to_fp16)[name = string("op_2937_cast_fp16")]; int32 var_2939 = const()[name = string("op_2939"), val = int32(-2)]; bool var_2940_interleave_0 = const()[name = string("op_2940_interleave_0"), val = bool(false)]; tensor var_2940_cast_fp16 = concat(axis = var_2939, interleave = var_2940_interleave_0, values = (var_2937_cast_fp16, var_2935_cast_fp16_0))[name = string("op_2940_cast_fp16")]; tensor var_2941_cast_fp16 = mul(x = var_2940_cast_fp16, y = var_878_cast_fp16)[name = string("op_2941_cast_fp16")]; tensor key_states_65_cast_fp16 = add(x = var_2934_cast_fp16, y = var_2941_cast_fp16)[name = string("key_states_65_cast_fp16")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; int32 concat_77_axis_0 = const()[name = string("concat_77_axis_0"), val = int32(0)]; bool concat_77_interleave_0 = const()[name = string("concat_77_interleave_0"), val = bool(false)]; tensor concat_77 = concat(axis = concat_77_axis_0, interleave = concat_77_interleave_0, values = (expand_dims_72, expand_dims_73, position_id, expand_dims_75))[name = string("concat_77")]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; tensor concat_78_values1_0 = const()[name = string("concat_78_values1_0"), val = tensor([0])]; tensor concat_78_values3_0 = const()[name = string("concat_78_values3_0"), val = tensor([0])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_76, concat_78_values1_0, cache_position_end, concat_78_values3_0))[name = string("concat_78")]; tensor key_states_67_perm_0 = const()[name = string("key_states_67_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_7_stride_0 = const()[name = string("key_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_67_cast_fp16 = transpose(perm = key_states_67_perm_0, x = key_states_65_cast_fp16)[name = string("transpose_495")]; tensor key_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = key_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = key_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_7_squeeze_mask_0, stride = key_cache_internal_tensor_assign_7_stride_0, update = key_states_67_cast_fp16, x = coreml_update_state_290)[name = string("key_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_7_cast_fp16, input = key_cache)[name = string("coreml_update_state_292_write_state")]; tensor coreml_update_state_292 = read_state(input = key_cache)[name = string("coreml_update_state_292")]; tensor value_states_39_perm_0 = const()[name = string("value_states_39_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_7_stride_0 = const()[name = string("value_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_39_cast_fp16 = transpose(perm = value_states_39_perm_0, x = var_2917_cast_fp16)[name = string("transpose_494")]; tensor value_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = value_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = value_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_7_squeeze_mask_0, stride = value_cache_internal_tensor_assign_7_stride_0, update = value_states_39_cast_fp16, x = coreml_update_state_291)[name = string("value_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_7_cast_fp16, input = value_cache)[name = string("coreml_update_state_293_write_state")]; tensor coreml_update_state_293 = read_state(input = value_cache)[name = string("coreml_update_state_293")]; tensor var_3011_begin_0 = const()[name = string("op_3011_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3011_end_0 = const()[name = string("op_3011_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3011_end_mask_0 = const()[name = string("op_3011_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3011_cast_fp16 = slice_by_index(begin = var_3011_begin_0, end = var_3011_end_0, end_mask = var_3011_end_mask_0, x = coreml_update_state_292)[name = string("op_3011_cast_fp16")]; tensor tile_12 = const()[name = string("tile_12"), val = tensor([1, 1])]; int32 var_3014_axis_0 = const()[name = string("op_3014_axis_0"), val = int32(1)]; tensor var_3014_cast_fp16_0, tensor var_3014_cast_fp16_1 = split(axis = var_3014_axis_0, split_sizes = tile_12, x = var_3011_cast_fp16)[name = string("op_3014_cast_fp16")]; tensor var_3021_begin_0 = const()[name = string("op_3021_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3021_end_0 = const()[name = string("op_3021_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3021_end_mask_0 = const()[name = string("op_3021_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3021_cast_fp16 = slice_by_index(begin = var_3021_begin_0, end = var_3021_end_0, end_mask = var_3021_end_mask_0, x = coreml_update_state_293)[name = string("op_3021_cast_fp16")]; tensor tile_13 = const()[name = string("tile_13"), val = tensor([1, 1])]; int32 var_3024_axis_0 = const()[name = string("op_3024_axis_0"), val = int32(1)]; tensor var_3024_cast_fp16_0, tensor var_3024_cast_fp16_1 = split(axis = var_3024_axis_0, split_sizes = tile_13, x = var_3021_cast_fp16)[name = string("op_3024_cast_fp16")]; tensor var_3027_split_sizes_0 = const()[name = string("op_3027_split_sizes_0"), val = tensor([8, 8])]; int32 var_3027_axis_0 = const()[name = string("op_3027_axis_0"), val = int32(1)]; tensor var_3027_0, tensor var_3027_1 = split(axis = var_3027_axis_0, split_sizes = var_3027_split_sizes_0, x = query_states_39_cast_fp16)[name = string("op_3027")]; bool attn_weights_97_transpose_x_0 = const()[name = string("attn_weights_97_transpose_x_0"), val = bool(false)]; bool attn_weights_97_transpose_y_0 = const()[name = string("attn_weights_97_transpose_y_0"), val = bool(false)]; tensor attn_weights_97_cast_fp16 = matmul(transpose_x = attn_weights_97_transpose_x_0, transpose_y = attn_weights_97_transpose_y_0, x = var_3014_cast_fp16_0, y = var_3027_0)[name = string("attn_weights_97_cast_fp16")]; fp16 var_3030_to_fp16 = const()[name = string("op_3030_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_99_cast_fp16 = mul(x = attn_weights_97_cast_fp16, y = var_3030_to_fp16)[name = string("attn_weights_99_cast_fp16")]; tensor attn_weights_101_cast_fp16 = add(x = attn_weights_99_cast_fp16, y = attn_mask_1)[name = string("attn_weights_101_cast_fp16")]; int32 var_3034 = const()[name = string("op_3034"), val = int32(-2)]; tensor attn_weights_103_cast_fp16 = softmax(axis = var_3034, x = attn_weights_101_cast_fp16)[name = string("attn_weights_103_cast_fp16")]; bool var_3040_transpose_x_1 = const()[name = string("op_3040_transpose_x_1"), val = bool(true)]; bool var_3040_transpose_y_1 = const()[name = string("op_3040_transpose_y_1"), val = bool(false)]; tensor var_3040_cast_fp16 = matmul(transpose_x = var_3040_transpose_x_1, transpose_y = var_3040_transpose_y_1, x = attn_weights_103_cast_fp16, y = var_3024_cast_fp16_0)[name = string("op_3040_cast_fp16")]; bool attn_weights_105_transpose_x_0 = const()[name = string("attn_weights_105_transpose_x_0"), val = bool(false)]; bool attn_weights_105_transpose_y_0 = const()[name = string("attn_weights_105_transpose_y_0"), val = bool(false)]; tensor attn_weights_105_cast_fp16 = matmul(transpose_x = attn_weights_105_transpose_x_0, transpose_y = attn_weights_105_transpose_y_0, x = var_3014_cast_fp16_1, y = var_3027_1)[name = string("attn_weights_105_cast_fp16")]; fp16 var_3042_to_fp16 = const()[name = string("op_3042_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_107_cast_fp16 = mul(x = attn_weights_105_cast_fp16, y = var_3042_to_fp16)[name = string("attn_weights_107_cast_fp16")]; tensor attn_weights_109_cast_fp16 = add(x = attn_weights_107_cast_fp16, y = attn_mask_1)[name = string("attn_weights_109_cast_fp16")]; int32 var_3046 = const()[name = string("op_3046"), val = int32(-2)]; tensor attn_weights_111_cast_fp16 = softmax(axis = var_3046, x = attn_weights_109_cast_fp16)[name = string("attn_weights_111_cast_fp16")]; bool attn_output_49_transpose_x_1 = const()[name = string("attn_output_49_transpose_x_1"), val = bool(true)]; bool attn_output_49_transpose_y_1 = const()[name = string("attn_output_49_transpose_y_1"), val = bool(false)]; tensor attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_1, transpose_y = attn_output_49_transpose_y_1, x = attn_weights_111_cast_fp16, y = var_3024_cast_fp16_1)[name = string("attn_output_49_cast_fp16")]; int32 var_3054 = const()[name = string("op_3054"), val = int32(1)]; bool attn_output_51_interleave_0 = const()[name = string("attn_output_51_interleave_0"), val = bool(false)]; tensor attn_output_51_cast_fp16 = concat(axis = var_3054, interleave = attn_output_51_interleave_0, values = (var_3040_cast_fp16, attn_output_49_cast_fp16))[name = string("attn_output_51_cast_fp16")]; tensor var_3058_perm_0 = const()[name = string("op_3058_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_83x = const()[name = string("concat_83x"), val = tensor([1, 2048, 1, -1])]; tensor var_3058_cast_fp16 = transpose(perm = var_3058_perm_0, x = attn_output_51_cast_fp16)[name = string("transpose_493")]; tensor attn_output_55_cast_fp16 = reshape(shape = concat_83x, x = var_3058_cast_fp16)[name = string("attn_output_55_cast_fp16")]; tensor hidden_states_63_strides_0 = const()[name = string("hidden_states_63_strides_0"), val = tensor([1, 1])]; string hidden_states_63_pad_type_0 = const()[name = string("hidden_states_63_pad_type_0"), val = string("valid")]; tensor hidden_states_63_pad_0 = const()[name = string("hidden_states_63_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_63_dilations_0 = const()[name = string("hidden_states_63_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_63_groups_0 = const()[name = string("hidden_states_63_groups_0"), val = int32(1)]; tensor hidden_states_63_cast_fp16 = conv(dilations = hidden_states_63_dilations_0, groups = hidden_states_63_groups_0, pad = hidden_states_63_pad_0, pad_type = hidden_states_63_pad_type_0, strides = hidden_states_63_strides_0, weight = layers_6_self_attn_o_proj_weight_cast_fp16, x = attn_output_55_cast_fp16)[name = string("hidden_states_63_cast_fp16")]; tensor hidden_states_65_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = hidden_states_63_cast_fp16)[name = string("hidden_states_65_cast_fp16")]; fp16 const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3091_cast_fp16 = mul(x = hidden_states_65_cast_fp16, y = const_68_promoted_to_fp16)[name = string("op_3091_cast_fp16")]; int32 var_3089 = const()[name = string("op_3089"), val = int32(1)]; bool doubled_53_interleave_0 = const()[name = string("doubled_53_interleave_0"), val = bool(false)]; tensor doubled_53_cast_fp16 = concat(axis = var_3089, interleave = doubled_53_interleave_0, values = (hidden_states_65_cast_fp16, var_3091_cast_fp16))[name = string("doubled_53_cast_fp16")]; tensor out_27_axes_0 = const()[name = string("out_27_axes_0"), val = tensor([1])]; tensor out_27_gamma_0_to_fp16 = const()[name = string("out_27_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327202816)))]; fp16 var_3101_to_fp16 = const()[name = string("op_3101_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_27_cast_fp16 = layer_norm(axes = out_27_axes_0, epsilon = var_3101_to_fp16, gamma = out_27_gamma_0_to_fp16, x = doubled_53_cast_fp16)[name = string("out_27_cast_fp16")]; tensor var_3112_split_sizes_0 = const()[name = string("op_3112_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3112_axis_0 = const()[name = string("op_3112_axis_0"), val = int32(1)]; tensor var_3112_cast_fp16_0, tensor var_3112_cast_fp16_1 = split(axis = var_3112_axis_0, split_sizes = var_3112_split_sizes_0, x = out_27_cast_fp16)[name = string("op_3112_cast_fp16")]; tensor input_13_strides_0 = const()[name = string("input_13_strides_0"), val = tensor([1, 1])]; string input_13_pad_type_0 = const()[name = string("input_13_pad_type_0"), val = string("valid")]; tensor input_13_pad_0 = const()[name = string("input_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_13_dilations_0 = const()[name = string("input_13_dilations_0"), val = tensor([1, 1])]; int32 input_13_groups_0 = const()[name = string("input_13_groups_0"), val = int32(1)]; tensor input_13_cast_fp16 = conv(dilations = input_13_dilations_0, groups = input_13_groups_0, pad = input_13_pad_0, pad_type = input_13_pad_type_0, strides = input_13_strides_0, weight = layers_6_mlp_gate_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("input_13_cast_fp16")]; tensor var_3129_cast_fp16 = silu(x = input_13_cast_fp16)[name = string("op_3129_cast_fp16")]; tensor var_3135_strides_0 = const()[name = string("op_3135_strides_0"), val = tensor([1, 1])]; string var_3135_pad_type_0 = const()[name = string("op_3135_pad_type_0"), val = string("valid")]; tensor var_3135_pad_0 = const()[name = string("op_3135_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3135_dilations_0 = const()[name = string("op_3135_dilations_0"), val = tensor([1, 1])]; int32 var_3135_groups_0 = const()[name = string("op_3135_groups_0"), val = int32(1)]; tensor var_3135_cast_fp16 = conv(dilations = var_3135_dilations_0, groups = var_3135_groups_0, pad = var_3135_pad_0, pad_type = var_3135_pad_type_0, strides = var_3135_strides_0, weight = layers_6_mlp_up_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("op_3135_cast_fp16")]; tensor x_69_cast_fp16 = mul(x = var_3129_cast_fp16, y = var_3135_cast_fp16)[name = string("x_69_cast_fp16")]; tensor hidden_states_67_strides_0 = const()[name = string("hidden_states_67_strides_0"), val = tensor([1, 1])]; string hidden_states_67_pad_type_0 = const()[name = string("hidden_states_67_pad_type_0"), val = string("valid")]; tensor hidden_states_67_pad_0 = const()[name = string("hidden_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_67_dilations_0 = const()[name = string("hidden_states_67_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_67_groups_0 = const()[name = string("hidden_states_67_groups_0"), val = int32(1)]; tensor hidden_states_67_cast_fp16 = conv(dilations = hidden_states_67_dilations_0, groups = hidden_states_67_groups_0, pad = hidden_states_67_pad_0, pad_type = hidden_states_67_pad_type_0, strides = hidden_states_67_strides_0, weight = layers_6_mlp_down_proj_weight_cast_fp16, x = x_69_cast_fp16)[name = string("hidden_states_67_cast_fp16")]; tensor hidden_states_69_cast_fp16 = add(x = hidden_states_65_cast_fp16, y = hidden_states_67_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3153_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_70_promoted_to_fp16)[name = string("op_3153_cast_fp16")]; int32 var_3151 = const()[name = string("op_3151"), val = int32(1)]; bool doubled_57_interleave_0 = const()[name = string("doubled_57_interleave_0"), val = bool(false)]; tensor doubled_57_cast_fp16 = concat(axis = var_3151, interleave = doubled_57_interleave_0, values = (hidden_states_69_cast_fp16, var_3153_cast_fp16))[name = string("doubled_57_cast_fp16")]; tensor out_29_axes_0 = const()[name = string("out_29_axes_0"), val = tensor([1])]; tensor out_29_gamma_0_to_fp16 = const()[name = string("out_29_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327211072)))]; fp16 var_3163_to_fp16 = const()[name = string("op_3163_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_29_cast_fp16 = layer_norm(axes = out_29_axes_0, epsilon = var_3163_to_fp16, gamma = out_29_gamma_0_to_fp16, x = doubled_57_cast_fp16)[name = string("out_29_cast_fp16")]; tensor var_3174_split_sizes_0 = const()[name = string("op_3174_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3174_axis_0 = const()[name = string("op_3174_axis_0"), val = int32(1)]; tensor var_3174_cast_fp16_0, tensor var_3174_cast_fp16_1 = split(axis = var_3174_axis_0, split_sizes = var_3174_split_sizes_0, x = out_29_cast_fp16)[name = string("op_3174_cast_fp16")]; tensor layers_7_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327219328)))]; tensor query_states_43_strides_0 = const()[name = string("query_states_43_strides_0"), val = tensor([1, 1])]; string query_states_43_pad_type_0 = const()[name = string("query_states_43_pad_type_0"), val = string("valid")]; tensor query_states_43_pad_0 = const()[name = string("query_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_43_dilations_0 = const()[name = string("query_states_43_dilations_0"), val = tensor([1, 1])]; int32 query_states_43_groups_0 = const()[name = string("query_states_43_groups_0"), val = int32(1)]; tensor query_states_43_cast_fp16 = conv(dilations = query_states_43_dilations_0, groups = query_states_43_groups_0, pad = query_states_43_pad_0, pad_type = query_states_43_pad_type_0, strides = query_states_43_strides_0, weight = layers_7_self_attn_q_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("query_states_43_cast_fp16")]; tensor layers_7_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1335608000)))]; tensor key_states_71_strides_0 = const()[name = string("key_states_71_strides_0"), val = tensor([1, 1])]; string key_states_71_pad_type_0 = const()[name = string("key_states_71_pad_type_0"), val = string("valid")]; tensor key_states_71_pad_0 = const()[name = string("key_states_71_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_71_dilations_0 = const()[name = string("key_states_71_dilations_0"), val = tensor([1, 1])]; int32 key_states_71_groups_0 = const()[name = string("key_states_71_groups_0"), val = int32(1)]; tensor key_states_71_cast_fp16 = conv(dilations = key_states_71_dilations_0, groups = key_states_71_groups_0, pad = key_states_71_pad_0, pad_type = key_states_71_pad_type_0, strides = key_states_71_strides_0, weight = layers_7_self_attn_k_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("key_states_71_cast_fp16")]; tensor value_states_43_strides_0 = const()[name = string("value_states_43_strides_0"), val = tensor([1, 1])]; string value_states_43_pad_type_0 = const()[name = string("value_states_43_pad_type_0"), val = string("valid")]; tensor value_states_43_pad_0 = const()[name = string("value_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_43_dilations_0 = const()[name = string("value_states_43_dilations_0"), val = tensor([1, 1])]; int32 value_states_43_groups_0 = const()[name = string("value_states_43_groups_0"), val = int32(1)]; tensor value_states_43_cast_fp16 = conv(dilations = value_states_43_dilations_0, groups = value_states_43_groups_0, pad = value_states_43_pad_0, pad_type = value_states_43_pad_type_0, strides = value_states_43_strides_0, weight = layers_7_self_attn_v_proj_weight_cast_fp16, x = var_3174_cast_fp16_0)[name = string("value_states_43_cast_fp16")]; tensor concat_84x = const()[name = string("concat_84x"), val = tensor([1, 16, 128, -1])]; tensor x_71_cast_fp16 = reshape(shape = concat_84x, x = query_states_43_cast_fp16)[name = string("x_71_cast_fp16")]; tensor concat_85x = const()[name = string("concat_85x"), val = tensor([1, 2, 128, -1])]; tensor var_3231_cast_fp16 = reshape(shape = concat_85x, x = key_states_71_cast_fp16)[name = string("op_3231_cast_fp16")]; tensor concat_86x = const()[name = string("concat_86x"), val = tensor([1, 2, 128, -1])]; tensor var_3238_cast_fp16 = reshape(shape = concat_86x, x = value_states_43_cast_fp16)[name = string("op_3238_cast_fp16")]; tensor var_3242_cast_fp16 = mul(x = x_71_cast_fp16, y = var_869_cast_fp16)[name = string("op_3242_cast_fp16")]; tensor var_3243_split_sizes_0 = const()[name = string("op_3243_split_sizes_0"), val = tensor([64, 64])]; int32 var_3243_axis_0 = const()[name = string("op_3243_axis_0"), val = int32(-2)]; tensor var_3243_cast_fp16_0, tensor var_3243_cast_fp16_1 = split(axis = var_3243_axis_0, split_sizes = var_3243_split_sizes_0, x = x_71_cast_fp16)[name = string("op_3243_cast_fp16")]; fp16 const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3245_cast_fp16 = mul(x = var_3243_cast_fp16_1, y = const_72_promoted_to_fp16)[name = string("op_3245_cast_fp16")]; int32 var_3247 = const()[name = string("op_3247"), val = int32(-2)]; bool var_3248_interleave_0 = const()[name = string("op_3248_interleave_0"), val = bool(false)]; tensor var_3248_cast_fp16 = concat(axis = var_3247, interleave = var_3248_interleave_0, values = (var_3245_cast_fp16, var_3243_cast_fp16_0))[name = string("op_3248_cast_fp16")]; tensor var_3249_cast_fp16 = mul(x = var_3248_cast_fp16, y = var_878_cast_fp16)[name = string("op_3249_cast_fp16")]; tensor query_states_45_cast_fp16 = add(x = var_3242_cast_fp16, y = var_3249_cast_fp16)[name = string("query_states_45_cast_fp16")]; tensor var_3255_cast_fp16 = mul(x = var_3231_cast_fp16, y = var_869_cast_fp16)[name = string("op_3255_cast_fp16")]; tensor var_3256_split_sizes_0 = const()[name = string("op_3256_split_sizes_0"), val = tensor([64, 64])]; int32 var_3256_axis_0 = const()[name = string("op_3256_axis_0"), val = int32(-2)]; tensor var_3256_cast_fp16_0, tensor var_3256_cast_fp16_1 = split(axis = var_3256_axis_0, split_sizes = var_3256_split_sizes_0, x = var_3231_cast_fp16)[name = string("op_3256_cast_fp16")]; fp16 const_73_promoted_to_fp16 = const()[name = string("const_73_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3258_cast_fp16 = mul(x = var_3256_cast_fp16_1, y = const_73_promoted_to_fp16)[name = string("op_3258_cast_fp16")]; int32 var_3260 = const()[name = string("op_3260"), val = int32(-2)]; bool var_3261_interleave_0 = const()[name = string("op_3261_interleave_0"), val = bool(false)]; tensor var_3261_cast_fp16 = concat(axis = var_3260, interleave = var_3261_interleave_0, values = (var_3258_cast_fp16, var_3256_cast_fp16_0))[name = string("op_3261_cast_fp16")]; tensor var_3262_cast_fp16 = mul(x = var_3261_cast_fp16, y = var_878_cast_fp16)[name = string("op_3262_cast_fp16")]; tensor key_states_75_cast_fp16 = add(x = var_3255_cast_fp16, y = var_3262_cast_fp16)[name = string("key_states_75_cast_fp16")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; int32 concat_89_axis_0 = const()[name = string("concat_89_axis_0"), val = int32(0)]; bool concat_89_interleave_0 = const()[name = string("concat_89_interleave_0"), val = bool(false)]; tensor concat_89 = concat(axis = concat_89_axis_0, interleave = concat_89_interleave_0, values = (expand_dims_84, expand_dims_85, position_id, expand_dims_87))[name = string("concat_89")]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; tensor concat_90_values1_0 = const()[name = string("concat_90_values1_0"), val = tensor([0])]; tensor concat_90_values3_0 = const()[name = string("concat_90_values3_0"), val = tensor([0])]; int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_88, concat_90_values1_0, cache_position_end, concat_90_values3_0))[name = string("concat_90")]; tensor key_states_77_perm_0 = const()[name = string("key_states_77_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_8_stride_0 = const()[name = string("key_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_77_cast_fp16 = transpose(perm = key_states_77_perm_0, x = key_states_75_cast_fp16)[name = string("transpose_492")]; tensor key_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = key_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = key_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_8_squeeze_mask_0, stride = key_cache_internal_tensor_assign_8_stride_0, update = key_states_77_cast_fp16, x = coreml_update_state_292)[name = string("key_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_8_cast_fp16, input = key_cache)[name = string("coreml_update_state_294_write_state")]; tensor coreml_update_state_294 = read_state(input = key_cache)[name = string("coreml_update_state_294")]; tensor value_states_45_perm_0 = const()[name = string("value_states_45_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_8_stride_0 = const()[name = string("value_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_45_cast_fp16 = transpose(perm = value_states_45_perm_0, x = var_3238_cast_fp16)[name = string("transpose_491")]; tensor value_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = value_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = value_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_8_squeeze_mask_0, stride = value_cache_internal_tensor_assign_8_stride_0, update = value_states_45_cast_fp16, x = coreml_update_state_293)[name = string("value_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_8_cast_fp16, input = value_cache)[name = string("coreml_update_state_295_write_state")]; tensor coreml_update_state_295 = read_state(input = value_cache)[name = string("coreml_update_state_295")]; tensor var_3332_begin_0 = const()[name = string("op_3332_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3332_end_0 = const()[name = string("op_3332_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3332_end_mask_0 = const()[name = string("op_3332_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3332_cast_fp16 = slice_by_index(begin = var_3332_begin_0, end = var_3332_end_0, end_mask = var_3332_end_mask_0, x = coreml_update_state_294)[name = string("op_3332_cast_fp16")]; tensor tile_14 = const()[name = string("tile_14"), val = tensor([1, 1])]; int32 var_3335_axis_0 = const()[name = string("op_3335_axis_0"), val = int32(1)]; tensor var_3335_cast_fp16_0, tensor var_3335_cast_fp16_1 = split(axis = var_3335_axis_0, split_sizes = tile_14, x = var_3332_cast_fp16)[name = string("op_3335_cast_fp16")]; tensor var_3342_begin_0 = const()[name = string("op_3342_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3342_end_0 = const()[name = string("op_3342_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3342_end_mask_0 = const()[name = string("op_3342_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3342_cast_fp16 = slice_by_index(begin = var_3342_begin_0, end = var_3342_end_0, end_mask = var_3342_end_mask_0, x = coreml_update_state_295)[name = string("op_3342_cast_fp16")]; tensor tile_15 = const()[name = string("tile_15"), val = tensor([1, 1])]; int32 var_3345_axis_0 = const()[name = string("op_3345_axis_0"), val = int32(1)]; tensor var_3345_cast_fp16_0, tensor var_3345_cast_fp16_1 = split(axis = var_3345_axis_0, split_sizes = tile_15, x = var_3342_cast_fp16)[name = string("op_3345_cast_fp16")]; tensor var_3348_split_sizes_0 = const()[name = string("op_3348_split_sizes_0"), val = tensor([8, 8])]; int32 var_3348_axis_0 = const()[name = string("op_3348_axis_0"), val = int32(1)]; tensor var_3348_0, tensor var_3348_1 = split(axis = var_3348_axis_0, split_sizes = var_3348_split_sizes_0, x = query_states_45_cast_fp16)[name = string("op_3348")]; bool attn_weights_113_transpose_x_0 = const()[name = string("attn_weights_113_transpose_x_0"), val = bool(false)]; bool attn_weights_113_transpose_y_0 = const()[name = string("attn_weights_113_transpose_y_0"), val = bool(false)]; tensor attn_weights_113_cast_fp16 = matmul(transpose_x = attn_weights_113_transpose_x_0, transpose_y = attn_weights_113_transpose_y_0, x = var_3335_cast_fp16_0, y = var_3348_0)[name = string("attn_weights_113_cast_fp16")]; fp16 var_3351_to_fp16 = const()[name = string("op_3351_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_115_cast_fp16 = mul(x = attn_weights_113_cast_fp16, y = var_3351_to_fp16)[name = string("attn_weights_115_cast_fp16")]; tensor attn_weights_117_cast_fp16 = add(x = attn_weights_115_cast_fp16, y = attn_mask_1)[name = string("attn_weights_117_cast_fp16")]; int32 var_3355 = const()[name = string("op_3355"), val = int32(-2)]; tensor attn_weights_119_cast_fp16 = softmax(axis = var_3355, x = attn_weights_117_cast_fp16)[name = string("attn_weights_119_cast_fp16")]; bool var_3361_transpose_x_1 = const()[name = string("op_3361_transpose_x_1"), val = bool(true)]; bool var_3361_transpose_y_1 = const()[name = string("op_3361_transpose_y_1"), val = bool(false)]; tensor var_3361_cast_fp16 = matmul(transpose_x = var_3361_transpose_x_1, transpose_y = var_3361_transpose_y_1, x = attn_weights_119_cast_fp16, y = var_3345_cast_fp16_0)[name = string("op_3361_cast_fp16")]; bool attn_weights_121_transpose_x_0 = const()[name = string("attn_weights_121_transpose_x_0"), val = bool(false)]; bool attn_weights_121_transpose_y_0 = const()[name = string("attn_weights_121_transpose_y_0"), val = bool(false)]; tensor attn_weights_121_cast_fp16 = matmul(transpose_x = attn_weights_121_transpose_x_0, transpose_y = attn_weights_121_transpose_y_0, x = var_3335_cast_fp16_1, y = var_3348_1)[name = string("attn_weights_121_cast_fp16")]; fp16 var_3363_to_fp16 = const()[name = string("op_3363_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_123_cast_fp16 = mul(x = attn_weights_121_cast_fp16, y = var_3363_to_fp16)[name = string("attn_weights_123_cast_fp16")]; tensor attn_weights_125_cast_fp16 = add(x = attn_weights_123_cast_fp16, y = attn_mask_1)[name = string("attn_weights_125_cast_fp16")]; int32 var_3367 = const()[name = string("op_3367"), val = int32(-2)]; tensor attn_weights_127_cast_fp16 = softmax(axis = var_3367, x = attn_weights_125_cast_fp16)[name = string("attn_weights_127_cast_fp16")]; bool attn_output_57_transpose_x_1 = const()[name = string("attn_output_57_transpose_x_1"), val = bool(true)]; bool attn_output_57_transpose_y_1 = const()[name = string("attn_output_57_transpose_y_1"), val = bool(false)]; tensor attn_output_57_cast_fp16 = matmul(transpose_x = attn_output_57_transpose_x_1, transpose_y = attn_output_57_transpose_y_1, x = attn_weights_127_cast_fp16, y = var_3345_cast_fp16_1)[name = string("attn_output_57_cast_fp16")]; int32 var_3375 = const()[name = string("op_3375"), val = int32(1)]; bool attn_output_59_interleave_0 = const()[name = string("attn_output_59_interleave_0"), val = bool(false)]; tensor attn_output_59_cast_fp16 = concat(axis = var_3375, interleave = attn_output_59_interleave_0, values = (var_3361_cast_fp16, attn_output_57_cast_fp16))[name = string("attn_output_59_cast_fp16")]; tensor var_3379_perm_0 = const()[name = string("op_3379_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_95x = const()[name = string("concat_95x"), val = tensor([1, 2048, 1, -1])]; tensor var_3379_cast_fp16 = transpose(perm = var_3379_perm_0, x = attn_output_59_cast_fp16)[name = string("transpose_490")]; tensor attn_output_63_cast_fp16 = reshape(shape = concat_95x, x = var_3379_cast_fp16)[name = string("attn_output_63_cast_fp16")]; tensor hidden_states_73_strides_0 = const()[name = string("hidden_states_73_strides_0"), val = tensor([1, 1])]; string hidden_states_73_pad_type_0 = const()[name = string("hidden_states_73_pad_type_0"), val = string("valid")]; tensor hidden_states_73_pad_0 = const()[name = string("hidden_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_73_dilations_0 = const()[name = string("hidden_states_73_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_73_groups_0 = const()[name = string("hidden_states_73_groups_0"), val = int32(1)]; tensor hidden_states_73_cast_fp16 = conv(dilations = hidden_states_73_dilations_0, groups = hidden_states_73_groups_0, pad = hidden_states_73_pad_0, pad_type = hidden_states_73_pad_type_0, strides = hidden_states_73_strides_0, weight = layers_7_self_attn_o_proj_weight_cast_fp16, x = attn_output_63_cast_fp16)[name = string("hidden_states_73_cast_fp16")]; tensor hidden_states_75_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = hidden_states_73_cast_fp16)[name = string("hidden_states_75_cast_fp16")]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3412_cast_fp16 = mul(x = hidden_states_75_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_3412_cast_fp16")]; int32 var_3410 = const()[name = string("op_3410"), val = int32(1)]; bool doubled_61_interleave_0 = const()[name = string("doubled_61_interleave_0"), val = bool(false)]; tensor doubled_61_cast_fp16 = concat(axis = var_3410, interleave = doubled_61_interleave_0, values = (hidden_states_75_cast_fp16, var_3412_cast_fp16))[name = string("doubled_61_cast_fp16")]; tensor out_31_axes_0 = const()[name = string("out_31_axes_0"), val = tensor([1])]; tensor out_31_gamma_0_to_fp16 = const()[name = string("out_31_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336656640)))]; fp16 var_3422_to_fp16 = const()[name = string("op_3422_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_31_cast_fp16 = layer_norm(axes = out_31_axes_0, epsilon = var_3422_to_fp16, gamma = out_31_gamma_0_to_fp16, x = doubled_61_cast_fp16)[name = string("out_31_cast_fp16")]; tensor var_3433_split_sizes_0 = const()[name = string("op_3433_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3433_axis_0 = const()[name = string("op_3433_axis_0"), val = int32(1)]; tensor var_3433_cast_fp16_0, tensor var_3433_cast_fp16_1 = split(axis = var_3433_axis_0, split_sizes = var_3433_split_sizes_0, x = out_31_cast_fp16)[name = string("op_3433_cast_fp16")]; tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; tensor input_15_cast_fp16 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = layers_7_mlp_gate_proj_weight_cast_fp16, x = var_3433_cast_fp16_0)[name = string("input_15_cast_fp16")]; tensor var_3450_cast_fp16 = silu(x = input_15_cast_fp16)[name = string("op_3450_cast_fp16")]; tensor layers_7_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336664896)))]; tensor var_3456_strides_0 = const()[name = string("op_3456_strides_0"), val = tensor([1, 1])]; string var_3456_pad_type_0 = const()[name = string("op_3456_pad_type_0"), val = string("valid")]; tensor var_3456_pad_0 = const()[name = string("op_3456_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3456_dilations_0 = const()[name = string("op_3456_dilations_0"), val = tensor([1, 1])]; int32 var_3456_groups_0 = const()[name = string("op_3456_groups_0"), val = int32(1)]; tensor var_3456_cast_fp16 = conv(dilations = var_3456_dilations_0, groups = var_3456_groups_0, pad = var_3456_pad_0, pad_type = var_3456_pad_type_0, strides = var_3456_strides_0, weight = layers_7_mlp_up_proj_weight_to_fp16, x = var_3433_cast_fp16_0)[name = string("op_3456_cast_fp16")]; tensor x_79_cast_fp16 = mul(x = var_3450_cast_fp16, y = var_3456_cast_fp16)[name = string("x_79_cast_fp16")]; tensor layers_7_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1361830784)))]; tensor hidden_states_77_strides_0 = const()[name = string("hidden_states_77_strides_0"), val = tensor([1, 1])]; string hidden_states_77_pad_type_0 = const()[name = string("hidden_states_77_pad_type_0"), val = string("valid")]; tensor hidden_states_77_pad_0 = const()[name = string("hidden_states_77_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_77_dilations_0 = const()[name = string("hidden_states_77_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_77_groups_0 = const()[name = string("hidden_states_77_groups_0"), val = int32(1)]; tensor hidden_states_77_cast_fp16 = conv(dilations = hidden_states_77_dilations_0, groups = hidden_states_77_groups_0, pad = hidden_states_77_pad_0, pad_type = hidden_states_77_pad_type_0, strides = hidden_states_77_strides_0, weight = layers_7_mlp_down_proj_weight_to_fp16, x = x_79_cast_fp16)[name = string("hidden_states_77_cast_fp16")]; tensor hidden_states_79_cast_fp16 = add(x = hidden_states_75_cast_fp16, y = hidden_states_77_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3474_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_80_promoted_to_fp16)[name = string("op_3474_cast_fp16")]; int32 var_3472 = const()[name = string("op_3472"), val = int32(1)]; bool doubled_65_interleave_0 = const()[name = string("doubled_65_interleave_0"), val = bool(false)]; tensor doubled_65_cast_fp16 = concat(axis = var_3472, interleave = doubled_65_interleave_0, values = (hidden_states_79_cast_fp16, var_3474_cast_fp16))[name = string("doubled_65_cast_fp16")]; tensor out_33_axes_0 = const()[name = string("out_33_axes_0"), val = tensor([1])]; tensor out_33_gamma_0_to_fp16 = const()[name = string("out_33_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1386996672)))]; fp16 var_3484_to_fp16 = const()[name = string("op_3484_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_33_cast_fp16 = layer_norm(axes = out_33_axes_0, epsilon = var_3484_to_fp16, gamma = out_33_gamma_0_to_fp16, x = doubled_65_cast_fp16)[name = string("out_33_cast_fp16")]; tensor var_3495_split_sizes_0 = const()[name = string("op_3495_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3495_axis_0 = const()[name = string("op_3495_axis_0"), val = int32(1)]; tensor var_3495_cast_fp16_0, tensor var_3495_cast_fp16_1 = split(axis = var_3495_axis_0, split_sizes = var_3495_split_sizes_0, x = out_33_cast_fp16)[name = string("op_3495_cast_fp16")]; tensor layers_8_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1387004928)))]; tensor query_states_49_strides_0 = const()[name = string("query_states_49_strides_0"), val = tensor([1, 1])]; string query_states_49_pad_type_0 = const()[name = string("query_states_49_pad_type_0"), val = string("valid")]; tensor query_states_49_pad_0 = const()[name = string("query_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_49_dilations_0 = const()[name = string("query_states_49_dilations_0"), val = tensor([1, 1])]; int32 query_states_49_groups_0 = const()[name = string("query_states_49_groups_0"), val = int32(1)]; tensor query_states_49_cast_fp16 = conv(dilations = query_states_49_dilations_0, groups = query_states_49_groups_0, pad = query_states_49_pad_0, pad_type = query_states_49_pad_type_0, strides = query_states_49_strides_0, weight = layers_8_self_attn_q_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("query_states_49_cast_fp16")]; tensor layers_8_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1395393600)))]; tensor key_states_81_strides_0 = const()[name = string("key_states_81_strides_0"), val = tensor([1, 1])]; string key_states_81_pad_type_0 = const()[name = string("key_states_81_pad_type_0"), val = string("valid")]; tensor key_states_81_pad_0 = const()[name = string("key_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_81_dilations_0 = const()[name = string("key_states_81_dilations_0"), val = tensor([1, 1])]; int32 key_states_81_groups_0 = const()[name = string("key_states_81_groups_0"), val = int32(1)]; tensor key_states_81_cast_fp16 = conv(dilations = key_states_81_dilations_0, groups = key_states_81_groups_0, pad = key_states_81_pad_0, pad_type = key_states_81_pad_type_0, strides = key_states_81_strides_0, weight = layers_8_self_attn_k_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("key_states_81_cast_fp16")]; tensor value_states_49_strides_0 = const()[name = string("value_states_49_strides_0"), val = tensor([1, 1])]; string value_states_49_pad_type_0 = const()[name = string("value_states_49_pad_type_0"), val = string("valid")]; tensor value_states_49_pad_0 = const()[name = string("value_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_49_dilations_0 = const()[name = string("value_states_49_dilations_0"), val = tensor([1, 1])]; int32 value_states_49_groups_0 = const()[name = string("value_states_49_groups_0"), val = int32(1)]; tensor value_states_49_cast_fp16 = conv(dilations = value_states_49_dilations_0, groups = value_states_49_groups_0, pad = value_states_49_pad_0, pad_type = value_states_49_pad_type_0, strides = value_states_49_strides_0, weight = layers_8_self_attn_v_proj_weight_cast_fp16, x = var_3495_cast_fp16_0)[name = string("value_states_49_cast_fp16")]; tensor concat_96x = const()[name = string("concat_96x"), val = tensor([1, 16, 128, -1])]; tensor x_81_cast_fp16 = reshape(shape = concat_96x, x = query_states_49_cast_fp16)[name = string("x_81_cast_fp16")]; tensor concat_97x = const()[name = string("concat_97x"), val = tensor([1, 2, 128, -1])]; tensor var_3552_cast_fp16 = reshape(shape = concat_97x, x = key_states_81_cast_fp16)[name = string("op_3552_cast_fp16")]; tensor concat_98x = const()[name = string("concat_98x"), val = tensor([1, 2, 128, -1])]; tensor var_3559_cast_fp16 = reshape(shape = concat_98x, x = value_states_49_cast_fp16)[name = string("op_3559_cast_fp16")]; tensor var_3563_cast_fp16 = mul(x = x_81_cast_fp16, y = var_869_cast_fp16)[name = string("op_3563_cast_fp16")]; tensor var_3564_split_sizes_0 = const()[name = string("op_3564_split_sizes_0"), val = tensor([64, 64])]; int32 var_3564_axis_0 = const()[name = string("op_3564_axis_0"), val = int32(-2)]; tensor var_3564_cast_fp16_0, tensor var_3564_cast_fp16_1 = split(axis = var_3564_axis_0, split_sizes = var_3564_split_sizes_0, x = x_81_cast_fp16)[name = string("op_3564_cast_fp16")]; fp16 const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3566_cast_fp16 = mul(x = var_3564_cast_fp16_1, y = const_82_promoted_to_fp16)[name = string("op_3566_cast_fp16")]; int32 var_3568 = const()[name = string("op_3568"), val = int32(-2)]; bool var_3569_interleave_0 = const()[name = string("op_3569_interleave_0"), val = bool(false)]; tensor var_3569_cast_fp16 = concat(axis = var_3568, interleave = var_3569_interleave_0, values = (var_3566_cast_fp16, var_3564_cast_fp16_0))[name = string("op_3569_cast_fp16")]; tensor var_3570_cast_fp16 = mul(x = var_3569_cast_fp16, y = var_878_cast_fp16)[name = string("op_3570_cast_fp16")]; tensor query_states_51_cast_fp16 = add(x = var_3563_cast_fp16, y = var_3570_cast_fp16)[name = string("query_states_51_cast_fp16")]; tensor var_3576_cast_fp16 = mul(x = var_3552_cast_fp16, y = var_869_cast_fp16)[name = string("op_3576_cast_fp16")]; tensor var_3577_split_sizes_0 = const()[name = string("op_3577_split_sizes_0"), val = tensor([64, 64])]; int32 var_3577_axis_0 = const()[name = string("op_3577_axis_0"), val = int32(-2)]; tensor var_3577_cast_fp16_0, tensor var_3577_cast_fp16_1 = split(axis = var_3577_axis_0, split_sizes = var_3577_split_sizes_0, x = var_3552_cast_fp16)[name = string("op_3577_cast_fp16")]; fp16 const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3579_cast_fp16 = mul(x = var_3577_cast_fp16_1, y = const_83_promoted_to_fp16)[name = string("op_3579_cast_fp16")]; int32 var_3581 = const()[name = string("op_3581"), val = int32(-2)]; bool var_3582_interleave_0 = const()[name = string("op_3582_interleave_0"), val = bool(false)]; tensor var_3582_cast_fp16 = concat(axis = var_3581, interleave = var_3582_interleave_0, values = (var_3579_cast_fp16, var_3577_cast_fp16_0))[name = string("op_3582_cast_fp16")]; tensor var_3583_cast_fp16 = mul(x = var_3582_cast_fp16, y = var_878_cast_fp16)[name = string("op_3583_cast_fp16")]; tensor key_states_85_cast_fp16 = add(x = var_3576_cast_fp16, y = var_3583_cast_fp16)[name = string("key_states_85_cast_fp16")]; tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; int32 concat_101_axis_0 = const()[name = string("concat_101_axis_0"), val = int32(0)]; bool concat_101_interleave_0 = const()[name = string("concat_101_interleave_0"), val = bool(false)]; tensor concat_101 = concat(axis = concat_101_axis_0, interleave = concat_101_interleave_0, values = (expand_dims_96, expand_dims_97, position_id, expand_dims_99))[name = string("concat_101")]; tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; tensor concat_102_values1_0 = const()[name = string("concat_102_values1_0"), val = tensor([0])]; tensor concat_102_values3_0 = const()[name = string("concat_102_values3_0"), val = tensor([0])]; int32 concat_102_axis_0 = const()[name = string("concat_102_axis_0"), val = int32(0)]; bool concat_102_interleave_0 = const()[name = string("concat_102_interleave_0"), val = bool(false)]; tensor concat_102 = concat(axis = concat_102_axis_0, interleave = concat_102_interleave_0, values = (expand_dims_100, concat_102_values1_0, cache_position_end, concat_102_values3_0))[name = string("concat_102")]; tensor key_states_87_perm_0 = const()[name = string("key_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_9_stride_0 = const()[name = string("key_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_87_cast_fp16 = transpose(perm = key_states_87_perm_0, x = key_states_85_cast_fp16)[name = string("transpose_489")]; tensor key_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = key_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = key_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_9_squeeze_mask_0, stride = key_cache_internal_tensor_assign_9_stride_0, update = key_states_87_cast_fp16, x = coreml_update_state_294)[name = string("key_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_9_cast_fp16, input = key_cache)[name = string("coreml_update_state_296_write_state")]; tensor coreml_update_state_296 = read_state(input = key_cache)[name = string("coreml_update_state_296")]; tensor value_states_51_perm_0 = const()[name = string("value_states_51_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_9_stride_0 = const()[name = string("value_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_51_cast_fp16 = transpose(perm = value_states_51_perm_0, x = var_3559_cast_fp16)[name = string("transpose_488")]; tensor value_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = value_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = value_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_9_squeeze_mask_0, stride = value_cache_internal_tensor_assign_9_stride_0, update = value_states_51_cast_fp16, x = coreml_update_state_295)[name = string("value_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_9_cast_fp16, input = value_cache)[name = string("coreml_update_state_297_write_state")]; tensor coreml_update_state_297 = read_state(input = value_cache)[name = string("coreml_update_state_297")]; tensor var_3653_begin_0 = const()[name = string("op_3653_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3653_end_0 = const()[name = string("op_3653_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3653_end_mask_0 = const()[name = string("op_3653_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3653_cast_fp16 = slice_by_index(begin = var_3653_begin_0, end = var_3653_end_0, end_mask = var_3653_end_mask_0, x = coreml_update_state_296)[name = string("op_3653_cast_fp16")]; tensor tile_16 = const()[name = string("tile_16"), val = tensor([1, 1])]; int32 var_3656_axis_0 = const()[name = string("op_3656_axis_0"), val = int32(1)]; tensor var_3656_cast_fp16_0, tensor var_3656_cast_fp16_1 = split(axis = var_3656_axis_0, split_sizes = tile_16, x = var_3653_cast_fp16)[name = string("op_3656_cast_fp16")]; tensor var_3663_begin_0 = const()[name = string("op_3663_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3663_end_0 = const()[name = string("op_3663_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3663_end_mask_0 = const()[name = string("op_3663_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3663_cast_fp16 = slice_by_index(begin = var_3663_begin_0, end = var_3663_end_0, end_mask = var_3663_end_mask_0, x = coreml_update_state_297)[name = string("op_3663_cast_fp16")]; tensor tile_17 = const()[name = string("tile_17"), val = tensor([1, 1])]; int32 var_3666_axis_0 = const()[name = string("op_3666_axis_0"), val = int32(1)]; tensor var_3666_cast_fp16_0, tensor var_3666_cast_fp16_1 = split(axis = var_3666_axis_0, split_sizes = tile_17, x = var_3663_cast_fp16)[name = string("op_3666_cast_fp16")]; tensor var_3669_split_sizes_0 = const()[name = string("op_3669_split_sizes_0"), val = tensor([8, 8])]; int32 var_3669_axis_0 = const()[name = string("op_3669_axis_0"), val = int32(1)]; tensor var_3669_0, tensor var_3669_1 = split(axis = var_3669_axis_0, split_sizes = var_3669_split_sizes_0, x = query_states_51_cast_fp16)[name = string("op_3669")]; bool attn_weights_129_transpose_x_0 = const()[name = string("attn_weights_129_transpose_x_0"), val = bool(false)]; bool attn_weights_129_transpose_y_0 = const()[name = string("attn_weights_129_transpose_y_0"), val = bool(false)]; tensor attn_weights_129_cast_fp16 = matmul(transpose_x = attn_weights_129_transpose_x_0, transpose_y = attn_weights_129_transpose_y_0, x = var_3656_cast_fp16_0, y = var_3669_0)[name = string("attn_weights_129_cast_fp16")]; fp16 var_3672_to_fp16 = const()[name = string("op_3672_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_131_cast_fp16 = mul(x = attn_weights_129_cast_fp16, y = var_3672_to_fp16)[name = string("attn_weights_131_cast_fp16")]; tensor attn_weights_133_cast_fp16 = add(x = attn_weights_131_cast_fp16, y = attn_mask_1)[name = string("attn_weights_133_cast_fp16")]; int32 var_3676 = const()[name = string("op_3676"), val = int32(-2)]; tensor attn_weights_135_cast_fp16 = softmax(axis = var_3676, x = attn_weights_133_cast_fp16)[name = string("attn_weights_135_cast_fp16")]; bool var_3682_transpose_x_1 = const()[name = string("op_3682_transpose_x_1"), val = bool(true)]; bool var_3682_transpose_y_1 = const()[name = string("op_3682_transpose_y_1"), val = bool(false)]; tensor var_3682_cast_fp16 = matmul(transpose_x = var_3682_transpose_x_1, transpose_y = var_3682_transpose_y_1, x = attn_weights_135_cast_fp16, y = var_3666_cast_fp16_0)[name = string("op_3682_cast_fp16")]; bool attn_weights_137_transpose_x_0 = const()[name = string("attn_weights_137_transpose_x_0"), val = bool(false)]; bool attn_weights_137_transpose_y_0 = const()[name = string("attn_weights_137_transpose_y_0"), val = bool(false)]; tensor attn_weights_137_cast_fp16 = matmul(transpose_x = attn_weights_137_transpose_x_0, transpose_y = attn_weights_137_transpose_y_0, x = var_3656_cast_fp16_1, y = var_3669_1)[name = string("attn_weights_137_cast_fp16")]; fp16 var_3684_to_fp16 = const()[name = string("op_3684_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_139_cast_fp16 = mul(x = attn_weights_137_cast_fp16, y = var_3684_to_fp16)[name = string("attn_weights_139_cast_fp16")]; tensor attn_weights_141_cast_fp16 = add(x = attn_weights_139_cast_fp16, y = attn_mask_1)[name = string("attn_weights_141_cast_fp16")]; int32 var_3688 = const()[name = string("op_3688"), val = int32(-2)]; tensor attn_weights_143_cast_fp16 = softmax(axis = var_3688, x = attn_weights_141_cast_fp16)[name = string("attn_weights_143_cast_fp16")]; bool attn_output_65_transpose_x_1 = const()[name = string("attn_output_65_transpose_x_1"), val = bool(true)]; bool attn_output_65_transpose_y_1 = const()[name = string("attn_output_65_transpose_y_1"), val = bool(false)]; tensor attn_output_65_cast_fp16 = matmul(transpose_x = attn_output_65_transpose_x_1, transpose_y = attn_output_65_transpose_y_1, x = attn_weights_143_cast_fp16, y = var_3666_cast_fp16_1)[name = string("attn_output_65_cast_fp16")]; int32 var_3696 = const()[name = string("op_3696"), val = int32(1)]; bool attn_output_67_interleave_0 = const()[name = string("attn_output_67_interleave_0"), val = bool(false)]; tensor attn_output_67_cast_fp16 = concat(axis = var_3696, interleave = attn_output_67_interleave_0, values = (var_3682_cast_fp16, attn_output_65_cast_fp16))[name = string("attn_output_67_cast_fp16")]; tensor var_3700_perm_0 = const()[name = string("op_3700_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_107x = const()[name = string("concat_107x"), val = tensor([1, 2048, 1, -1])]; tensor var_3700_cast_fp16 = transpose(perm = var_3700_perm_0, x = attn_output_67_cast_fp16)[name = string("transpose_487")]; tensor attn_output_71_cast_fp16 = reshape(shape = concat_107x, x = var_3700_cast_fp16)[name = string("attn_output_71_cast_fp16")]; tensor hidden_states_83_strides_0 = const()[name = string("hidden_states_83_strides_0"), val = tensor([1, 1])]; string hidden_states_83_pad_type_0 = const()[name = string("hidden_states_83_pad_type_0"), val = string("valid")]; tensor hidden_states_83_pad_0 = const()[name = string("hidden_states_83_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_83_dilations_0 = const()[name = string("hidden_states_83_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_83_groups_0 = const()[name = string("hidden_states_83_groups_0"), val = int32(1)]; tensor hidden_states_83_cast_fp16 = conv(dilations = hidden_states_83_dilations_0, groups = hidden_states_83_groups_0, pad = hidden_states_83_pad_0, pad_type = hidden_states_83_pad_type_0, strides = hidden_states_83_strides_0, weight = layers_8_self_attn_o_proj_weight_cast_fp16, x = attn_output_71_cast_fp16)[name = string("hidden_states_83_cast_fp16")]; tensor hidden_states_85_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = hidden_states_83_cast_fp16)[name = string("hidden_states_85_cast_fp16")]; fp16 const_88_promoted_to_fp16 = const()[name = string("const_88_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3733_cast_fp16 = mul(x = hidden_states_85_cast_fp16, y = const_88_promoted_to_fp16)[name = string("op_3733_cast_fp16")]; int32 var_3731 = const()[name = string("op_3731"), val = int32(1)]; bool doubled_69_interleave_0 = const()[name = string("doubled_69_interleave_0"), val = bool(false)]; tensor doubled_69_cast_fp16 = concat(axis = var_3731, interleave = doubled_69_interleave_0, values = (hidden_states_85_cast_fp16, var_3733_cast_fp16))[name = string("doubled_69_cast_fp16")]; tensor out_35_axes_0 = const()[name = string("out_35_axes_0"), val = tensor([1])]; tensor out_35_gamma_0_to_fp16 = const()[name = string("out_35_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396442240)))]; fp16 var_3743_to_fp16 = const()[name = string("op_3743_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_35_cast_fp16 = layer_norm(axes = out_35_axes_0, epsilon = var_3743_to_fp16, gamma = out_35_gamma_0_to_fp16, x = doubled_69_cast_fp16)[name = string("out_35_cast_fp16")]; tensor var_3754_split_sizes_0 = const()[name = string("op_3754_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3754_axis_0 = const()[name = string("op_3754_axis_0"), val = int32(1)]; tensor var_3754_cast_fp16_0, tensor var_3754_cast_fp16_1 = split(axis = var_3754_axis_0, split_sizes = var_3754_split_sizes_0, x = out_35_cast_fp16)[name = string("op_3754_cast_fp16")]; tensor input_17_strides_0 = const()[name = string("input_17_strides_0"), val = tensor([1, 1])]; string input_17_pad_type_0 = const()[name = string("input_17_pad_type_0"), val = string("valid")]; tensor input_17_pad_0 = const()[name = string("input_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_17_dilations_0 = const()[name = string("input_17_dilations_0"), val = tensor([1, 1])]; int32 input_17_groups_0 = const()[name = string("input_17_groups_0"), val = int32(1)]; tensor input_17_cast_fp16 = conv(dilations = input_17_dilations_0, groups = input_17_groups_0, pad = input_17_pad_0, pad_type = input_17_pad_type_0, strides = input_17_strides_0, weight = layers_8_mlp_gate_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("input_17_cast_fp16")]; tensor var_3771_cast_fp16 = silu(x = input_17_cast_fp16)[name = string("op_3771_cast_fp16")]; tensor var_3777_strides_0 = const()[name = string("op_3777_strides_0"), val = tensor([1, 1])]; string var_3777_pad_type_0 = const()[name = string("op_3777_pad_type_0"), val = string("valid")]; tensor var_3777_pad_0 = const()[name = string("op_3777_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3777_dilations_0 = const()[name = string("op_3777_dilations_0"), val = tensor([1, 1])]; int32 var_3777_groups_0 = const()[name = string("op_3777_groups_0"), val = int32(1)]; tensor var_3777_cast_fp16 = conv(dilations = var_3777_dilations_0, groups = var_3777_groups_0, pad = var_3777_pad_0, pad_type = var_3777_pad_type_0, strides = var_3777_strides_0, weight = layers_8_mlp_up_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("op_3777_cast_fp16")]; tensor x_89_cast_fp16 = mul(x = var_3771_cast_fp16, y = var_3777_cast_fp16)[name = string("x_89_cast_fp16")]; tensor hidden_states_87_strides_0 = const()[name = string("hidden_states_87_strides_0"), val = tensor([1, 1])]; string hidden_states_87_pad_type_0 = const()[name = string("hidden_states_87_pad_type_0"), val = string("valid")]; tensor hidden_states_87_pad_0 = const()[name = string("hidden_states_87_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_87_dilations_0 = const()[name = string("hidden_states_87_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_87_groups_0 = const()[name = string("hidden_states_87_groups_0"), val = int32(1)]; tensor hidden_states_87_cast_fp16 = conv(dilations = hidden_states_87_dilations_0, groups = hidden_states_87_groups_0, pad = hidden_states_87_pad_0, pad_type = hidden_states_87_pad_type_0, strides = hidden_states_87_strides_0, weight = layers_8_mlp_down_proj_weight_cast_fp16, x = x_89_cast_fp16)[name = string("hidden_states_87_cast_fp16")]; tensor hidden_states_89_cast_fp16 = add(x = hidden_states_85_cast_fp16, y = hidden_states_87_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3795_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_3795_cast_fp16")]; int32 var_3793 = const()[name = string("op_3793"), val = int32(1)]; bool doubled_73_interleave_0 = const()[name = string("doubled_73_interleave_0"), val = bool(false)]; tensor doubled_73_cast_fp16 = concat(axis = var_3793, interleave = doubled_73_interleave_0, values = (hidden_states_89_cast_fp16, var_3795_cast_fp16))[name = string("doubled_73_cast_fp16")]; tensor out_37_axes_0 = const()[name = string("out_37_axes_0"), val = tensor([1])]; tensor out_37_gamma_0_to_fp16 = const()[name = string("out_37_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396450496)))]; fp16 var_3805_to_fp16 = const()[name = string("op_3805_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_37_cast_fp16 = layer_norm(axes = out_37_axes_0, epsilon = var_3805_to_fp16, gamma = out_37_gamma_0_to_fp16, x = doubled_73_cast_fp16)[name = string("out_37_cast_fp16")]; tensor var_3816_split_sizes_0 = const()[name = string("op_3816_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3816_axis_0 = const()[name = string("op_3816_axis_0"), val = int32(1)]; tensor var_3816_cast_fp16_0, tensor var_3816_cast_fp16_1 = split(axis = var_3816_axis_0, split_sizes = var_3816_split_sizes_0, x = out_37_cast_fp16)[name = string("op_3816_cast_fp16")]; tensor layers_9_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396458752)))]; tensor query_states_55_strides_0 = const()[name = string("query_states_55_strides_0"), val = tensor([1, 1])]; string query_states_55_pad_type_0 = const()[name = string("query_states_55_pad_type_0"), val = string("valid")]; tensor query_states_55_pad_0 = const()[name = string("query_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_55_dilations_0 = const()[name = string("query_states_55_dilations_0"), val = tensor([1, 1])]; int32 query_states_55_groups_0 = const()[name = string("query_states_55_groups_0"), val = int32(1)]; tensor query_states_55_cast_fp16 = conv(dilations = query_states_55_dilations_0, groups = query_states_55_groups_0, pad = query_states_55_pad_0, pad_type = query_states_55_pad_type_0, strides = query_states_55_strides_0, weight = layers_9_self_attn_q_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("query_states_55_cast_fp16")]; tensor layers_9_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1404847424)))]; tensor key_states_91_strides_0 = const()[name = string("key_states_91_strides_0"), val = tensor([1, 1])]; string key_states_91_pad_type_0 = const()[name = string("key_states_91_pad_type_0"), val = string("valid")]; tensor key_states_91_pad_0 = const()[name = string("key_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_91_dilations_0 = const()[name = string("key_states_91_dilations_0"), val = tensor([1, 1])]; int32 key_states_91_groups_0 = const()[name = string("key_states_91_groups_0"), val = int32(1)]; tensor key_states_91_cast_fp16 = conv(dilations = key_states_91_dilations_0, groups = key_states_91_groups_0, pad = key_states_91_pad_0, pad_type = key_states_91_pad_type_0, strides = key_states_91_strides_0, weight = layers_9_self_attn_k_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("key_states_91_cast_fp16")]; tensor value_states_55_strides_0 = const()[name = string("value_states_55_strides_0"), val = tensor([1, 1])]; string value_states_55_pad_type_0 = const()[name = string("value_states_55_pad_type_0"), val = string("valid")]; tensor value_states_55_pad_0 = const()[name = string("value_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_55_dilations_0 = const()[name = string("value_states_55_dilations_0"), val = tensor([1, 1])]; int32 value_states_55_groups_0 = const()[name = string("value_states_55_groups_0"), val = int32(1)]; tensor value_states_55_cast_fp16 = conv(dilations = value_states_55_dilations_0, groups = value_states_55_groups_0, pad = value_states_55_pad_0, pad_type = value_states_55_pad_type_0, strides = value_states_55_strides_0, weight = layers_9_self_attn_v_proj_weight_cast_fp16, x = var_3816_cast_fp16_0)[name = string("value_states_55_cast_fp16")]; tensor concat_108x = const()[name = string("concat_108x"), val = tensor([1, 16, 128, -1])]; tensor x_91_cast_fp16 = reshape(shape = concat_108x, x = query_states_55_cast_fp16)[name = string("x_91_cast_fp16")]; tensor concat_109x = const()[name = string("concat_109x"), val = tensor([1, 2, 128, -1])]; tensor var_3873_cast_fp16 = reshape(shape = concat_109x, x = key_states_91_cast_fp16)[name = string("op_3873_cast_fp16")]; tensor concat_110x = const()[name = string("concat_110x"), val = tensor([1, 2, 128, -1])]; tensor var_3880_cast_fp16 = reshape(shape = concat_110x, x = value_states_55_cast_fp16)[name = string("op_3880_cast_fp16")]; tensor var_3884_cast_fp16 = mul(x = x_91_cast_fp16, y = var_869_cast_fp16)[name = string("op_3884_cast_fp16")]; tensor var_3885_split_sizes_0 = const()[name = string("op_3885_split_sizes_0"), val = tensor([64, 64])]; int32 var_3885_axis_0 = const()[name = string("op_3885_axis_0"), val = int32(-2)]; tensor var_3885_cast_fp16_0, tensor var_3885_cast_fp16_1 = split(axis = var_3885_axis_0, split_sizes = var_3885_split_sizes_0, x = x_91_cast_fp16)[name = string("op_3885_cast_fp16")]; fp16 const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3887_cast_fp16 = mul(x = var_3885_cast_fp16_1, y = const_92_promoted_to_fp16)[name = string("op_3887_cast_fp16")]; int32 var_3889 = const()[name = string("op_3889"), val = int32(-2)]; bool var_3890_interleave_0 = const()[name = string("op_3890_interleave_0"), val = bool(false)]; tensor var_3890_cast_fp16 = concat(axis = var_3889, interleave = var_3890_interleave_0, values = (var_3887_cast_fp16, var_3885_cast_fp16_0))[name = string("op_3890_cast_fp16")]; tensor var_3891_cast_fp16 = mul(x = var_3890_cast_fp16, y = var_878_cast_fp16)[name = string("op_3891_cast_fp16")]; tensor query_states_57_cast_fp16 = add(x = var_3884_cast_fp16, y = var_3891_cast_fp16)[name = string("query_states_57_cast_fp16")]; tensor var_3897_cast_fp16 = mul(x = var_3873_cast_fp16, y = var_869_cast_fp16)[name = string("op_3897_cast_fp16")]; tensor var_3898_split_sizes_0 = const()[name = string("op_3898_split_sizes_0"), val = tensor([64, 64])]; int32 var_3898_axis_0 = const()[name = string("op_3898_axis_0"), val = int32(-2)]; tensor var_3898_cast_fp16_0, tensor var_3898_cast_fp16_1 = split(axis = var_3898_axis_0, split_sizes = var_3898_split_sizes_0, x = var_3873_cast_fp16)[name = string("op_3898_cast_fp16")]; fp16 const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3900_cast_fp16 = mul(x = var_3898_cast_fp16_1, y = const_93_promoted_to_fp16)[name = string("op_3900_cast_fp16")]; int32 var_3902 = const()[name = string("op_3902"), val = int32(-2)]; bool var_3903_interleave_0 = const()[name = string("op_3903_interleave_0"), val = bool(false)]; tensor var_3903_cast_fp16 = concat(axis = var_3902, interleave = var_3903_interleave_0, values = (var_3900_cast_fp16, var_3898_cast_fp16_0))[name = string("op_3903_cast_fp16")]; tensor var_3904_cast_fp16 = mul(x = var_3903_cast_fp16, y = var_878_cast_fp16)[name = string("op_3904_cast_fp16")]; tensor key_states_95_cast_fp16 = add(x = var_3897_cast_fp16, y = var_3904_cast_fp16)[name = string("key_states_95_cast_fp16")]; tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; int32 concat_113_axis_0 = const()[name = string("concat_113_axis_0"), val = int32(0)]; bool concat_113_interleave_0 = const()[name = string("concat_113_interleave_0"), val = bool(false)]; tensor concat_113 = concat(axis = concat_113_axis_0, interleave = concat_113_interleave_0, values = (expand_dims_108, expand_dims_109, position_id, expand_dims_111))[name = string("concat_113")]; tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; tensor concat_114_values1_0 = const()[name = string("concat_114_values1_0"), val = tensor([0])]; tensor concat_114_values3_0 = const()[name = string("concat_114_values3_0"), val = tensor([0])]; int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_112, concat_114_values1_0, cache_position_end, concat_114_values3_0))[name = string("concat_114")]; tensor key_states_97_perm_0 = const()[name = string("key_states_97_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_10_stride_0 = const()[name = string("key_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_97_cast_fp16 = transpose(perm = key_states_97_perm_0, x = key_states_95_cast_fp16)[name = string("transpose_486")]; tensor key_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = key_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = key_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_10_squeeze_mask_0, stride = key_cache_internal_tensor_assign_10_stride_0, update = key_states_97_cast_fp16, x = coreml_update_state_296)[name = string("key_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_10_cast_fp16, input = key_cache)[name = string("coreml_update_state_298_write_state")]; tensor coreml_update_state_298 = read_state(input = key_cache)[name = string("coreml_update_state_298")]; tensor value_states_57_perm_0 = const()[name = string("value_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_10_stride_0 = const()[name = string("value_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_57_cast_fp16 = transpose(perm = value_states_57_perm_0, x = var_3880_cast_fp16)[name = string("transpose_485")]; tensor value_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = value_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = value_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_10_squeeze_mask_0, stride = value_cache_internal_tensor_assign_10_stride_0, update = value_states_57_cast_fp16, x = coreml_update_state_297)[name = string("value_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_10_cast_fp16, input = value_cache)[name = string("coreml_update_state_299_write_state")]; tensor coreml_update_state_299 = read_state(input = value_cache)[name = string("coreml_update_state_299")]; tensor var_3974_begin_0 = const()[name = string("op_3974_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3974_end_0 = const()[name = string("op_3974_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3974_end_mask_0 = const()[name = string("op_3974_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3974_cast_fp16 = slice_by_index(begin = var_3974_begin_0, end = var_3974_end_0, end_mask = var_3974_end_mask_0, x = coreml_update_state_298)[name = string("op_3974_cast_fp16")]; tensor tile_18 = const()[name = string("tile_18"), val = tensor([1, 1])]; int32 var_3977_axis_0 = const()[name = string("op_3977_axis_0"), val = int32(1)]; tensor var_3977_cast_fp16_0, tensor var_3977_cast_fp16_1 = split(axis = var_3977_axis_0, split_sizes = tile_18, x = var_3974_cast_fp16)[name = string("op_3977_cast_fp16")]; tensor var_3984_begin_0 = const()[name = string("op_3984_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3984_end_0 = const()[name = string("op_3984_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3984_end_mask_0 = const()[name = string("op_3984_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3984_cast_fp16 = slice_by_index(begin = var_3984_begin_0, end = var_3984_end_0, end_mask = var_3984_end_mask_0, x = coreml_update_state_299)[name = string("op_3984_cast_fp16")]; tensor tile_19 = const()[name = string("tile_19"), val = tensor([1, 1])]; int32 var_3987_axis_0 = const()[name = string("op_3987_axis_0"), val = int32(1)]; tensor var_3987_cast_fp16_0, tensor var_3987_cast_fp16_1 = split(axis = var_3987_axis_0, split_sizes = tile_19, x = var_3984_cast_fp16)[name = string("op_3987_cast_fp16")]; tensor var_3990_split_sizes_0 = const()[name = string("op_3990_split_sizes_0"), val = tensor([8, 8])]; int32 var_3990_axis_0 = const()[name = string("op_3990_axis_0"), val = int32(1)]; tensor var_3990_0, tensor var_3990_1 = split(axis = var_3990_axis_0, split_sizes = var_3990_split_sizes_0, x = query_states_57_cast_fp16)[name = string("op_3990")]; bool attn_weights_145_transpose_x_0 = const()[name = string("attn_weights_145_transpose_x_0"), val = bool(false)]; bool attn_weights_145_transpose_y_0 = const()[name = string("attn_weights_145_transpose_y_0"), val = bool(false)]; tensor attn_weights_145_cast_fp16 = matmul(transpose_x = attn_weights_145_transpose_x_0, transpose_y = attn_weights_145_transpose_y_0, x = var_3977_cast_fp16_0, y = var_3990_0)[name = string("attn_weights_145_cast_fp16")]; fp16 var_3993_to_fp16 = const()[name = string("op_3993_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_147_cast_fp16 = mul(x = attn_weights_145_cast_fp16, y = var_3993_to_fp16)[name = string("attn_weights_147_cast_fp16")]; tensor attn_weights_149_cast_fp16 = add(x = attn_weights_147_cast_fp16, y = attn_mask_1)[name = string("attn_weights_149_cast_fp16")]; int32 var_3997 = const()[name = string("op_3997"), val = int32(-2)]; tensor attn_weights_151_cast_fp16 = softmax(axis = var_3997, x = attn_weights_149_cast_fp16)[name = string("attn_weights_151_cast_fp16")]; bool var_4003_transpose_x_1 = const()[name = string("op_4003_transpose_x_1"), val = bool(true)]; bool var_4003_transpose_y_1 = const()[name = string("op_4003_transpose_y_1"), val = bool(false)]; tensor var_4003_cast_fp16 = matmul(transpose_x = var_4003_transpose_x_1, transpose_y = var_4003_transpose_y_1, x = attn_weights_151_cast_fp16, y = var_3987_cast_fp16_0)[name = string("op_4003_cast_fp16")]; bool attn_weights_153_transpose_x_0 = const()[name = string("attn_weights_153_transpose_x_0"), val = bool(false)]; bool attn_weights_153_transpose_y_0 = const()[name = string("attn_weights_153_transpose_y_0"), val = bool(false)]; tensor attn_weights_153_cast_fp16 = matmul(transpose_x = attn_weights_153_transpose_x_0, transpose_y = attn_weights_153_transpose_y_0, x = var_3977_cast_fp16_1, y = var_3990_1)[name = string("attn_weights_153_cast_fp16")]; fp16 var_4005_to_fp16 = const()[name = string("op_4005_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_155_cast_fp16 = mul(x = attn_weights_153_cast_fp16, y = var_4005_to_fp16)[name = string("attn_weights_155_cast_fp16")]; tensor attn_weights_157_cast_fp16 = add(x = attn_weights_155_cast_fp16, y = attn_mask_1)[name = string("attn_weights_157_cast_fp16")]; int32 var_4009 = const()[name = string("op_4009"), val = int32(-2)]; tensor attn_weights_159_cast_fp16 = softmax(axis = var_4009, x = attn_weights_157_cast_fp16)[name = string("attn_weights_159_cast_fp16")]; bool attn_output_73_transpose_x_1 = const()[name = string("attn_output_73_transpose_x_1"), val = bool(true)]; bool attn_output_73_transpose_y_1 = const()[name = string("attn_output_73_transpose_y_1"), val = bool(false)]; tensor attn_output_73_cast_fp16 = matmul(transpose_x = attn_output_73_transpose_x_1, transpose_y = attn_output_73_transpose_y_1, x = attn_weights_159_cast_fp16, y = var_3987_cast_fp16_1)[name = string("attn_output_73_cast_fp16")]; int32 var_4017 = const()[name = string("op_4017"), val = int32(1)]; bool attn_output_75_interleave_0 = const()[name = string("attn_output_75_interleave_0"), val = bool(false)]; tensor attn_output_75_cast_fp16 = concat(axis = var_4017, interleave = attn_output_75_interleave_0, values = (var_4003_cast_fp16, attn_output_73_cast_fp16))[name = string("attn_output_75_cast_fp16")]; tensor var_4021_perm_0 = const()[name = string("op_4021_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_119x = const()[name = string("concat_119x"), val = tensor([1, 2048, 1, -1])]; tensor var_4021_cast_fp16 = transpose(perm = var_4021_perm_0, x = attn_output_75_cast_fp16)[name = string("transpose_484")]; tensor attn_output_79_cast_fp16 = reshape(shape = concat_119x, x = var_4021_cast_fp16)[name = string("attn_output_79_cast_fp16")]; tensor hidden_states_93_strides_0 = const()[name = string("hidden_states_93_strides_0"), val = tensor([1, 1])]; string hidden_states_93_pad_type_0 = const()[name = string("hidden_states_93_pad_type_0"), val = string("valid")]; tensor hidden_states_93_pad_0 = const()[name = string("hidden_states_93_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_93_dilations_0 = const()[name = string("hidden_states_93_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_93_groups_0 = const()[name = string("hidden_states_93_groups_0"), val = int32(1)]; tensor hidden_states_93_cast_fp16 = conv(dilations = hidden_states_93_dilations_0, groups = hidden_states_93_groups_0, pad = hidden_states_93_pad_0, pad_type = hidden_states_93_pad_type_0, strides = hidden_states_93_strides_0, weight = layers_9_self_attn_o_proj_weight_cast_fp16, x = attn_output_79_cast_fp16)[name = string("hidden_states_93_cast_fp16")]; tensor hidden_states_95_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = hidden_states_93_cast_fp16)[name = string("hidden_states_95_cast_fp16")]; fp16 const_98_promoted_to_fp16 = const()[name = string("const_98_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4054_cast_fp16 = mul(x = hidden_states_95_cast_fp16, y = const_98_promoted_to_fp16)[name = string("op_4054_cast_fp16")]; int32 var_4052 = const()[name = string("op_4052"), val = int32(1)]; bool doubled_77_interleave_0 = const()[name = string("doubled_77_interleave_0"), val = bool(false)]; tensor doubled_77_cast_fp16 = concat(axis = var_4052, interleave = doubled_77_interleave_0, values = (hidden_states_95_cast_fp16, var_4054_cast_fp16))[name = string("doubled_77_cast_fp16")]; tensor out_39_axes_0 = const()[name = string("out_39_axes_0"), val = tensor([1])]; tensor out_39_gamma_0_to_fp16 = const()[name = string("out_39_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405896064)))]; fp16 var_4064_to_fp16 = const()[name = string("op_4064_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_39_cast_fp16 = layer_norm(axes = out_39_axes_0, epsilon = var_4064_to_fp16, gamma = out_39_gamma_0_to_fp16, x = doubled_77_cast_fp16)[name = string("out_39_cast_fp16")]; tensor var_4075_split_sizes_0 = const()[name = string("op_4075_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4075_axis_0 = const()[name = string("op_4075_axis_0"), val = int32(1)]; tensor var_4075_cast_fp16_0, tensor var_4075_cast_fp16_1 = split(axis = var_4075_axis_0, split_sizes = var_4075_split_sizes_0, x = out_39_cast_fp16)[name = string("op_4075_cast_fp16")]; tensor input_19_strides_0 = const()[name = string("input_19_strides_0"), val = tensor([1, 1])]; string input_19_pad_type_0 = const()[name = string("input_19_pad_type_0"), val = string("valid")]; tensor input_19_pad_0 = const()[name = string("input_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_19_dilations_0 = const()[name = string("input_19_dilations_0"), val = tensor([1, 1])]; int32 input_19_groups_0 = const()[name = string("input_19_groups_0"), val = int32(1)]; tensor input_19_cast_fp16 = conv(dilations = input_19_dilations_0, groups = input_19_groups_0, pad = input_19_pad_0, pad_type = input_19_pad_type_0, strides = input_19_strides_0, weight = layers_9_mlp_gate_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("input_19_cast_fp16")]; tensor var_4092_cast_fp16 = silu(x = input_19_cast_fp16)[name = string("op_4092_cast_fp16")]; tensor var_4098_strides_0 = const()[name = string("op_4098_strides_0"), val = tensor([1, 1])]; string var_4098_pad_type_0 = const()[name = string("op_4098_pad_type_0"), val = string("valid")]; tensor var_4098_pad_0 = const()[name = string("op_4098_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4098_dilations_0 = const()[name = string("op_4098_dilations_0"), val = tensor([1, 1])]; int32 var_4098_groups_0 = const()[name = string("op_4098_groups_0"), val = int32(1)]; tensor var_4098_cast_fp16 = conv(dilations = var_4098_dilations_0, groups = var_4098_groups_0, pad = var_4098_pad_0, pad_type = var_4098_pad_type_0, strides = var_4098_strides_0, weight = layers_9_mlp_up_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("op_4098_cast_fp16")]; tensor x_99_cast_fp16 = mul(x = var_4092_cast_fp16, y = var_4098_cast_fp16)[name = string("x_99_cast_fp16")]; tensor hidden_states_97_strides_0 = const()[name = string("hidden_states_97_strides_0"), val = tensor([1, 1])]; string hidden_states_97_pad_type_0 = const()[name = string("hidden_states_97_pad_type_0"), val = string("valid")]; tensor hidden_states_97_pad_0 = const()[name = string("hidden_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_97_dilations_0 = const()[name = string("hidden_states_97_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_97_groups_0 = const()[name = string("hidden_states_97_groups_0"), val = int32(1)]; tensor hidden_states_97_cast_fp16 = conv(dilations = hidden_states_97_dilations_0, groups = hidden_states_97_groups_0, pad = hidden_states_97_pad_0, pad_type = hidden_states_97_pad_type_0, strides = hidden_states_97_strides_0, weight = layers_9_mlp_down_proj_weight_cast_fp16, x = x_99_cast_fp16)[name = string("hidden_states_97_cast_fp16")]; tensor hidden_states_99_cast_fp16 = add(x = hidden_states_95_cast_fp16, y = hidden_states_97_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; fp16 const_100_promoted_to_fp16 = const()[name = string("const_100_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4116_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_100_promoted_to_fp16)[name = string("op_4116_cast_fp16")]; int32 var_4114 = const()[name = string("op_4114"), val = int32(1)]; bool doubled_81_interleave_0 = const()[name = string("doubled_81_interleave_0"), val = bool(false)]; tensor doubled_81_cast_fp16 = concat(axis = var_4114, interleave = doubled_81_interleave_0, values = (hidden_states_99_cast_fp16, var_4116_cast_fp16))[name = string("doubled_81_cast_fp16")]; tensor out_41_axes_0 = const()[name = string("out_41_axes_0"), val = tensor([1])]; tensor out_41_gamma_0_to_fp16 = const()[name = string("out_41_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405904320)))]; fp16 var_4126_to_fp16 = const()[name = string("op_4126_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_41_cast_fp16 = layer_norm(axes = out_41_axes_0, epsilon = var_4126_to_fp16, gamma = out_41_gamma_0_to_fp16, x = doubled_81_cast_fp16)[name = string("out_41_cast_fp16")]; tensor var_4137_split_sizes_0 = const()[name = string("op_4137_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4137_axis_0 = const()[name = string("op_4137_axis_0"), val = int32(1)]; tensor var_4137_cast_fp16_0, tensor var_4137_cast_fp16_1 = split(axis = var_4137_axis_0, split_sizes = var_4137_split_sizes_0, x = out_41_cast_fp16)[name = string("op_4137_cast_fp16")]; tensor layers_10_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405912576)))]; tensor query_states_61_strides_0 = const()[name = string("query_states_61_strides_0"), val = tensor([1, 1])]; string query_states_61_pad_type_0 = const()[name = string("query_states_61_pad_type_0"), val = string("valid")]; tensor query_states_61_pad_0 = const()[name = string("query_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_61_dilations_0 = const()[name = string("query_states_61_dilations_0"), val = tensor([1, 1])]; int32 query_states_61_groups_0 = const()[name = string("query_states_61_groups_0"), val = int32(1)]; tensor query_states_61_cast_fp16 = conv(dilations = query_states_61_dilations_0, groups = query_states_61_groups_0, pad = query_states_61_pad_0, pad_type = query_states_61_pad_type_0, strides = query_states_61_strides_0, weight = layers_10_self_attn_q_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("query_states_61_cast_fp16")]; tensor layers_10_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1414301248)))]; tensor key_states_101_strides_0 = const()[name = string("key_states_101_strides_0"), val = tensor([1, 1])]; string key_states_101_pad_type_0 = const()[name = string("key_states_101_pad_type_0"), val = string("valid")]; tensor key_states_101_pad_0 = const()[name = string("key_states_101_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_101_dilations_0 = const()[name = string("key_states_101_dilations_0"), val = tensor([1, 1])]; int32 key_states_101_groups_0 = const()[name = string("key_states_101_groups_0"), val = int32(1)]; tensor key_states_101_cast_fp16 = conv(dilations = key_states_101_dilations_0, groups = key_states_101_groups_0, pad = key_states_101_pad_0, pad_type = key_states_101_pad_type_0, strides = key_states_101_strides_0, weight = layers_10_self_attn_k_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("key_states_101_cast_fp16")]; tensor value_states_61_strides_0 = const()[name = string("value_states_61_strides_0"), val = tensor([1, 1])]; string value_states_61_pad_type_0 = const()[name = string("value_states_61_pad_type_0"), val = string("valid")]; tensor value_states_61_pad_0 = const()[name = string("value_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_61_dilations_0 = const()[name = string("value_states_61_dilations_0"), val = tensor([1, 1])]; int32 value_states_61_groups_0 = const()[name = string("value_states_61_groups_0"), val = int32(1)]; tensor value_states_61_cast_fp16 = conv(dilations = value_states_61_dilations_0, groups = value_states_61_groups_0, pad = value_states_61_pad_0, pad_type = value_states_61_pad_type_0, strides = value_states_61_strides_0, weight = layers_10_self_attn_v_proj_weight_cast_fp16, x = var_4137_cast_fp16_0)[name = string("value_states_61_cast_fp16")]; tensor concat_120x = const()[name = string("concat_120x"), val = tensor([1, 16, 128, -1])]; tensor x_101_cast_fp16 = reshape(shape = concat_120x, x = query_states_61_cast_fp16)[name = string("x_101_cast_fp16")]; tensor concat_121x = const()[name = string("concat_121x"), val = tensor([1, 2, 128, -1])]; tensor var_4194_cast_fp16 = reshape(shape = concat_121x, x = key_states_101_cast_fp16)[name = string("op_4194_cast_fp16")]; tensor concat_122x = const()[name = string("concat_122x"), val = tensor([1, 2, 128, -1])]; tensor var_4201_cast_fp16 = reshape(shape = concat_122x, x = value_states_61_cast_fp16)[name = string("op_4201_cast_fp16")]; tensor var_4205_cast_fp16 = mul(x = x_101_cast_fp16, y = var_869_cast_fp16)[name = string("op_4205_cast_fp16")]; tensor var_4206_split_sizes_0 = const()[name = string("op_4206_split_sizes_0"), val = tensor([64, 64])]; int32 var_4206_axis_0 = const()[name = string("op_4206_axis_0"), val = int32(-2)]; tensor var_4206_cast_fp16_0, tensor var_4206_cast_fp16_1 = split(axis = var_4206_axis_0, split_sizes = var_4206_split_sizes_0, x = x_101_cast_fp16)[name = string("op_4206_cast_fp16")]; fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4208_cast_fp16 = mul(x = var_4206_cast_fp16_1, y = const_102_promoted_to_fp16)[name = string("op_4208_cast_fp16")]; int32 var_4210 = const()[name = string("op_4210"), val = int32(-2)]; bool var_4211_interleave_0 = const()[name = string("op_4211_interleave_0"), val = bool(false)]; tensor var_4211_cast_fp16 = concat(axis = var_4210, interleave = var_4211_interleave_0, values = (var_4208_cast_fp16, var_4206_cast_fp16_0))[name = string("op_4211_cast_fp16")]; tensor var_4212_cast_fp16 = mul(x = var_4211_cast_fp16, y = var_878_cast_fp16)[name = string("op_4212_cast_fp16")]; tensor query_states_63_cast_fp16 = add(x = var_4205_cast_fp16, y = var_4212_cast_fp16)[name = string("query_states_63_cast_fp16")]; tensor var_4218_cast_fp16 = mul(x = var_4194_cast_fp16, y = var_869_cast_fp16)[name = string("op_4218_cast_fp16")]; tensor var_4219_split_sizes_0 = const()[name = string("op_4219_split_sizes_0"), val = tensor([64, 64])]; int32 var_4219_axis_0 = const()[name = string("op_4219_axis_0"), val = int32(-2)]; tensor var_4219_cast_fp16_0, tensor var_4219_cast_fp16_1 = split(axis = var_4219_axis_0, split_sizes = var_4219_split_sizes_0, x = var_4194_cast_fp16)[name = string("op_4219_cast_fp16")]; fp16 const_103_promoted_to_fp16 = const()[name = string("const_103_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4221_cast_fp16 = mul(x = var_4219_cast_fp16_1, y = const_103_promoted_to_fp16)[name = string("op_4221_cast_fp16")]; int32 var_4223 = const()[name = string("op_4223"), val = int32(-2)]; bool var_4224_interleave_0 = const()[name = string("op_4224_interleave_0"), val = bool(false)]; tensor var_4224_cast_fp16 = concat(axis = var_4223, interleave = var_4224_interleave_0, values = (var_4221_cast_fp16, var_4219_cast_fp16_0))[name = string("op_4224_cast_fp16")]; tensor var_4225_cast_fp16 = mul(x = var_4224_cast_fp16, y = var_878_cast_fp16)[name = string("op_4225_cast_fp16")]; tensor key_states_105_cast_fp16 = add(x = var_4218_cast_fp16, y = var_4225_cast_fp16)[name = string("key_states_105_cast_fp16")]; tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; int32 concat_125_axis_0 = const()[name = string("concat_125_axis_0"), val = int32(0)]; bool concat_125_interleave_0 = const()[name = string("concat_125_interleave_0"), val = bool(false)]; tensor concat_125 = concat(axis = concat_125_axis_0, interleave = concat_125_interleave_0, values = (expand_dims_120, expand_dims_121, position_id, expand_dims_123))[name = string("concat_125")]; tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; tensor concat_126_values1_0 = const()[name = string("concat_126_values1_0"), val = tensor([0])]; tensor concat_126_values3_0 = const()[name = string("concat_126_values3_0"), val = tensor([0])]; int32 concat_126_axis_0 = const()[name = string("concat_126_axis_0"), val = int32(0)]; bool concat_126_interleave_0 = const()[name = string("concat_126_interleave_0"), val = bool(false)]; tensor concat_126 = concat(axis = concat_126_axis_0, interleave = concat_126_interleave_0, values = (expand_dims_124, concat_126_values1_0, cache_position_end, concat_126_values3_0))[name = string("concat_126")]; tensor key_states_107_perm_0 = const()[name = string("key_states_107_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_11_stride_0 = const()[name = string("key_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_107_cast_fp16 = transpose(perm = key_states_107_perm_0, x = key_states_105_cast_fp16)[name = string("transpose_483")]; tensor key_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = key_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = key_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_11_squeeze_mask_0, stride = key_cache_internal_tensor_assign_11_stride_0, update = key_states_107_cast_fp16, x = coreml_update_state_298)[name = string("key_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_11_cast_fp16, input = key_cache)[name = string("coreml_update_state_300_write_state")]; tensor coreml_update_state_300 = read_state(input = key_cache)[name = string("coreml_update_state_300")]; tensor value_states_63_perm_0 = const()[name = string("value_states_63_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_11_stride_0 = const()[name = string("value_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_63_cast_fp16 = transpose(perm = value_states_63_perm_0, x = var_4201_cast_fp16)[name = string("transpose_482")]; tensor value_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = value_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = value_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_11_squeeze_mask_0, stride = value_cache_internal_tensor_assign_11_stride_0, update = value_states_63_cast_fp16, x = coreml_update_state_299)[name = string("value_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_11_cast_fp16, input = value_cache)[name = string("coreml_update_state_301_write_state")]; tensor coreml_update_state_301 = read_state(input = value_cache)[name = string("coreml_update_state_301")]; tensor var_4295_begin_0 = const()[name = string("op_4295_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4295_end_0 = const()[name = string("op_4295_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4295_end_mask_0 = const()[name = string("op_4295_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4295_cast_fp16 = slice_by_index(begin = var_4295_begin_0, end = var_4295_end_0, end_mask = var_4295_end_mask_0, x = coreml_update_state_300)[name = string("op_4295_cast_fp16")]; tensor tile_20 = const()[name = string("tile_20"), val = tensor([1, 1])]; int32 var_4298_axis_0 = const()[name = string("op_4298_axis_0"), val = int32(1)]; tensor var_4298_cast_fp16_0, tensor var_4298_cast_fp16_1 = split(axis = var_4298_axis_0, split_sizes = tile_20, x = var_4295_cast_fp16)[name = string("op_4298_cast_fp16")]; tensor var_4305_begin_0 = const()[name = string("op_4305_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4305_end_0 = const()[name = string("op_4305_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4305_end_mask_0 = const()[name = string("op_4305_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4305_cast_fp16 = slice_by_index(begin = var_4305_begin_0, end = var_4305_end_0, end_mask = var_4305_end_mask_0, x = coreml_update_state_301)[name = string("op_4305_cast_fp16")]; tensor tile_21 = const()[name = string("tile_21"), val = tensor([1, 1])]; int32 var_4308_axis_0 = const()[name = string("op_4308_axis_0"), val = int32(1)]; tensor var_4308_cast_fp16_0, tensor var_4308_cast_fp16_1 = split(axis = var_4308_axis_0, split_sizes = tile_21, x = var_4305_cast_fp16)[name = string("op_4308_cast_fp16")]; tensor var_4311_split_sizes_0 = const()[name = string("op_4311_split_sizes_0"), val = tensor([8, 8])]; int32 var_4311_axis_0 = const()[name = string("op_4311_axis_0"), val = int32(1)]; tensor var_4311_0, tensor var_4311_1 = split(axis = var_4311_axis_0, split_sizes = var_4311_split_sizes_0, x = query_states_63_cast_fp16)[name = string("op_4311")]; bool attn_weights_161_transpose_x_0 = const()[name = string("attn_weights_161_transpose_x_0"), val = bool(false)]; bool attn_weights_161_transpose_y_0 = const()[name = string("attn_weights_161_transpose_y_0"), val = bool(false)]; tensor attn_weights_161_cast_fp16 = matmul(transpose_x = attn_weights_161_transpose_x_0, transpose_y = attn_weights_161_transpose_y_0, x = var_4298_cast_fp16_0, y = var_4311_0)[name = string("attn_weights_161_cast_fp16")]; fp16 var_4314_to_fp16 = const()[name = string("op_4314_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_163_cast_fp16 = mul(x = attn_weights_161_cast_fp16, y = var_4314_to_fp16)[name = string("attn_weights_163_cast_fp16")]; tensor attn_weights_165_cast_fp16 = add(x = attn_weights_163_cast_fp16, y = attn_mask_1)[name = string("attn_weights_165_cast_fp16")]; int32 var_4318 = const()[name = string("op_4318"), val = int32(-2)]; tensor attn_weights_167_cast_fp16 = softmax(axis = var_4318, x = attn_weights_165_cast_fp16)[name = string("attn_weights_167_cast_fp16")]; bool var_4324_transpose_x_1 = const()[name = string("op_4324_transpose_x_1"), val = bool(true)]; bool var_4324_transpose_y_1 = const()[name = string("op_4324_transpose_y_1"), val = bool(false)]; tensor var_4324_cast_fp16 = matmul(transpose_x = var_4324_transpose_x_1, transpose_y = var_4324_transpose_y_1, x = attn_weights_167_cast_fp16, y = var_4308_cast_fp16_0)[name = string("op_4324_cast_fp16")]; bool attn_weights_169_transpose_x_0 = const()[name = string("attn_weights_169_transpose_x_0"), val = bool(false)]; bool attn_weights_169_transpose_y_0 = const()[name = string("attn_weights_169_transpose_y_0"), val = bool(false)]; tensor attn_weights_169_cast_fp16 = matmul(transpose_x = attn_weights_169_transpose_x_0, transpose_y = attn_weights_169_transpose_y_0, x = var_4298_cast_fp16_1, y = var_4311_1)[name = string("attn_weights_169_cast_fp16")]; fp16 var_4326_to_fp16 = const()[name = string("op_4326_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_171_cast_fp16 = mul(x = attn_weights_169_cast_fp16, y = var_4326_to_fp16)[name = string("attn_weights_171_cast_fp16")]; tensor attn_weights_173_cast_fp16 = add(x = attn_weights_171_cast_fp16, y = attn_mask_1)[name = string("attn_weights_173_cast_fp16")]; int32 var_4330 = const()[name = string("op_4330"), val = int32(-2)]; tensor attn_weights_175_cast_fp16 = softmax(axis = var_4330, x = attn_weights_173_cast_fp16)[name = string("attn_weights_175_cast_fp16")]; bool attn_output_81_transpose_x_1 = const()[name = string("attn_output_81_transpose_x_1"), val = bool(true)]; bool attn_output_81_transpose_y_1 = const()[name = string("attn_output_81_transpose_y_1"), val = bool(false)]; tensor attn_output_81_cast_fp16 = matmul(transpose_x = attn_output_81_transpose_x_1, transpose_y = attn_output_81_transpose_y_1, x = attn_weights_175_cast_fp16, y = var_4308_cast_fp16_1)[name = string("attn_output_81_cast_fp16")]; int32 var_4338 = const()[name = string("op_4338"), val = int32(1)]; bool attn_output_83_interleave_0 = const()[name = string("attn_output_83_interleave_0"), val = bool(false)]; tensor attn_output_83_cast_fp16 = concat(axis = var_4338, interleave = attn_output_83_interleave_0, values = (var_4324_cast_fp16, attn_output_81_cast_fp16))[name = string("attn_output_83_cast_fp16")]; tensor var_4342_perm_0 = const()[name = string("op_4342_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_131x = const()[name = string("concat_131x"), val = tensor([1, 2048, 1, -1])]; tensor var_4342_cast_fp16 = transpose(perm = var_4342_perm_0, x = attn_output_83_cast_fp16)[name = string("transpose_481")]; tensor attn_output_87_cast_fp16 = reshape(shape = concat_131x, x = var_4342_cast_fp16)[name = string("attn_output_87_cast_fp16")]; tensor hidden_states_103_strides_0 = const()[name = string("hidden_states_103_strides_0"), val = tensor([1, 1])]; string hidden_states_103_pad_type_0 = const()[name = string("hidden_states_103_pad_type_0"), val = string("valid")]; tensor hidden_states_103_pad_0 = const()[name = string("hidden_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_103_dilations_0 = const()[name = string("hidden_states_103_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_103_groups_0 = const()[name = string("hidden_states_103_groups_0"), val = int32(1)]; tensor hidden_states_103_cast_fp16 = conv(dilations = hidden_states_103_dilations_0, groups = hidden_states_103_groups_0, pad = hidden_states_103_pad_0, pad_type = hidden_states_103_pad_type_0, strides = hidden_states_103_strides_0, weight = layers_10_self_attn_o_proj_weight_cast_fp16, x = attn_output_87_cast_fp16)[name = string("hidden_states_103_cast_fp16")]; tensor hidden_states_105_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = hidden_states_103_cast_fp16)[name = string("hidden_states_105_cast_fp16")]; fp16 const_108_promoted_to_fp16 = const()[name = string("const_108_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4375_cast_fp16 = mul(x = hidden_states_105_cast_fp16, y = const_108_promoted_to_fp16)[name = string("op_4375_cast_fp16")]; int32 var_4373 = const()[name = string("op_4373"), val = int32(1)]; bool doubled_85_interleave_0 = const()[name = string("doubled_85_interleave_0"), val = bool(false)]; tensor doubled_85_cast_fp16 = concat(axis = var_4373, interleave = doubled_85_interleave_0, values = (hidden_states_105_cast_fp16, var_4375_cast_fp16))[name = string("doubled_85_cast_fp16")]; tensor out_43_axes_0 = const()[name = string("out_43_axes_0"), val = tensor([1])]; tensor out_43_gamma_0_to_fp16 = const()[name = string("out_43_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415349888)))]; fp16 var_4385_to_fp16 = const()[name = string("op_4385_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_43_cast_fp16 = layer_norm(axes = out_43_axes_0, epsilon = var_4385_to_fp16, gamma = out_43_gamma_0_to_fp16, x = doubled_85_cast_fp16)[name = string("out_43_cast_fp16")]; tensor var_4396_split_sizes_0 = const()[name = string("op_4396_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4396_axis_0 = const()[name = string("op_4396_axis_0"), val = int32(1)]; tensor var_4396_cast_fp16_0, tensor var_4396_cast_fp16_1 = split(axis = var_4396_axis_0, split_sizes = var_4396_split_sizes_0, x = out_43_cast_fp16)[name = string("op_4396_cast_fp16")]; tensor input_21_strides_0 = const()[name = string("input_21_strides_0"), val = tensor([1, 1])]; string input_21_pad_type_0 = const()[name = string("input_21_pad_type_0"), val = string("valid")]; tensor input_21_pad_0 = const()[name = string("input_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_21_dilations_0 = const()[name = string("input_21_dilations_0"), val = tensor([1, 1])]; int32 input_21_groups_0 = const()[name = string("input_21_groups_0"), val = int32(1)]; tensor input_21_cast_fp16 = conv(dilations = input_21_dilations_0, groups = input_21_groups_0, pad = input_21_pad_0, pad_type = input_21_pad_type_0, strides = input_21_strides_0, weight = layers_10_mlp_gate_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("input_21_cast_fp16")]; tensor var_4413_cast_fp16 = silu(x = input_21_cast_fp16)[name = string("op_4413_cast_fp16")]; tensor var_4419_strides_0 = const()[name = string("op_4419_strides_0"), val = tensor([1, 1])]; string var_4419_pad_type_0 = const()[name = string("op_4419_pad_type_0"), val = string("valid")]; tensor var_4419_pad_0 = const()[name = string("op_4419_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4419_dilations_0 = const()[name = string("op_4419_dilations_0"), val = tensor([1, 1])]; int32 var_4419_groups_0 = const()[name = string("op_4419_groups_0"), val = int32(1)]; tensor var_4419_cast_fp16 = conv(dilations = var_4419_dilations_0, groups = var_4419_groups_0, pad = var_4419_pad_0, pad_type = var_4419_pad_type_0, strides = var_4419_strides_0, weight = layers_10_mlp_up_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("op_4419_cast_fp16")]; tensor x_109_cast_fp16 = mul(x = var_4413_cast_fp16, y = var_4419_cast_fp16)[name = string("x_109_cast_fp16")]; tensor hidden_states_107_strides_0 = const()[name = string("hidden_states_107_strides_0"), val = tensor([1, 1])]; string hidden_states_107_pad_type_0 = const()[name = string("hidden_states_107_pad_type_0"), val = string("valid")]; tensor hidden_states_107_pad_0 = const()[name = string("hidden_states_107_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_107_dilations_0 = const()[name = string("hidden_states_107_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_107_groups_0 = const()[name = string("hidden_states_107_groups_0"), val = int32(1)]; tensor hidden_states_107_cast_fp16 = conv(dilations = hidden_states_107_dilations_0, groups = hidden_states_107_groups_0, pad = hidden_states_107_pad_0, pad_type = hidden_states_107_pad_type_0, strides = hidden_states_107_strides_0, weight = layers_10_mlp_down_proj_weight_cast_fp16, x = x_109_cast_fp16)[name = string("hidden_states_107_cast_fp16")]; tensor hidden_states_109_cast_fp16 = add(x = hidden_states_105_cast_fp16, y = hidden_states_107_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; fp16 const_110_promoted_to_fp16 = const()[name = string("const_110_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4437_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_110_promoted_to_fp16)[name = string("op_4437_cast_fp16")]; int32 var_4435 = const()[name = string("op_4435"), val = int32(1)]; bool doubled_89_interleave_0 = const()[name = string("doubled_89_interleave_0"), val = bool(false)]; tensor doubled_89_cast_fp16 = concat(axis = var_4435, interleave = doubled_89_interleave_0, values = (hidden_states_109_cast_fp16, var_4437_cast_fp16))[name = string("doubled_89_cast_fp16")]; tensor out_45_axes_0 = const()[name = string("out_45_axes_0"), val = tensor([1])]; tensor out_45_gamma_0_to_fp16 = const()[name = string("out_45_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415358144)))]; fp16 var_4447_to_fp16 = const()[name = string("op_4447_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_45_cast_fp16 = layer_norm(axes = out_45_axes_0, epsilon = var_4447_to_fp16, gamma = out_45_gamma_0_to_fp16, x = doubled_89_cast_fp16)[name = string("out_45_cast_fp16")]; tensor var_4458_split_sizes_0 = const()[name = string("op_4458_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4458_axis_0 = const()[name = string("op_4458_axis_0"), val = int32(1)]; tensor var_4458_cast_fp16_0, tensor var_4458_cast_fp16_1 = split(axis = var_4458_axis_0, split_sizes = var_4458_split_sizes_0, x = out_45_cast_fp16)[name = string("op_4458_cast_fp16")]; tensor query_states_67_strides_0 = const()[name = string("query_states_67_strides_0"), val = tensor([1, 1])]; string query_states_67_pad_type_0 = const()[name = string("query_states_67_pad_type_0"), val = string("valid")]; tensor query_states_67_pad_0 = const()[name = string("query_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_67_dilations_0 = const()[name = string("query_states_67_dilations_0"), val = tensor([1, 1])]; int32 query_states_67_groups_0 = const()[name = string("query_states_67_groups_0"), val = int32(1)]; tensor query_states_67_cast_fp16 = conv(dilations = query_states_67_dilations_0, groups = query_states_67_groups_0, pad = query_states_67_pad_0, pad_type = query_states_67_pad_type_0, strides = query_states_67_strides_0, weight = layers_11_self_attn_q_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("query_states_67_cast_fp16")]; tensor key_states_111_strides_0 = const()[name = string("key_states_111_strides_0"), val = tensor([1, 1])]; string key_states_111_pad_type_0 = const()[name = string("key_states_111_pad_type_0"), val = string("valid")]; tensor key_states_111_pad_0 = const()[name = string("key_states_111_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_111_dilations_0 = const()[name = string("key_states_111_dilations_0"), val = tensor([1, 1])]; int32 key_states_111_groups_0 = const()[name = string("key_states_111_groups_0"), val = int32(1)]; tensor key_states_111_cast_fp16 = conv(dilations = key_states_111_dilations_0, groups = key_states_111_groups_0, pad = key_states_111_pad_0, pad_type = key_states_111_pad_type_0, strides = key_states_111_strides_0, weight = layers_11_self_attn_k_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("key_states_111_cast_fp16")]; tensor value_states_67_strides_0 = const()[name = string("value_states_67_strides_0"), val = tensor([1, 1])]; string value_states_67_pad_type_0 = const()[name = string("value_states_67_pad_type_0"), val = string("valid")]; tensor value_states_67_pad_0 = const()[name = string("value_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_67_dilations_0 = const()[name = string("value_states_67_dilations_0"), val = tensor([1, 1])]; int32 value_states_67_groups_0 = const()[name = string("value_states_67_groups_0"), val = int32(1)]; tensor value_states_67_cast_fp16 = conv(dilations = value_states_67_dilations_0, groups = value_states_67_groups_0, pad = value_states_67_pad_0, pad_type = value_states_67_pad_type_0, strides = value_states_67_strides_0, weight = layers_11_self_attn_v_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("value_states_67_cast_fp16")]; tensor concat_132x = const()[name = string("concat_132x"), val = tensor([1, 16, 128, -1])]; tensor x_111_cast_fp16 = reshape(shape = concat_132x, x = query_states_67_cast_fp16)[name = string("x_111_cast_fp16")]; tensor concat_133x = const()[name = string("concat_133x"), val = tensor([1, 2, 128, -1])]; tensor var_4515_cast_fp16 = reshape(shape = concat_133x, x = key_states_111_cast_fp16)[name = string("op_4515_cast_fp16")]; tensor concat_134x = const()[name = string("concat_134x"), val = tensor([1, 2, 128, -1])]; tensor var_4522_cast_fp16 = reshape(shape = concat_134x, x = value_states_67_cast_fp16)[name = string("op_4522_cast_fp16")]; tensor var_4526_cast_fp16 = mul(x = x_111_cast_fp16, y = var_869_cast_fp16)[name = string("op_4526_cast_fp16")]; tensor var_4527_split_sizes_0 = const()[name = string("op_4527_split_sizes_0"), val = tensor([64, 64])]; int32 var_4527_axis_0 = const()[name = string("op_4527_axis_0"), val = int32(-2)]; tensor var_4527_cast_fp16_0, tensor var_4527_cast_fp16_1 = split(axis = var_4527_axis_0, split_sizes = var_4527_split_sizes_0, x = x_111_cast_fp16)[name = string("op_4527_cast_fp16")]; fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4529_cast_fp16 = mul(x = var_4527_cast_fp16_1, y = const_112_promoted_to_fp16)[name = string("op_4529_cast_fp16")]; int32 var_4531 = const()[name = string("op_4531"), val = int32(-2)]; bool var_4532_interleave_0 = const()[name = string("op_4532_interleave_0"), val = bool(false)]; tensor var_4532_cast_fp16 = concat(axis = var_4531, interleave = var_4532_interleave_0, values = (var_4529_cast_fp16, var_4527_cast_fp16_0))[name = string("op_4532_cast_fp16")]; tensor var_4533_cast_fp16 = mul(x = var_4532_cast_fp16, y = var_878_cast_fp16)[name = string("op_4533_cast_fp16")]; tensor query_states_69_cast_fp16 = add(x = var_4526_cast_fp16, y = var_4533_cast_fp16)[name = string("query_states_69_cast_fp16")]; tensor var_4539_cast_fp16 = mul(x = var_4515_cast_fp16, y = var_869_cast_fp16)[name = string("op_4539_cast_fp16")]; tensor var_4540_split_sizes_0 = const()[name = string("op_4540_split_sizes_0"), val = tensor([64, 64])]; int32 var_4540_axis_0 = const()[name = string("op_4540_axis_0"), val = int32(-2)]; tensor var_4540_cast_fp16_0, tensor var_4540_cast_fp16_1 = split(axis = var_4540_axis_0, split_sizes = var_4540_split_sizes_0, x = var_4515_cast_fp16)[name = string("op_4540_cast_fp16")]; fp16 const_113_promoted_to_fp16 = const()[name = string("const_113_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4542_cast_fp16 = mul(x = var_4540_cast_fp16_1, y = const_113_promoted_to_fp16)[name = string("op_4542_cast_fp16")]; int32 var_4544 = const()[name = string("op_4544"), val = int32(-2)]; bool var_4545_interleave_0 = const()[name = string("op_4545_interleave_0"), val = bool(false)]; tensor var_4545_cast_fp16 = concat(axis = var_4544, interleave = var_4545_interleave_0, values = (var_4542_cast_fp16, var_4540_cast_fp16_0))[name = string("op_4545_cast_fp16")]; tensor var_4546_cast_fp16 = mul(x = var_4545_cast_fp16, y = var_878_cast_fp16)[name = string("op_4546_cast_fp16")]; tensor key_states_115_cast_fp16 = add(x = var_4539_cast_fp16, y = var_4546_cast_fp16)[name = string("key_states_115_cast_fp16")]; tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; int32 concat_137_axis_0 = const()[name = string("concat_137_axis_0"), val = int32(0)]; bool concat_137_interleave_0 = const()[name = string("concat_137_interleave_0"), val = bool(false)]; tensor concat_137 = concat(axis = concat_137_axis_0, interleave = concat_137_interleave_0, values = (expand_dims_132, expand_dims_133, position_id, expand_dims_135))[name = string("concat_137")]; tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; tensor concat_138_values1_0 = const()[name = string("concat_138_values1_0"), val = tensor([0])]; tensor concat_138_values3_0 = const()[name = string("concat_138_values3_0"), val = tensor([0])]; int32 concat_138_axis_0 = const()[name = string("concat_138_axis_0"), val = int32(0)]; bool concat_138_interleave_0 = const()[name = string("concat_138_interleave_0"), val = bool(false)]; tensor concat_138 = concat(axis = concat_138_axis_0, interleave = concat_138_interleave_0, values = (expand_dims_136, concat_138_values1_0, cache_position_end, concat_138_values3_0))[name = string("concat_138")]; tensor key_states_117_perm_0 = const()[name = string("key_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_12_stride_0 = const()[name = string("key_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_117_cast_fp16 = transpose(perm = key_states_117_perm_0, x = key_states_115_cast_fp16)[name = string("transpose_480")]; tensor key_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = key_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = key_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_12_squeeze_mask_0, stride = key_cache_internal_tensor_assign_12_stride_0, update = key_states_117_cast_fp16, x = coreml_update_state_300)[name = string("key_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_12_cast_fp16, input = key_cache)[name = string("coreml_update_state_302_write_state")]; tensor coreml_update_state_302 = read_state(input = key_cache)[name = string("coreml_update_state_302")]; tensor value_states_69_perm_0 = const()[name = string("value_states_69_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_12_stride_0 = const()[name = string("value_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_69_cast_fp16 = transpose(perm = value_states_69_perm_0, x = var_4522_cast_fp16)[name = string("transpose_479")]; tensor value_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = value_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = value_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_12_squeeze_mask_0, stride = value_cache_internal_tensor_assign_12_stride_0, update = value_states_69_cast_fp16, x = coreml_update_state_301)[name = string("value_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_12_cast_fp16, input = value_cache)[name = string("coreml_update_state_303_write_state")]; tensor coreml_update_state_303 = read_state(input = value_cache)[name = string("coreml_update_state_303")]; tensor var_4616_begin_0 = const()[name = string("op_4616_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4616_end_0 = const()[name = string("op_4616_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4616_end_mask_0 = const()[name = string("op_4616_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4616_cast_fp16 = slice_by_index(begin = var_4616_begin_0, end = var_4616_end_0, end_mask = var_4616_end_mask_0, x = coreml_update_state_302)[name = string("op_4616_cast_fp16")]; tensor tile_22 = const()[name = string("tile_22"), val = tensor([1, 1])]; int32 var_4619_axis_0 = const()[name = string("op_4619_axis_0"), val = int32(1)]; tensor var_4619_cast_fp16_0, tensor var_4619_cast_fp16_1 = split(axis = var_4619_axis_0, split_sizes = tile_22, x = var_4616_cast_fp16)[name = string("op_4619_cast_fp16")]; tensor var_4626_begin_0 = const()[name = string("op_4626_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4626_end_0 = const()[name = string("op_4626_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4626_end_mask_0 = const()[name = string("op_4626_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4626_cast_fp16 = slice_by_index(begin = var_4626_begin_0, end = var_4626_end_0, end_mask = var_4626_end_mask_0, x = coreml_update_state_303)[name = string("op_4626_cast_fp16")]; tensor tile_23 = const()[name = string("tile_23"), val = tensor([1, 1])]; int32 var_4629_axis_0 = const()[name = string("op_4629_axis_0"), val = int32(1)]; tensor var_4629_cast_fp16_0, tensor var_4629_cast_fp16_1 = split(axis = var_4629_axis_0, split_sizes = tile_23, x = var_4626_cast_fp16)[name = string("op_4629_cast_fp16")]; tensor var_4632_split_sizes_0 = const()[name = string("op_4632_split_sizes_0"), val = tensor([8, 8])]; int32 var_4632_axis_0 = const()[name = string("op_4632_axis_0"), val = int32(1)]; tensor var_4632_0, tensor var_4632_1 = split(axis = var_4632_axis_0, split_sizes = var_4632_split_sizes_0, x = query_states_69_cast_fp16)[name = string("op_4632")]; bool attn_weights_177_transpose_x_0 = const()[name = string("attn_weights_177_transpose_x_0"), val = bool(false)]; bool attn_weights_177_transpose_y_0 = const()[name = string("attn_weights_177_transpose_y_0"), val = bool(false)]; tensor attn_weights_177_cast_fp16 = matmul(transpose_x = attn_weights_177_transpose_x_0, transpose_y = attn_weights_177_transpose_y_0, x = var_4619_cast_fp16_0, y = var_4632_0)[name = string("attn_weights_177_cast_fp16")]; fp16 var_4635_to_fp16 = const()[name = string("op_4635_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_179_cast_fp16 = mul(x = attn_weights_177_cast_fp16, y = var_4635_to_fp16)[name = string("attn_weights_179_cast_fp16")]; tensor attn_weights_181_cast_fp16 = add(x = attn_weights_179_cast_fp16, y = attn_mask_1)[name = string("attn_weights_181_cast_fp16")]; int32 var_4639 = const()[name = string("op_4639"), val = int32(-2)]; tensor attn_weights_183_cast_fp16 = softmax(axis = var_4639, x = attn_weights_181_cast_fp16)[name = string("attn_weights_183_cast_fp16")]; bool var_4645_transpose_x_1 = const()[name = string("op_4645_transpose_x_1"), val = bool(true)]; bool var_4645_transpose_y_1 = const()[name = string("op_4645_transpose_y_1"), val = bool(false)]; tensor var_4645_cast_fp16 = matmul(transpose_x = var_4645_transpose_x_1, transpose_y = var_4645_transpose_y_1, x = attn_weights_183_cast_fp16, y = var_4629_cast_fp16_0)[name = string("op_4645_cast_fp16")]; bool attn_weights_185_transpose_x_0 = const()[name = string("attn_weights_185_transpose_x_0"), val = bool(false)]; bool attn_weights_185_transpose_y_0 = const()[name = string("attn_weights_185_transpose_y_0"), val = bool(false)]; tensor attn_weights_185_cast_fp16 = matmul(transpose_x = attn_weights_185_transpose_x_0, transpose_y = attn_weights_185_transpose_y_0, x = var_4619_cast_fp16_1, y = var_4632_1)[name = string("attn_weights_185_cast_fp16")]; fp16 var_4647_to_fp16 = const()[name = string("op_4647_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_187_cast_fp16 = mul(x = attn_weights_185_cast_fp16, y = var_4647_to_fp16)[name = string("attn_weights_187_cast_fp16")]; tensor attn_weights_189_cast_fp16 = add(x = attn_weights_187_cast_fp16, y = attn_mask_1)[name = string("attn_weights_189_cast_fp16")]; int32 var_4651 = const()[name = string("op_4651"), val = int32(-2)]; tensor attn_weights_191_cast_fp16 = softmax(axis = var_4651, x = attn_weights_189_cast_fp16)[name = string("attn_weights_191_cast_fp16")]; bool attn_output_89_transpose_x_1 = const()[name = string("attn_output_89_transpose_x_1"), val = bool(true)]; bool attn_output_89_transpose_y_1 = const()[name = string("attn_output_89_transpose_y_1"), val = bool(false)]; tensor attn_output_89_cast_fp16 = matmul(transpose_x = attn_output_89_transpose_x_1, transpose_y = attn_output_89_transpose_y_1, x = attn_weights_191_cast_fp16, y = var_4629_cast_fp16_1)[name = string("attn_output_89_cast_fp16")]; int32 var_4659 = const()[name = string("op_4659"), val = int32(1)]; bool attn_output_91_interleave_0 = const()[name = string("attn_output_91_interleave_0"), val = bool(false)]; tensor attn_output_91_cast_fp16 = concat(axis = var_4659, interleave = attn_output_91_interleave_0, values = (var_4645_cast_fp16, attn_output_89_cast_fp16))[name = string("attn_output_91_cast_fp16")]; tensor var_4663_perm_0 = const()[name = string("op_4663_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_143x = const()[name = string("concat_143x"), val = tensor([1, 2048, 1, -1])]; tensor var_4663_cast_fp16 = transpose(perm = var_4663_perm_0, x = attn_output_91_cast_fp16)[name = string("transpose_478")]; tensor attn_output_95_cast_fp16 = reshape(shape = concat_143x, x = var_4663_cast_fp16)[name = string("attn_output_95_cast_fp16")]; tensor hidden_states_113_strides_0 = const()[name = string("hidden_states_113_strides_0"), val = tensor([1, 1])]; string hidden_states_113_pad_type_0 = const()[name = string("hidden_states_113_pad_type_0"), val = string("valid")]; tensor hidden_states_113_pad_0 = const()[name = string("hidden_states_113_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_113_dilations_0 = const()[name = string("hidden_states_113_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_113_groups_0 = const()[name = string("hidden_states_113_groups_0"), val = int32(1)]; tensor hidden_states_113_cast_fp16 = conv(dilations = hidden_states_113_dilations_0, groups = hidden_states_113_groups_0, pad = hidden_states_113_pad_0, pad_type = hidden_states_113_pad_type_0, strides = hidden_states_113_strides_0, weight = layers_11_self_attn_o_proj_weight_cast_fp16, x = attn_output_95_cast_fp16)[name = string("hidden_states_113_cast_fp16")]; tensor hidden_states_115_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = hidden_states_113_cast_fp16)[name = string("hidden_states_115_cast_fp16")]; fp16 const_118_promoted_to_fp16 = const()[name = string("const_118_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4696_cast_fp16 = mul(x = hidden_states_115_cast_fp16, y = const_118_promoted_to_fp16)[name = string("op_4696_cast_fp16")]; int32 var_4694 = const()[name = string("op_4694"), val = int32(1)]; bool doubled_93_interleave_0 = const()[name = string("doubled_93_interleave_0"), val = bool(false)]; tensor doubled_93_cast_fp16 = concat(axis = var_4694, interleave = doubled_93_interleave_0, values = (hidden_states_115_cast_fp16, var_4696_cast_fp16))[name = string("doubled_93_cast_fp16")]; tensor out_47_axes_0 = const()[name = string("out_47_axes_0"), val = tensor([1])]; tensor out_47_gamma_0_to_fp16 = const()[name = string("out_47_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415366400)))]; fp16 var_4706_to_fp16 = const()[name = string("op_4706_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_47_cast_fp16 = layer_norm(axes = out_47_axes_0, epsilon = var_4706_to_fp16, gamma = out_47_gamma_0_to_fp16, x = doubled_93_cast_fp16)[name = string("out_47_cast_fp16")]; tensor var_4717_split_sizes_0 = const()[name = string("op_4717_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4717_axis_0 = const()[name = string("op_4717_axis_0"), val = int32(1)]; tensor var_4717_cast_fp16_0, tensor var_4717_cast_fp16_1 = split(axis = var_4717_axis_0, split_sizes = var_4717_split_sizes_0, x = out_47_cast_fp16)[name = string("op_4717_cast_fp16")]; tensor input_23_strides_0 = const()[name = string("input_23_strides_0"), val = tensor([1, 1])]; string input_23_pad_type_0 = const()[name = string("input_23_pad_type_0"), val = string("valid")]; tensor input_23_pad_0 = const()[name = string("input_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_23_dilations_0 = const()[name = string("input_23_dilations_0"), val = tensor([1, 1])]; int32 input_23_groups_0 = const()[name = string("input_23_groups_0"), val = int32(1)]; tensor input_23_cast_fp16 = conv(dilations = input_23_dilations_0, groups = input_23_groups_0, pad = input_23_pad_0, pad_type = input_23_pad_type_0, strides = input_23_strides_0, weight = layers_11_mlp_gate_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("input_23_cast_fp16")]; tensor var_4734_cast_fp16 = silu(x = input_23_cast_fp16)[name = string("op_4734_cast_fp16")]; tensor var_4740_strides_0 = const()[name = string("op_4740_strides_0"), val = tensor([1, 1])]; string var_4740_pad_type_0 = const()[name = string("op_4740_pad_type_0"), val = string("valid")]; tensor var_4740_pad_0 = const()[name = string("op_4740_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4740_dilations_0 = const()[name = string("op_4740_dilations_0"), val = tensor([1, 1])]; int32 var_4740_groups_0 = const()[name = string("op_4740_groups_0"), val = int32(1)]; tensor var_4740_cast_fp16 = conv(dilations = var_4740_dilations_0, groups = var_4740_groups_0, pad = var_4740_pad_0, pad_type = var_4740_pad_type_0, strides = var_4740_strides_0, weight = layers_11_mlp_up_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("op_4740_cast_fp16")]; tensor x_119_cast_fp16 = mul(x = var_4734_cast_fp16, y = var_4740_cast_fp16)[name = string("x_119_cast_fp16")]; tensor hidden_states_117_strides_0 = const()[name = string("hidden_states_117_strides_0"), val = tensor([1, 1])]; string hidden_states_117_pad_type_0 = const()[name = string("hidden_states_117_pad_type_0"), val = string("valid")]; tensor hidden_states_117_pad_0 = const()[name = string("hidden_states_117_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_117_dilations_0 = const()[name = string("hidden_states_117_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_117_groups_0 = const()[name = string("hidden_states_117_groups_0"), val = int32(1)]; tensor hidden_states_117_cast_fp16 = conv(dilations = hidden_states_117_dilations_0, groups = hidden_states_117_groups_0, pad = hidden_states_117_pad_0, pad_type = hidden_states_117_pad_type_0, strides = hidden_states_117_strides_0, weight = layers_11_mlp_down_proj_weight_cast_fp16, x = x_119_cast_fp16)[name = string("hidden_states_117_cast_fp16")]; tensor hidden_states_119_cast_fp16 = add(x = hidden_states_115_cast_fp16, y = hidden_states_117_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; fp16 const_120_promoted_to_fp16 = const()[name = string("const_120_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4758_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_120_promoted_to_fp16)[name = string("op_4758_cast_fp16")]; int32 var_4756 = const()[name = string("op_4756"), val = int32(1)]; bool doubled_97_interleave_0 = const()[name = string("doubled_97_interleave_0"), val = bool(false)]; tensor doubled_97_cast_fp16 = concat(axis = var_4756, interleave = doubled_97_interleave_0, values = (hidden_states_119_cast_fp16, var_4758_cast_fp16))[name = string("doubled_97_cast_fp16")]; tensor out_49_axes_0 = const()[name = string("out_49_axes_0"), val = tensor([1])]; tensor out_49_gamma_0_to_fp16 = const()[name = string("out_49_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415374656)))]; fp16 var_4768_to_fp16 = const()[name = string("op_4768_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_49_cast_fp16 = layer_norm(axes = out_49_axes_0, epsilon = var_4768_to_fp16, gamma = out_49_gamma_0_to_fp16, x = doubled_97_cast_fp16)[name = string("out_49_cast_fp16")]; tensor var_4779_split_sizes_0 = const()[name = string("op_4779_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4779_axis_0 = const()[name = string("op_4779_axis_0"), val = int32(1)]; tensor var_4779_cast_fp16_0, tensor var_4779_cast_fp16_1 = split(axis = var_4779_axis_0, split_sizes = var_4779_split_sizes_0, x = out_49_cast_fp16)[name = string("op_4779_cast_fp16")]; tensor query_states_73_strides_0 = const()[name = string("query_states_73_strides_0"), val = tensor([1, 1])]; string query_states_73_pad_type_0 = const()[name = string("query_states_73_pad_type_0"), val = string("valid")]; tensor query_states_73_pad_0 = const()[name = string("query_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_73_dilations_0 = const()[name = string("query_states_73_dilations_0"), val = tensor([1, 1])]; int32 query_states_73_groups_0 = const()[name = string("query_states_73_groups_0"), val = int32(1)]; tensor query_states_73_cast_fp16 = conv(dilations = query_states_73_dilations_0, groups = query_states_73_groups_0, pad = query_states_73_pad_0, pad_type = query_states_73_pad_type_0, strides = query_states_73_strides_0, weight = layers_12_self_attn_q_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("query_states_73_cast_fp16")]; tensor key_states_121_strides_0 = const()[name = string("key_states_121_strides_0"), val = tensor([1, 1])]; string key_states_121_pad_type_0 = const()[name = string("key_states_121_pad_type_0"), val = string("valid")]; tensor key_states_121_pad_0 = const()[name = string("key_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_121_dilations_0 = const()[name = string("key_states_121_dilations_0"), val = tensor([1, 1])]; int32 key_states_121_groups_0 = const()[name = string("key_states_121_groups_0"), val = int32(1)]; tensor key_states_121_cast_fp16 = conv(dilations = key_states_121_dilations_0, groups = key_states_121_groups_0, pad = key_states_121_pad_0, pad_type = key_states_121_pad_type_0, strides = key_states_121_strides_0, weight = layers_12_self_attn_k_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("key_states_121_cast_fp16")]; tensor value_states_73_strides_0 = const()[name = string("value_states_73_strides_0"), val = tensor([1, 1])]; string value_states_73_pad_type_0 = const()[name = string("value_states_73_pad_type_0"), val = string("valid")]; tensor value_states_73_pad_0 = const()[name = string("value_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_73_dilations_0 = const()[name = string("value_states_73_dilations_0"), val = tensor([1, 1])]; int32 value_states_73_groups_0 = const()[name = string("value_states_73_groups_0"), val = int32(1)]; tensor value_states_73_cast_fp16 = conv(dilations = value_states_73_dilations_0, groups = value_states_73_groups_0, pad = value_states_73_pad_0, pad_type = value_states_73_pad_type_0, strides = value_states_73_strides_0, weight = layers_12_self_attn_v_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("value_states_73_cast_fp16")]; tensor concat_144x = const()[name = string("concat_144x"), val = tensor([1, 16, 128, -1])]; tensor x_121_cast_fp16 = reshape(shape = concat_144x, x = query_states_73_cast_fp16)[name = string("x_121_cast_fp16")]; tensor concat_145x = const()[name = string("concat_145x"), val = tensor([1, 2, 128, -1])]; tensor var_4836_cast_fp16 = reshape(shape = concat_145x, x = key_states_121_cast_fp16)[name = string("op_4836_cast_fp16")]; tensor concat_146x = const()[name = string("concat_146x"), val = tensor([1, 2, 128, -1])]; tensor var_4843_cast_fp16 = reshape(shape = concat_146x, x = value_states_73_cast_fp16)[name = string("op_4843_cast_fp16")]; tensor var_4847_cast_fp16 = mul(x = x_121_cast_fp16, y = var_869_cast_fp16)[name = string("op_4847_cast_fp16")]; tensor var_4848_split_sizes_0 = const()[name = string("op_4848_split_sizes_0"), val = tensor([64, 64])]; int32 var_4848_axis_0 = const()[name = string("op_4848_axis_0"), val = int32(-2)]; tensor var_4848_cast_fp16_0, tensor var_4848_cast_fp16_1 = split(axis = var_4848_axis_0, split_sizes = var_4848_split_sizes_0, x = x_121_cast_fp16)[name = string("op_4848_cast_fp16")]; fp16 const_122_promoted_to_fp16 = const()[name = string("const_122_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4850_cast_fp16 = mul(x = var_4848_cast_fp16_1, y = const_122_promoted_to_fp16)[name = string("op_4850_cast_fp16")]; int32 var_4852 = const()[name = string("op_4852"), val = int32(-2)]; bool var_4853_interleave_0 = const()[name = string("op_4853_interleave_0"), val = bool(false)]; tensor var_4853_cast_fp16 = concat(axis = var_4852, interleave = var_4853_interleave_0, values = (var_4850_cast_fp16, var_4848_cast_fp16_0))[name = string("op_4853_cast_fp16")]; tensor var_4854_cast_fp16 = mul(x = var_4853_cast_fp16, y = var_878_cast_fp16)[name = string("op_4854_cast_fp16")]; tensor query_states_75_cast_fp16 = add(x = var_4847_cast_fp16, y = var_4854_cast_fp16)[name = string("query_states_75_cast_fp16")]; tensor var_4860_cast_fp16 = mul(x = var_4836_cast_fp16, y = var_869_cast_fp16)[name = string("op_4860_cast_fp16")]; tensor var_4861_split_sizes_0 = const()[name = string("op_4861_split_sizes_0"), val = tensor([64, 64])]; int32 var_4861_axis_0 = const()[name = string("op_4861_axis_0"), val = int32(-2)]; tensor var_4861_cast_fp16_0, tensor var_4861_cast_fp16_1 = split(axis = var_4861_axis_0, split_sizes = var_4861_split_sizes_0, x = var_4836_cast_fp16)[name = string("op_4861_cast_fp16")]; fp16 const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4863_cast_fp16 = mul(x = var_4861_cast_fp16_1, y = const_123_promoted_to_fp16)[name = string("op_4863_cast_fp16")]; int32 var_4865 = const()[name = string("op_4865"), val = int32(-2)]; bool var_4866_interleave_0 = const()[name = string("op_4866_interleave_0"), val = bool(false)]; tensor var_4866_cast_fp16 = concat(axis = var_4865, interleave = var_4866_interleave_0, values = (var_4863_cast_fp16, var_4861_cast_fp16_0))[name = string("op_4866_cast_fp16")]; tensor var_4867_cast_fp16 = mul(x = var_4866_cast_fp16, y = var_878_cast_fp16)[name = string("op_4867_cast_fp16")]; tensor key_states_125_cast_fp16 = add(x = var_4860_cast_fp16, y = var_4867_cast_fp16)[name = string("key_states_125_cast_fp16")]; tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; int32 concat_149_axis_0 = const()[name = string("concat_149_axis_0"), val = int32(0)]; bool concat_149_interleave_0 = const()[name = string("concat_149_interleave_0"), val = bool(false)]; tensor concat_149 = concat(axis = concat_149_axis_0, interleave = concat_149_interleave_0, values = (expand_dims_144, expand_dims_145, position_id, expand_dims_147))[name = string("concat_149")]; tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; tensor concat_150_values1_0 = const()[name = string("concat_150_values1_0"), val = tensor([0])]; tensor concat_150_values3_0 = const()[name = string("concat_150_values3_0"), val = tensor([0])]; int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_148, concat_150_values1_0, cache_position_end, concat_150_values3_0))[name = string("concat_150")]; tensor key_states_127_perm_0 = const()[name = string("key_states_127_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_13_stride_0 = const()[name = string("key_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_127_cast_fp16 = transpose(perm = key_states_127_perm_0, x = key_states_125_cast_fp16)[name = string("transpose_477")]; tensor key_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = key_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = key_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_13_squeeze_mask_0, stride = key_cache_internal_tensor_assign_13_stride_0, update = key_states_127_cast_fp16, x = coreml_update_state_302)[name = string("key_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_13_cast_fp16, input = key_cache)[name = string("coreml_update_state_304_write_state")]; tensor coreml_update_state_304 = read_state(input = key_cache)[name = string("coreml_update_state_304")]; tensor value_states_75_perm_0 = const()[name = string("value_states_75_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_13_stride_0 = const()[name = string("value_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_75_cast_fp16 = transpose(perm = value_states_75_perm_0, x = var_4843_cast_fp16)[name = string("transpose_476")]; tensor value_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = value_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = value_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_13_squeeze_mask_0, stride = value_cache_internal_tensor_assign_13_stride_0, update = value_states_75_cast_fp16, x = coreml_update_state_303)[name = string("value_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_13_cast_fp16, input = value_cache)[name = string("coreml_update_state_305_write_state")]; tensor coreml_update_state_305 = read_state(input = value_cache)[name = string("coreml_update_state_305")]; tensor var_4937_begin_0 = const()[name = string("op_4937_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4937_end_0 = const()[name = string("op_4937_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4937_end_mask_0 = const()[name = string("op_4937_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4937_cast_fp16 = slice_by_index(begin = var_4937_begin_0, end = var_4937_end_0, end_mask = var_4937_end_mask_0, x = coreml_update_state_304)[name = string("op_4937_cast_fp16")]; tensor tile_24 = const()[name = string("tile_24"), val = tensor([1, 1])]; int32 var_4940_axis_0 = const()[name = string("op_4940_axis_0"), val = int32(1)]; tensor var_4940_cast_fp16_0, tensor var_4940_cast_fp16_1 = split(axis = var_4940_axis_0, split_sizes = tile_24, x = var_4937_cast_fp16)[name = string("op_4940_cast_fp16")]; tensor var_4947_begin_0 = const()[name = string("op_4947_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4947_end_0 = const()[name = string("op_4947_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4947_end_mask_0 = const()[name = string("op_4947_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4947_cast_fp16 = slice_by_index(begin = var_4947_begin_0, end = var_4947_end_0, end_mask = var_4947_end_mask_0, x = coreml_update_state_305)[name = string("op_4947_cast_fp16")]; tensor tile_25 = const()[name = string("tile_25"), val = tensor([1, 1])]; int32 var_4950_axis_0 = const()[name = string("op_4950_axis_0"), val = int32(1)]; tensor var_4950_cast_fp16_0, tensor var_4950_cast_fp16_1 = split(axis = var_4950_axis_0, split_sizes = tile_25, x = var_4947_cast_fp16)[name = string("op_4950_cast_fp16")]; tensor var_4953_split_sizes_0 = const()[name = string("op_4953_split_sizes_0"), val = tensor([8, 8])]; int32 var_4953_axis_0 = const()[name = string("op_4953_axis_0"), val = int32(1)]; tensor var_4953_0, tensor var_4953_1 = split(axis = var_4953_axis_0, split_sizes = var_4953_split_sizes_0, x = query_states_75_cast_fp16)[name = string("op_4953")]; bool attn_weights_193_transpose_x_0 = const()[name = string("attn_weights_193_transpose_x_0"), val = bool(false)]; bool attn_weights_193_transpose_y_0 = const()[name = string("attn_weights_193_transpose_y_0"), val = bool(false)]; tensor attn_weights_193_cast_fp16 = matmul(transpose_x = attn_weights_193_transpose_x_0, transpose_y = attn_weights_193_transpose_y_0, x = var_4940_cast_fp16_0, y = var_4953_0)[name = string("attn_weights_193_cast_fp16")]; fp16 var_4956_to_fp16 = const()[name = string("op_4956_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_195_cast_fp16 = mul(x = attn_weights_193_cast_fp16, y = var_4956_to_fp16)[name = string("attn_weights_195_cast_fp16")]; tensor attn_weights_197_cast_fp16 = add(x = attn_weights_195_cast_fp16, y = attn_mask_1)[name = string("attn_weights_197_cast_fp16")]; int32 var_4960 = const()[name = string("op_4960"), val = int32(-2)]; tensor attn_weights_199_cast_fp16 = softmax(axis = var_4960, x = attn_weights_197_cast_fp16)[name = string("attn_weights_199_cast_fp16")]; bool var_4966_transpose_x_1 = const()[name = string("op_4966_transpose_x_1"), val = bool(true)]; bool var_4966_transpose_y_1 = const()[name = string("op_4966_transpose_y_1"), val = bool(false)]; tensor var_4966_cast_fp16 = matmul(transpose_x = var_4966_transpose_x_1, transpose_y = var_4966_transpose_y_1, x = attn_weights_199_cast_fp16, y = var_4950_cast_fp16_0)[name = string("op_4966_cast_fp16")]; bool attn_weights_201_transpose_x_0 = const()[name = string("attn_weights_201_transpose_x_0"), val = bool(false)]; bool attn_weights_201_transpose_y_0 = const()[name = string("attn_weights_201_transpose_y_0"), val = bool(false)]; tensor attn_weights_201_cast_fp16 = matmul(transpose_x = attn_weights_201_transpose_x_0, transpose_y = attn_weights_201_transpose_y_0, x = var_4940_cast_fp16_1, y = var_4953_1)[name = string("attn_weights_201_cast_fp16")]; fp16 var_4968_to_fp16 = const()[name = string("op_4968_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_203_cast_fp16 = mul(x = attn_weights_201_cast_fp16, y = var_4968_to_fp16)[name = string("attn_weights_203_cast_fp16")]; tensor attn_weights_205_cast_fp16 = add(x = attn_weights_203_cast_fp16, y = attn_mask_1)[name = string("attn_weights_205_cast_fp16")]; int32 var_4972 = const()[name = string("op_4972"), val = int32(-2)]; tensor attn_weights_207_cast_fp16 = softmax(axis = var_4972, x = attn_weights_205_cast_fp16)[name = string("attn_weights_207_cast_fp16")]; bool attn_output_97_transpose_x_1 = const()[name = string("attn_output_97_transpose_x_1"), val = bool(true)]; bool attn_output_97_transpose_y_1 = const()[name = string("attn_output_97_transpose_y_1"), val = bool(false)]; tensor attn_output_97_cast_fp16 = matmul(transpose_x = attn_output_97_transpose_x_1, transpose_y = attn_output_97_transpose_y_1, x = attn_weights_207_cast_fp16, y = var_4950_cast_fp16_1)[name = string("attn_output_97_cast_fp16")]; int32 var_4980 = const()[name = string("op_4980"), val = int32(1)]; bool attn_output_99_interleave_0 = const()[name = string("attn_output_99_interleave_0"), val = bool(false)]; tensor attn_output_99_cast_fp16 = concat(axis = var_4980, interleave = attn_output_99_interleave_0, values = (var_4966_cast_fp16, attn_output_97_cast_fp16))[name = string("attn_output_99_cast_fp16")]; tensor var_4984_perm_0 = const()[name = string("op_4984_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_155x = const()[name = string("concat_155x"), val = tensor([1, 2048, 1, -1])]; tensor var_4984_cast_fp16 = transpose(perm = var_4984_perm_0, x = attn_output_99_cast_fp16)[name = string("transpose_475")]; tensor attn_output_103_cast_fp16 = reshape(shape = concat_155x, x = var_4984_cast_fp16)[name = string("attn_output_103_cast_fp16")]; tensor hidden_states_123_strides_0 = const()[name = string("hidden_states_123_strides_0"), val = tensor([1, 1])]; string hidden_states_123_pad_type_0 = const()[name = string("hidden_states_123_pad_type_0"), val = string("valid")]; tensor hidden_states_123_pad_0 = const()[name = string("hidden_states_123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_123_dilations_0 = const()[name = string("hidden_states_123_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_123_groups_0 = const()[name = string("hidden_states_123_groups_0"), val = int32(1)]; tensor hidden_states_123_cast_fp16 = conv(dilations = hidden_states_123_dilations_0, groups = hidden_states_123_groups_0, pad = hidden_states_123_pad_0, pad_type = hidden_states_123_pad_type_0, strides = hidden_states_123_strides_0, weight = layers_12_self_attn_o_proj_weight_cast_fp16, x = attn_output_103_cast_fp16)[name = string("hidden_states_123_cast_fp16")]; tensor hidden_states_125_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = hidden_states_123_cast_fp16)[name = string("hidden_states_125_cast_fp16")]; fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5017_cast_fp16 = mul(x = hidden_states_125_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_5017_cast_fp16")]; int32 var_5015 = const()[name = string("op_5015"), val = int32(1)]; bool doubled_101_interleave_0 = const()[name = string("doubled_101_interleave_0"), val = bool(false)]; tensor doubled_101_cast_fp16 = concat(axis = var_5015, interleave = doubled_101_interleave_0, values = (hidden_states_125_cast_fp16, var_5017_cast_fp16))[name = string("doubled_101_cast_fp16")]; tensor out_51_axes_0 = const()[name = string("out_51_axes_0"), val = tensor([1])]; tensor out_51_gamma_0_to_fp16 = const()[name = string("out_51_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415382912)))]; fp16 var_5027_to_fp16 = const()[name = string("op_5027_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_51_cast_fp16 = layer_norm(axes = out_51_axes_0, epsilon = var_5027_to_fp16, gamma = out_51_gamma_0_to_fp16, x = doubled_101_cast_fp16)[name = string("out_51_cast_fp16")]; tensor var_5038_split_sizes_0 = const()[name = string("op_5038_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5038_axis_0 = const()[name = string("op_5038_axis_0"), val = int32(1)]; tensor var_5038_cast_fp16_0, tensor var_5038_cast_fp16_1 = split(axis = var_5038_axis_0, split_sizes = var_5038_split_sizes_0, x = out_51_cast_fp16)[name = string("op_5038_cast_fp16")]; tensor input_25_strides_0 = const()[name = string("input_25_strides_0"), val = tensor([1, 1])]; string input_25_pad_type_0 = const()[name = string("input_25_pad_type_0"), val = string("valid")]; tensor input_25_pad_0 = const()[name = string("input_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_25_dilations_0 = const()[name = string("input_25_dilations_0"), val = tensor([1, 1])]; int32 input_25_groups_0 = const()[name = string("input_25_groups_0"), val = int32(1)]; tensor input_25_cast_fp16 = conv(dilations = input_25_dilations_0, groups = input_25_groups_0, pad = input_25_pad_0, pad_type = input_25_pad_type_0, strides = input_25_strides_0, weight = layers_12_mlp_gate_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("input_25_cast_fp16")]; tensor var_5055_cast_fp16 = silu(x = input_25_cast_fp16)[name = string("op_5055_cast_fp16")]; tensor var_5061_strides_0 = const()[name = string("op_5061_strides_0"), val = tensor([1, 1])]; string var_5061_pad_type_0 = const()[name = string("op_5061_pad_type_0"), val = string("valid")]; tensor var_5061_pad_0 = const()[name = string("op_5061_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5061_dilations_0 = const()[name = string("op_5061_dilations_0"), val = tensor([1, 1])]; int32 var_5061_groups_0 = const()[name = string("op_5061_groups_0"), val = int32(1)]; tensor var_5061_cast_fp16 = conv(dilations = var_5061_dilations_0, groups = var_5061_groups_0, pad = var_5061_pad_0, pad_type = var_5061_pad_type_0, strides = var_5061_strides_0, weight = layers_12_mlp_up_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("op_5061_cast_fp16")]; tensor x_129_cast_fp16 = mul(x = var_5055_cast_fp16, y = var_5061_cast_fp16)[name = string("x_129_cast_fp16")]; tensor hidden_states_127_strides_0 = const()[name = string("hidden_states_127_strides_0"), val = tensor([1, 1])]; string hidden_states_127_pad_type_0 = const()[name = string("hidden_states_127_pad_type_0"), val = string("valid")]; tensor hidden_states_127_pad_0 = const()[name = string("hidden_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_127_dilations_0 = const()[name = string("hidden_states_127_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_127_groups_0 = const()[name = string("hidden_states_127_groups_0"), val = int32(1)]; tensor hidden_states_127_cast_fp16 = conv(dilations = hidden_states_127_dilations_0, groups = hidden_states_127_groups_0, pad = hidden_states_127_pad_0, pad_type = hidden_states_127_pad_type_0, strides = hidden_states_127_strides_0, weight = layers_12_mlp_down_proj_weight_cast_fp16, x = x_129_cast_fp16)[name = string("hidden_states_127_cast_fp16")]; tensor hidden_states_129_cast_fp16 = add(x = hidden_states_125_cast_fp16, y = hidden_states_127_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5079_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_130_promoted_to_fp16)[name = string("op_5079_cast_fp16")]; int32 var_5077 = const()[name = string("op_5077"), val = int32(1)]; bool doubled_105_interleave_0 = const()[name = string("doubled_105_interleave_0"), val = bool(false)]; tensor doubled_105_cast_fp16 = concat(axis = var_5077, interleave = doubled_105_interleave_0, values = (hidden_states_129_cast_fp16, var_5079_cast_fp16))[name = string("doubled_105_cast_fp16")]; tensor out_53_axes_0 = const()[name = string("out_53_axes_0"), val = tensor([1])]; tensor out_53_gamma_0_to_fp16 = const()[name = string("out_53_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415391168)))]; fp16 var_5089_to_fp16 = const()[name = string("op_5089_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_53_cast_fp16 = layer_norm(axes = out_53_axes_0, epsilon = var_5089_to_fp16, gamma = out_53_gamma_0_to_fp16, x = doubled_105_cast_fp16)[name = string("out_53_cast_fp16")]; tensor var_5100_split_sizes_0 = const()[name = string("op_5100_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5100_axis_0 = const()[name = string("op_5100_axis_0"), val = int32(1)]; tensor var_5100_cast_fp16_0, tensor var_5100_cast_fp16_1 = split(axis = var_5100_axis_0, split_sizes = var_5100_split_sizes_0, x = out_53_cast_fp16)[name = string("op_5100_cast_fp16")]; tensor query_states_79_strides_0 = const()[name = string("query_states_79_strides_0"), val = tensor([1, 1])]; string query_states_79_pad_type_0 = const()[name = string("query_states_79_pad_type_0"), val = string("valid")]; tensor query_states_79_pad_0 = const()[name = string("query_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_79_dilations_0 = const()[name = string("query_states_79_dilations_0"), val = tensor([1, 1])]; int32 query_states_79_groups_0 = const()[name = string("query_states_79_groups_0"), val = int32(1)]; tensor query_states_79_cast_fp16 = conv(dilations = query_states_79_dilations_0, groups = query_states_79_groups_0, pad = query_states_79_pad_0, pad_type = query_states_79_pad_type_0, strides = query_states_79_strides_0, weight = layers_13_self_attn_q_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("query_states_79_cast_fp16")]; tensor key_states_131_strides_0 = const()[name = string("key_states_131_strides_0"), val = tensor([1, 1])]; string key_states_131_pad_type_0 = const()[name = string("key_states_131_pad_type_0"), val = string("valid")]; tensor key_states_131_pad_0 = const()[name = string("key_states_131_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_131_dilations_0 = const()[name = string("key_states_131_dilations_0"), val = tensor([1, 1])]; int32 key_states_131_groups_0 = const()[name = string("key_states_131_groups_0"), val = int32(1)]; tensor key_states_131_cast_fp16 = conv(dilations = key_states_131_dilations_0, groups = key_states_131_groups_0, pad = key_states_131_pad_0, pad_type = key_states_131_pad_type_0, strides = key_states_131_strides_0, weight = layers_13_self_attn_k_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("key_states_131_cast_fp16")]; tensor value_states_79_strides_0 = const()[name = string("value_states_79_strides_0"), val = tensor([1, 1])]; string value_states_79_pad_type_0 = const()[name = string("value_states_79_pad_type_0"), val = string("valid")]; tensor value_states_79_pad_0 = const()[name = string("value_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_79_dilations_0 = const()[name = string("value_states_79_dilations_0"), val = tensor([1, 1])]; int32 value_states_79_groups_0 = const()[name = string("value_states_79_groups_0"), val = int32(1)]; tensor value_states_79_cast_fp16 = conv(dilations = value_states_79_dilations_0, groups = value_states_79_groups_0, pad = value_states_79_pad_0, pad_type = value_states_79_pad_type_0, strides = value_states_79_strides_0, weight = layers_13_self_attn_v_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("value_states_79_cast_fp16")]; tensor concat_156x = const()[name = string("concat_156x"), val = tensor([1, 16, 128, -1])]; tensor x_131_cast_fp16 = reshape(shape = concat_156x, x = query_states_79_cast_fp16)[name = string("x_131_cast_fp16")]; tensor concat_157x = const()[name = string("concat_157x"), val = tensor([1, 2, 128, -1])]; tensor var_5157_cast_fp16 = reshape(shape = concat_157x, x = key_states_131_cast_fp16)[name = string("op_5157_cast_fp16")]; tensor concat_158x = const()[name = string("concat_158x"), val = tensor([1, 2, 128, -1])]; tensor var_5164_cast_fp16 = reshape(shape = concat_158x, x = value_states_79_cast_fp16)[name = string("op_5164_cast_fp16")]; tensor var_5168_cast_fp16 = mul(x = x_131_cast_fp16, y = var_869_cast_fp16)[name = string("op_5168_cast_fp16")]; tensor var_5169_split_sizes_0 = const()[name = string("op_5169_split_sizes_0"), val = tensor([64, 64])]; int32 var_5169_axis_0 = const()[name = string("op_5169_axis_0"), val = int32(-2)]; tensor var_5169_cast_fp16_0, tensor var_5169_cast_fp16_1 = split(axis = var_5169_axis_0, split_sizes = var_5169_split_sizes_0, x = x_131_cast_fp16)[name = string("op_5169_cast_fp16")]; fp16 const_132_promoted_to_fp16 = const()[name = string("const_132_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5171_cast_fp16 = mul(x = var_5169_cast_fp16_1, y = const_132_promoted_to_fp16)[name = string("op_5171_cast_fp16")]; int32 var_5173 = const()[name = string("op_5173"), val = int32(-2)]; bool var_5174_interleave_0 = const()[name = string("op_5174_interleave_0"), val = bool(false)]; tensor var_5174_cast_fp16 = concat(axis = var_5173, interleave = var_5174_interleave_0, values = (var_5171_cast_fp16, var_5169_cast_fp16_0))[name = string("op_5174_cast_fp16")]; tensor var_5175_cast_fp16 = mul(x = var_5174_cast_fp16, y = var_878_cast_fp16)[name = string("op_5175_cast_fp16")]; tensor query_states_81_cast_fp16 = add(x = var_5168_cast_fp16, y = var_5175_cast_fp16)[name = string("query_states_81_cast_fp16")]; tensor var_5181_cast_fp16 = mul(x = var_5157_cast_fp16, y = var_869_cast_fp16)[name = string("op_5181_cast_fp16")]; tensor var_5182_split_sizes_0 = const()[name = string("op_5182_split_sizes_0"), val = tensor([64, 64])]; int32 var_5182_axis_0 = const()[name = string("op_5182_axis_0"), val = int32(-2)]; tensor var_5182_cast_fp16_0, tensor var_5182_cast_fp16_1 = split(axis = var_5182_axis_0, split_sizes = var_5182_split_sizes_0, x = var_5157_cast_fp16)[name = string("op_5182_cast_fp16")]; fp16 const_133_promoted_to_fp16 = const()[name = string("const_133_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5184_cast_fp16 = mul(x = var_5182_cast_fp16_1, y = const_133_promoted_to_fp16)[name = string("op_5184_cast_fp16")]; int32 var_5186 = const()[name = string("op_5186"), val = int32(-2)]; bool var_5187_interleave_0 = const()[name = string("op_5187_interleave_0"), val = bool(false)]; tensor var_5187_cast_fp16 = concat(axis = var_5186, interleave = var_5187_interleave_0, values = (var_5184_cast_fp16, var_5182_cast_fp16_0))[name = string("op_5187_cast_fp16")]; tensor var_5188_cast_fp16 = mul(x = var_5187_cast_fp16, y = var_878_cast_fp16)[name = string("op_5188_cast_fp16")]; tensor key_states_135_cast_fp16 = add(x = var_5181_cast_fp16, y = var_5188_cast_fp16)[name = string("key_states_135_cast_fp16")]; tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; int32 concat_161_axis_0 = const()[name = string("concat_161_axis_0"), val = int32(0)]; bool concat_161_interleave_0 = const()[name = string("concat_161_interleave_0"), val = bool(false)]; tensor concat_161 = concat(axis = concat_161_axis_0, interleave = concat_161_interleave_0, values = (expand_dims_156, expand_dims_157, position_id, expand_dims_159))[name = string("concat_161")]; tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; tensor concat_162_values1_0 = const()[name = string("concat_162_values1_0"), val = tensor([0])]; tensor concat_162_values3_0 = const()[name = string("concat_162_values3_0"), val = tensor([0])]; int32 concat_162_axis_0 = const()[name = string("concat_162_axis_0"), val = int32(0)]; bool concat_162_interleave_0 = const()[name = string("concat_162_interleave_0"), val = bool(false)]; tensor concat_162 = concat(axis = concat_162_axis_0, interleave = concat_162_interleave_0, values = (expand_dims_160, concat_162_values1_0, cache_position_end, concat_162_values3_0))[name = string("concat_162")]; tensor key_states_137_perm_0 = const()[name = string("key_states_137_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_14_stride_0 = const()[name = string("key_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_137_cast_fp16 = transpose(perm = key_states_137_perm_0, x = key_states_135_cast_fp16)[name = string("transpose_474")]; tensor key_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = key_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = key_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_14_squeeze_mask_0, stride = key_cache_internal_tensor_assign_14_stride_0, update = key_states_137_cast_fp16, x = coreml_update_state_304)[name = string("key_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_14_cast_fp16, input = key_cache)[name = string("coreml_update_state_306_write_state")]; tensor coreml_update_state_306 = read_state(input = key_cache)[name = string("coreml_update_state_306")]; tensor value_states_81_perm_0 = const()[name = string("value_states_81_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_14_stride_0 = const()[name = string("value_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_81_cast_fp16 = transpose(perm = value_states_81_perm_0, x = var_5164_cast_fp16)[name = string("transpose_473")]; tensor value_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = value_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = value_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_14_squeeze_mask_0, stride = value_cache_internal_tensor_assign_14_stride_0, update = value_states_81_cast_fp16, x = coreml_update_state_305)[name = string("value_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_14_cast_fp16, input = value_cache)[name = string("coreml_update_state_307_write_state")]; tensor coreml_update_state_307 = read_state(input = value_cache)[name = string("coreml_update_state_307")]; tensor var_5258_begin_0 = const()[name = string("op_5258_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5258_end_0 = const()[name = string("op_5258_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5258_end_mask_0 = const()[name = string("op_5258_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5258_cast_fp16 = slice_by_index(begin = var_5258_begin_0, end = var_5258_end_0, end_mask = var_5258_end_mask_0, x = coreml_update_state_306)[name = string("op_5258_cast_fp16")]; tensor tile_26 = const()[name = string("tile_26"), val = tensor([1, 1])]; int32 var_5261_axis_0 = const()[name = string("op_5261_axis_0"), val = int32(1)]; tensor var_5261_cast_fp16_0, tensor var_5261_cast_fp16_1 = split(axis = var_5261_axis_0, split_sizes = tile_26, x = var_5258_cast_fp16)[name = string("op_5261_cast_fp16")]; tensor var_5268_begin_0 = const()[name = string("op_5268_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5268_end_0 = const()[name = string("op_5268_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5268_end_mask_0 = const()[name = string("op_5268_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5268_cast_fp16 = slice_by_index(begin = var_5268_begin_0, end = var_5268_end_0, end_mask = var_5268_end_mask_0, x = coreml_update_state_307)[name = string("op_5268_cast_fp16")]; tensor tile_27 = const()[name = string("tile_27"), val = tensor([1, 1])]; int32 var_5271_axis_0 = const()[name = string("op_5271_axis_0"), val = int32(1)]; tensor var_5271_cast_fp16_0, tensor var_5271_cast_fp16_1 = split(axis = var_5271_axis_0, split_sizes = tile_27, x = var_5268_cast_fp16)[name = string("op_5271_cast_fp16")]; tensor var_5274_split_sizes_0 = const()[name = string("op_5274_split_sizes_0"), val = tensor([8, 8])]; int32 var_5274_axis_0 = const()[name = string("op_5274_axis_0"), val = int32(1)]; tensor var_5274_0, tensor var_5274_1 = split(axis = var_5274_axis_0, split_sizes = var_5274_split_sizes_0, x = query_states_81_cast_fp16)[name = string("op_5274")]; bool attn_weights_209_transpose_x_0 = const()[name = string("attn_weights_209_transpose_x_0"), val = bool(false)]; bool attn_weights_209_transpose_y_0 = const()[name = string("attn_weights_209_transpose_y_0"), val = bool(false)]; tensor attn_weights_209_cast_fp16 = matmul(transpose_x = attn_weights_209_transpose_x_0, transpose_y = attn_weights_209_transpose_y_0, x = var_5261_cast_fp16_0, y = var_5274_0)[name = string("attn_weights_209_cast_fp16")]; fp16 var_5277_to_fp16 = const()[name = string("op_5277_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_211_cast_fp16 = mul(x = attn_weights_209_cast_fp16, y = var_5277_to_fp16)[name = string("attn_weights_211_cast_fp16")]; tensor attn_weights_213_cast_fp16 = add(x = attn_weights_211_cast_fp16, y = attn_mask_1)[name = string("attn_weights_213_cast_fp16")]; int32 var_5281 = const()[name = string("op_5281"), val = int32(-2)]; tensor attn_weights_215_cast_fp16 = softmax(axis = var_5281, x = attn_weights_213_cast_fp16)[name = string("attn_weights_215_cast_fp16")]; bool var_5287_transpose_x_1 = const()[name = string("op_5287_transpose_x_1"), val = bool(true)]; bool var_5287_transpose_y_1 = const()[name = string("op_5287_transpose_y_1"), val = bool(false)]; tensor var_5287_cast_fp16 = matmul(transpose_x = var_5287_transpose_x_1, transpose_y = var_5287_transpose_y_1, x = attn_weights_215_cast_fp16, y = var_5271_cast_fp16_0)[name = string("op_5287_cast_fp16")]; bool attn_weights_217_transpose_x_0 = const()[name = string("attn_weights_217_transpose_x_0"), val = bool(false)]; bool attn_weights_217_transpose_y_0 = const()[name = string("attn_weights_217_transpose_y_0"), val = bool(false)]; tensor attn_weights_217_cast_fp16 = matmul(transpose_x = attn_weights_217_transpose_x_0, transpose_y = attn_weights_217_transpose_y_0, x = var_5261_cast_fp16_1, y = var_5274_1)[name = string("attn_weights_217_cast_fp16")]; fp16 var_5289_to_fp16 = const()[name = string("op_5289_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_219_cast_fp16 = mul(x = attn_weights_217_cast_fp16, y = var_5289_to_fp16)[name = string("attn_weights_219_cast_fp16")]; tensor attn_weights_221_cast_fp16 = add(x = attn_weights_219_cast_fp16, y = attn_mask_1)[name = string("attn_weights_221_cast_fp16")]; int32 var_5293 = const()[name = string("op_5293"), val = int32(-2)]; tensor attn_weights_223_cast_fp16 = softmax(axis = var_5293, x = attn_weights_221_cast_fp16)[name = string("attn_weights_223_cast_fp16")]; bool attn_output_105_transpose_x_1 = const()[name = string("attn_output_105_transpose_x_1"), val = bool(true)]; bool attn_output_105_transpose_y_1 = const()[name = string("attn_output_105_transpose_y_1"), val = bool(false)]; tensor attn_output_105_cast_fp16 = matmul(transpose_x = attn_output_105_transpose_x_1, transpose_y = attn_output_105_transpose_y_1, x = attn_weights_223_cast_fp16, y = var_5271_cast_fp16_1)[name = string("attn_output_105_cast_fp16")]; int32 var_5301 = const()[name = string("op_5301"), val = int32(1)]; bool attn_output_107_interleave_0 = const()[name = string("attn_output_107_interleave_0"), val = bool(false)]; tensor attn_output_107_cast_fp16 = concat(axis = var_5301, interleave = attn_output_107_interleave_0, values = (var_5287_cast_fp16, attn_output_105_cast_fp16))[name = string("attn_output_107_cast_fp16")]; tensor var_5305_perm_0 = const()[name = string("op_5305_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_167x = const()[name = string("concat_167x"), val = tensor([1, 2048, 1, -1])]; tensor var_5305_cast_fp16 = transpose(perm = var_5305_perm_0, x = attn_output_107_cast_fp16)[name = string("transpose_472")]; tensor attn_output_111_cast_fp16 = reshape(shape = concat_167x, x = var_5305_cast_fp16)[name = string("attn_output_111_cast_fp16")]; tensor hidden_states_133_strides_0 = const()[name = string("hidden_states_133_strides_0"), val = tensor([1, 1])]; string hidden_states_133_pad_type_0 = const()[name = string("hidden_states_133_pad_type_0"), val = string("valid")]; tensor hidden_states_133_pad_0 = const()[name = string("hidden_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_133_dilations_0 = const()[name = string("hidden_states_133_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_133_groups_0 = const()[name = string("hidden_states_133_groups_0"), val = int32(1)]; tensor hidden_states_133_cast_fp16 = conv(dilations = hidden_states_133_dilations_0, groups = hidden_states_133_groups_0, pad = hidden_states_133_pad_0, pad_type = hidden_states_133_pad_type_0, strides = hidden_states_133_strides_0, weight = layers_13_self_attn_o_proj_weight_cast_fp16, x = attn_output_111_cast_fp16)[name = string("hidden_states_133_cast_fp16")]; tensor hidden_states_135_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = hidden_states_133_cast_fp16)[name = string("hidden_states_135_cast_fp16")]; fp16 const_138_promoted_to_fp16 = const()[name = string("const_138_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5338_cast_fp16 = mul(x = hidden_states_135_cast_fp16, y = const_138_promoted_to_fp16)[name = string("op_5338_cast_fp16")]; int32 var_5336 = const()[name = string("op_5336"), val = int32(1)]; bool doubled_109_interleave_0 = const()[name = string("doubled_109_interleave_0"), val = bool(false)]; tensor doubled_109_cast_fp16 = concat(axis = var_5336, interleave = doubled_109_interleave_0, values = (hidden_states_135_cast_fp16, var_5338_cast_fp16))[name = string("doubled_109_cast_fp16")]; tensor out_55_axes_0 = const()[name = string("out_55_axes_0"), val = tensor([1])]; tensor out_55_gamma_0_to_fp16 = const()[name = string("out_55_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415399424)))]; fp16 var_5348_to_fp16 = const()[name = string("op_5348_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_55_cast_fp16 = layer_norm(axes = out_55_axes_0, epsilon = var_5348_to_fp16, gamma = out_55_gamma_0_to_fp16, x = doubled_109_cast_fp16)[name = string("out_55_cast_fp16")]; tensor var_5359_split_sizes_0 = const()[name = string("op_5359_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5359_axis_0 = const()[name = string("op_5359_axis_0"), val = int32(1)]; tensor var_5359_cast_fp16_0, tensor var_5359_cast_fp16_1 = split(axis = var_5359_axis_0, split_sizes = var_5359_split_sizes_0, x = out_55_cast_fp16)[name = string("op_5359_cast_fp16")]; tensor input_27_strides_0 = const()[name = string("input_27_strides_0"), val = tensor([1, 1])]; string input_27_pad_type_0 = const()[name = string("input_27_pad_type_0"), val = string("valid")]; tensor input_27_pad_0 = const()[name = string("input_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_27_dilations_0 = const()[name = string("input_27_dilations_0"), val = tensor([1, 1])]; int32 input_27_groups_0 = const()[name = string("input_27_groups_0"), val = int32(1)]; tensor input_27_cast_fp16 = conv(dilations = input_27_dilations_0, groups = input_27_groups_0, pad = input_27_pad_0, pad_type = input_27_pad_type_0, strides = input_27_strides_0, weight = layers_13_mlp_gate_proj_weight_cast_fp16, x = var_5359_cast_fp16_0)[name = string("input_27_cast_fp16")]; tensor var_5376_cast_fp16 = silu(x = input_27_cast_fp16)[name = string("op_5376_cast_fp16")]; tensor layers_13_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_13_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415407680)))]; tensor var_5382_strides_0 = const()[name = string("op_5382_strides_0"), val = tensor([1, 1])]; string var_5382_pad_type_0 = const()[name = string("op_5382_pad_type_0"), val = string("valid")]; tensor var_5382_pad_0 = const()[name = string("op_5382_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5382_dilations_0 = const()[name = string("op_5382_dilations_0"), val = tensor([1, 1])]; int32 var_5382_groups_0 = const()[name = string("op_5382_groups_0"), val = int32(1)]; tensor var_5382_cast_fp16 = conv(dilations = var_5382_dilations_0, groups = var_5382_groups_0, pad = var_5382_pad_0, pad_type = var_5382_pad_type_0, strides = var_5382_strides_0, weight = layers_13_mlp_up_proj_weight_to_fp16, x = var_5359_cast_fp16_0)[name = string("op_5382_cast_fp16")]; tensor x_139_cast_fp16 = mul(x = var_5376_cast_fp16, y = var_5382_cast_fp16)[name = string("x_139_cast_fp16")]; tensor hidden_states_137_strides_0 = const()[name = string("hidden_states_137_strides_0"), val = tensor([1, 1])]; string hidden_states_137_pad_type_0 = const()[name = string("hidden_states_137_pad_type_0"), val = string("valid")]; tensor hidden_states_137_pad_0 = const()[name = string("hidden_states_137_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_137_dilations_0 = const()[name = string("hidden_states_137_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_137_groups_0 = const()[name = string("hidden_states_137_groups_0"), val = int32(1)]; tensor hidden_states_137_cast_fp16 = conv(dilations = hidden_states_137_dilations_0, groups = hidden_states_137_groups_0, pad = hidden_states_137_pad_0, pad_type = hidden_states_137_pad_type_0, strides = hidden_states_137_strides_0, weight = layers_13_mlp_down_proj_weight_cast_fp16, x = x_139_cast_fp16)[name = string("hidden_states_137_cast_fp16")]; tensor hidden_states_139_cast_fp16 = add(x = hidden_states_135_cast_fp16, y = hidden_states_137_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; fp16 const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5400_cast_fp16 = mul(x = hidden_states_139_cast_fp16, y = const_140_promoted_to_fp16)[name = string("op_5400_cast_fp16")]; int32 var_5398 = const()[name = string("op_5398"), val = int32(1)]; bool doubled_113_interleave_0 = const()[name = string("doubled_113_interleave_0"), val = bool(false)]; tensor doubled_113_cast_fp16 = concat(axis = var_5398, interleave = doubled_113_interleave_0, values = (hidden_states_139_cast_fp16, var_5400_cast_fp16))[name = string("doubled_113_cast_fp16")]; tensor out_57_axes_0 = const()[name = string("out_57_axes_0"), val = tensor([1])]; tensor out_57_gamma_0_to_fp16 = const()[name = string("out_57_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440573568)))]; fp16 var_5410_to_fp16 = const()[name = string("op_5410_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_57_cast_fp16 = layer_norm(axes = out_57_axes_0, epsilon = var_5410_to_fp16, gamma = out_57_gamma_0_to_fp16, x = doubled_113_cast_fp16)[name = string("out_57_cast_fp16")]; tensor var_5421_split_sizes_0 = const()[name = string("op_5421_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5421_axis_0 = const()[name = string("op_5421_axis_0"), val = int32(1)]; tensor var_5421_cast_fp16_0, tensor var_5421_cast_fp16_1 = split(axis = var_5421_axis_0, split_sizes = var_5421_split_sizes_0, x = out_57_cast_fp16)[name = string("op_5421_cast_fp16")]; tensor query_states_85_strides_0 = const()[name = string("query_states_85_strides_0"), val = tensor([1, 1])]; string query_states_85_pad_type_0 = const()[name = string("query_states_85_pad_type_0"), val = string("valid")]; tensor query_states_85_pad_0 = const()[name = string("query_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_85_dilations_0 = const()[name = string("query_states_85_dilations_0"), val = tensor([1, 1])]; int32 query_states_85_groups_0 = const()[name = string("query_states_85_groups_0"), val = int32(1)]; tensor query_states_85_cast_fp16 = conv(dilations = query_states_85_dilations_0, groups = query_states_85_groups_0, pad = query_states_85_pad_0, pad_type = query_states_85_pad_type_0, strides = query_states_85_strides_0, weight = layers_14_self_attn_q_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("query_states_85_cast_fp16")]; tensor layers_14_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_14_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440581824)))]; tensor key_states_141_strides_0 = const()[name = string("key_states_141_strides_0"), val = tensor([1, 1])]; string key_states_141_pad_type_0 = const()[name = string("key_states_141_pad_type_0"), val = string("valid")]; tensor key_states_141_pad_0 = const()[name = string("key_states_141_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_141_dilations_0 = const()[name = string("key_states_141_dilations_0"), val = tensor([1, 1])]; int32 key_states_141_groups_0 = const()[name = string("key_states_141_groups_0"), val = int32(1)]; tensor key_states_141_cast_fp16 = conv(dilations = key_states_141_dilations_0, groups = key_states_141_groups_0, pad = key_states_141_pad_0, pad_type = key_states_141_pad_type_0, strides = key_states_141_strides_0, weight = layers_14_self_attn_k_proj_weight_to_fp16, x = var_5421_cast_fp16_0)[name = string("key_states_141_cast_fp16")]; tensor value_states_85_strides_0 = const()[name = string("value_states_85_strides_0"), val = tensor([1, 1])]; string value_states_85_pad_type_0 = const()[name = string("value_states_85_pad_type_0"), val = string("valid")]; tensor value_states_85_pad_0 = const()[name = string("value_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_85_dilations_0 = const()[name = string("value_states_85_dilations_0"), val = tensor([1, 1])]; int32 value_states_85_groups_0 = const()[name = string("value_states_85_groups_0"), val = int32(1)]; tensor value_states_85_cast_fp16 = conv(dilations = value_states_85_dilations_0, groups = value_states_85_groups_0, pad = value_states_85_pad_0, pad_type = value_states_85_pad_type_0, strides = value_states_85_strides_0, weight = layers_14_self_attn_v_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("value_states_85_cast_fp16")]; tensor concat_168x = const()[name = string("concat_168x"), val = tensor([1, 16, 128, -1])]; tensor x_141_cast_fp16 = reshape(shape = concat_168x, x = query_states_85_cast_fp16)[name = string("x_141_cast_fp16")]; tensor concat_169x = const()[name = string("concat_169x"), val = tensor([1, 2, 128, -1])]; tensor var_5478_cast_fp16 = reshape(shape = concat_169x, x = key_states_141_cast_fp16)[name = string("op_5478_cast_fp16")]; tensor concat_170x = const()[name = string("concat_170x"), val = tensor([1, 2, 128, -1])]; tensor var_5485_cast_fp16 = reshape(shape = concat_170x, x = value_states_85_cast_fp16)[name = string("op_5485_cast_fp16")]; tensor var_5489_cast_fp16 = mul(x = x_141_cast_fp16, y = var_869_cast_fp16)[name = string("op_5489_cast_fp16")]; tensor var_5490_split_sizes_0 = const()[name = string("op_5490_split_sizes_0"), val = tensor([64, 64])]; int32 var_5490_axis_0 = const()[name = string("op_5490_axis_0"), val = int32(-2)]; tensor var_5490_cast_fp16_0, tensor var_5490_cast_fp16_1 = split(axis = var_5490_axis_0, split_sizes = var_5490_split_sizes_0, x = x_141_cast_fp16)[name = string("op_5490_cast_fp16")]; fp16 const_142_promoted_to_fp16 = const()[name = string("const_142_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5492_cast_fp16 = mul(x = var_5490_cast_fp16_1, y = const_142_promoted_to_fp16)[name = string("op_5492_cast_fp16")]; int32 var_5494 = const()[name = string("op_5494"), val = int32(-2)]; bool var_5495_interleave_0 = const()[name = string("op_5495_interleave_0"), val = bool(false)]; tensor var_5495_cast_fp16 = concat(axis = var_5494, interleave = var_5495_interleave_0, values = (var_5492_cast_fp16, var_5490_cast_fp16_0))[name = string("op_5495_cast_fp16")]; tensor var_5496_cast_fp16 = mul(x = var_5495_cast_fp16, y = var_878_cast_fp16)[name = string("op_5496_cast_fp16")]; tensor query_states_87_cast_fp16 = add(x = var_5489_cast_fp16, y = var_5496_cast_fp16)[name = string("query_states_87_cast_fp16")]; tensor var_5502_cast_fp16 = mul(x = var_5478_cast_fp16, y = var_869_cast_fp16)[name = string("op_5502_cast_fp16")]; tensor var_5503_split_sizes_0 = const()[name = string("op_5503_split_sizes_0"), val = tensor([64, 64])]; int32 var_5503_axis_0 = const()[name = string("op_5503_axis_0"), val = int32(-2)]; tensor var_5503_cast_fp16_0, tensor var_5503_cast_fp16_1 = split(axis = var_5503_axis_0, split_sizes = var_5503_split_sizes_0, x = var_5478_cast_fp16)[name = string("op_5503_cast_fp16")]; fp16 const_143_promoted_to_fp16 = const()[name = string("const_143_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5505_cast_fp16 = mul(x = var_5503_cast_fp16_1, y = const_143_promoted_to_fp16)[name = string("op_5505_cast_fp16")]; int32 var_5507 = const()[name = string("op_5507"), val = int32(-2)]; bool var_5508_interleave_0 = const()[name = string("op_5508_interleave_0"), val = bool(false)]; tensor var_5508_cast_fp16 = concat(axis = var_5507, interleave = var_5508_interleave_0, values = (var_5505_cast_fp16, var_5503_cast_fp16_0))[name = string("op_5508_cast_fp16")]; tensor var_5509_cast_fp16 = mul(x = var_5508_cast_fp16, y = var_878_cast_fp16)[name = string("op_5509_cast_fp16")]; tensor key_states_145_cast_fp16 = add(x = var_5502_cast_fp16, y = var_5509_cast_fp16)[name = string("key_states_145_cast_fp16")]; tensor expand_dims_168 = const()[name = string("expand_dims_168"), val = tensor([14])]; tensor expand_dims_169 = const()[name = string("expand_dims_169"), val = tensor([0])]; tensor expand_dims_171 = const()[name = string("expand_dims_171"), val = tensor([0])]; int32 concat_173_axis_0 = const()[name = string("concat_173_axis_0"), val = int32(0)]; bool concat_173_interleave_0 = const()[name = string("concat_173_interleave_0"), val = bool(false)]; tensor concat_173 = concat(axis = concat_173_axis_0, interleave = concat_173_interleave_0, values = (expand_dims_168, expand_dims_169, position_id, expand_dims_171))[name = string("concat_173")]; tensor expand_dims_172 = const()[name = string("expand_dims_172"), val = tensor([15])]; tensor concat_174_values1_0 = const()[name = string("concat_174_values1_0"), val = tensor([0])]; tensor concat_174_values3_0 = const()[name = string("concat_174_values3_0"), val = tensor([0])]; int32 concat_174_axis_0 = const()[name = string("concat_174_axis_0"), val = int32(0)]; bool concat_174_interleave_0 = const()[name = string("concat_174_interleave_0"), val = bool(false)]; tensor concat_174 = concat(axis = concat_174_axis_0, interleave = concat_174_interleave_0, values = (expand_dims_172, concat_174_values1_0, cache_position_end, concat_174_values3_0))[name = string("concat_174")]; tensor key_states_147_perm_0 = const()[name = string("key_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_15_stride_0 = const()[name = string("key_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_147_cast_fp16 = transpose(perm = key_states_147_perm_0, x = key_states_145_cast_fp16)[name = string("transpose_471")]; tensor key_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = key_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = key_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_15_squeeze_mask_0, stride = key_cache_internal_tensor_assign_15_stride_0, update = key_states_147_cast_fp16, x = coreml_update_state_306)[name = string("key_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_15_cast_fp16, input = key_cache)[name = string("coreml_update_state_308_write_state")]; tensor coreml_update_state_308 = read_state(input = key_cache)[name = string("coreml_update_state_308")]; tensor value_states_87_perm_0 = const()[name = string("value_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_15_stride_0 = const()[name = string("value_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_87_cast_fp16 = transpose(perm = value_states_87_perm_0, x = var_5485_cast_fp16)[name = string("transpose_470")]; tensor value_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = value_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = value_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_15_squeeze_mask_0, stride = value_cache_internal_tensor_assign_15_stride_0, update = value_states_87_cast_fp16, x = coreml_update_state_307)[name = string("value_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_15_cast_fp16, input = value_cache)[name = string("coreml_update_state_309_write_state")]; tensor coreml_update_state_309 = read_state(input = value_cache)[name = string("coreml_update_state_309")]; tensor var_5579_begin_0 = const()[name = string("op_5579_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5579_end_0 = const()[name = string("op_5579_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5579_end_mask_0 = const()[name = string("op_5579_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5579_cast_fp16 = slice_by_index(begin = var_5579_begin_0, end = var_5579_end_0, end_mask = var_5579_end_mask_0, x = coreml_update_state_308)[name = string("op_5579_cast_fp16")]; tensor tile_28 = const()[name = string("tile_28"), val = tensor([1, 1])]; int32 var_5582_axis_0 = const()[name = string("op_5582_axis_0"), val = int32(1)]; tensor var_5582_cast_fp16_0, tensor var_5582_cast_fp16_1 = split(axis = var_5582_axis_0, split_sizes = tile_28, x = var_5579_cast_fp16)[name = string("op_5582_cast_fp16")]; tensor var_5589_begin_0 = const()[name = string("op_5589_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5589_end_0 = const()[name = string("op_5589_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5589_end_mask_0 = const()[name = string("op_5589_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5589_cast_fp16 = slice_by_index(begin = var_5589_begin_0, end = var_5589_end_0, end_mask = var_5589_end_mask_0, x = coreml_update_state_309)[name = string("op_5589_cast_fp16")]; tensor tile_29 = const()[name = string("tile_29"), val = tensor([1, 1])]; int32 var_5592_axis_0 = const()[name = string("op_5592_axis_0"), val = int32(1)]; tensor var_5592_cast_fp16_0, tensor var_5592_cast_fp16_1 = split(axis = var_5592_axis_0, split_sizes = tile_29, x = var_5589_cast_fp16)[name = string("op_5592_cast_fp16")]; tensor var_5595_split_sizes_0 = const()[name = string("op_5595_split_sizes_0"), val = tensor([8, 8])]; int32 var_5595_axis_0 = const()[name = string("op_5595_axis_0"), val = int32(1)]; tensor var_5595_0, tensor var_5595_1 = split(axis = var_5595_axis_0, split_sizes = var_5595_split_sizes_0, x = query_states_87_cast_fp16)[name = string("op_5595")]; bool attn_weights_225_transpose_x_0 = const()[name = string("attn_weights_225_transpose_x_0"), val = bool(false)]; bool attn_weights_225_transpose_y_0 = const()[name = string("attn_weights_225_transpose_y_0"), val = bool(false)]; tensor attn_weights_225_cast_fp16 = matmul(transpose_x = attn_weights_225_transpose_x_0, transpose_y = attn_weights_225_transpose_y_0, x = var_5582_cast_fp16_0, y = var_5595_0)[name = string("attn_weights_225_cast_fp16")]; fp16 var_5598_to_fp16 = const()[name = string("op_5598_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_227_cast_fp16 = mul(x = attn_weights_225_cast_fp16, y = var_5598_to_fp16)[name = string("attn_weights_227_cast_fp16")]; tensor attn_weights_229_cast_fp16 = add(x = attn_weights_227_cast_fp16, y = attn_mask_1)[name = string("attn_weights_229_cast_fp16")]; int32 var_5602 = const()[name = string("op_5602"), val = int32(-2)]; tensor attn_weights_231_cast_fp16 = softmax(axis = var_5602, x = attn_weights_229_cast_fp16)[name = string("attn_weights_231_cast_fp16")]; bool var_5608_transpose_x_1 = const()[name = string("op_5608_transpose_x_1"), val = bool(true)]; bool var_5608_transpose_y_1 = const()[name = string("op_5608_transpose_y_1"), val = bool(false)]; tensor var_5608_cast_fp16 = matmul(transpose_x = var_5608_transpose_x_1, transpose_y = var_5608_transpose_y_1, x = attn_weights_231_cast_fp16, y = var_5592_cast_fp16_0)[name = string("op_5608_cast_fp16")]; bool attn_weights_233_transpose_x_0 = const()[name = string("attn_weights_233_transpose_x_0"), val = bool(false)]; bool attn_weights_233_transpose_y_0 = const()[name = string("attn_weights_233_transpose_y_0"), val = bool(false)]; tensor attn_weights_233_cast_fp16 = matmul(transpose_x = attn_weights_233_transpose_x_0, transpose_y = attn_weights_233_transpose_y_0, x = var_5582_cast_fp16_1, y = var_5595_1)[name = string("attn_weights_233_cast_fp16")]; fp16 var_5610_to_fp16 = const()[name = string("op_5610_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_235_cast_fp16 = mul(x = attn_weights_233_cast_fp16, y = var_5610_to_fp16)[name = string("attn_weights_235_cast_fp16")]; tensor attn_weights_237_cast_fp16 = add(x = attn_weights_235_cast_fp16, y = attn_mask_1)[name = string("attn_weights_237_cast_fp16")]; int32 var_5614 = const()[name = string("op_5614"), val = int32(-2)]; tensor attn_weights_239_cast_fp16 = softmax(axis = var_5614, x = attn_weights_237_cast_fp16)[name = string("attn_weights_239_cast_fp16")]; bool attn_output_113_transpose_x_1 = const()[name = string("attn_output_113_transpose_x_1"), val = bool(true)]; bool attn_output_113_transpose_y_1 = const()[name = string("attn_output_113_transpose_y_1"), val = bool(false)]; tensor attn_output_113_cast_fp16 = matmul(transpose_x = attn_output_113_transpose_x_1, transpose_y = attn_output_113_transpose_y_1, x = attn_weights_239_cast_fp16, y = var_5592_cast_fp16_1)[name = string("attn_output_113_cast_fp16")]; int32 var_5622 = const()[name = string("op_5622"), val = int32(1)]; bool attn_output_115_interleave_0 = const()[name = string("attn_output_115_interleave_0"), val = bool(false)]; tensor attn_output_115_cast_fp16 = concat(axis = var_5622, interleave = attn_output_115_interleave_0, values = (var_5608_cast_fp16, attn_output_113_cast_fp16))[name = string("attn_output_115_cast_fp16")]; tensor var_5626_perm_0 = const()[name = string("op_5626_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_179x = const()[name = string("concat_179x"), val = tensor([1, 2048, 1, -1])]; tensor var_5626_cast_fp16 = transpose(perm = var_5626_perm_0, x = attn_output_115_cast_fp16)[name = string("transpose_469")]; tensor attn_output_119_cast_fp16 = reshape(shape = concat_179x, x = var_5626_cast_fp16)[name = string("attn_output_119_cast_fp16")]; tensor hidden_states_143_strides_0 = const()[name = string("hidden_states_143_strides_0"), val = tensor([1, 1])]; string hidden_states_143_pad_type_0 = const()[name = string("hidden_states_143_pad_type_0"), val = string("valid")]; tensor hidden_states_143_pad_0 = const()[name = string("hidden_states_143_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_143_dilations_0 = const()[name = string("hidden_states_143_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_143_groups_0 = const()[name = string("hidden_states_143_groups_0"), val = int32(1)]; tensor hidden_states_143_cast_fp16 = conv(dilations = hidden_states_143_dilations_0, groups = hidden_states_143_groups_0, pad = hidden_states_143_pad_0, pad_type = hidden_states_143_pad_type_0, strides = hidden_states_143_strides_0, weight = layers_14_self_attn_o_proj_weight_cast_fp16, x = attn_output_119_cast_fp16)[name = string("hidden_states_143_cast_fp16")]; tensor hidden_states_145_cast_fp16 = add(x = hidden_states_139_cast_fp16, y = hidden_states_143_cast_fp16)[name = string("hidden_states_145_cast_fp16")]; fp16 const_148_promoted_to_fp16 = const()[name = string("const_148_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5659_cast_fp16 = mul(x = hidden_states_145_cast_fp16, y = const_148_promoted_to_fp16)[name = string("op_5659_cast_fp16")]; int32 var_5657 = const()[name = string("op_5657"), val = int32(1)]; bool doubled_117_interleave_0 = const()[name = string("doubled_117_interleave_0"), val = bool(false)]; tensor doubled_117_cast_fp16 = concat(axis = var_5657, interleave = doubled_117_interleave_0, values = (hidden_states_145_cast_fp16, var_5659_cast_fp16))[name = string("doubled_117_cast_fp16")]; tensor out_59_axes_0 = const()[name = string("out_59_axes_0"), val = tensor([1])]; tensor out_59_gamma_0_to_fp16 = const()[name = string("out_59_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441630464)))]; fp16 var_5669_to_fp16 = const()[name = string("op_5669_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_59_cast_fp16 = layer_norm(axes = out_59_axes_0, epsilon = var_5669_to_fp16, gamma = out_59_gamma_0_to_fp16, x = doubled_117_cast_fp16)[name = string("out_59_cast_fp16")]; tensor var_5680_split_sizes_0 = const()[name = string("op_5680_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5680_axis_0 = const()[name = string("op_5680_axis_0"), val = int32(1)]; tensor var_5680_cast_fp16_0, tensor var_5680_cast_fp16_1 = split(axis = var_5680_axis_0, split_sizes = var_5680_split_sizes_0, x = out_59_cast_fp16)[name = string("op_5680_cast_fp16")]; tensor input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor([1, 1])]; string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; tensor input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor([1, 1])]; int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; tensor input_29_cast_fp16 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_14_mlp_gate_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("input_29_cast_fp16")]; tensor var_5697_cast_fp16 = silu(x = input_29_cast_fp16)[name = string("op_5697_cast_fp16")]; tensor var_5703_strides_0 = const()[name = string("op_5703_strides_0"), val = tensor([1, 1])]; string var_5703_pad_type_0 = const()[name = string("op_5703_pad_type_0"), val = string("valid")]; tensor var_5703_pad_0 = const()[name = string("op_5703_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5703_dilations_0 = const()[name = string("op_5703_dilations_0"), val = tensor([1, 1])]; int32 var_5703_groups_0 = const()[name = string("op_5703_groups_0"), val = int32(1)]; tensor var_5703_cast_fp16 = conv(dilations = var_5703_dilations_0, groups = var_5703_groups_0, pad = var_5703_pad_0, pad_type = var_5703_pad_type_0, strides = var_5703_strides_0, weight = layers_14_mlp_up_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("op_5703_cast_fp16")]; tensor x_149_cast_fp16 = mul(x = var_5697_cast_fp16, y = var_5703_cast_fp16)[name = string("x_149_cast_fp16")]; tensor hidden_states_147_strides_0 = const()[name = string("hidden_states_147_strides_0"), val = tensor([1, 1])]; string hidden_states_147_pad_type_0 = const()[name = string("hidden_states_147_pad_type_0"), val = string("valid")]; tensor hidden_states_147_pad_0 = const()[name = string("hidden_states_147_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_147_dilations_0 = const()[name = string("hidden_states_147_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_147_groups_0 = const()[name = string("hidden_states_147_groups_0"), val = int32(1)]; tensor hidden_states_147_cast_fp16 = conv(dilations = hidden_states_147_dilations_0, groups = hidden_states_147_groups_0, pad = hidden_states_147_pad_0, pad_type = hidden_states_147_pad_type_0, strides = hidden_states_147_strides_0, weight = layers_14_mlp_down_proj_weight_cast_fp16, x = x_149_cast_fp16)[name = string("hidden_states_147_cast_fp16")]; tensor hidden_states_149_cast_fp16 = add(x = hidden_states_145_cast_fp16, y = hidden_states_147_cast_fp16)[name = string("hidden_states_149_cast_fp16")]; fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5721_cast_fp16 = mul(x = hidden_states_149_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_5721_cast_fp16")]; int32 var_5719 = const()[name = string("op_5719"), val = int32(1)]; bool doubled_121_interleave_0 = const()[name = string("doubled_121_interleave_0"), val = bool(false)]; tensor doubled_121_cast_fp16 = concat(axis = var_5719, interleave = doubled_121_interleave_0, values = (hidden_states_149_cast_fp16, var_5721_cast_fp16))[name = string("doubled_121_cast_fp16")]; tensor out_61_axes_0 = const()[name = string("out_61_axes_0"), val = tensor([1])]; tensor out_61_gamma_0_to_fp16 = const()[name = string("out_61_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441638720)))]; fp16 var_5731_to_fp16 = const()[name = string("op_5731_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_61_cast_fp16 = layer_norm(axes = out_61_axes_0, epsilon = var_5731_to_fp16, gamma = out_61_gamma_0_to_fp16, x = doubled_121_cast_fp16)[name = string("out_61_cast_fp16")]; tensor var_5742_split_sizes_0 = const()[name = string("op_5742_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5742_axis_0 = const()[name = string("op_5742_axis_0"), val = int32(1)]; tensor var_5742_cast_fp16_0, tensor var_5742_cast_fp16_1 = split(axis = var_5742_axis_0, split_sizes = var_5742_split_sizes_0, x = out_61_cast_fp16)[name = string("op_5742_cast_fp16")]; tensor query_states_91_strides_0 = const()[name = string("query_states_91_strides_0"), val = tensor([1, 1])]; string query_states_91_pad_type_0 = const()[name = string("query_states_91_pad_type_0"), val = string("valid")]; tensor query_states_91_pad_0 = const()[name = string("query_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_91_dilations_0 = const()[name = string("query_states_91_dilations_0"), val = tensor([1, 1])]; int32 query_states_91_groups_0 = const()[name = string("query_states_91_groups_0"), val = int32(1)]; tensor query_states_91_cast_fp16 = conv(dilations = query_states_91_dilations_0, groups = query_states_91_groups_0, pad = query_states_91_pad_0, pad_type = query_states_91_pad_type_0, strides = query_states_91_strides_0, weight = layers_15_self_attn_q_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("query_states_91_cast_fp16")]; tensor key_states_151_strides_0 = const()[name = string("key_states_151_strides_0"), val = tensor([1, 1])]; string key_states_151_pad_type_0 = const()[name = string("key_states_151_pad_type_0"), val = string("valid")]; tensor key_states_151_pad_0 = const()[name = string("key_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_151_dilations_0 = const()[name = string("key_states_151_dilations_0"), val = tensor([1, 1])]; int32 key_states_151_groups_0 = const()[name = string("key_states_151_groups_0"), val = int32(1)]; tensor key_states_151_cast_fp16 = conv(dilations = key_states_151_dilations_0, groups = key_states_151_groups_0, pad = key_states_151_pad_0, pad_type = key_states_151_pad_type_0, strides = key_states_151_strides_0, weight = layers_15_self_attn_k_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("key_states_151_cast_fp16")]; tensor value_states_91_strides_0 = const()[name = string("value_states_91_strides_0"), val = tensor([1, 1])]; string value_states_91_pad_type_0 = const()[name = string("value_states_91_pad_type_0"), val = string("valid")]; tensor value_states_91_pad_0 = const()[name = string("value_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_91_dilations_0 = const()[name = string("value_states_91_dilations_0"), val = tensor([1, 1])]; int32 value_states_91_groups_0 = const()[name = string("value_states_91_groups_0"), val = int32(1)]; tensor value_states_91_cast_fp16 = conv(dilations = value_states_91_dilations_0, groups = value_states_91_groups_0, pad = value_states_91_pad_0, pad_type = value_states_91_pad_type_0, strides = value_states_91_strides_0, weight = layers_15_self_attn_v_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("value_states_91_cast_fp16")]; tensor concat_180x = const()[name = string("concat_180x"), val = tensor([1, 16, 128, -1])]; tensor x_151_cast_fp16 = reshape(shape = concat_180x, x = query_states_91_cast_fp16)[name = string("x_151_cast_fp16")]; tensor concat_181x = const()[name = string("concat_181x"), val = tensor([1, 2, 128, -1])]; tensor var_5799_cast_fp16 = reshape(shape = concat_181x, x = key_states_151_cast_fp16)[name = string("op_5799_cast_fp16")]; tensor concat_182x = const()[name = string("concat_182x"), val = tensor([1, 2, 128, -1])]; tensor var_5806_cast_fp16 = reshape(shape = concat_182x, x = value_states_91_cast_fp16)[name = string("op_5806_cast_fp16")]; tensor var_5810_cast_fp16 = mul(x = x_151_cast_fp16, y = var_869_cast_fp16)[name = string("op_5810_cast_fp16")]; tensor var_5811_split_sizes_0 = const()[name = string("op_5811_split_sizes_0"), val = tensor([64, 64])]; int32 var_5811_axis_0 = const()[name = string("op_5811_axis_0"), val = int32(-2)]; tensor var_5811_cast_fp16_0, tensor var_5811_cast_fp16_1 = split(axis = var_5811_axis_0, split_sizes = var_5811_split_sizes_0, x = x_151_cast_fp16)[name = string("op_5811_cast_fp16")]; fp16 const_152_promoted_to_fp16 = const()[name = string("const_152_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5813_cast_fp16 = mul(x = var_5811_cast_fp16_1, y = const_152_promoted_to_fp16)[name = string("op_5813_cast_fp16")]; int32 var_5815 = const()[name = string("op_5815"), val = int32(-2)]; bool var_5816_interleave_0 = const()[name = string("op_5816_interleave_0"), val = bool(false)]; tensor var_5816_cast_fp16 = concat(axis = var_5815, interleave = var_5816_interleave_0, values = (var_5813_cast_fp16, var_5811_cast_fp16_0))[name = string("op_5816_cast_fp16")]; tensor var_5817_cast_fp16 = mul(x = var_5816_cast_fp16, y = var_878_cast_fp16)[name = string("op_5817_cast_fp16")]; tensor query_states_93_cast_fp16 = add(x = var_5810_cast_fp16, y = var_5817_cast_fp16)[name = string("query_states_93_cast_fp16")]; tensor var_5823_cast_fp16 = mul(x = var_5799_cast_fp16, y = var_869_cast_fp16)[name = string("op_5823_cast_fp16")]; tensor var_5824_split_sizes_0 = const()[name = string("op_5824_split_sizes_0"), val = tensor([64, 64])]; int32 var_5824_axis_0 = const()[name = string("op_5824_axis_0"), val = int32(-2)]; tensor var_5824_cast_fp16_0, tensor var_5824_cast_fp16_1 = split(axis = var_5824_axis_0, split_sizes = var_5824_split_sizes_0, x = var_5799_cast_fp16)[name = string("op_5824_cast_fp16")]; fp16 const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5826_cast_fp16 = mul(x = var_5824_cast_fp16_1, y = const_153_promoted_to_fp16)[name = string("op_5826_cast_fp16")]; int32 var_5828 = const()[name = string("op_5828"), val = int32(-2)]; bool var_5829_interleave_0 = const()[name = string("op_5829_interleave_0"), val = bool(false)]; tensor var_5829_cast_fp16 = concat(axis = var_5828, interleave = var_5829_interleave_0, values = (var_5826_cast_fp16, var_5824_cast_fp16_0))[name = string("op_5829_cast_fp16")]; tensor var_5830_cast_fp16 = mul(x = var_5829_cast_fp16, y = var_878_cast_fp16)[name = string("op_5830_cast_fp16")]; tensor key_states_155_cast_fp16 = add(x = var_5823_cast_fp16, y = var_5830_cast_fp16)[name = string("key_states_155_cast_fp16")]; tensor expand_dims_180 = const()[name = string("expand_dims_180"), val = tensor([15])]; tensor expand_dims_181 = const()[name = string("expand_dims_181"), val = tensor([0])]; tensor expand_dims_183 = const()[name = string("expand_dims_183"), val = tensor([0])]; int32 concat_185_axis_0 = const()[name = string("concat_185_axis_0"), val = int32(0)]; bool concat_185_interleave_0 = const()[name = string("concat_185_interleave_0"), val = bool(false)]; tensor concat_185 = concat(axis = concat_185_axis_0, interleave = concat_185_interleave_0, values = (expand_dims_180, expand_dims_181, position_id, expand_dims_183))[name = string("concat_185")]; tensor expand_dims_184 = const()[name = string("expand_dims_184"), val = tensor([16])]; tensor concat_186_values1_0 = const()[name = string("concat_186_values1_0"), val = tensor([0])]; tensor concat_186_values3_0 = const()[name = string("concat_186_values3_0"), val = tensor([0])]; int32 concat_186_axis_0 = const()[name = string("concat_186_axis_0"), val = int32(0)]; bool concat_186_interleave_0 = const()[name = string("concat_186_interleave_0"), val = bool(false)]; tensor concat_186 = concat(axis = concat_186_axis_0, interleave = concat_186_interleave_0, values = (expand_dims_184, concat_186_values1_0, cache_position_end, concat_186_values3_0))[name = string("concat_186")]; tensor key_states_157_perm_0 = const()[name = string("key_states_157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_16_stride_0 = const()[name = string("key_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_157_cast_fp16 = transpose(perm = key_states_157_perm_0, x = key_states_155_cast_fp16)[name = string("transpose_468")]; tensor key_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = key_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = key_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_16_squeeze_mask_0, stride = key_cache_internal_tensor_assign_16_stride_0, update = key_states_157_cast_fp16, x = coreml_update_state_308)[name = string("key_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_16_cast_fp16, input = key_cache)[name = string("coreml_update_state_310_write_state")]; tensor coreml_update_state_310 = read_state(input = key_cache)[name = string("coreml_update_state_310")]; tensor value_states_93_perm_0 = const()[name = string("value_states_93_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_16_stride_0 = const()[name = string("value_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_93_cast_fp16 = transpose(perm = value_states_93_perm_0, x = var_5806_cast_fp16)[name = string("transpose_467")]; tensor value_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = value_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = value_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_16_squeeze_mask_0, stride = value_cache_internal_tensor_assign_16_stride_0, update = value_states_93_cast_fp16, x = coreml_update_state_309)[name = string("value_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_16_cast_fp16, input = value_cache)[name = string("coreml_update_state_311_write_state")]; tensor coreml_update_state_311 = read_state(input = value_cache)[name = string("coreml_update_state_311")]; tensor var_5900_begin_0 = const()[name = string("op_5900_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5900_end_0 = const()[name = string("op_5900_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5900_end_mask_0 = const()[name = string("op_5900_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5900_cast_fp16 = slice_by_index(begin = var_5900_begin_0, end = var_5900_end_0, end_mask = var_5900_end_mask_0, x = coreml_update_state_310)[name = string("op_5900_cast_fp16")]; tensor tile_30 = const()[name = string("tile_30"), val = tensor([1, 1])]; int32 var_5903_axis_0 = const()[name = string("op_5903_axis_0"), val = int32(1)]; tensor var_5903_cast_fp16_0, tensor var_5903_cast_fp16_1 = split(axis = var_5903_axis_0, split_sizes = tile_30, x = var_5900_cast_fp16)[name = string("op_5903_cast_fp16")]; tensor var_5910_begin_0 = const()[name = string("op_5910_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5910_end_0 = const()[name = string("op_5910_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5910_end_mask_0 = const()[name = string("op_5910_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5910_cast_fp16 = slice_by_index(begin = var_5910_begin_0, end = var_5910_end_0, end_mask = var_5910_end_mask_0, x = coreml_update_state_311)[name = string("op_5910_cast_fp16")]; tensor tile_31 = const()[name = string("tile_31"), val = tensor([1, 1])]; int32 var_5913_axis_0 = const()[name = string("op_5913_axis_0"), val = int32(1)]; tensor var_5913_cast_fp16_0, tensor var_5913_cast_fp16_1 = split(axis = var_5913_axis_0, split_sizes = tile_31, x = var_5910_cast_fp16)[name = string("op_5913_cast_fp16")]; tensor var_5916_split_sizes_0 = const()[name = string("op_5916_split_sizes_0"), val = tensor([8, 8])]; int32 var_5916_axis_0 = const()[name = string("op_5916_axis_0"), val = int32(1)]; tensor var_5916_0, tensor var_5916_1 = split(axis = var_5916_axis_0, split_sizes = var_5916_split_sizes_0, x = query_states_93_cast_fp16)[name = string("op_5916")]; bool attn_weights_241_transpose_x_0 = const()[name = string("attn_weights_241_transpose_x_0"), val = bool(false)]; bool attn_weights_241_transpose_y_0 = const()[name = string("attn_weights_241_transpose_y_0"), val = bool(false)]; tensor attn_weights_241_cast_fp16 = matmul(transpose_x = attn_weights_241_transpose_x_0, transpose_y = attn_weights_241_transpose_y_0, x = var_5903_cast_fp16_0, y = var_5916_0)[name = string("attn_weights_241_cast_fp16")]; fp16 var_5919_to_fp16 = const()[name = string("op_5919_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_243_cast_fp16 = mul(x = attn_weights_241_cast_fp16, y = var_5919_to_fp16)[name = string("attn_weights_243_cast_fp16")]; tensor attn_weights_245_cast_fp16 = add(x = attn_weights_243_cast_fp16, y = attn_mask_1)[name = string("attn_weights_245_cast_fp16")]; int32 var_5923 = const()[name = string("op_5923"), val = int32(-2)]; tensor attn_weights_247_cast_fp16 = softmax(axis = var_5923, x = attn_weights_245_cast_fp16)[name = string("attn_weights_247_cast_fp16")]; bool var_5929_transpose_x_1 = const()[name = string("op_5929_transpose_x_1"), val = bool(true)]; bool var_5929_transpose_y_1 = const()[name = string("op_5929_transpose_y_1"), val = bool(false)]; tensor var_5929_cast_fp16 = matmul(transpose_x = var_5929_transpose_x_1, transpose_y = var_5929_transpose_y_1, x = attn_weights_247_cast_fp16, y = var_5913_cast_fp16_0)[name = string("op_5929_cast_fp16")]; bool attn_weights_249_transpose_x_0 = const()[name = string("attn_weights_249_transpose_x_0"), val = bool(false)]; bool attn_weights_249_transpose_y_0 = const()[name = string("attn_weights_249_transpose_y_0"), val = bool(false)]; tensor attn_weights_249_cast_fp16 = matmul(transpose_x = attn_weights_249_transpose_x_0, transpose_y = attn_weights_249_transpose_y_0, x = var_5903_cast_fp16_1, y = var_5916_1)[name = string("attn_weights_249_cast_fp16")]; fp16 var_5931_to_fp16 = const()[name = string("op_5931_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_251_cast_fp16 = mul(x = attn_weights_249_cast_fp16, y = var_5931_to_fp16)[name = string("attn_weights_251_cast_fp16")]; tensor attn_weights_253_cast_fp16 = add(x = attn_weights_251_cast_fp16, y = attn_mask_1)[name = string("attn_weights_253_cast_fp16")]; int32 var_5935 = const()[name = string("op_5935"), val = int32(-2)]; tensor attn_weights_255_cast_fp16 = softmax(axis = var_5935, x = attn_weights_253_cast_fp16)[name = string("attn_weights_255_cast_fp16")]; bool attn_output_121_transpose_x_1 = const()[name = string("attn_output_121_transpose_x_1"), val = bool(true)]; bool attn_output_121_transpose_y_1 = const()[name = string("attn_output_121_transpose_y_1"), val = bool(false)]; tensor attn_output_121_cast_fp16 = matmul(transpose_x = attn_output_121_transpose_x_1, transpose_y = attn_output_121_transpose_y_1, x = attn_weights_255_cast_fp16, y = var_5913_cast_fp16_1)[name = string("attn_output_121_cast_fp16")]; int32 var_5943 = const()[name = string("op_5943"), val = int32(1)]; bool attn_output_123_interleave_0 = const()[name = string("attn_output_123_interleave_0"), val = bool(false)]; tensor attn_output_123_cast_fp16 = concat(axis = var_5943, interleave = attn_output_123_interleave_0, values = (var_5929_cast_fp16, attn_output_121_cast_fp16))[name = string("attn_output_123_cast_fp16")]; tensor var_5947_perm_0 = const()[name = string("op_5947_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_191x = const()[name = string("concat_191x"), val = tensor([1, 2048, 1, -1])]; tensor var_5947_cast_fp16 = transpose(perm = var_5947_perm_0, x = attn_output_123_cast_fp16)[name = string("transpose_466")]; tensor attn_output_127_cast_fp16 = reshape(shape = concat_191x, x = var_5947_cast_fp16)[name = string("attn_output_127_cast_fp16")]; tensor hidden_states_153_strides_0 = const()[name = string("hidden_states_153_strides_0"), val = tensor([1, 1])]; string hidden_states_153_pad_type_0 = const()[name = string("hidden_states_153_pad_type_0"), val = string("valid")]; tensor hidden_states_153_pad_0 = const()[name = string("hidden_states_153_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_153_dilations_0 = const()[name = string("hidden_states_153_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_153_groups_0 = const()[name = string("hidden_states_153_groups_0"), val = int32(1)]; tensor hidden_states_153_cast_fp16 = conv(dilations = hidden_states_153_dilations_0, groups = hidden_states_153_groups_0, pad = hidden_states_153_pad_0, pad_type = hidden_states_153_pad_type_0, strides = hidden_states_153_strides_0, weight = layers_15_self_attn_o_proj_weight_cast_fp16, x = attn_output_127_cast_fp16)[name = string("hidden_states_153_cast_fp16")]; tensor hidden_states_155_cast_fp16 = add(x = hidden_states_149_cast_fp16, y = hidden_states_153_cast_fp16)[name = string("hidden_states_155_cast_fp16")]; fp16 const_158_promoted_to_fp16 = const()[name = string("const_158_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5980_cast_fp16 = mul(x = hidden_states_155_cast_fp16, y = const_158_promoted_to_fp16)[name = string("op_5980_cast_fp16")]; int32 var_5978 = const()[name = string("op_5978"), val = int32(1)]; bool doubled_125_interleave_0 = const()[name = string("doubled_125_interleave_0"), val = bool(false)]; tensor doubled_125_cast_fp16 = concat(axis = var_5978, interleave = doubled_125_interleave_0, values = (hidden_states_155_cast_fp16, var_5980_cast_fp16))[name = string("doubled_125_cast_fp16")]; tensor out_63_axes_0 = const()[name = string("out_63_axes_0"), val = tensor([1])]; tensor out_63_gamma_0_to_fp16 = const()[name = string("out_63_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441646976)))]; fp16 var_5990_to_fp16 = const()[name = string("op_5990_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_63_cast_fp16 = layer_norm(axes = out_63_axes_0, epsilon = var_5990_to_fp16, gamma = out_63_gamma_0_to_fp16, x = doubled_125_cast_fp16)[name = string("out_63_cast_fp16")]; tensor var_6001_split_sizes_0 = const()[name = string("op_6001_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6001_axis_0 = const()[name = string("op_6001_axis_0"), val = int32(1)]; tensor var_6001_cast_fp16_0, tensor var_6001_cast_fp16_1 = split(axis = var_6001_axis_0, split_sizes = var_6001_split_sizes_0, x = out_63_cast_fp16)[name = string("op_6001_cast_fp16")]; tensor input_31_strides_0 = const()[name = string("input_31_strides_0"), val = tensor([1, 1])]; string input_31_pad_type_0 = const()[name = string("input_31_pad_type_0"), val = string("valid")]; tensor input_31_pad_0 = const()[name = string("input_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_31_dilations_0 = const()[name = string("input_31_dilations_0"), val = tensor([1, 1])]; int32 input_31_groups_0 = const()[name = string("input_31_groups_0"), val = int32(1)]; tensor input_31_cast_fp16 = conv(dilations = input_31_dilations_0, groups = input_31_groups_0, pad = input_31_pad_0, pad_type = input_31_pad_type_0, strides = input_31_strides_0, weight = layers_15_mlp_gate_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("input_31_cast_fp16")]; tensor var_6018_cast_fp16 = silu(x = input_31_cast_fp16)[name = string("op_6018_cast_fp16")]; tensor var_6024_strides_0 = const()[name = string("op_6024_strides_0"), val = tensor([1, 1])]; string var_6024_pad_type_0 = const()[name = string("op_6024_pad_type_0"), val = string("valid")]; tensor var_6024_pad_0 = const()[name = string("op_6024_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6024_dilations_0 = const()[name = string("op_6024_dilations_0"), val = tensor([1, 1])]; int32 var_6024_groups_0 = const()[name = string("op_6024_groups_0"), val = int32(1)]; tensor var_6024_cast_fp16 = conv(dilations = var_6024_dilations_0, groups = var_6024_groups_0, pad = var_6024_pad_0, pad_type = var_6024_pad_type_0, strides = var_6024_strides_0, weight = layers_15_mlp_up_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("op_6024_cast_fp16")]; tensor x_159_cast_fp16 = mul(x = var_6018_cast_fp16, y = var_6024_cast_fp16)[name = string("x_159_cast_fp16")]; tensor hidden_states_157_strides_0 = const()[name = string("hidden_states_157_strides_0"), val = tensor([1, 1])]; string hidden_states_157_pad_type_0 = const()[name = string("hidden_states_157_pad_type_0"), val = string("valid")]; tensor hidden_states_157_pad_0 = const()[name = string("hidden_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_157_dilations_0 = const()[name = string("hidden_states_157_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_157_groups_0 = const()[name = string("hidden_states_157_groups_0"), val = int32(1)]; tensor hidden_states_157_cast_fp16 = conv(dilations = hidden_states_157_dilations_0, groups = hidden_states_157_groups_0, pad = hidden_states_157_pad_0, pad_type = hidden_states_157_pad_type_0, strides = hidden_states_157_strides_0, weight = layers_15_mlp_down_proj_weight_cast_fp16, x = x_159_cast_fp16)[name = string("hidden_states_157_cast_fp16")]; tensor hidden_states_159_cast_fp16 = add(x = hidden_states_155_cast_fp16, y = hidden_states_157_cast_fp16)[name = string("hidden_states_159_cast_fp16")]; fp16 const_160_promoted_to_fp16 = const()[name = string("const_160_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6042_cast_fp16 = mul(x = hidden_states_159_cast_fp16, y = const_160_promoted_to_fp16)[name = string("op_6042_cast_fp16")]; int32 var_6040 = const()[name = string("op_6040"), val = int32(1)]; bool doubled_129_interleave_0 = const()[name = string("doubled_129_interleave_0"), val = bool(false)]; tensor doubled_129_cast_fp16 = concat(axis = var_6040, interleave = doubled_129_interleave_0, values = (hidden_states_159_cast_fp16, var_6042_cast_fp16))[name = string("doubled_129_cast_fp16")]; tensor out_65_axes_0 = const()[name = string("out_65_axes_0"), val = tensor([1])]; tensor out_65_gamma_0_to_fp16 = const()[name = string("out_65_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441655232)))]; fp16 var_6052_to_fp16 = const()[name = string("op_6052_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_65_cast_fp16 = layer_norm(axes = out_65_axes_0, epsilon = var_6052_to_fp16, gamma = out_65_gamma_0_to_fp16, x = doubled_129_cast_fp16)[name = string("out_65_cast_fp16")]; tensor var_6063_split_sizes_0 = const()[name = string("op_6063_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6063_axis_0 = const()[name = string("op_6063_axis_0"), val = int32(1)]; tensor var_6063_cast_fp16_0, tensor var_6063_cast_fp16_1 = split(axis = var_6063_axis_0, split_sizes = var_6063_split_sizes_0, x = out_65_cast_fp16)[name = string("op_6063_cast_fp16")]; tensor query_states_97_strides_0 = const()[name = string("query_states_97_strides_0"), val = tensor([1, 1])]; string query_states_97_pad_type_0 = const()[name = string("query_states_97_pad_type_0"), val = string("valid")]; tensor query_states_97_pad_0 = const()[name = string("query_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_97_dilations_0 = const()[name = string("query_states_97_dilations_0"), val = tensor([1, 1])]; int32 query_states_97_groups_0 = const()[name = string("query_states_97_groups_0"), val = int32(1)]; tensor query_states_97_cast_fp16 = conv(dilations = query_states_97_dilations_0, groups = query_states_97_groups_0, pad = query_states_97_pad_0, pad_type = query_states_97_pad_type_0, strides = query_states_97_strides_0, weight = layers_16_self_attn_q_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("query_states_97_cast_fp16")]; tensor key_states_161_strides_0 = const()[name = string("key_states_161_strides_0"), val = tensor([1, 1])]; string key_states_161_pad_type_0 = const()[name = string("key_states_161_pad_type_0"), val = string("valid")]; tensor key_states_161_pad_0 = const()[name = string("key_states_161_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_161_dilations_0 = const()[name = string("key_states_161_dilations_0"), val = tensor([1, 1])]; int32 key_states_161_groups_0 = const()[name = string("key_states_161_groups_0"), val = int32(1)]; tensor key_states_161_cast_fp16 = conv(dilations = key_states_161_dilations_0, groups = key_states_161_groups_0, pad = key_states_161_pad_0, pad_type = key_states_161_pad_type_0, strides = key_states_161_strides_0, weight = layers_16_self_attn_k_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("key_states_161_cast_fp16")]; tensor value_states_97_strides_0 = const()[name = string("value_states_97_strides_0"), val = tensor([1, 1])]; string value_states_97_pad_type_0 = const()[name = string("value_states_97_pad_type_0"), val = string("valid")]; tensor value_states_97_pad_0 = const()[name = string("value_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_97_dilations_0 = const()[name = string("value_states_97_dilations_0"), val = tensor([1, 1])]; int32 value_states_97_groups_0 = const()[name = string("value_states_97_groups_0"), val = int32(1)]; tensor value_states_97_cast_fp16 = conv(dilations = value_states_97_dilations_0, groups = value_states_97_groups_0, pad = value_states_97_pad_0, pad_type = value_states_97_pad_type_0, strides = value_states_97_strides_0, weight = layers_16_self_attn_v_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("value_states_97_cast_fp16")]; tensor concat_192x = const()[name = string("concat_192x"), val = tensor([1, 16, 128, -1])]; tensor x_161_cast_fp16 = reshape(shape = concat_192x, x = query_states_97_cast_fp16)[name = string("x_161_cast_fp16")]; tensor concat_193x = const()[name = string("concat_193x"), val = tensor([1, 2, 128, -1])]; tensor var_6120_cast_fp16 = reshape(shape = concat_193x, x = key_states_161_cast_fp16)[name = string("op_6120_cast_fp16")]; tensor concat_194x = const()[name = string("concat_194x"), val = tensor([1, 2, 128, -1])]; tensor var_6127_cast_fp16 = reshape(shape = concat_194x, x = value_states_97_cast_fp16)[name = string("op_6127_cast_fp16")]; tensor var_6131_cast_fp16 = mul(x = x_161_cast_fp16, y = var_869_cast_fp16)[name = string("op_6131_cast_fp16")]; tensor var_6132_split_sizes_0 = const()[name = string("op_6132_split_sizes_0"), val = tensor([64, 64])]; int32 var_6132_axis_0 = const()[name = string("op_6132_axis_0"), val = int32(-2)]; tensor var_6132_cast_fp16_0, tensor var_6132_cast_fp16_1 = split(axis = var_6132_axis_0, split_sizes = var_6132_split_sizes_0, x = x_161_cast_fp16)[name = string("op_6132_cast_fp16")]; fp16 const_162_promoted_to_fp16 = const()[name = string("const_162_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6134_cast_fp16 = mul(x = var_6132_cast_fp16_1, y = const_162_promoted_to_fp16)[name = string("op_6134_cast_fp16")]; int32 var_6136 = const()[name = string("op_6136"), val = int32(-2)]; bool var_6137_interleave_0 = const()[name = string("op_6137_interleave_0"), val = bool(false)]; tensor var_6137_cast_fp16 = concat(axis = var_6136, interleave = var_6137_interleave_0, values = (var_6134_cast_fp16, var_6132_cast_fp16_0))[name = string("op_6137_cast_fp16")]; tensor var_6138_cast_fp16 = mul(x = var_6137_cast_fp16, y = var_878_cast_fp16)[name = string("op_6138_cast_fp16")]; tensor query_states_99_cast_fp16 = add(x = var_6131_cast_fp16, y = var_6138_cast_fp16)[name = string("query_states_99_cast_fp16")]; tensor var_6144_cast_fp16 = mul(x = var_6120_cast_fp16, y = var_869_cast_fp16)[name = string("op_6144_cast_fp16")]; tensor var_6145_split_sizes_0 = const()[name = string("op_6145_split_sizes_0"), val = tensor([64, 64])]; int32 var_6145_axis_0 = const()[name = string("op_6145_axis_0"), val = int32(-2)]; tensor var_6145_cast_fp16_0, tensor var_6145_cast_fp16_1 = split(axis = var_6145_axis_0, split_sizes = var_6145_split_sizes_0, x = var_6120_cast_fp16)[name = string("op_6145_cast_fp16")]; fp16 const_163_promoted_to_fp16 = const()[name = string("const_163_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6147_cast_fp16 = mul(x = var_6145_cast_fp16_1, y = const_163_promoted_to_fp16)[name = string("op_6147_cast_fp16")]; int32 var_6149 = const()[name = string("op_6149"), val = int32(-2)]; bool var_6150_interleave_0 = const()[name = string("op_6150_interleave_0"), val = bool(false)]; tensor var_6150_cast_fp16 = concat(axis = var_6149, interleave = var_6150_interleave_0, values = (var_6147_cast_fp16, var_6145_cast_fp16_0))[name = string("op_6150_cast_fp16")]; tensor var_6151_cast_fp16 = mul(x = var_6150_cast_fp16, y = var_878_cast_fp16)[name = string("op_6151_cast_fp16")]; tensor key_states_165_cast_fp16 = add(x = var_6144_cast_fp16, y = var_6151_cast_fp16)[name = string("key_states_165_cast_fp16")]; tensor expand_dims_192 = const()[name = string("expand_dims_192"), val = tensor([16])]; tensor expand_dims_193 = const()[name = string("expand_dims_193"), val = tensor([0])]; tensor expand_dims_195 = const()[name = string("expand_dims_195"), val = tensor([0])]; int32 concat_197_axis_0 = const()[name = string("concat_197_axis_0"), val = int32(0)]; bool concat_197_interleave_0 = const()[name = string("concat_197_interleave_0"), val = bool(false)]; tensor concat_197 = concat(axis = concat_197_axis_0, interleave = concat_197_interleave_0, values = (expand_dims_192, expand_dims_193, position_id, expand_dims_195))[name = string("concat_197")]; tensor expand_dims_196 = const()[name = string("expand_dims_196"), val = tensor([17])]; tensor concat_198_values1_0 = const()[name = string("concat_198_values1_0"), val = tensor([0])]; tensor concat_198_values3_0 = const()[name = string("concat_198_values3_0"), val = tensor([0])]; int32 concat_198_axis_0 = const()[name = string("concat_198_axis_0"), val = int32(0)]; bool concat_198_interleave_0 = const()[name = string("concat_198_interleave_0"), val = bool(false)]; tensor concat_198 = concat(axis = concat_198_axis_0, interleave = concat_198_interleave_0, values = (expand_dims_196, concat_198_values1_0, cache_position_end, concat_198_values3_0))[name = string("concat_198")]; tensor key_states_167_perm_0 = const()[name = string("key_states_167_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_17_stride_0 = const()[name = string("key_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_167_cast_fp16 = transpose(perm = key_states_167_perm_0, x = key_states_165_cast_fp16)[name = string("transpose_465")]; tensor key_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = key_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = key_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_17_squeeze_mask_0, stride = key_cache_internal_tensor_assign_17_stride_0, update = key_states_167_cast_fp16, x = coreml_update_state_310)[name = string("key_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_17_cast_fp16, input = key_cache)[name = string("coreml_update_state_312_write_state")]; tensor coreml_update_state_312 = read_state(input = key_cache)[name = string("coreml_update_state_312")]; tensor value_states_99_perm_0 = const()[name = string("value_states_99_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_17_stride_0 = const()[name = string("value_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_99_cast_fp16 = transpose(perm = value_states_99_perm_0, x = var_6127_cast_fp16)[name = string("transpose_464")]; tensor value_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = value_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = value_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_17_squeeze_mask_0, stride = value_cache_internal_tensor_assign_17_stride_0, update = value_states_99_cast_fp16, x = coreml_update_state_311)[name = string("value_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_17_cast_fp16, input = value_cache)[name = string("coreml_update_state_313_write_state")]; tensor coreml_update_state_313 = read_state(input = value_cache)[name = string("coreml_update_state_313")]; tensor var_6221_begin_0 = const()[name = string("op_6221_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6221_end_0 = const()[name = string("op_6221_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6221_end_mask_0 = const()[name = string("op_6221_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6221_cast_fp16 = slice_by_index(begin = var_6221_begin_0, end = var_6221_end_0, end_mask = var_6221_end_mask_0, x = coreml_update_state_312)[name = string("op_6221_cast_fp16")]; tensor tile_32 = const()[name = string("tile_32"), val = tensor([1, 1])]; int32 var_6224_axis_0 = const()[name = string("op_6224_axis_0"), val = int32(1)]; tensor var_6224_cast_fp16_0, tensor var_6224_cast_fp16_1 = split(axis = var_6224_axis_0, split_sizes = tile_32, x = var_6221_cast_fp16)[name = string("op_6224_cast_fp16")]; tensor var_6231_begin_0 = const()[name = string("op_6231_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6231_end_0 = const()[name = string("op_6231_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6231_end_mask_0 = const()[name = string("op_6231_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6231_cast_fp16 = slice_by_index(begin = var_6231_begin_0, end = var_6231_end_0, end_mask = var_6231_end_mask_0, x = coreml_update_state_313)[name = string("op_6231_cast_fp16")]; tensor tile_33 = const()[name = string("tile_33"), val = tensor([1, 1])]; int32 var_6234_axis_0 = const()[name = string("op_6234_axis_0"), val = int32(1)]; tensor var_6234_cast_fp16_0, tensor var_6234_cast_fp16_1 = split(axis = var_6234_axis_0, split_sizes = tile_33, x = var_6231_cast_fp16)[name = string("op_6234_cast_fp16")]; tensor var_6237_split_sizes_0 = const()[name = string("op_6237_split_sizes_0"), val = tensor([8, 8])]; int32 var_6237_axis_0 = const()[name = string("op_6237_axis_0"), val = int32(1)]; tensor var_6237_0, tensor var_6237_1 = split(axis = var_6237_axis_0, split_sizes = var_6237_split_sizes_0, x = query_states_99_cast_fp16)[name = string("op_6237")]; bool attn_weights_257_transpose_x_0 = const()[name = string("attn_weights_257_transpose_x_0"), val = bool(false)]; bool attn_weights_257_transpose_y_0 = const()[name = string("attn_weights_257_transpose_y_0"), val = bool(false)]; tensor attn_weights_257_cast_fp16 = matmul(transpose_x = attn_weights_257_transpose_x_0, transpose_y = attn_weights_257_transpose_y_0, x = var_6224_cast_fp16_0, y = var_6237_0)[name = string("attn_weights_257_cast_fp16")]; fp16 var_6240_to_fp16 = const()[name = string("op_6240_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_259_cast_fp16 = mul(x = attn_weights_257_cast_fp16, y = var_6240_to_fp16)[name = string("attn_weights_259_cast_fp16")]; tensor attn_weights_261_cast_fp16 = add(x = attn_weights_259_cast_fp16, y = attn_mask_1)[name = string("attn_weights_261_cast_fp16")]; int32 var_6244 = const()[name = string("op_6244"), val = int32(-2)]; tensor attn_weights_263_cast_fp16 = softmax(axis = var_6244, x = attn_weights_261_cast_fp16)[name = string("attn_weights_263_cast_fp16")]; bool var_6250_transpose_x_1 = const()[name = string("op_6250_transpose_x_1"), val = bool(true)]; bool var_6250_transpose_y_1 = const()[name = string("op_6250_transpose_y_1"), val = bool(false)]; tensor var_6250_cast_fp16 = matmul(transpose_x = var_6250_transpose_x_1, transpose_y = var_6250_transpose_y_1, x = attn_weights_263_cast_fp16, y = var_6234_cast_fp16_0)[name = string("op_6250_cast_fp16")]; bool attn_weights_265_transpose_x_0 = const()[name = string("attn_weights_265_transpose_x_0"), val = bool(false)]; bool attn_weights_265_transpose_y_0 = const()[name = string("attn_weights_265_transpose_y_0"), val = bool(false)]; tensor attn_weights_265_cast_fp16 = matmul(transpose_x = attn_weights_265_transpose_x_0, transpose_y = attn_weights_265_transpose_y_0, x = var_6224_cast_fp16_1, y = var_6237_1)[name = string("attn_weights_265_cast_fp16")]; fp16 var_6252_to_fp16 = const()[name = string("op_6252_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_267_cast_fp16 = mul(x = attn_weights_265_cast_fp16, y = var_6252_to_fp16)[name = string("attn_weights_267_cast_fp16")]; tensor attn_weights_269_cast_fp16 = add(x = attn_weights_267_cast_fp16, y = attn_mask_1)[name = string("attn_weights_269_cast_fp16")]; int32 var_6256 = const()[name = string("op_6256"), val = int32(-2)]; tensor attn_weights_271_cast_fp16 = softmax(axis = var_6256, x = attn_weights_269_cast_fp16)[name = string("attn_weights_271_cast_fp16")]; bool attn_output_129_transpose_x_1 = const()[name = string("attn_output_129_transpose_x_1"), val = bool(true)]; bool attn_output_129_transpose_y_1 = const()[name = string("attn_output_129_transpose_y_1"), val = bool(false)]; tensor attn_output_129_cast_fp16 = matmul(transpose_x = attn_output_129_transpose_x_1, transpose_y = attn_output_129_transpose_y_1, x = attn_weights_271_cast_fp16, y = var_6234_cast_fp16_1)[name = string("attn_output_129_cast_fp16")]; int32 var_6264 = const()[name = string("op_6264"), val = int32(1)]; bool attn_output_131_interleave_0 = const()[name = string("attn_output_131_interleave_0"), val = bool(false)]; tensor attn_output_131_cast_fp16 = concat(axis = var_6264, interleave = attn_output_131_interleave_0, values = (var_6250_cast_fp16, attn_output_129_cast_fp16))[name = string("attn_output_131_cast_fp16")]; tensor var_6268_perm_0 = const()[name = string("op_6268_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_203x = const()[name = string("concat_203x"), val = tensor([1, 2048, 1, -1])]; tensor var_6268_cast_fp16 = transpose(perm = var_6268_perm_0, x = attn_output_131_cast_fp16)[name = string("transpose_463")]; tensor attn_output_135_cast_fp16 = reshape(shape = concat_203x, x = var_6268_cast_fp16)[name = string("attn_output_135_cast_fp16")]; tensor hidden_states_163_strides_0 = const()[name = string("hidden_states_163_strides_0"), val = tensor([1, 1])]; string hidden_states_163_pad_type_0 = const()[name = string("hidden_states_163_pad_type_0"), val = string("valid")]; tensor hidden_states_163_pad_0 = const()[name = string("hidden_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_163_dilations_0 = const()[name = string("hidden_states_163_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_163_groups_0 = const()[name = string("hidden_states_163_groups_0"), val = int32(1)]; tensor hidden_states_163_cast_fp16 = conv(dilations = hidden_states_163_dilations_0, groups = hidden_states_163_groups_0, pad = hidden_states_163_pad_0, pad_type = hidden_states_163_pad_type_0, strides = hidden_states_163_strides_0, weight = layers_16_self_attn_o_proj_weight_cast_fp16, x = attn_output_135_cast_fp16)[name = string("hidden_states_163_cast_fp16")]; tensor hidden_states_165_cast_fp16 = add(x = hidden_states_159_cast_fp16, y = hidden_states_163_cast_fp16)[name = string("hidden_states_165_cast_fp16")]; fp16 const_168_promoted_to_fp16 = const()[name = string("const_168_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6301_cast_fp16 = mul(x = hidden_states_165_cast_fp16, y = const_168_promoted_to_fp16)[name = string("op_6301_cast_fp16")]; int32 var_6299 = const()[name = string("op_6299"), val = int32(1)]; bool doubled_133_interleave_0 = const()[name = string("doubled_133_interleave_0"), val = bool(false)]; tensor doubled_133_cast_fp16 = concat(axis = var_6299, interleave = doubled_133_interleave_0, values = (hidden_states_165_cast_fp16, var_6301_cast_fp16))[name = string("doubled_133_cast_fp16")]; tensor out_67_axes_0 = const()[name = string("out_67_axes_0"), val = tensor([1])]; tensor out_67_gamma_0_to_fp16 = const()[name = string("out_67_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441663488)))]; fp16 var_6311_to_fp16 = const()[name = string("op_6311_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_67_cast_fp16 = layer_norm(axes = out_67_axes_0, epsilon = var_6311_to_fp16, gamma = out_67_gamma_0_to_fp16, x = doubled_133_cast_fp16)[name = string("out_67_cast_fp16")]; tensor var_6322_split_sizes_0 = const()[name = string("op_6322_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6322_axis_0 = const()[name = string("op_6322_axis_0"), val = int32(1)]; tensor var_6322_cast_fp16_0, tensor var_6322_cast_fp16_1 = split(axis = var_6322_axis_0, split_sizes = var_6322_split_sizes_0, x = out_67_cast_fp16)[name = string("op_6322_cast_fp16")]; tensor layers_16_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441671744)))]; tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; tensor input_33_cast_fp16 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = layers_16_mlp_gate_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("input_33_cast_fp16")]; tensor var_6339_cast_fp16 = silu(x = input_33_cast_fp16)[name = string("op_6339_cast_fp16")]; tensor layers_16_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1466837632)))]; tensor var_6345_strides_0 = const()[name = string("op_6345_strides_0"), val = tensor([1, 1])]; string var_6345_pad_type_0 = const()[name = string("op_6345_pad_type_0"), val = string("valid")]; tensor var_6345_pad_0 = const()[name = string("op_6345_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6345_dilations_0 = const()[name = string("op_6345_dilations_0"), val = tensor([1, 1])]; int32 var_6345_groups_0 = const()[name = string("op_6345_groups_0"), val = int32(1)]; tensor var_6345_cast_fp16 = conv(dilations = var_6345_dilations_0, groups = var_6345_groups_0, pad = var_6345_pad_0, pad_type = var_6345_pad_type_0, strides = var_6345_strides_0, weight = layers_16_mlp_up_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("op_6345_cast_fp16")]; tensor x_169_cast_fp16 = mul(x = var_6339_cast_fp16, y = var_6345_cast_fp16)[name = string("x_169_cast_fp16")]; tensor hidden_states_167_strides_0 = const()[name = string("hidden_states_167_strides_0"), val = tensor([1, 1])]; string hidden_states_167_pad_type_0 = const()[name = string("hidden_states_167_pad_type_0"), val = string("valid")]; tensor hidden_states_167_pad_0 = const()[name = string("hidden_states_167_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_167_dilations_0 = const()[name = string("hidden_states_167_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_167_groups_0 = const()[name = string("hidden_states_167_groups_0"), val = int32(1)]; tensor hidden_states_167_cast_fp16 = conv(dilations = hidden_states_167_dilations_0, groups = hidden_states_167_groups_0, pad = hidden_states_167_pad_0, pad_type = hidden_states_167_pad_type_0, strides = hidden_states_167_strides_0, weight = layers_16_mlp_down_proj_weight_cast_fp16, x = x_169_cast_fp16)[name = string("hidden_states_167_cast_fp16")]; tensor hidden_states_169_cast_fp16 = add(x = hidden_states_165_cast_fp16, y = hidden_states_167_cast_fp16)[name = string("hidden_states_169_cast_fp16")]; fp16 const_170_promoted_to_fp16 = const()[name = string("const_170_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6363_cast_fp16 = mul(x = hidden_states_169_cast_fp16, y = const_170_promoted_to_fp16)[name = string("op_6363_cast_fp16")]; int32 var_6361 = const()[name = string("op_6361"), val = int32(1)]; bool doubled_137_interleave_0 = const()[name = string("doubled_137_interleave_0"), val = bool(false)]; tensor doubled_137_cast_fp16 = concat(axis = var_6361, interleave = doubled_137_interleave_0, values = (hidden_states_169_cast_fp16, var_6363_cast_fp16))[name = string("doubled_137_cast_fp16")]; tensor out_69_axes_0 = const()[name = string("out_69_axes_0"), val = tensor([1])]; tensor out_69_gamma_0_to_fp16 = const()[name = string("out_69_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492003520)))]; fp16 var_6373_to_fp16 = const()[name = string("op_6373_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_69_cast_fp16 = layer_norm(axes = out_69_axes_0, epsilon = var_6373_to_fp16, gamma = out_69_gamma_0_to_fp16, x = doubled_137_cast_fp16)[name = string("out_69_cast_fp16")]; tensor var_6384_split_sizes_0 = const()[name = string("op_6384_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6384_axis_0 = const()[name = string("op_6384_axis_0"), val = int32(1)]; tensor var_6384_cast_fp16_0, tensor var_6384_cast_fp16_1 = split(axis = var_6384_axis_0, split_sizes = var_6384_split_sizes_0, x = out_69_cast_fp16)[name = string("op_6384_cast_fp16")]; tensor query_states_103_strides_0 = const()[name = string("query_states_103_strides_0"), val = tensor([1, 1])]; string query_states_103_pad_type_0 = const()[name = string("query_states_103_pad_type_0"), val = string("valid")]; tensor query_states_103_pad_0 = const()[name = string("query_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_103_dilations_0 = const()[name = string("query_states_103_dilations_0"), val = tensor([1, 1])]; int32 query_states_103_groups_0 = const()[name = string("query_states_103_groups_0"), val = int32(1)]; tensor query_states_103_cast_fp16 = conv(dilations = query_states_103_dilations_0, groups = query_states_103_groups_0, pad = query_states_103_pad_0, pad_type = query_states_103_pad_type_0, strides = query_states_103_strides_0, weight = layers_17_self_attn_q_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("query_states_103_cast_fp16")]; tensor key_states_171_strides_0 = const()[name = string("key_states_171_strides_0"), val = tensor([1, 1])]; string key_states_171_pad_type_0 = const()[name = string("key_states_171_pad_type_0"), val = string("valid")]; tensor key_states_171_pad_0 = const()[name = string("key_states_171_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_171_dilations_0 = const()[name = string("key_states_171_dilations_0"), val = tensor([1, 1])]; int32 key_states_171_groups_0 = const()[name = string("key_states_171_groups_0"), val = int32(1)]; tensor key_states_171_cast_fp16 = conv(dilations = key_states_171_dilations_0, groups = key_states_171_groups_0, pad = key_states_171_pad_0, pad_type = key_states_171_pad_type_0, strides = key_states_171_strides_0, weight = layers_17_self_attn_k_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("key_states_171_cast_fp16")]; tensor value_states_103_strides_0 = const()[name = string("value_states_103_strides_0"), val = tensor([1, 1])]; string value_states_103_pad_type_0 = const()[name = string("value_states_103_pad_type_0"), val = string("valid")]; tensor value_states_103_pad_0 = const()[name = string("value_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_103_dilations_0 = const()[name = string("value_states_103_dilations_0"), val = tensor([1, 1])]; int32 value_states_103_groups_0 = const()[name = string("value_states_103_groups_0"), val = int32(1)]; tensor value_states_103_cast_fp16 = conv(dilations = value_states_103_dilations_0, groups = value_states_103_groups_0, pad = value_states_103_pad_0, pad_type = value_states_103_pad_type_0, strides = value_states_103_strides_0, weight = layers_17_self_attn_v_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("value_states_103_cast_fp16")]; tensor concat_204x = const()[name = string("concat_204x"), val = tensor([1, 16, 128, -1])]; tensor x_171_cast_fp16 = reshape(shape = concat_204x, x = query_states_103_cast_fp16)[name = string("x_171_cast_fp16")]; tensor concat_205x = const()[name = string("concat_205x"), val = tensor([1, 2, 128, -1])]; tensor var_6441_cast_fp16 = reshape(shape = concat_205x, x = key_states_171_cast_fp16)[name = string("op_6441_cast_fp16")]; tensor concat_206x = const()[name = string("concat_206x"), val = tensor([1, 2, 128, -1])]; tensor var_6448_cast_fp16 = reshape(shape = concat_206x, x = value_states_103_cast_fp16)[name = string("op_6448_cast_fp16")]; tensor var_6452_cast_fp16 = mul(x = x_171_cast_fp16, y = var_869_cast_fp16)[name = string("op_6452_cast_fp16")]; tensor var_6453_split_sizes_0 = const()[name = string("op_6453_split_sizes_0"), val = tensor([64, 64])]; int32 var_6453_axis_0 = const()[name = string("op_6453_axis_0"), val = int32(-2)]; tensor var_6453_cast_fp16_0, tensor var_6453_cast_fp16_1 = split(axis = var_6453_axis_0, split_sizes = var_6453_split_sizes_0, x = x_171_cast_fp16)[name = string("op_6453_cast_fp16")]; fp16 const_172_promoted_to_fp16 = const()[name = string("const_172_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6455_cast_fp16 = mul(x = var_6453_cast_fp16_1, y = const_172_promoted_to_fp16)[name = string("op_6455_cast_fp16")]; int32 var_6457 = const()[name = string("op_6457"), val = int32(-2)]; bool var_6458_interleave_0 = const()[name = string("op_6458_interleave_0"), val = bool(false)]; tensor var_6458_cast_fp16 = concat(axis = var_6457, interleave = var_6458_interleave_0, values = (var_6455_cast_fp16, var_6453_cast_fp16_0))[name = string("op_6458_cast_fp16")]; tensor var_6459_cast_fp16 = mul(x = var_6458_cast_fp16, y = var_878_cast_fp16)[name = string("op_6459_cast_fp16")]; tensor query_states_105_cast_fp16 = add(x = var_6452_cast_fp16, y = var_6459_cast_fp16)[name = string("query_states_105_cast_fp16")]; tensor var_6465_cast_fp16 = mul(x = var_6441_cast_fp16, y = var_869_cast_fp16)[name = string("op_6465_cast_fp16")]; tensor var_6466_split_sizes_0 = const()[name = string("op_6466_split_sizes_0"), val = tensor([64, 64])]; int32 var_6466_axis_0 = const()[name = string("op_6466_axis_0"), val = int32(-2)]; tensor var_6466_cast_fp16_0, tensor var_6466_cast_fp16_1 = split(axis = var_6466_axis_0, split_sizes = var_6466_split_sizes_0, x = var_6441_cast_fp16)[name = string("op_6466_cast_fp16")]; fp16 const_173_promoted_to_fp16 = const()[name = string("const_173_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6468_cast_fp16 = mul(x = var_6466_cast_fp16_1, y = const_173_promoted_to_fp16)[name = string("op_6468_cast_fp16")]; int32 var_6470 = const()[name = string("op_6470"), val = int32(-2)]; bool var_6471_interleave_0 = const()[name = string("op_6471_interleave_0"), val = bool(false)]; tensor var_6471_cast_fp16 = concat(axis = var_6470, interleave = var_6471_interleave_0, values = (var_6468_cast_fp16, var_6466_cast_fp16_0))[name = string("op_6471_cast_fp16")]; tensor var_6472_cast_fp16 = mul(x = var_6471_cast_fp16, y = var_878_cast_fp16)[name = string("op_6472_cast_fp16")]; tensor key_states_175_cast_fp16 = add(x = var_6465_cast_fp16, y = var_6472_cast_fp16)[name = string("key_states_175_cast_fp16")]; tensor expand_dims_204 = const()[name = string("expand_dims_204"), val = tensor([17])]; tensor expand_dims_205 = const()[name = string("expand_dims_205"), val = tensor([0])]; tensor expand_dims_207 = const()[name = string("expand_dims_207"), val = tensor([0])]; int32 concat_209_axis_0 = const()[name = string("concat_209_axis_0"), val = int32(0)]; bool concat_209_interleave_0 = const()[name = string("concat_209_interleave_0"), val = bool(false)]; tensor concat_209 = concat(axis = concat_209_axis_0, interleave = concat_209_interleave_0, values = (expand_dims_204, expand_dims_205, position_id, expand_dims_207))[name = string("concat_209")]; tensor expand_dims_208 = const()[name = string("expand_dims_208"), val = tensor([18])]; tensor concat_210_values1_0 = const()[name = string("concat_210_values1_0"), val = tensor([0])]; tensor concat_210_values3_0 = const()[name = string("concat_210_values3_0"), val = tensor([0])]; int32 concat_210_axis_0 = const()[name = string("concat_210_axis_0"), val = int32(0)]; bool concat_210_interleave_0 = const()[name = string("concat_210_interleave_0"), val = bool(false)]; tensor concat_210 = concat(axis = concat_210_axis_0, interleave = concat_210_interleave_0, values = (expand_dims_208, concat_210_values1_0, cache_position_end, concat_210_values3_0))[name = string("concat_210")]; tensor key_states_177_perm_0 = const()[name = string("key_states_177_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_18_stride_0 = const()[name = string("key_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_177_cast_fp16 = transpose(perm = key_states_177_perm_0, x = key_states_175_cast_fp16)[name = string("transpose_462")]; tensor key_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = key_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = key_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_18_squeeze_mask_0, stride = key_cache_internal_tensor_assign_18_stride_0, update = key_states_177_cast_fp16, x = coreml_update_state_312)[name = string("key_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_18_cast_fp16, input = key_cache)[name = string("coreml_update_state_314_write_state")]; tensor coreml_update_state_314 = read_state(input = key_cache)[name = string("coreml_update_state_314")]; tensor value_states_105_perm_0 = const()[name = string("value_states_105_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_18_stride_0 = const()[name = string("value_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_105_cast_fp16 = transpose(perm = value_states_105_perm_0, x = var_6448_cast_fp16)[name = string("transpose_461")]; tensor value_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = value_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = value_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_18_squeeze_mask_0, stride = value_cache_internal_tensor_assign_18_stride_0, update = value_states_105_cast_fp16, x = coreml_update_state_313)[name = string("value_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_18_cast_fp16, input = value_cache)[name = string("coreml_update_state_315_write_state")]; tensor coreml_update_state_315 = read_state(input = value_cache)[name = string("coreml_update_state_315")]; tensor var_6542_begin_0 = const()[name = string("op_6542_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6542_end_0 = const()[name = string("op_6542_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6542_end_mask_0 = const()[name = string("op_6542_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6542_cast_fp16 = slice_by_index(begin = var_6542_begin_0, end = var_6542_end_0, end_mask = var_6542_end_mask_0, x = coreml_update_state_314)[name = string("op_6542_cast_fp16")]; tensor tile_34 = const()[name = string("tile_34"), val = tensor([1, 1])]; int32 var_6545_axis_0 = const()[name = string("op_6545_axis_0"), val = int32(1)]; tensor var_6545_cast_fp16_0, tensor var_6545_cast_fp16_1 = split(axis = var_6545_axis_0, split_sizes = tile_34, x = var_6542_cast_fp16)[name = string("op_6545_cast_fp16")]; tensor var_6552_begin_0 = const()[name = string("op_6552_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6552_end_0 = const()[name = string("op_6552_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6552_end_mask_0 = const()[name = string("op_6552_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6552_cast_fp16 = slice_by_index(begin = var_6552_begin_0, end = var_6552_end_0, end_mask = var_6552_end_mask_0, x = coreml_update_state_315)[name = string("op_6552_cast_fp16")]; tensor tile_35 = const()[name = string("tile_35"), val = tensor([1, 1])]; int32 var_6555_axis_0 = const()[name = string("op_6555_axis_0"), val = int32(1)]; tensor var_6555_cast_fp16_0, tensor var_6555_cast_fp16_1 = split(axis = var_6555_axis_0, split_sizes = tile_35, x = var_6552_cast_fp16)[name = string("op_6555_cast_fp16")]; tensor var_6558_split_sizes_0 = const()[name = string("op_6558_split_sizes_0"), val = tensor([8, 8])]; int32 var_6558_axis_0 = const()[name = string("op_6558_axis_0"), val = int32(1)]; tensor var_6558_0, tensor var_6558_1 = split(axis = var_6558_axis_0, split_sizes = var_6558_split_sizes_0, x = query_states_105_cast_fp16)[name = string("op_6558")]; bool attn_weights_273_transpose_x_0 = const()[name = string("attn_weights_273_transpose_x_0"), val = bool(false)]; bool attn_weights_273_transpose_y_0 = const()[name = string("attn_weights_273_transpose_y_0"), val = bool(false)]; tensor attn_weights_273_cast_fp16 = matmul(transpose_x = attn_weights_273_transpose_x_0, transpose_y = attn_weights_273_transpose_y_0, x = var_6545_cast_fp16_0, y = var_6558_0)[name = string("attn_weights_273_cast_fp16")]; fp16 var_6561_to_fp16 = const()[name = string("op_6561_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_275_cast_fp16 = mul(x = attn_weights_273_cast_fp16, y = var_6561_to_fp16)[name = string("attn_weights_275_cast_fp16")]; tensor attn_weights_277_cast_fp16 = add(x = attn_weights_275_cast_fp16, y = attn_mask_1)[name = string("attn_weights_277_cast_fp16")]; int32 var_6565 = const()[name = string("op_6565"), val = int32(-2)]; tensor attn_weights_279_cast_fp16 = softmax(axis = var_6565, x = attn_weights_277_cast_fp16)[name = string("attn_weights_279_cast_fp16")]; bool var_6571_transpose_x_1 = const()[name = string("op_6571_transpose_x_1"), val = bool(true)]; bool var_6571_transpose_y_1 = const()[name = string("op_6571_transpose_y_1"), val = bool(false)]; tensor var_6571_cast_fp16 = matmul(transpose_x = var_6571_transpose_x_1, transpose_y = var_6571_transpose_y_1, x = attn_weights_279_cast_fp16, y = var_6555_cast_fp16_0)[name = string("op_6571_cast_fp16")]; bool attn_weights_281_transpose_x_0 = const()[name = string("attn_weights_281_transpose_x_0"), val = bool(false)]; bool attn_weights_281_transpose_y_0 = const()[name = string("attn_weights_281_transpose_y_0"), val = bool(false)]; tensor attn_weights_281_cast_fp16 = matmul(transpose_x = attn_weights_281_transpose_x_0, transpose_y = attn_weights_281_transpose_y_0, x = var_6545_cast_fp16_1, y = var_6558_1)[name = string("attn_weights_281_cast_fp16")]; fp16 var_6573_to_fp16 = const()[name = string("op_6573_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_283_cast_fp16 = mul(x = attn_weights_281_cast_fp16, y = var_6573_to_fp16)[name = string("attn_weights_283_cast_fp16")]; tensor attn_weights_285_cast_fp16 = add(x = attn_weights_283_cast_fp16, y = attn_mask_1)[name = string("attn_weights_285_cast_fp16")]; int32 var_6577 = const()[name = string("op_6577"), val = int32(-2)]; tensor attn_weights_287_cast_fp16 = softmax(axis = var_6577, x = attn_weights_285_cast_fp16)[name = string("attn_weights_287_cast_fp16")]; bool attn_output_137_transpose_x_1 = const()[name = string("attn_output_137_transpose_x_1"), val = bool(true)]; bool attn_output_137_transpose_y_1 = const()[name = string("attn_output_137_transpose_y_1"), val = bool(false)]; tensor attn_output_137_cast_fp16 = matmul(transpose_x = attn_output_137_transpose_x_1, transpose_y = attn_output_137_transpose_y_1, x = attn_weights_287_cast_fp16, y = var_6555_cast_fp16_1)[name = string("attn_output_137_cast_fp16")]; int32 var_6585 = const()[name = string("op_6585"), val = int32(1)]; bool attn_output_139_interleave_0 = const()[name = string("attn_output_139_interleave_0"), val = bool(false)]; tensor attn_output_139_cast_fp16 = concat(axis = var_6585, interleave = attn_output_139_interleave_0, values = (var_6571_cast_fp16, attn_output_137_cast_fp16))[name = string("attn_output_139_cast_fp16")]; tensor var_6589_perm_0 = const()[name = string("op_6589_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_215x = const()[name = string("concat_215x"), val = tensor([1, 2048, 1, -1])]; tensor var_6589_cast_fp16 = transpose(perm = var_6589_perm_0, x = attn_output_139_cast_fp16)[name = string("transpose_460")]; tensor attn_output_143_cast_fp16 = reshape(shape = concat_215x, x = var_6589_cast_fp16)[name = string("attn_output_143_cast_fp16")]; tensor hidden_states_173_strides_0 = const()[name = string("hidden_states_173_strides_0"), val = tensor([1, 1])]; string hidden_states_173_pad_type_0 = const()[name = string("hidden_states_173_pad_type_0"), val = string("valid")]; tensor hidden_states_173_pad_0 = const()[name = string("hidden_states_173_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_173_dilations_0 = const()[name = string("hidden_states_173_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_173_groups_0 = const()[name = string("hidden_states_173_groups_0"), val = int32(1)]; tensor hidden_states_173_cast_fp16 = conv(dilations = hidden_states_173_dilations_0, groups = hidden_states_173_groups_0, pad = hidden_states_173_pad_0, pad_type = hidden_states_173_pad_type_0, strides = hidden_states_173_strides_0, weight = layers_17_self_attn_o_proj_weight_cast_fp16, x = attn_output_143_cast_fp16)[name = string("hidden_states_173_cast_fp16")]; tensor hidden_states_175_cast_fp16 = add(x = hidden_states_169_cast_fp16, y = hidden_states_173_cast_fp16)[name = string("hidden_states_175_cast_fp16")]; fp16 const_178_promoted_to_fp16 = const()[name = string("const_178_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6622_cast_fp16 = mul(x = hidden_states_175_cast_fp16, y = const_178_promoted_to_fp16)[name = string("op_6622_cast_fp16")]; int32 var_6620 = const()[name = string("op_6620"), val = int32(1)]; bool doubled_141_interleave_0 = const()[name = string("doubled_141_interleave_0"), val = bool(false)]; tensor doubled_141_cast_fp16 = concat(axis = var_6620, interleave = doubled_141_interleave_0, values = (hidden_states_175_cast_fp16, var_6622_cast_fp16))[name = string("doubled_141_cast_fp16")]; tensor out_71_axes_0 = const()[name = string("out_71_axes_0"), val = tensor([1])]; tensor out_71_gamma_0_to_fp16 = const()[name = string("out_71_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492011776)))]; fp16 var_6632_to_fp16 = const()[name = string("op_6632_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_71_cast_fp16 = layer_norm(axes = out_71_axes_0, epsilon = var_6632_to_fp16, gamma = out_71_gamma_0_to_fp16, x = doubled_141_cast_fp16)[name = string("out_71_cast_fp16")]; tensor var_6643_split_sizes_0 = const()[name = string("op_6643_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6643_axis_0 = const()[name = string("op_6643_axis_0"), val = int32(1)]; tensor var_6643_cast_fp16_0, tensor var_6643_cast_fp16_1 = split(axis = var_6643_axis_0, split_sizes = var_6643_split_sizes_0, x = out_71_cast_fp16)[name = string("op_6643_cast_fp16")]; tensor input_35_strides_0 = const()[name = string("input_35_strides_0"), val = tensor([1, 1])]; string input_35_pad_type_0 = const()[name = string("input_35_pad_type_0"), val = string("valid")]; tensor input_35_pad_0 = const()[name = string("input_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_35_dilations_0 = const()[name = string("input_35_dilations_0"), val = tensor([1, 1])]; int32 input_35_groups_0 = const()[name = string("input_35_groups_0"), val = int32(1)]; tensor input_35_cast_fp16 = conv(dilations = input_35_dilations_0, groups = input_35_groups_0, pad = input_35_pad_0, pad_type = input_35_pad_type_0, strides = input_35_strides_0, weight = layers_17_mlp_gate_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("input_35_cast_fp16")]; tensor var_6660_cast_fp16 = silu(x = input_35_cast_fp16)[name = string("op_6660_cast_fp16")]; tensor var_6666_strides_0 = const()[name = string("op_6666_strides_0"), val = tensor([1, 1])]; string var_6666_pad_type_0 = const()[name = string("op_6666_pad_type_0"), val = string("valid")]; tensor var_6666_pad_0 = const()[name = string("op_6666_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6666_dilations_0 = const()[name = string("op_6666_dilations_0"), val = tensor([1, 1])]; int32 var_6666_groups_0 = const()[name = string("op_6666_groups_0"), val = int32(1)]; tensor var_6666_cast_fp16 = conv(dilations = var_6666_dilations_0, groups = var_6666_groups_0, pad = var_6666_pad_0, pad_type = var_6666_pad_type_0, strides = var_6666_strides_0, weight = layers_17_mlp_up_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("op_6666_cast_fp16")]; tensor x_179_cast_fp16 = mul(x = var_6660_cast_fp16, y = var_6666_cast_fp16)[name = string("x_179_cast_fp16")]; tensor hidden_states_177_strides_0 = const()[name = string("hidden_states_177_strides_0"), val = tensor([1, 1])]; string hidden_states_177_pad_type_0 = const()[name = string("hidden_states_177_pad_type_0"), val = string("valid")]; tensor hidden_states_177_pad_0 = const()[name = string("hidden_states_177_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_177_dilations_0 = const()[name = string("hidden_states_177_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_177_groups_0 = const()[name = string("hidden_states_177_groups_0"), val = int32(1)]; tensor hidden_states_177_cast_fp16 = conv(dilations = hidden_states_177_dilations_0, groups = hidden_states_177_groups_0, pad = hidden_states_177_pad_0, pad_type = hidden_states_177_pad_type_0, strides = hidden_states_177_strides_0, weight = layers_17_mlp_down_proj_weight_cast_fp16, x = x_179_cast_fp16)[name = string("hidden_states_177_cast_fp16")]; tensor hidden_states_179_cast_fp16 = add(x = hidden_states_175_cast_fp16, y = hidden_states_177_cast_fp16)[name = string("hidden_states_179_cast_fp16")]; fp16 const_180_promoted_to_fp16 = const()[name = string("const_180_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6684_cast_fp16 = mul(x = hidden_states_179_cast_fp16, y = const_180_promoted_to_fp16)[name = string("op_6684_cast_fp16")]; int32 var_6682 = const()[name = string("op_6682"), val = int32(1)]; bool doubled_145_interleave_0 = const()[name = string("doubled_145_interleave_0"), val = bool(false)]; tensor doubled_145_cast_fp16 = concat(axis = var_6682, interleave = doubled_145_interleave_0, values = (hidden_states_179_cast_fp16, var_6684_cast_fp16))[name = string("doubled_145_cast_fp16")]; tensor out_73_axes_0 = const()[name = string("out_73_axes_0"), val = tensor([1])]; tensor out_73_gamma_0_to_fp16 = const()[name = string("out_73_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492020032)))]; fp16 var_6694_to_fp16 = const()[name = string("op_6694_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_73_cast_fp16 = layer_norm(axes = out_73_axes_0, epsilon = var_6694_to_fp16, gamma = out_73_gamma_0_to_fp16, x = doubled_145_cast_fp16)[name = string("out_73_cast_fp16")]; tensor var_6705_split_sizes_0 = const()[name = string("op_6705_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6705_axis_0 = const()[name = string("op_6705_axis_0"), val = int32(1)]; tensor var_6705_cast_fp16_0, tensor var_6705_cast_fp16_1 = split(axis = var_6705_axis_0, split_sizes = var_6705_split_sizes_0, x = out_73_cast_fp16)[name = string("op_6705_cast_fp16")]; tensor query_states_109_strides_0 = const()[name = string("query_states_109_strides_0"), val = tensor([1, 1])]; string query_states_109_pad_type_0 = const()[name = string("query_states_109_pad_type_0"), val = string("valid")]; tensor query_states_109_pad_0 = const()[name = string("query_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_109_dilations_0 = const()[name = string("query_states_109_dilations_0"), val = tensor([1, 1])]; int32 query_states_109_groups_0 = const()[name = string("query_states_109_groups_0"), val = int32(1)]; tensor query_states_109_cast_fp16 = conv(dilations = query_states_109_dilations_0, groups = query_states_109_groups_0, pad = query_states_109_pad_0, pad_type = query_states_109_pad_type_0, strides = query_states_109_strides_0, weight = layers_18_self_attn_q_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("query_states_109_cast_fp16")]; tensor key_states_181_strides_0 = const()[name = string("key_states_181_strides_0"), val = tensor([1, 1])]; string key_states_181_pad_type_0 = const()[name = string("key_states_181_pad_type_0"), val = string("valid")]; tensor key_states_181_pad_0 = const()[name = string("key_states_181_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_181_dilations_0 = const()[name = string("key_states_181_dilations_0"), val = tensor([1, 1])]; int32 key_states_181_groups_0 = const()[name = string("key_states_181_groups_0"), val = int32(1)]; tensor key_states_181_cast_fp16 = conv(dilations = key_states_181_dilations_0, groups = key_states_181_groups_0, pad = key_states_181_pad_0, pad_type = key_states_181_pad_type_0, strides = key_states_181_strides_0, weight = layers_18_self_attn_k_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("key_states_181_cast_fp16")]; tensor value_states_109_strides_0 = const()[name = string("value_states_109_strides_0"), val = tensor([1, 1])]; string value_states_109_pad_type_0 = const()[name = string("value_states_109_pad_type_0"), val = string("valid")]; tensor value_states_109_pad_0 = const()[name = string("value_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_109_dilations_0 = const()[name = string("value_states_109_dilations_0"), val = tensor([1, 1])]; int32 value_states_109_groups_0 = const()[name = string("value_states_109_groups_0"), val = int32(1)]; tensor value_states_109_cast_fp16 = conv(dilations = value_states_109_dilations_0, groups = value_states_109_groups_0, pad = value_states_109_pad_0, pad_type = value_states_109_pad_type_0, strides = value_states_109_strides_0, weight = layers_18_self_attn_v_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("value_states_109_cast_fp16")]; tensor concat_216x = const()[name = string("concat_216x"), val = tensor([1, 16, 128, -1])]; tensor x_181_cast_fp16 = reshape(shape = concat_216x, x = query_states_109_cast_fp16)[name = string("x_181_cast_fp16")]; tensor concat_217x = const()[name = string("concat_217x"), val = tensor([1, 2, 128, -1])]; tensor var_6762_cast_fp16 = reshape(shape = concat_217x, x = key_states_181_cast_fp16)[name = string("op_6762_cast_fp16")]; tensor concat_218x = const()[name = string("concat_218x"), val = tensor([1, 2, 128, -1])]; tensor var_6769_cast_fp16 = reshape(shape = concat_218x, x = value_states_109_cast_fp16)[name = string("op_6769_cast_fp16")]; tensor var_6773_cast_fp16 = mul(x = x_181_cast_fp16, y = var_869_cast_fp16)[name = string("op_6773_cast_fp16")]; tensor var_6774_split_sizes_0 = const()[name = string("op_6774_split_sizes_0"), val = tensor([64, 64])]; int32 var_6774_axis_0 = const()[name = string("op_6774_axis_0"), val = int32(-2)]; tensor var_6774_cast_fp16_0, tensor var_6774_cast_fp16_1 = split(axis = var_6774_axis_0, split_sizes = var_6774_split_sizes_0, x = x_181_cast_fp16)[name = string("op_6774_cast_fp16")]; fp16 const_182_promoted_to_fp16 = const()[name = string("const_182_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6776_cast_fp16 = mul(x = var_6774_cast_fp16_1, y = const_182_promoted_to_fp16)[name = string("op_6776_cast_fp16")]; int32 var_6778 = const()[name = string("op_6778"), val = int32(-2)]; bool var_6779_interleave_0 = const()[name = string("op_6779_interleave_0"), val = bool(false)]; tensor var_6779_cast_fp16 = concat(axis = var_6778, interleave = var_6779_interleave_0, values = (var_6776_cast_fp16, var_6774_cast_fp16_0))[name = string("op_6779_cast_fp16")]; tensor var_6780_cast_fp16 = mul(x = var_6779_cast_fp16, y = var_878_cast_fp16)[name = string("op_6780_cast_fp16")]; tensor query_states_111_cast_fp16 = add(x = var_6773_cast_fp16, y = var_6780_cast_fp16)[name = string("query_states_111_cast_fp16")]; tensor var_6786_cast_fp16 = mul(x = var_6762_cast_fp16, y = var_869_cast_fp16)[name = string("op_6786_cast_fp16")]; tensor var_6787_split_sizes_0 = const()[name = string("op_6787_split_sizes_0"), val = tensor([64, 64])]; int32 var_6787_axis_0 = const()[name = string("op_6787_axis_0"), val = int32(-2)]; tensor var_6787_cast_fp16_0, tensor var_6787_cast_fp16_1 = split(axis = var_6787_axis_0, split_sizes = var_6787_split_sizes_0, x = var_6762_cast_fp16)[name = string("op_6787_cast_fp16")]; fp16 const_183_promoted_to_fp16 = const()[name = string("const_183_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6789_cast_fp16 = mul(x = var_6787_cast_fp16_1, y = const_183_promoted_to_fp16)[name = string("op_6789_cast_fp16")]; int32 var_6791 = const()[name = string("op_6791"), val = int32(-2)]; bool var_6792_interleave_0 = const()[name = string("op_6792_interleave_0"), val = bool(false)]; tensor var_6792_cast_fp16 = concat(axis = var_6791, interleave = var_6792_interleave_0, values = (var_6789_cast_fp16, var_6787_cast_fp16_0))[name = string("op_6792_cast_fp16")]; tensor var_6793_cast_fp16 = mul(x = var_6792_cast_fp16, y = var_878_cast_fp16)[name = string("op_6793_cast_fp16")]; tensor key_states_185_cast_fp16 = add(x = var_6786_cast_fp16, y = var_6793_cast_fp16)[name = string("key_states_185_cast_fp16")]; tensor expand_dims_216 = const()[name = string("expand_dims_216"), val = tensor([18])]; tensor expand_dims_217 = const()[name = string("expand_dims_217"), val = tensor([0])]; tensor expand_dims_219 = const()[name = string("expand_dims_219"), val = tensor([0])]; int32 concat_221_axis_0 = const()[name = string("concat_221_axis_0"), val = int32(0)]; bool concat_221_interleave_0 = const()[name = string("concat_221_interleave_0"), val = bool(false)]; tensor concat_221 = concat(axis = concat_221_axis_0, interleave = concat_221_interleave_0, values = (expand_dims_216, expand_dims_217, position_id, expand_dims_219))[name = string("concat_221")]; tensor expand_dims_220 = const()[name = string("expand_dims_220"), val = tensor([19])]; tensor concat_222_values1_0 = const()[name = string("concat_222_values1_0"), val = tensor([0])]; tensor concat_222_values3_0 = const()[name = string("concat_222_values3_0"), val = tensor([0])]; int32 concat_222_axis_0 = const()[name = string("concat_222_axis_0"), val = int32(0)]; bool concat_222_interleave_0 = const()[name = string("concat_222_interleave_0"), val = bool(false)]; tensor concat_222 = concat(axis = concat_222_axis_0, interleave = concat_222_interleave_0, values = (expand_dims_220, concat_222_values1_0, cache_position_end, concat_222_values3_0))[name = string("concat_222")]; tensor key_states_187_perm_0 = const()[name = string("key_states_187_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_19_stride_0 = const()[name = string("key_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_187_cast_fp16 = transpose(perm = key_states_187_perm_0, x = key_states_185_cast_fp16)[name = string("transpose_459")]; tensor key_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = key_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = key_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_19_squeeze_mask_0, stride = key_cache_internal_tensor_assign_19_stride_0, update = key_states_187_cast_fp16, x = coreml_update_state_314)[name = string("key_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_19_cast_fp16, input = key_cache)[name = string("coreml_update_state_316_write_state")]; tensor coreml_update_state_316 = read_state(input = key_cache)[name = string("coreml_update_state_316")]; tensor value_states_111_perm_0 = const()[name = string("value_states_111_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_19_stride_0 = const()[name = string("value_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_111_cast_fp16 = transpose(perm = value_states_111_perm_0, x = var_6769_cast_fp16)[name = string("transpose_458")]; tensor value_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = value_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = value_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_19_squeeze_mask_0, stride = value_cache_internal_tensor_assign_19_stride_0, update = value_states_111_cast_fp16, x = coreml_update_state_315)[name = string("value_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_19_cast_fp16, input = value_cache)[name = string("coreml_update_state_317_write_state")]; tensor coreml_update_state_317 = read_state(input = value_cache)[name = string("coreml_update_state_317")]; tensor var_6863_begin_0 = const()[name = string("op_6863_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6863_end_0 = const()[name = string("op_6863_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6863_end_mask_0 = const()[name = string("op_6863_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6863_cast_fp16 = slice_by_index(begin = var_6863_begin_0, end = var_6863_end_0, end_mask = var_6863_end_mask_0, x = coreml_update_state_316)[name = string("op_6863_cast_fp16")]; tensor tile_36 = const()[name = string("tile_36"), val = tensor([1, 1])]; int32 var_6866_axis_0 = const()[name = string("op_6866_axis_0"), val = int32(1)]; tensor var_6866_cast_fp16_0, tensor var_6866_cast_fp16_1 = split(axis = var_6866_axis_0, split_sizes = tile_36, x = var_6863_cast_fp16)[name = string("op_6866_cast_fp16")]; tensor var_6873_begin_0 = const()[name = string("op_6873_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6873_end_0 = const()[name = string("op_6873_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6873_end_mask_0 = const()[name = string("op_6873_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6873_cast_fp16 = slice_by_index(begin = var_6873_begin_0, end = var_6873_end_0, end_mask = var_6873_end_mask_0, x = coreml_update_state_317)[name = string("op_6873_cast_fp16")]; tensor tile_37 = const()[name = string("tile_37"), val = tensor([1, 1])]; int32 var_6876_axis_0 = const()[name = string("op_6876_axis_0"), val = int32(1)]; tensor var_6876_cast_fp16_0, tensor var_6876_cast_fp16_1 = split(axis = var_6876_axis_0, split_sizes = tile_37, x = var_6873_cast_fp16)[name = string("op_6876_cast_fp16")]; tensor var_6879_split_sizes_0 = const()[name = string("op_6879_split_sizes_0"), val = tensor([8, 8])]; int32 var_6879_axis_0 = const()[name = string("op_6879_axis_0"), val = int32(1)]; tensor var_6879_0, tensor var_6879_1 = split(axis = var_6879_axis_0, split_sizes = var_6879_split_sizes_0, x = query_states_111_cast_fp16)[name = string("op_6879")]; bool attn_weights_289_transpose_x_0 = const()[name = string("attn_weights_289_transpose_x_0"), val = bool(false)]; bool attn_weights_289_transpose_y_0 = const()[name = string("attn_weights_289_transpose_y_0"), val = bool(false)]; tensor attn_weights_289_cast_fp16 = matmul(transpose_x = attn_weights_289_transpose_x_0, transpose_y = attn_weights_289_transpose_y_0, x = var_6866_cast_fp16_0, y = var_6879_0)[name = string("attn_weights_289_cast_fp16")]; fp16 var_6882_to_fp16 = const()[name = string("op_6882_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_291_cast_fp16 = mul(x = attn_weights_289_cast_fp16, y = var_6882_to_fp16)[name = string("attn_weights_291_cast_fp16")]; tensor attn_weights_293_cast_fp16 = add(x = attn_weights_291_cast_fp16, y = attn_mask_1)[name = string("attn_weights_293_cast_fp16")]; int32 var_6886 = const()[name = string("op_6886"), val = int32(-2)]; tensor attn_weights_295_cast_fp16 = softmax(axis = var_6886, x = attn_weights_293_cast_fp16)[name = string("attn_weights_295_cast_fp16")]; bool var_6892_transpose_x_1 = const()[name = string("op_6892_transpose_x_1"), val = bool(true)]; bool var_6892_transpose_y_1 = const()[name = string("op_6892_transpose_y_1"), val = bool(false)]; tensor var_6892_cast_fp16 = matmul(transpose_x = var_6892_transpose_x_1, transpose_y = var_6892_transpose_y_1, x = attn_weights_295_cast_fp16, y = var_6876_cast_fp16_0)[name = string("op_6892_cast_fp16")]; bool attn_weights_297_transpose_x_0 = const()[name = string("attn_weights_297_transpose_x_0"), val = bool(false)]; bool attn_weights_297_transpose_y_0 = const()[name = string("attn_weights_297_transpose_y_0"), val = bool(false)]; tensor attn_weights_297_cast_fp16 = matmul(transpose_x = attn_weights_297_transpose_x_0, transpose_y = attn_weights_297_transpose_y_0, x = var_6866_cast_fp16_1, y = var_6879_1)[name = string("attn_weights_297_cast_fp16")]; fp16 var_6894_to_fp16 = const()[name = string("op_6894_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_299_cast_fp16 = mul(x = attn_weights_297_cast_fp16, y = var_6894_to_fp16)[name = string("attn_weights_299_cast_fp16")]; tensor attn_weights_301_cast_fp16 = add(x = attn_weights_299_cast_fp16, y = attn_mask_1)[name = string("attn_weights_301_cast_fp16")]; int32 var_6898 = const()[name = string("op_6898"), val = int32(-2)]; tensor attn_weights_303_cast_fp16 = softmax(axis = var_6898, x = attn_weights_301_cast_fp16)[name = string("attn_weights_303_cast_fp16")]; bool attn_output_145_transpose_x_1 = const()[name = string("attn_output_145_transpose_x_1"), val = bool(true)]; bool attn_output_145_transpose_y_1 = const()[name = string("attn_output_145_transpose_y_1"), val = bool(false)]; tensor attn_output_145_cast_fp16 = matmul(transpose_x = attn_output_145_transpose_x_1, transpose_y = attn_output_145_transpose_y_1, x = attn_weights_303_cast_fp16, y = var_6876_cast_fp16_1)[name = string("attn_output_145_cast_fp16")]; int32 var_6906 = const()[name = string("op_6906"), val = int32(1)]; bool attn_output_147_interleave_0 = const()[name = string("attn_output_147_interleave_0"), val = bool(false)]; tensor attn_output_147_cast_fp16 = concat(axis = var_6906, interleave = attn_output_147_interleave_0, values = (var_6892_cast_fp16, attn_output_145_cast_fp16))[name = string("attn_output_147_cast_fp16")]; tensor var_6910_perm_0 = const()[name = string("op_6910_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_227x = const()[name = string("concat_227x"), val = tensor([1, 2048, 1, -1])]; tensor var_6910_cast_fp16 = transpose(perm = var_6910_perm_0, x = attn_output_147_cast_fp16)[name = string("transpose_457")]; tensor attn_output_151_cast_fp16 = reshape(shape = concat_227x, x = var_6910_cast_fp16)[name = string("attn_output_151_cast_fp16")]; tensor hidden_states_183_strides_0 = const()[name = string("hidden_states_183_strides_0"), val = tensor([1, 1])]; string hidden_states_183_pad_type_0 = const()[name = string("hidden_states_183_pad_type_0"), val = string("valid")]; tensor hidden_states_183_pad_0 = const()[name = string("hidden_states_183_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_183_dilations_0 = const()[name = string("hidden_states_183_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_183_groups_0 = const()[name = string("hidden_states_183_groups_0"), val = int32(1)]; tensor hidden_states_183_cast_fp16 = conv(dilations = hidden_states_183_dilations_0, groups = hidden_states_183_groups_0, pad = hidden_states_183_pad_0, pad_type = hidden_states_183_pad_type_0, strides = hidden_states_183_strides_0, weight = layers_18_self_attn_o_proj_weight_cast_fp16, x = attn_output_151_cast_fp16)[name = string("hidden_states_183_cast_fp16")]; tensor hidden_states_185_cast_fp16 = add(x = hidden_states_179_cast_fp16, y = hidden_states_183_cast_fp16)[name = string("hidden_states_185_cast_fp16")]; fp16 const_188_promoted_to_fp16 = const()[name = string("const_188_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6943_cast_fp16 = mul(x = hidden_states_185_cast_fp16, y = const_188_promoted_to_fp16)[name = string("op_6943_cast_fp16")]; int32 var_6941 = const()[name = string("op_6941"), val = int32(1)]; bool doubled_149_interleave_0 = const()[name = string("doubled_149_interleave_0"), val = bool(false)]; tensor doubled_149_cast_fp16 = concat(axis = var_6941, interleave = doubled_149_interleave_0, values = (hidden_states_185_cast_fp16, var_6943_cast_fp16))[name = string("doubled_149_cast_fp16")]; tensor out_75_axes_0 = const()[name = string("out_75_axes_0"), val = tensor([1])]; tensor out_75_gamma_0_to_fp16 = const()[name = string("out_75_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492028288)))]; fp16 var_6953_to_fp16 = const()[name = string("op_6953_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_75_cast_fp16 = layer_norm(axes = out_75_axes_0, epsilon = var_6953_to_fp16, gamma = out_75_gamma_0_to_fp16, x = doubled_149_cast_fp16)[name = string("out_75_cast_fp16")]; tensor var_6964_split_sizes_0 = const()[name = string("op_6964_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6964_axis_0 = const()[name = string("op_6964_axis_0"), val = int32(1)]; tensor var_6964_cast_fp16_0, tensor var_6964_cast_fp16_1 = split(axis = var_6964_axis_0, split_sizes = var_6964_split_sizes_0, x = out_75_cast_fp16)[name = string("op_6964_cast_fp16")]; tensor input_37_strides_0 = const()[name = string("input_37_strides_0"), val = tensor([1, 1])]; string input_37_pad_type_0 = const()[name = string("input_37_pad_type_0"), val = string("valid")]; tensor input_37_pad_0 = const()[name = string("input_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_37_dilations_0 = const()[name = string("input_37_dilations_0"), val = tensor([1, 1])]; int32 input_37_groups_0 = const()[name = string("input_37_groups_0"), val = int32(1)]; tensor input_37_cast_fp16 = conv(dilations = input_37_dilations_0, groups = input_37_groups_0, pad = input_37_pad_0, pad_type = input_37_pad_type_0, strides = input_37_strides_0, weight = layers_18_mlp_gate_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("input_37_cast_fp16")]; tensor var_6981_cast_fp16 = silu(x = input_37_cast_fp16)[name = string("op_6981_cast_fp16")]; tensor var_6987_strides_0 = const()[name = string("op_6987_strides_0"), val = tensor([1, 1])]; string var_6987_pad_type_0 = const()[name = string("op_6987_pad_type_0"), val = string("valid")]; tensor var_6987_pad_0 = const()[name = string("op_6987_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6987_dilations_0 = const()[name = string("op_6987_dilations_0"), val = tensor([1, 1])]; int32 var_6987_groups_0 = const()[name = string("op_6987_groups_0"), val = int32(1)]; tensor var_6987_cast_fp16 = conv(dilations = var_6987_dilations_0, groups = var_6987_groups_0, pad = var_6987_pad_0, pad_type = var_6987_pad_type_0, strides = var_6987_strides_0, weight = layers_18_mlp_up_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("op_6987_cast_fp16")]; tensor x_189_cast_fp16 = mul(x = var_6981_cast_fp16, y = var_6987_cast_fp16)[name = string("x_189_cast_fp16")]; tensor hidden_states_187_strides_0 = const()[name = string("hidden_states_187_strides_0"), val = tensor([1, 1])]; string hidden_states_187_pad_type_0 = const()[name = string("hidden_states_187_pad_type_0"), val = string("valid")]; tensor hidden_states_187_pad_0 = const()[name = string("hidden_states_187_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_187_dilations_0 = const()[name = string("hidden_states_187_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_187_groups_0 = const()[name = string("hidden_states_187_groups_0"), val = int32(1)]; tensor hidden_states_187_cast_fp16 = conv(dilations = hidden_states_187_dilations_0, groups = hidden_states_187_groups_0, pad = hidden_states_187_pad_0, pad_type = hidden_states_187_pad_type_0, strides = hidden_states_187_strides_0, weight = layers_18_mlp_down_proj_weight_cast_fp16, x = x_189_cast_fp16)[name = string("hidden_states_187_cast_fp16")]; tensor hidden_states_189_cast_fp16 = add(x = hidden_states_185_cast_fp16, y = hidden_states_187_cast_fp16)[name = string("hidden_states_189_cast_fp16")]; fp16 const_190_promoted_to_fp16 = const()[name = string("const_190_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7005_cast_fp16 = mul(x = hidden_states_189_cast_fp16, y = const_190_promoted_to_fp16)[name = string("op_7005_cast_fp16")]; int32 var_7003 = const()[name = string("op_7003"), val = int32(1)]; bool doubled_153_interleave_0 = const()[name = string("doubled_153_interleave_0"), val = bool(false)]; tensor doubled_153_cast_fp16 = concat(axis = var_7003, interleave = doubled_153_interleave_0, values = (hidden_states_189_cast_fp16, var_7005_cast_fp16))[name = string("doubled_153_cast_fp16")]; tensor out_77_axes_0 = const()[name = string("out_77_axes_0"), val = tensor([1])]; tensor out_77_gamma_0_to_fp16 = const()[name = string("out_77_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492036544)))]; fp16 var_7015_to_fp16 = const()[name = string("op_7015_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_77_cast_fp16 = layer_norm(axes = out_77_axes_0, epsilon = var_7015_to_fp16, gamma = out_77_gamma_0_to_fp16, x = doubled_153_cast_fp16)[name = string("out_77_cast_fp16")]; tensor var_7026_split_sizes_0 = const()[name = string("op_7026_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7026_axis_0 = const()[name = string("op_7026_axis_0"), val = int32(1)]; tensor var_7026_cast_fp16_0, tensor var_7026_cast_fp16_1 = split(axis = var_7026_axis_0, split_sizes = var_7026_split_sizes_0, x = out_77_cast_fp16)[name = string("op_7026_cast_fp16")]; tensor query_states_115_strides_0 = const()[name = string("query_states_115_strides_0"), val = tensor([1, 1])]; string query_states_115_pad_type_0 = const()[name = string("query_states_115_pad_type_0"), val = string("valid")]; tensor query_states_115_pad_0 = const()[name = string("query_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_115_dilations_0 = const()[name = string("query_states_115_dilations_0"), val = tensor([1, 1])]; int32 query_states_115_groups_0 = const()[name = string("query_states_115_groups_0"), val = int32(1)]; tensor query_states_115_cast_fp16 = conv(dilations = query_states_115_dilations_0, groups = query_states_115_groups_0, pad = query_states_115_pad_0, pad_type = query_states_115_pad_type_0, strides = query_states_115_strides_0, weight = layers_19_self_attn_q_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("query_states_115_cast_fp16")]; tensor key_states_191_strides_0 = const()[name = string("key_states_191_strides_0"), val = tensor([1, 1])]; string key_states_191_pad_type_0 = const()[name = string("key_states_191_pad_type_0"), val = string("valid")]; tensor key_states_191_pad_0 = const()[name = string("key_states_191_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_191_dilations_0 = const()[name = string("key_states_191_dilations_0"), val = tensor([1, 1])]; int32 key_states_191_groups_0 = const()[name = string("key_states_191_groups_0"), val = int32(1)]; tensor key_states_191_cast_fp16 = conv(dilations = key_states_191_dilations_0, groups = key_states_191_groups_0, pad = key_states_191_pad_0, pad_type = key_states_191_pad_type_0, strides = key_states_191_strides_0, weight = layers_19_self_attn_k_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("key_states_191_cast_fp16")]; tensor layers_19_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492044800)))]; tensor value_states_115_strides_0 = const()[name = string("value_states_115_strides_0"), val = tensor([1, 1])]; string value_states_115_pad_type_0 = const()[name = string("value_states_115_pad_type_0"), val = string("valid")]; tensor value_states_115_pad_0 = const()[name = string("value_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_115_dilations_0 = const()[name = string("value_states_115_dilations_0"), val = tensor([1, 1])]; int32 value_states_115_groups_0 = const()[name = string("value_states_115_groups_0"), val = int32(1)]; tensor value_states_115_cast_fp16 = conv(dilations = value_states_115_dilations_0, groups = value_states_115_groups_0, pad = value_states_115_pad_0, pad_type = value_states_115_pad_type_0, strides = value_states_115_strides_0, weight = layers_19_self_attn_v_proj_weight_to_fp16, x = var_7026_cast_fp16_0)[name = string("value_states_115_cast_fp16")]; tensor concat_228x = const()[name = string("concat_228x"), val = tensor([1, 16, 128, -1])]; tensor x_191_cast_fp16 = reshape(shape = concat_228x, x = query_states_115_cast_fp16)[name = string("x_191_cast_fp16")]; tensor concat_229x = const()[name = string("concat_229x"), val = tensor([1, 2, 128, -1])]; tensor var_7083_cast_fp16 = reshape(shape = concat_229x, x = key_states_191_cast_fp16)[name = string("op_7083_cast_fp16")]; tensor concat_230x = const()[name = string("concat_230x"), val = tensor([1, 2, 128, -1])]; tensor var_7090_cast_fp16 = reshape(shape = concat_230x, x = value_states_115_cast_fp16)[name = string("op_7090_cast_fp16")]; tensor var_7094_cast_fp16 = mul(x = x_191_cast_fp16, y = var_869_cast_fp16)[name = string("op_7094_cast_fp16")]; tensor var_7095_split_sizes_0 = const()[name = string("op_7095_split_sizes_0"), val = tensor([64, 64])]; int32 var_7095_axis_0 = const()[name = string("op_7095_axis_0"), val = int32(-2)]; tensor var_7095_cast_fp16_0, tensor var_7095_cast_fp16_1 = split(axis = var_7095_axis_0, split_sizes = var_7095_split_sizes_0, x = x_191_cast_fp16)[name = string("op_7095_cast_fp16")]; fp16 const_192_promoted_to_fp16 = const()[name = string("const_192_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7097_cast_fp16 = mul(x = var_7095_cast_fp16_1, y = const_192_promoted_to_fp16)[name = string("op_7097_cast_fp16")]; int32 var_7099 = const()[name = string("op_7099"), val = int32(-2)]; bool var_7100_interleave_0 = const()[name = string("op_7100_interleave_0"), val = bool(false)]; tensor var_7100_cast_fp16 = concat(axis = var_7099, interleave = var_7100_interleave_0, values = (var_7097_cast_fp16, var_7095_cast_fp16_0))[name = string("op_7100_cast_fp16")]; tensor var_7101_cast_fp16 = mul(x = var_7100_cast_fp16, y = var_878_cast_fp16)[name = string("op_7101_cast_fp16")]; tensor query_states_117_cast_fp16 = add(x = var_7094_cast_fp16, y = var_7101_cast_fp16)[name = string("query_states_117_cast_fp16")]; tensor var_7107_cast_fp16 = mul(x = var_7083_cast_fp16, y = var_869_cast_fp16)[name = string("op_7107_cast_fp16")]; tensor var_7108_split_sizes_0 = const()[name = string("op_7108_split_sizes_0"), val = tensor([64, 64])]; int32 var_7108_axis_0 = const()[name = string("op_7108_axis_0"), val = int32(-2)]; tensor var_7108_cast_fp16_0, tensor var_7108_cast_fp16_1 = split(axis = var_7108_axis_0, split_sizes = var_7108_split_sizes_0, x = var_7083_cast_fp16)[name = string("op_7108_cast_fp16")]; fp16 const_193_promoted_to_fp16 = const()[name = string("const_193_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7110_cast_fp16 = mul(x = var_7108_cast_fp16_1, y = const_193_promoted_to_fp16)[name = string("op_7110_cast_fp16")]; int32 var_7112 = const()[name = string("op_7112"), val = int32(-2)]; bool var_7113_interleave_0 = const()[name = string("op_7113_interleave_0"), val = bool(false)]; tensor var_7113_cast_fp16 = concat(axis = var_7112, interleave = var_7113_interleave_0, values = (var_7110_cast_fp16, var_7108_cast_fp16_0))[name = string("op_7113_cast_fp16")]; tensor var_7114_cast_fp16 = mul(x = var_7113_cast_fp16, y = var_878_cast_fp16)[name = string("op_7114_cast_fp16")]; tensor key_states_195_cast_fp16 = add(x = var_7107_cast_fp16, y = var_7114_cast_fp16)[name = string("key_states_195_cast_fp16")]; tensor expand_dims_228 = const()[name = string("expand_dims_228"), val = tensor([19])]; tensor expand_dims_229 = const()[name = string("expand_dims_229"), val = tensor([0])]; tensor expand_dims_231 = const()[name = string("expand_dims_231"), val = tensor([0])]; int32 concat_233_axis_0 = const()[name = string("concat_233_axis_0"), val = int32(0)]; bool concat_233_interleave_0 = const()[name = string("concat_233_interleave_0"), val = bool(false)]; tensor concat_233 = concat(axis = concat_233_axis_0, interleave = concat_233_interleave_0, values = (expand_dims_228, expand_dims_229, position_id, expand_dims_231))[name = string("concat_233")]; tensor expand_dims_232 = const()[name = string("expand_dims_232"), val = tensor([20])]; tensor concat_234_values1_0 = const()[name = string("concat_234_values1_0"), val = tensor([0])]; tensor concat_234_values3_0 = const()[name = string("concat_234_values3_0"), val = tensor([0])]; int32 concat_234_axis_0 = const()[name = string("concat_234_axis_0"), val = int32(0)]; bool concat_234_interleave_0 = const()[name = string("concat_234_interleave_0"), val = bool(false)]; tensor concat_234 = concat(axis = concat_234_axis_0, interleave = concat_234_interleave_0, values = (expand_dims_232, concat_234_values1_0, cache_position_end, concat_234_values3_0))[name = string("concat_234")]; tensor key_states_197_perm_0 = const()[name = string("key_states_197_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_20_stride_0 = const()[name = string("key_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_197_cast_fp16 = transpose(perm = key_states_197_perm_0, x = key_states_195_cast_fp16)[name = string("transpose_456")]; tensor key_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = key_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = key_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_20_squeeze_mask_0, stride = key_cache_internal_tensor_assign_20_stride_0, update = key_states_197_cast_fp16, x = coreml_update_state_316)[name = string("key_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_20_cast_fp16, input = key_cache)[name = string("coreml_update_state_318_write_state")]; tensor coreml_update_state_318 = read_state(input = key_cache)[name = string("coreml_update_state_318")]; tensor value_states_117_perm_0 = const()[name = string("value_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_20_stride_0 = const()[name = string("value_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_117_cast_fp16 = transpose(perm = value_states_117_perm_0, x = var_7090_cast_fp16)[name = string("transpose_455")]; tensor value_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = value_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = value_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_20_squeeze_mask_0, stride = value_cache_internal_tensor_assign_20_stride_0, update = value_states_117_cast_fp16, x = coreml_update_state_317)[name = string("value_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_20_cast_fp16, input = value_cache)[name = string("coreml_update_state_319_write_state")]; tensor coreml_update_state_319 = read_state(input = value_cache)[name = string("coreml_update_state_319")]; tensor var_7184_begin_0 = const()[name = string("op_7184_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7184_end_0 = const()[name = string("op_7184_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7184_end_mask_0 = const()[name = string("op_7184_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7184_cast_fp16 = slice_by_index(begin = var_7184_begin_0, end = var_7184_end_0, end_mask = var_7184_end_mask_0, x = coreml_update_state_318)[name = string("op_7184_cast_fp16")]; tensor tile_38 = const()[name = string("tile_38"), val = tensor([1, 1])]; int32 var_7187_axis_0 = const()[name = string("op_7187_axis_0"), val = int32(1)]; tensor var_7187_cast_fp16_0, tensor var_7187_cast_fp16_1 = split(axis = var_7187_axis_0, split_sizes = tile_38, x = var_7184_cast_fp16)[name = string("op_7187_cast_fp16")]; tensor var_7194_begin_0 = const()[name = string("op_7194_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7194_end_0 = const()[name = string("op_7194_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7194_end_mask_0 = const()[name = string("op_7194_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7194_cast_fp16 = slice_by_index(begin = var_7194_begin_0, end = var_7194_end_0, end_mask = var_7194_end_mask_0, x = coreml_update_state_319)[name = string("op_7194_cast_fp16")]; tensor tile_39 = const()[name = string("tile_39"), val = tensor([1, 1])]; int32 var_7197_axis_0 = const()[name = string("op_7197_axis_0"), val = int32(1)]; tensor var_7197_cast_fp16_0, tensor var_7197_cast_fp16_1 = split(axis = var_7197_axis_0, split_sizes = tile_39, x = var_7194_cast_fp16)[name = string("op_7197_cast_fp16")]; tensor var_7200_split_sizes_0 = const()[name = string("op_7200_split_sizes_0"), val = tensor([8, 8])]; int32 var_7200_axis_0 = const()[name = string("op_7200_axis_0"), val = int32(1)]; tensor var_7200_0, tensor var_7200_1 = split(axis = var_7200_axis_0, split_sizes = var_7200_split_sizes_0, x = query_states_117_cast_fp16)[name = string("op_7200")]; bool attn_weights_305_transpose_x_0 = const()[name = string("attn_weights_305_transpose_x_0"), val = bool(false)]; bool attn_weights_305_transpose_y_0 = const()[name = string("attn_weights_305_transpose_y_0"), val = bool(false)]; tensor attn_weights_305_cast_fp16 = matmul(transpose_x = attn_weights_305_transpose_x_0, transpose_y = attn_weights_305_transpose_y_0, x = var_7187_cast_fp16_0, y = var_7200_0)[name = string("attn_weights_305_cast_fp16")]; fp16 var_7203_to_fp16 = const()[name = string("op_7203_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_307_cast_fp16 = mul(x = attn_weights_305_cast_fp16, y = var_7203_to_fp16)[name = string("attn_weights_307_cast_fp16")]; tensor attn_weights_309_cast_fp16 = add(x = attn_weights_307_cast_fp16, y = attn_mask_1)[name = string("attn_weights_309_cast_fp16")]; int32 var_7207 = const()[name = string("op_7207"), val = int32(-2)]; tensor attn_weights_311_cast_fp16 = softmax(axis = var_7207, x = attn_weights_309_cast_fp16)[name = string("attn_weights_311_cast_fp16")]; bool var_7213_transpose_x_1 = const()[name = string("op_7213_transpose_x_1"), val = bool(true)]; bool var_7213_transpose_y_1 = const()[name = string("op_7213_transpose_y_1"), val = bool(false)]; tensor var_7213_cast_fp16 = matmul(transpose_x = var_7213_transpose_x_1, transpose_y = var_7213_transpose_y_1, x = attn_weights_311_cast_fp16, y = var_7197_cast_fp16_0)[name = string("op_7213_cast_fp16")]; bool attn_weights_313_transpose_x_0 = const()[name = string("attn_weights_313_transpose_x_0"), val = bool(false)]; bool attn_weights_313_transpose_y_0 = const()[name = string("attn_weights_313_transpose_y_0"), val = bool(false)]; tensor attn_weights_313_cast_fp16 = matmul(transpose_x = attn_weights_313_transpose_x_0, transpose_y = attn_weights_313_transpose_y_0, x = var_7187_cast_fp16_1, y = var_7200_1)[name = string("attn_weights_313_cast_fp16")]; fp16 var_7215_to_fp16 = const()[name = string("op_7215_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_315_cast_fp16 = mul(x = attn_weights_313_cast_fp16, y = var_7215_to_fp16)[name = string("attn_weights_315_cast_fp16")]; tensor attn_weights_317_cast_fp16 = add(x = attn_weights_315_cast_fp16, y = attn_mask_1)[name = string("attn_weights_317_cast_fp16")]; int32 var_7219 = const()[name = string("op_7219"), val = int32(-2)]; tensor attn_weights_319_cast_fp16 = softmax(axis = var_7219, x = attn_weights_317_cast_fp16)[name = string("attn_weights_319_cast_fp16")]; bool attn_output_153_transpose_x_1 = const()[name = string("attn_output_153_transpose_x_1"), val = bool(true)]; bool attn_output_153_transpose_y_1 = const()[name = string("attn_output_153_transpose_y_1"), val = bool(false)]; tensor attn_output_153_cast_fp16 = matmul(transpose_x = attn_output_153_transpose_x_1, transpose_y = attn_output_153_transpose_y_1, x = attn_weights_319_cast_fp16, y = var_7197_cast_fp16_1)[name = string("attn_output_153_cast_fp16")]; int32 var_7227 = const()[name = string("op_7227"), val = int32(1)]; bool attn_output_155_interleave_0 = const()[name = string("attn_output_155_interleave_0"), val = bool(false)]; tensor attn_output_155_cast_fp16 = concat(axis = var_7227, interleave = attn_output_155_interleave_0, values = (var_7213_cast_fp16, attn_output_153_cast_fp16))[name = string("attn_output_155_cast_fp16")]; tensor var_7231_perm_0 = const()[name = string("op_7231_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_239x = const()[name = string("concat_239x"), val = tensor([1, 2048, 1, -1])]; tensor var_7231_cast_fp16 = transpose(perm = var_7231_perm_0, x = attn_output_155_cast_fp16)[name = string("transpose_454")]; tensor attn_output_159_cast_fp16 = reshape(shape = concat_239x, x = var_7231_cast_fp16)[name = string("attn_output_159_cast_fp16")]; tensor layers_19_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1493093440)))]; tensor hidden_states_193_strides_0 = const()[name = string("hidden_states_193_strides_0"), val = tensor([1, 1])]; string hidden_states_193_pad_type_0 = const()[name = string("hidden_states_193_pad_type_0"), val = string("valid")]; tensor hidden_states_193_pad_0 = const()[name = string("hidden_states_193_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_193_dilations_0 = const()[name = string("hidden_states_193_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_193_groups_0 = const()[name = string("hidden_states_193_groups_0"), val = int32(1)]; tensor hidden_states_193_cast_fp16 = conv(dilations = hidden_states_193_dilations_0, groups = hidden_states_193_groups_0, pad = hidden_states_193_pad_0, pad_type = hidden_states_193_pad_type_0, strides = hidden_states_193_strides_0, weight = layers_19_self_attn_o_proj_weight_to_fp16, x = attn_output_159_cast_fp16)[name = string("hidden_states_193_cast_fp16")]; tensor hidden_states_195_cast_fp16 = add(x = hidden_states_189_cast_fp16, y = hidden_states_193_cast_fp16)[name = string("hidden_states_195_cast_fp16")]; fp16 const_198_promoted_to_fp16 = const()[name = string("const_198_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7264_cast_fp16 = mul(x = hidden_states_195_cast_fp16, y = const_198_promoted_to_fp16)[name = string("op_7264_cast_fp16")]; int32 var_7262 = const()[name = string("op_7262"), val = int32(1)]; bool doubled_157_interleave_0 = const()[name = string("doubled_157_interleave_0"), val = bool(false)]; tensor doubled_157_cast_fp16 = concat(axis = var_7262, interleave = doubled_157_interleave_0, values = (hidden_states_195_cast_fp16, var_7264_cast_fp16))[name = string("doubled_157_cast_fp16")]; tensor out_79_axes_0 = const()[name = string("out_79_axes_0"), val = tensor([1])]; tensor out_79_gamma_0_to_fp16 = const()[name = string("out_79_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501482112)))]; fp16 var_7274_to_fp16 = const()[name = string("op_7274_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_79_cast_fp16 = layer_norm(axes = out_79_axes_0, epsilon = var_7274_to_fp16, gamma = out_79_gamma_0_to_fp16, x = doubled_157_cast_fp16)[name = string("out_79_cast_fp16")]; tensor var_7285_split_sizes_0 = const()[name = string("op_7285_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7285_axis_0 = const()[name = string("op_7285_axis_0"), val = int32(1)]; tensor var_7285_cast_fp16_0, tensor var_7285_cast_fp16_1 = split(axis = var_7285_axis_0, split_sizes = var_7285_split_sizes_0, x = out_79_cast_fp16)[name = string("op_7285_cast_fp16")]; tensor input_39_strides_0 = const()[name = string("input_39_strides_0"), val = tensor([1, 1])]; string input_39_pad_type_0 = const()[name = string("input_39_pad_type_0"), val = string("valid")]; tensor input_39_pad_0 = const()[name = string("input_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_39_dilations_0 = const()[name = string("input_39_dilations_0"), val = tensor([1, 1])]; int32 input_39_groups_0 = const()[name = string("input_39_groups_0"), val = int32(1)]; tensor input_39_cast_fp16 = conv(dilations = input_39_dilations_0, groups = input_39_groups_0, pad = input_39_pad_0, pad_type = input_39_pad_type_0, strides = input_39_strides_0, weight = layers_19_mlp_gate_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("input_39_cast_fp16")]; tensor var_7302_cast_fp16 = silu(x = input_39_cast_fp16)[name = string("op_7302_cast_fp16")]; tensor var_7308_strides_0 = const()[name = string("op_7308_strides_0"), val = tensor([1, 1])]; string var_7308_pad_type_0 = const()[name = string("op_7308_pad_type_0"), val = string("valid")]; tensor var_7308_pad_0 = const()[name = string("op_7308_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7308_dilations_0 = const()[name = string("op_7308_dilations_0"), val = tensor([1, 1])]; int32 var_7308_groups_0 = const()[name = string("op_7308_groups_0"), val = int32(1)]; tensor var_7308_cast_fp16 = conv(dilations = var_7308_dilations_0, groups = var_7308_groups_0, pad = var_7308_pad_0, pad_type = var_7308_pad_type_0, strides = var_7308_strides_0, weight = layers_19_mlp_up_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("op_7308_cast_fp16")]; tensor x_199_cast_fp16 = mul(x = var_7302_cast_fp16, y = var_7308_cast_fp16)[name = string("x_199_cast_fp16")]; tensor hidden_states_197_strides_0 = const()[name = string("hidden_states_197_strides_0"), val = tensor([1, 1])]; string hidden_states_197_pad_type_0 = const()[name = string("hidden_states_197_pad_type_0"), val = string("valid")]; tensor hidden_states_197_pad_0 = const()[name = string("hidden_states_197_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_197_dilations_0 = const()[name = string("hidden_states_197_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_197_groups_0 = const()[name = string("hidden_states_197_groups_0"), val = int32(1)]; tensor hidden_states_197_cast_fp16 = conv(dilations = hidden_states_197_dilations_0, groups = hidden_states_197_groups_0, pad = hidden_states_197_pad_0, pad_type = hidden_states_197_pad_type_0, strides = hidden_states_197_strides_0, weight = layers_19_mlp_down_proj_weight_cast_fp16, x = x_199_cast_fp16)[name = string("hidden_states_197_cast_fp16")]; tensor hidden_states_199_cast_fp16 = add(x = hidden_states_195_cast_fp16, y = hidden_states_197_cast_fp16)[name = string("hidden_states_199_cast_fp16")]; fp16 const_200_promoted_to_fp16 = const()[name = string("const_200_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7326_cast_fp16 = mul(x = hidden_states_199_cast_fp16, y = const_200_promoted_to_fp16)[name = string("op_7326_cast_fp16")]; int32 var_7324 = const()[name = string("op_7324"), val = int32(1)]; bool doubled_161_interleave_0 = const()[name = string("doubled_161_interleave_0"), val = bool(false)]; tensor doubled_161_cast_fp16 = concat(axis = var_7324, interleave = doubled_161_interleave_0, values = (hidden_states_199_cast_fp16, var_7326_cast_fp16))[name = string("doubled_161_cast_fp16")]; tensor out_81_axes_0 = const()[name = string("out_81_axes_0"), val = tensor([1])]; tensor out_81_gamma_0_to_fp16 = const()[name = string("out_81_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501490368)))]; fp16 var_7336_to_fp16 = const()[name = string("op_7336_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_81_cast_fp16 = layer_norm(axes = out_81_axes_0, epsilon = var_7336_to_fp16, gamma = out_81_gamma_0_to_fp16, x = doubled_161_cast_fp16)[name = string("out_81_cast_fp16")]; tensor var_7347_split_sizes_0 = const()[name = string("op_7347_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7347_axis_0 = const()[name = string("op_7347_axis_0"), val = int32(1)]; tensor var_7347_cast_fp16_0, tensor var_7347_cast_fp16_1 = split(axis = var_7347_axis_0, split_sizes = var_7347_split_sizes_0, x = out_81_cast_fp16)[name = string("op_7347_cast_fp16")]; tensor query_states_121_strides_0 = const()[name = string("query_states_121_strides_0"), val = tensor([1, 1])]; string query_states_121_pad_type_0 = const()[name = string("query_states_121_pad_type_0"), val = string("valid")]; tensor query_states_121_pad_0 = const()[name = string("query_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_121_dilations_0 = const()[name = string("query_states_121_dilations_0"), val = tensor([1, 1])]; int32 query_states_121_groups_0 = const()[name = string("query_states_121_groups_0"), val = int32(1)]; tensor query_states_121_cast_fp16 = conv(dilations = query_states_121_dilations_0, groups = query_states_121_groups_0, pad = query_states_121_pad_0, pad_type = query_states_121_pad_type_0, strides = query_states_121_strides_0, weight = layers_20_self_attn_q_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("query_states_121_cast_fp16")]; tensor key_states_201_strides_0 = const()[name = string("key_states_201_strides_0"), val = tensor([1, 1])]; string key_states_201_pad_type_0 = const()[name = string("key_states_201_pad_type_0"), val = string("valid")]; tensor key_states_201_pad_0 = const()[name = string("key_states_201_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_201_dilations_0 = const()[name = string("key_states_201_dilations_0"), val = tensor([1, 1])]; int32 key_states_201_groups_0 = const()[name = string("key_states_201_groups_0"), val = int32(1)]; tensor key_states_201_cast_fp16 = conv(dilations = key_states_201_dilations_0, groups = key_states_201_groups_0, pad = key_states_201_pad_0, pad_type = key_states_201_pad_type_0, strides = key_states_201_strides_0, weight = layers_20_self_attn_k_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("key_states_201_cast_fp16")]; tensor layers_20_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_20_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501498624)))]; tensor value_states_121_strides_0 = const()[name = string("value_states_121_strides_0"), val = tensor([1, 1])]; string value_states_121_pad_type_0 = const()[name = string("value_states_121_pad_type_0"), val = string("valid")]; tensor value_states_121_pad_0 = const()[name = string("value_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_121_dilations_0 = const()[name = string("value_states_121_dilations_0"), val = tensor([1, 1])]; int32 value_states_121_groups_0 = const()[name = string("value_states_121_groups_0"), val = int32(1)]; tensor value_states_121_cast_fp16 = conv(dilations = value_states_121_dilations_0, groups = value_states_121_groups_0, pad = value_states_121_pad_0, pad_type = value_states_121_pad_type_0, strides = value_states_121_strides_0, weight = layers_20_self_attn_v_proj_weight_to_fp16, x = var_7347_cast_fp16_0)[name = string("value_states_121_cast_fp16")]; tensor concat_240x = const()[name = string("concat_240x"), val = tensor([1, 16, 128, -1])]; tensor x_201_cast_fp16 = reshape(shape = concat_240x, x = query_states_121_cast_fp16)[name = string("x_201_cast_fp16")]; tensor concat_241x = const()[name = string("concat_241x"), val = tensor([1, 2, 128, -1])]; tensor var_7404_cast_fp16 = reshape(shape = concat_241x, x = key_states_201_cast_fp16)[name = string("op_7404_cast_fp16")]; tensor concat_242x = const()[name = string("concat_242x"), val = tensor([1, 2, 128, -1])]; tensor var_7411_cast_fp16 = reshape(shape = concat_242x, x = value_states_121_cast_fp16)[name = string("op_7411_cast_fp16")]; tensor var_7415_cast_fp16 = mul(x = x_201_cast_fp16, y = var_869_cast_fp16)[name = string("op_7415_cast_fp16")]; tensor var_7416_split_sizes_0 = const()[name = string("op_7416_split_sizes_0"), val = tensor([64, 64])]; int32 var_7416_axis_0 = const()[name = string("op_7416_axis_0"), val = int32(-2)]; tensor var_7416_cast_fp16_0, tensor var_7416_cast_fp16_1 = split(axis = var_7416_axis_0, split_sizes = var_7416_split_sizes_0, x = x_201_cast_fp16)[name = string("op_7416_cast_fp16")]; fp16 const_202_promoted_to_fp16 = const()[name = string("const_202_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7418_cast_fp16 = mul(x = var_7416_cast_fp16_1, y = const_202_promoted_to_fp16)[name = string("op_7418_cast_fp16")]; int32 var_7420 = const()[name = string("op_7420"), val = int32(-2)]; bool var_7421_interleave_0 = const()[name = string("op_7421_interleave_0"), val = bool(false)]; tensor var_7421_cast_fp16 = concat(axis = var_7420, interleave = var_7421_interleave_0, values = (var_7418_cast_fp16, var_7416_cast_fp16_0))[name = string("op_7421_cast_fp16")]; tensor var_7422_cast_fp16 = mul(x = var_7421_cast_fp16, y = var_878_cast_fp16)[name = string("op_7422_cast_fp16")]; tensor query_states_123_cast_fp16 = add(x = var_7415_cast_fp16, y = var_7422_cast_fp16)[name = string("query_states_123_cast_fp16")]; tensor var_7428_cast_fp16 = mul(x = var_7404_cast_fp16, y = var_869_cast_fp16)[name = string("op_7428_cast_fp16")]; tensor var_7429_split_sizes_0 = const()[name = string("op_7429_split_sizes_0"), val = tensor([64, 64])]; int32 var_7429_axis_0 = const()[name = string("op_7429_axis_0"), val = int32(-2)]; tensor var_7429_cast_fp16_0, tensor var_7429_cast_fp16_1 = split(axis = var_7429_axis_0, split_sizes = var_7429_split_sizes_0, x = var_7404_cast_fp16)[name = string("op_7429_cast_fp16")]; fp16 const_203_promoted_to_fp16 = const()[name = string("const_203_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7431_cast_fp16 = mul(x = var_7429_cast_fp16_1, y = const_203_promoted_to_fp16)[name = string("op_7431_cast_fp16")]; int32 var_7433 = const()[name = string("op_7433"), val = int32(-2)]; bool var_7434_interleave_0 = const()[name = string("op_7434_interleave_0"), val = bool(false)]; tensor var_7434_cast_fp16 = concat(axis = var_7433, interleave = var_7434_interleave_0, values = (var_7431_cast_fp16, var_7429_cast_fp16_0))[name = string("op_7434_cast_fp16")]; tensor var_7435_cast_fp16 = mul(x = var_7434_cast_fp16, y = var_878_cast_fp16)[name = string("op_7435_cast_fp16")]; tensor key_states_205_cast_fp16 = add(x = var_7428_cast_fp16, y = var_7435_cast_fp16)[name = string("key_states_205_cast_fp16")]; tensor expand_dims_240 = const()[name = string("expand_dims_240"), val = tensor([20])]; tensor expand_dims_241 = const()[name = string("expand_dims_241"), val = tensor([0])]; tensor expand_dims_243 = const()[name = string("expand_dims_243"), val = tensor([0])]; int32 concat_245_axis_0 = const()[name = string("concat_245_axis_0"), val = int32(0)]; bool concat_245_interleave_0 = const()[name = string("concat_245_interleave_0"), val = bool(false)]; tensor concat_245 = concat(axis = concat_245_axis_0, interleave = concat_245_interleave_0, values = (expand_dims_240, expand_dims_241, position_id, expand_dims_243))[name = string("concat_245")]; tensor expand_dims_244 = const()[name = string("expand_dims_244"), val = tensor([21])]; tensor concat_246_values1_0 = const()[name = string("concat_246_values1_0"), val = tensor([0])]; tensor concat_246_values3_0 = const()[name = string("concat_246_values3_0"), val = tensor([0])]; int32 concat_246_axis_0 = const()[name = string("concat_246_axis_0"), val = int32(0)]; bool concat_246_interleave_0 = const()[name = string("concat_246_interleave_0"), val = bool(false)]; tensor concat_246 = concat(axis = concat_246_axis_0, interleave = concat_246_interleave_0, values = (expand_dims_244, concat_246_values1_0, cache_position_end, concat_246_values3_0))[name = string("concat_246")]; tensor key_states_207_perm_0 = const()[name = string("key_states_207_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_21_stride_0 = const()[name = string("key_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_207_cast_fp16 = transpose(perm = key_states_207_perm_0, x = key_states_205_cast_fp16)[name = string("transpose_453")]; tensor key_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = key_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = key_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_21_squeeze_mask_0, stride = key_cache_internal_tensor_assign_21_stride_0, update = key_states_207_cast_fp16, x = coreml_update_state_318)[name = string("key_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_21_cast_fp16, input = key_cache)[name = string("coreml_update_state_320_write_state")]; tensor coreml_update_state_320 = read_state(input = key_cache)[name = string("coreml_update_state_320")]; tensor value_states_123_perm_0 = const()[name = string("value_states_123_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_21_stride_0 = const()[name = string("value_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_123_cast_fp16 = transpose(perm = value_states_123_perm_0, x = var_7411_cast_fp16)[name = string("transpose_452")]; tensor value_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = value_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = value_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_21_squeeze_mask_0, stride = value_cache_internal_tensor_assign_21_stride_0, update = value_states_123_cast_fp16, x = coreml_update_state_319)[name = string("value_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_21_cast_fp16, input = value_cache)[name = string("coreml_update_state_321_write_state")]; tensor coreml_update_state_321 = read_state(input = value_cache)[name = string("coreml_update_state_321")]; tensor var_7505_begin_0 = const()[name = string("op_7505_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7505_end_0 = const()[name = string("op_7505_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7505_end_mask_0 = const()[name = string("op_7505_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7505_cast_fp16 = slice_by_index(begin = var_7505_begin_0, end = var_7505_end_0, end_mask = var_7505_end_mask_0, x = coreml_update_state_320)[name = string("op_7505_cast_fp16")]; tensor tile_40 = const()[name = string("tile_40"), val = tensor([1, 1])]; int32 var_7508_axis_0 = const()[name = string("op_7508_axis_0"), val = int32(1)]; tensor var_7508_cast_fp16_0, tensor var_7508_cast_fp16_1 = split(axis = var_7508_axis_0, split_sizes = tile_40, x = var_7505_cast_fp16)[name = string("op_7508_cast_fp16")]; tensor var_7515_begin_0 = const()[name = string("op_7515_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7515_end_0 = const()[name = string("op_7515_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7515_end_mask_0 = const()[name = string("op_7515_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7515_cast_fp16 = slice_by_index(begin = var_7515_begin_0, end = var_7515_end_0, end_mask = var_7515_end_mask_0, x = coreml_update_state_321)[name = string("op_7515_cast_fp16")]; tensor tile_41 = const()[name = string("tile_41"), val = tensor([1, 1])]; int32 var_7518_axis_0 = const()[name = string("op_7518_axis_0"), val = int32(1)]; tensor var_7518_cast_fp16_0, tensor var_7518_cast_fp16_1 = split(axis = var_7518_axis_0, split_sizes = tile_41, x = var_7515_cast_fp16)[name = string("op_7518_cast_fp16")]; tensor var_7521_split_sizes_0 = const()[name = string("op_7521_split_sizes_0"), val = tensor([8, 8])]; int32 var_7521_axis_0 = const()[name = string("op_7521_axis_0"), val = int32(1)]; tensor var_7521_0, tensor var_7521_1 = split(axis = var_7521_axis_0, split_sizes = var_7521_split_sizes_0, x = query_states_123_cast_fp16)[name = string("op_7521")]; bool attn_weights_321_transpose_x_0 = const()[name = string("attn_weights_321_transpose_x_0"), val = bool(false)]; bool attn_weights_321_transpose_y_0 = const()[name = string("attn_weights_321_transpose_y_0"), val = bool(false)]; tensor attn_weights_321_cast_fp16 = matmul(transpose_x = attn_weights_321_transpose_x_0, transpose_y = attn_weights_321_transpose_y_0, x = var_7508_cast_fp16_0, y = var_7521_0)[name = string("attn_weights_321_cast_fp16")]; fp16 var_7524_to_fp16 = const()[name = string("op_7524_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_323_cast_fp16 = mul(x = attn_weights_321_cast_fp16, y = var_7524_to_fp16)[name = string("attn_weights_323_cast_fp16")]; tensor attn_weights_325_cast_fp16 = add(x = attn_weights_323_cast_fp16, y = attn_mask_1)[name = string("attn_weights_325_cast_fp16")]; int32 var_7528 = const()[name = string("op_7528"), val = int32(-2)]; tensor attn_weights_327_cast_fp16 = softmax(axis = var_7528, x = attn_weights_325_cast_fp16)[name = string("attn_weights_327_cast_fp16")]; bool var_7534_transpose_x_1 = const()[name = string("op_7534_transpose_x_1"), val = bool(true)]; bool var_7534_transpose_y_1 = const()[name = string("op_7534_transpose_y_1"), val = bool(false)]; tensor var_7534_cast_fp16 = matmul(transpose_x = var_7534_transpose_x_1, transpose_y = var_7534_transpose_y_1, x = attn_weights_327_cast_fp16, y = var_7518_cast_fp16_0)[name = string("op_7534_cast_fp16")]; bool attn_weights_329_transpose_x_0 = const()[name = string("attn_weights_329_transpose_x_0"), val = bool(false)]; bool attn_weights_329_transpose_y_0 = const()[name = string("attn_weights_329_transpose_y_0"), val = bool(false)]; tensor attn_weights_329_cast_fp16 = matmul(transpose_x = attn_weights_329_transpose_x_0, transpose_y = attn_weights_329_transpose_y_0, x = var_7508_cast_fp16_1, y = var_7521_1)[name = string("attn_weights_329_cast_fp16")]; fp16 var_7536_to_fp16 = const()[name = string("op_7536_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_331_cast_fp16 = mul(x = attn_weights_329_cast_fp16, y = var_7536_to_fp16)[name = string("attn_weights_331_cast_fp16")]; tensor attn_weights_333_cast_fp16 = add(x = attn_weights_331_cast_fp16, y = attn_mask_1)[name = string("attn_weights_333_cast_fp16")]; int32 var_7540 = const()[name = string("op_7540"), val = int32(-2)]; tensor attn_weights_335_cast_fp16 = softmax(axis = var_7540, x = attn_weights_333_cast_fp16)[name = string("attn_weights_335_cast_fp16")]; bool attn_output_161_transpose_x_1 = const()[name = string("attn_output_161_transpose_x_1"), val = bool(true)]; bool attn_output_161_transpose_y_1 = const()[name = string("attn_output_161_transpose_y_1"), val = bool(false)]; tensor attn_output_161_cast_fp16 = matmul(transpose_x = attn_output_161_transpose_x_1, transpose_y = attn_output_161_transpose_y_1, x = attn_weights_335_cast_fp16, y = var_7518_cast_fp16_1)[name = string("attn_output_161_cast_fp16")]; int32 var_7548 = const()[name = string("op_7548"), val = int32(1)]; bool attn_output_163_interleave_0 = const()[name = string("attn_output_163_interleave_0"), val = bool(false)]; tensor attn_output_163_cast_fp16 = concat(axis = var_7548, interleave = attn_output_163_interleave_0, values = (var_7534_cast_fp16, attn_output_161_cast_fp16))[name = string("attn_output_163_cast_fp16")]; tensor var_7552_perm_0 = const()[name = string("op_7552_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_251x = const()[name = string("concat_251x"), val = tensor([1, 2048, 1, -1])]; tensor var_7552_cast_fp16 = transpose(perm = var_7552_perm_0, x = attn_output_163_cast_fp16)[name = string("transpose_451")]; tensor attn_output_167_cast_fp16 = reshape(shape = concat_251x, x = var_7552_cast_fp16)[name = string("attn_output_167_cast_fp16")]; tensor hidden_states_203_strides_0 = const()[name = string("hidden_states_203_strides_0"), val = tensor([1, 1])]; string hidden_states_203_pad_type_0 = const()[name = string("hidden_states_203_pad_type_0"), val = string("valid")]; tensor hidden_states_203_pad_0 = const()[name = string("hidden_states_203_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_203_dilations_0 = const()[name = string("hidden_states_203_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_203_groups_0 = const()[name = string("hidden_states_203_groups_0"), val = int32(1)]; tensor hidden_states_203_cast_fp16 = conv(dilations = hidden_states_203_dilations_0, groups = hidden_states_203_groups_0, pad = hidden_states_203_pad_0, pad_type = hidden_states_203_pad_type_0, strides = hidden_states_203_strides_0, weight = layers_20_self_attn_o_proj_weight_cast_fp16, x = attn_output_167_cast_fp16)[name = string("hidden_states_203_cast_fp16")]; tensor hidden_states_205_cast_fp16 = add(x = hidden_states_199_cast_fp16, y = hidden_states_203_cast_fp16)[name = string("hidden_states_205_cast_fp16")]; fp16 const_208_promoted_to_fp16 = const()[name = string("const_208_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7585_cast_fp16 = mul(x = hidden_states_205_cast_fp16, y = const_208_promoted_to_fp16)[name = string("op_7585_cast_fp16")]; int32 var_7583 = const()[name = string("op_7583"), val = int32(1)]; bool doubled_165_interleave_0 = const()[name = string("doubled_165_interleave_0"), val = bool(false)]; tensor doubled_165_cast_fp16 = concat(axis = var_7583, interleave = doubled_165_interleave_0, values = (hidden_states_205_cast_fp16, var_7585_cast_fp16))[name = string("doubled_165_cast_fp16")]; tensor out_83_axes_0 = const()[name = string("out_83_axes_0"), val = tensor([1])]; tensor out_83_gamma_0_to_fp16 = const()[name = string("out_83_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502547264)))]; fp16 var_7595_to_fp16 = const()[name = string("op_7595_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_83_cast_fp16 = layer_norm(axes = out_83_axes_0, epsilon = var_7595_to_fp16, gamma = out_83_gamma_0_to_fp16, x = doubled_165_cast_fp16)[name = string("out_83_cast_fp16")]; tensor var_7606_split_sizes_0 = const()[name = string("op_7606_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7606_axis_0 = const()[name = string("op_7606_axis_0"), val = int32(1)]; tensor var_7606_cast_fp16_0, tensor var_7606_cast_fp16_1 = split(axis = var_7606_axis_0, split_sizes = var_7606_split_sizes_0, x = out_83_cast_fp16)[name = string("op_7606_cast_fp16")]; tensor input_41_strides_0 = const()[name = string("input_41_strides_0"), val = tensor([1, 1])]; string input_41_pad_type_0 = const()[name = string("input_41_pad_type_0"), val = string("valid")]; tensor input_41_pad_0 = const()[name = string("input_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_41_dilations_0 = const()[name = string("input_41_dilations_0"), val = tensor([1, 1])]; int32 input_41_groups_0 = const()[name = string("input_41_groups_0"), val = int32(1)]; tensor input_41_cast_fp16 = conv(dilations = input_41_dilations_0, groups = input_41_groups_0, pad = input_41_pad_0, pad_type = input_41_pad_type_0, strides = input_41_strides_0, weight = layers_20_mlp_gate_proj_weight_cast_fp16, x = var_7606_cast_fp16_0)[name = string("input_41_cast_fp16")]; tensor var_7623_cast_fp16 = silu(x = input_41_cast_fp16)[name = string("op_7623_cast_fp16")]; tensor layers_20_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_20_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502555520)))]; tensor var_7629_strides_0 = const()[name = string("op_7629_strides_0"), val = tensor([1, 1])]; string var_7629_pad_type_0 = const()[name = string("op_7629_pad_type_0"), val = string("valid")]; tensor var_7629_pad_0 = const()[name = string("op_7629_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7629_dilations_0 = const()[name = string("op_7629_dilations_0"), val = tensor([1, 1])]; int32 var_7629_groups_0 = const()[name = string("op_7629_groups_0"), val = int32(1)]; tensor var_7629_cast_fp16 = conv(dilations = var_7629_dilations_0, groups = var_7629_groups_0, pad = var_7629_pad_0, pad_type = var_7629_pad_type_0, strides = var_7629_strides_0, weight = layers_20_mlp_up_proj_weight_to_fp16, x = var_7606_cast_fp16_0)[name = string("op_7629_cast_fp16")]; tensor x_209_cast_fp16 = mul(x = var_7623_cast_fp16, y = var_7629_cast_fp16)[name = string("x_209_cast_fp16")]; tensor hidden_states_207_strides_0 = const()[name = string("hidden_states_207_strides_0"), val = tensor([1, 1])]; string hidden_states_207_pad_type_0 = const()[name = string("hidden_states_207_pad_type_0"), val = string("valid")]; tensor hidden_states_207_pad_0 = const()[name = string("hidden_states_207_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_207_dilations_0 = const()[name = string("hidden_states_207_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_207_groups_0 = const()[name = string("hidden_states_207_groups_0"), val = int32(1)]; tensor hidden_states_207_cast_fp16 = conv(dilations = hidden_states_207_dilations_0, groups = hidden_states_207_groups_0, pad = hidden_states_207_pad_0, pad_type = hidden_states_207_pad_type_0, strides = hidden_states_207_strides_0, weight = layers_20_mlp_down_proj_weight_cast_fp16, x = x_209_cast_fp16)[name = string("hidden_states_207_cast_fp16")]; tensor hidden_states_209_cast_fp16 = add(x = hidden_states_205_cast_fp16, y = hidden_states_207_cast_fp16)[name = string("hidden_states_209_cast_fp16")]; fp16 const_210_promoted_to_fp16 = const()[name = string("const_210_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7647_cast_fp16 = mul(x = hidden_states_209_cast_fp16, y = const_210_promoted_to_fp16)[name = string("op_7647_cast_fp16")]; int32 var_7645 = const()[name = string("op_7645"), val = int32(1)]; bool doubled_169_interleave_0 = const()[name = string("doubled_169_interleave_0"), val = bool(false)]; tensor doubled_169_cast_fp16 = concat(axis = var_7645, interleave = doubled_169_interleave_0, values = (hidden_states_209_cast_fp16, var_7647_cast_fp16))[name = string("doubled_169_cast_fp16")]; tensor out_85_axes_0 = const()[name = string("out_85_axes_0"), val = tensor([1])]; tensor out_85_gamma_0_to_fp16 = const()[name = string("out_85_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527721408)))]; fp16 var_7657_to_fp16 = const()[name = string("op_7657_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_85_cast_fp16 = layer_norm(axes = out_85_axes_0, epsilon = var_7657_to_fp16, gamma = out_85_gamma_0_to_fp16, x = doubled_169_cast_fp16)[name = string("out_85_cast_fp16")]; tensor var_7668_split_sizes_0 = const()[name = string("op_7668_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7668_axis_0 = const()[name = string("op_7668_axis_0"), val = int32(1)]; tensor var_7668_cast_fp16_0, tensor var_7668_cast_fp16_1 = split(axis = var_7668_axis_0, split_sizes = var_7668_split_sizes_0, x = out_85_cast_fp16)[name = string("op_7668_cast_fp16")]; tensor query_states_127_strides_0 = const()[name = string("query_states_127_strides_0"), val = tensor([1, 1])]; string query_states_127_pad_type_0 = const()[name = string("query_states_127_pad_type_0"), val = string("valid")]; tensor query_states_127_pad_0 = const()[name = string("query_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_127_dilations_0 = const()[name = string("query_states_127_dilations_0"), val = tensor([1, 1])]; int32 query_states_127_groups_0 = const()[name = string("query_states_127_groups_0"), val = int32(1)]; tensor query_states_127_cast_fp16 = conv(dilations = query_states_127_dilations_0, groups = query_states_127_groups_0, pad = query_states_127_pad_0, pad_type = query_states_127_pad_type_0, strides = query_states_127_strides_0, weight = layers_21_self_attn_q_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("query_states_127_cast_fp16")]; tensor key_states_211_strides_0 = const()[name = string("key_states_211_strides_0"), val = tensor([1, 1])]; string key_states_211_pad_type_0 = const()[name = string("key_states_211_pad_type_0"), val = string("valid")]; tensor key_states_211_pad_0 = const()[name = string("key_states_211_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_211_dilations_0 = const()[name = string("key_states_211_dilations_0"), val = tensor([1, 1])]; int32 key_states_211_groups_0 = const()[name = string("key_states_211_groups_0"), val = int32(1)]; tensor key_states_211_cast_fp16 = conv(dilations = key_states_211_dilations_0, groups = key_states_211_groups_0, pad = key_states_211_pad_0, pad_type = key_states_211_pad_type_0, strides = key_states_211_strides_0, weight = layers_21_self_attn_k_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("key_states_211_cast_fp16")]; tensor layers_21_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_21_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527729664)))]; tensor value_states_127_strides_0 = const()[name = string("value_states_127_strides_0"), val = tensor([1, 1])]; string value_states_127_pad_type_0 = const()[name = string("value_states_127_pad_type_0"), val = string("valid")]; tensor value_states_127_pad_0 = const()[name = string("value_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_127_dilations_0 = const()[name = string("value_states_127_dilations_0"), val = tensor([1, 1])]; int32 value_states_127_groups_0 = const()[name = string("value_states_127_groups_0"), val = int32(1)]; tensor value_states_127_cast_fp16 = conv(dilations = value_states_127_dilations_0, groups = value_states_127_groups_0, pad = value_states_127_pad_0, pad_type = value_states_127_pad_type_0, strides = value_states_127_strides_0, weight = layers_21_self_attn_v_proj_weight_to_fp16, x = var_7668_cast_fp16_0)[name = string("value_states_127_cast_fp16")]; tensor concat_252x = const()[name = string("concat_252x"), val = tensor([1, 16, 128, -1])]; tensor x_211_cast_fp16 = reshape(shape = concat_252x, x = query_states_127_cast_fp16)[name = string("x_211_cast_fp16")]; tensor concat_253x = const()[name = string("concat_253x"), val = tensor([1, 2, 128, -1])]; tensor var_7725_cast_fp16 = reshape(shape = concat_253x, x = key_states_211_cast_fp16)[name = string("op_7725_cast_fp16")]; tensor concat_254x = const()[name = string("concat_254x"), val = tensor([1, 2, 128, -1])]; tensor var_7732_cast_fp16 = reshape(shape = concat_254x, x = value_states_127_cast_fp16)[name = string("op_7732_cast_fp16")]; tensor var_7736_cast_fp16 = mul(x = x_211_cast_fp16, y = var_869_cast_fp16)[name = string("op_7736_cast_fp16")]; tensor var_7737_split_sizes_0 = const()[name = string("op_7737_split_sizes_0"), val = tensor([64, 64])]; int32 var_7737_axis_0 = const()[name = string("op_7737_axis_0"), val = int32(-2)]; tensor var_7737_cast_fp16_0, tensor var_7737_cast_fp16_1 = split(axis = var_7737_axis_0, split_sizes = var_7737_split_sizes_0, x = x_211_cast_fp16)[name = string("op_7737_cast_fp16")]; fp16 const_212_promoted_to_fp16 = const()[name = string("const_212_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7739_cast_fp16 = mul(x = var_7737_cast_fp16_1, y = const_212_promoted_to_fp16)[name = string("op_7739_cast_fp16")]; int32 var_7741 = const()[name = string("op_7741"), val = int32(-2)]; bool var_7742_interleave_0 = const()[name = string("op_7742_interleave_0"), val = bool(false)]; tensor var_7742_cast_fp16 = concat(axis = var_7741, interleave = var_7742_interleave_0, values = (var_7739_cast_fp16, var_7737_cast_fp16_0))[name = string("op_7742_cast_fp16")]; tensor var_7743_cast_fp16 = mul(x = var_7742_cast_fp16, y = var_878_cast_fp16)[name = string("op_7743_cast_fp16")]; tensor query_states_129_cast_fp16 = add(x = var_7736_cast_fp16, y = var_7743_cast_fp16)[name = string("query_states_129_cast_fp16")]; tensor var_7749_cast_fp16 = mul(x = var_7725_cast_fp16, y = var_869_cast_fp16)[name = string("op_7749_cast_fp16")]; tensor var_7750_split_sizes_0 = const()[name = string("op_7750_split_sizes_0"), val = tensor([64, 64])]; int32 var_7750_axis_0 = const()[name = string("op_7750_axis_0"), val = int32(-2)]; tensor var_7750_cast_fp16_0, tensor var_7750_cast_fp16_1 = split(axis = var_7750_axis_0, split_sizes = var_7750_split_sizes_0, x = var_7725_cast_fp16)[name = string("op_7750_cast_fp16")]; fp16 const_213_promoted_to_fp16 = const()[name = string("const_213_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7752_cast_fp16 = mul(x = var_7750_cast_fp16_1, y = const_213_promoted_to_fp16)[name = string("op_7752_cast_fp16")]; int32 var_7754 = const()[name = string("op_7754"), val = int32(-2)]; bool var_7755_interleave_0 = const()[name = string("op_7755_interleave_0"), val = bool(false)]; tensor var_7755_cast_fp16 = concat(axis = var_7754, interleave = var_7755_interleave_0, values = (var_7752_cast_fp16, var_7750_cast_fp16_0))[name = string("op_7755_cast_fp16")]; tensor var_7756_cast_fp16 = mul(x = var_7755_cast_fp16, y = var_878_cast_fp16)[name = string("op_7756_cast_fp16")]; tensor key_states_215_cast_fp16 = add(x = var_7749_cast_fp16, y = var_7756_cast_fp16)[name = string("key_states_215_cast_fp16")]; tensor expand_dims_252 = const()[name = string("expand_dims_252"), val = tensor([21])]; tensor expand_dims_253 = const()[name = string("expand_dims_253"), val = tensor([0])]; tensor expand_dims_255 = const()[name = string("expand_dims_255"), val = tensor([0])]; int32 concat_257_axis_0 = const()[name = string("concat_257_axis_0"), val = int32(0)]; bool concat_257_interleave_0 = const()[name = string("concat_257_interleave_0"), val = bool(false)]; tensor concat_257 = concat(axis = concat_257_axis_0, interleave = concat_257_interleave_0, values = (expand_dims_252, expand_dims_253, position_id, expand_dims_255))[name = string("concat_257")]; tensor expand_dims_256 = const()[name = string("expand_dims_256"), val = tensor([22])]; tensor concat_258_values1_0 = const()[name = string("concat_258_values1_0"), val = tensor([0])]; tensor concat_258_values3_0 = const()[name = string("concat_258_values3_0"), val = tensor([0])]; int32 concat_258_axis_0 = const()[name = string("concat_258_axis_0"), val = int32(0)]; bool concat_258_interleave_0 = const()[name = string("concat_258_interleave_0"), val = bool(false)]; tensor concat_258 = concat(axis = concat_258_axis_0, interleave = concat_258_interleave_0, values = (expand_dims_256, concat_258_values1_0, cache_position_end, concat_258_values3_0))[name = string("concat_258")]; tensor key_states_217_perm_0 = const()[name = string("key_states_217_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_22_stride_0 = const()[name = string("key_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_217_cast_fp16 = transpose(perm = key_states_217_perm_0, x = key_states_215_cast_fp16)[name = string("transpose_450")]; tensor key_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = key_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = key_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_22_squeeze_mask_0, stride = key_cache_internal_tensor_assign_22_stride_0, update = key_states_217_cast_fp16, x = coreml_update_state_320)[name = string("key_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_22_cast_fp16, input = key_cache)[name = string("coreml_update_state_322_write_state")]; tensor coreml_update_state_322 = read_state(input = key_cache)[name = string("coreml_update_state_322")]; tensor value_states_129_perm_0 = const()[name = string("value_states_129_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_22_stride_0 = const()[name = string("value_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_129_cast_fp16 = transpose(perm = value_states_129_perm_0, x = var_7732_cast_fp16)[name = string("transpose_449")]; tensor value_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = value_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = value_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_22_squeeze_mask_0, stride = value_cache_internal_tensor_assign_22_stride_0, update = value_states_129_cast_fp16, x = coreml_update_state_321)[name = string("value_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_22_cast_fp16, input = value_cache)[name = string("coreml_update_state_323_write_state")]; tensor coreml_update_state_323 = read_state(input = value_cache)[name = string("coreml_update_state_323")]; tensor var_7826_begin_0 = const()[name = string("op_7826_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7826_end_0 = const()[name = string("op_7826_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7826_end_mask_0 = const()[name = string("op_7826_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7826_cast_fp16 = slice_by_index(begin = var_7826_begin_0, end = var_7826_end_0, end_mask = var_7826_end_mask_0, x = coreml_update_state_322)[name = string("op_7826_cast_fp16")]; tensor tile_42 = const()[name = string("tile_42"), val = tensor([1, 1])]; int32 var_7829_axis_0 = const()[name = string("op_7829_axis_0"), val = int32(1)]; tensor var_7829_cast_fp16_0, tensor var_7829_cast_fp16_1 = split(axis = var_7829_axis_0, split_sizes = tile_42, x = var_7826_cast_fp16)[name = string("op_7829_cast_fp16")]; tensor var_7836_begin_0 = const()[name = string("op_7836_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7836_end_0 = const()[name = string("op_7836_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7836_end_mask_0 = const()[name = string("op_7836_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7836_cast_fp16 = slice_by_index(begin = var_7836_begin_0, end = var_7836_end_0, end_mask = var_7836_end_mask_0, x = coreml_update_state_323)[name = string("op_7836_cast_fp16")]; tensor tile_43 = const()[name = string("tile_43"), val = tensor([1, 1])]; int32 var_7839_axis_0 = const()[name = string("op_7839_axis_0"), val = int32(1)]; tensor var_7839_cast_fp16_0, tensor var_7839_cast_fp16_1 = split(axis = var_7839_axis_0, split_sizes = tile_43, x = var_7836_cast_fp16)[name = string("op_7839_cast_fp16")]; tensor var_7842_split_sizes_0 = const()[name = string("op_7842_split_sizes_0"), val = tensor([8, 8])]; int32 var_7842_axis_0 = const()[name = string("op_7842_axis_0"), val = int32(1)]; tensor var_7842_0, tensor var_7842_1 = split(axis = var_7842_axis_0, split_sizes = var_7842_split_sizes_0, x = query_states_129_cast_fp16)[name = string("op_7842")]; bool attn_weights_337_transpose_x_0 = const()[name = string("attn_weights_337_transpose_x_0"), val = bool(false)]; bool attn_weights_337_transpose_y_0 = const()[name = string("attn_weights_337_transpose_y_0"), val = bool(false)]; tensor attn_weights_337_cast_fp16 = matmul(transpose_x = attn_weights_337_transpose_x_0, transpose_y = attn_weights_337_transpose_y_0, x = var_7829_cast_fp16_0, y = var_7842_0)[name = string("attn_weights_337_cast_fp16")]; fp16 var_7845_to_fp16 = const()[name = string("op_7845_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_339_cast_fp16 = mul(x = attn_weights_337_cast_fp16, y = var_7845_to_fp16)[name = string("attn_weights_339_cast_fp16")]; tensor attn_weights_341_cast_fp16 = add(x = attn_weights_339_cast_fp16, y = attn_mask_1)[name = string("attn_weights_341_cast_fp16")]; int32 var_7849 = const()[name = string("op_7849"), val = int32(-2)]; tensor attn_weights_343_cast_fp16 = softmax(axis = var_7849, x = attn_weights_341_cast_fp16)[name = string("attn_weights_343_cast_fp16")]; bool var_7855_transpose_x_1 = const()[name = string("op_7855_transpose_x_1"), val = bool(true)]; bool var_7855_transpose_y_1 = const()[name = string("op_7855_transpose_y_1"), val = bool(false)]; tensor var_7855_cast_fp16 = matmul(transpose_x = var_7855_transpose_x_1, transpose_y = var_7855_transpose_y_1, x = attn_weights_343_cast_fp16, y = var_7839_cast_fp16_0)[name = string("op_7855_cast_fp16")]; bool attn_weights_345_transpose_x_0 = const()[name = string("attn_weights_345_transpose_x_0"), val = bool(false)]; bool attn_weights_345_transpose_y_0 = const()[name = string("attn_weights_345_transpose_y_0"), val = bool(false)]; tensor attn_weights_345_cast_fp16 = matmul(transpose_x = attn_weights_345_transpose_x_0, transpose_y = attn_weights_345_transpose_y_0, x = var_7829_cast_fp16_1, y = var_7842_1)[name = string("attn_weights_345_cast_fp16")]; fp16 var_7857_to_fp16 = const()[name = string("op_7857_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_347_cast_fp16 = mul(x = attn_weights_345_cast_fp16, y = var_7857_to_fp16)[name = string("attn_weights_347_cast_fp16")]; tensor attn_weights_349_cast_fp16 = add(x = attn_weights_347_cast_fp16, y = attn_mask_1)[name = string("attn_weights_349_cast_fp16")]; int32 var_7861 = const()[name = string("op_7861"), val = int32(-2)]; tensor attn_weights_351_cast_fp16 = softmax(axis = var_7861, x = attn_weights_349_cast_fp16)[name = string("attn_weights_351_cast_fp16")]; bool attn_output_169_transpose_x_1 = const()[name = string("attn_output_169_transpose_x_1"), val = bool(true)]; bool attn_output_169_transpose_y_1 = const()[name = string("attn_output_169_transpose_y_1"), val = bool(false)]; tensor attn_output_169_cast_fp16 = matmul(transpose_x = attn_output_169_transpose_x_1, transpose_y = attn_output_169_transpose_y_1, x = attn_weights_351_cast_fp16, y = var_7839_cast_fp16_1)[name = string("attn_output_169_cast_fp16")]; int32 var_7869 = const()[name = string("op_7869"), val = int32(1)]; bool attn_output_171_interleave_0 = const()[name = string("attn_output_171_interleave_0"), val = bool(false)]; tensor attn_output_171_cast_fp16 = concat(axis = var_7869, interleave = attn_output_171_interleave_0, values = (var_7855_cast_fp16, attn_output_169_cast_fp16))[name = string("attn_output_171_cast_fp16")]; tensor var_7873_perm_0 = const()[name = string("op_7873_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_263x = const()[name = string("concat_263x"), val = tensor([1, 2048, 1, -1])]; tensor var_7873_cast_fp16 = transpose(perm = var_7873_perm_0, x = attn_output_171_cast_fp16)[name = string("transpose_448")]; tensor attn_output_175_cast_fp16 = reshape(shape = concat_263x, x = var_7873_cast_fp16)[name = string("attn_output_175_cast_fp16")]; tensor hidden_states_213_strides_0 = const()[name = string("hidden_states_213_strides_0"), val = tensor([1, 1])]; string hidden_states_213_pad_type_0 = const()[name = string("hidden_states_213_pad_type_0"), val = string("valid")]; tensor hidden_states_213_pad_0 = const()[name = string("hidden_states_213_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_213_dilations_0 = const()[name = string("hidden_states_213_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_213_groups_0 = const()[name = string("hidden_states_213_groups_0"), val = int32(1)]; tensor hidden_states_213_cast_fp16 = conv(dilations = hidden_states_213_dilations_0, groups = hidden_states_213_groups_0, pad = hidden_states_213_pad_0, pad_type = hidden_states_213_pad_type_0, strides = hidden_states_213_strides_0, weight = layers_21_self_attn_o_proj_weight_cast_fp16, x = attn_output_175_cast_fp16)[name = string("hidden_states_213_cast_fp16")]; tensor hidden_states_215_cast_fp16 = add(x = hidden_states_209_cast_fp16, y = hidden_states_213_cast_fp16)[name = string("hidden_states_215_cast_fp16")]; fp16 const_218_promoted_to_fp16 = const()[name = string("const_218_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7906_cast_fp16 = mul(x = hidden_states_215_cast_fp16, y = const_218_promoted_to_fp16)[name = string("op_7906_cast_fp16")]; int32 var_7904 = const()[name = string("op_7904"), val = int32(1)]; bool doubled_173_interleave_0 = const()[name = string("doubled_173_interleave_0"), val = bool(false)]; tensor doubled_173_cast_fp16 = concat(axis = var_7904, interleave = doubled_173_interleave_0, values = (hidden_states_215_cast_fp16, var_7906_cast_fp16))[name = string("doubled_173_cast_fp16")]; tensor out_87_axes_0 = const()[name = string("out_87_axes_0"), val = tensor([1])]; tensor out_87_gamma_0_to_fp16 = const()[name = string("out_87_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528778304)))]; fp16 var_7916_to_fp16 = const()[name = string("op_7916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_87_cast_fp16 = layer_norm(axes = out_87_axes_0, epsilon = var_7916_to_fp16, gamma = out_87_gamma_0_to_fp16, x = doubled_173_cast_fp16)[name = string("out_87_cast_fp16")]; tensor var_7927_split_sizes_0 = const()[name = string("op_7927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7927_axis_0 = const()[name = string("op_7927_axis_0"), val = int32(1)]; tensor var_7927_cast_fp16_0, tensor var_7927_cast_fp16_1 = split(axis = var_7927_axis_0, split_sizes = var_7927_split_sizes_0, x = out_87_cast_fp16)[name = string("op_7927_cast_fp16")]; tensor input_43_strides_0 = const()[name = string("input_43_strides_0"), val = tensor([1, 1])]; string input_43_pad_type_0 = const()[name = string("input_43_pad_type_0"), val = string("valid")]; tensor input_43_pad_0 = const()[name = string("input_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_43_dilations_0 = const()[name = string("input_43_dilations_0"), val = tensor([1, 1])]; int32 input_43_groups_0 = const()[name = string("input_43_groups_0"), val = int32(1)]; tensor input_43_cast_fp16 = conv(dilations = input_43_dilations_0, groups = input_43_groups_0, pad = input_43_pad_0, pad_type = input_43_pad_type_0, strides = input_43_strides_0, weight = layers_21_mlp_gate_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("input_43_cast_fp16")]; tensor var_7944_cast_fp16 = silu(x = input_43_cast_fp16)[name = string("op_7944_cast_fp16")]; tensor var_7950_strides_0 = const()[name = string("op_7950_strides_0"), val = tensor([1, 1])]; string var_7950_pad_type_0 = const()[name = string("op_7950_pad_type_0"), val = string("valid")]; tensor var_7950_pad_0 = const()[name = string("op_7950_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7950_dilations_0 = const()[name = string("op_7950_dilations_0"), val = tensor([1, 1])]; int32 var_7950_groups_0 = const()[name = string("op_7950_groups_0"), val = int32(1)]; tensor var_7950_cast_fp16 = conv(dilations = var_7950_dilations_0, groups = var_7950_groups_0, pad = var_7950_pad_0, pad_type = var_7950_pad_type_0, strides = var_7950_strides_0, weight = layers_21_mlp_up_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("op_7950_cast_fp16")]; tensor x_219_cast_fp16 = mul(x = var_7944_cast_fp16, y = var_7950_cast_fp16)[name = string("x_219_cast_fp16")]; tensor hidden_states_217_strides_0 = const()[name = string("hidden_states_217_strides_0"), val = tensor([1, 1])]; string hidden_states_217_pad_type_0 = const()[name = string("hidden_states_217_pad_type_0"), val = string("valid")]; tensor hidden_states_217_pad_0 = const()[name = string("hidden_states_217_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_217_dilations_0 = const()[name = string("hidden_states_217_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_217_groups_0 = const()[name = string("hidden_states_217_groups_0"), val = int32(1)]; tensor hidden_states_217_cast_fp16 = conv(dilations = hidden_states_217_dilations_0, groups = hidden_states_217_groups_0, pad = hidden_states_217_pad_0, pad_type = hidden_states_217_pad_type_0, strides = hidden_states_217_strides_0, weight = layers_21_mlp_down_proj_weight_cast_fp16, x = x_219_cast_fp16)[name = string("hidden_states_217_cast_fp16")]; tensor hidden_states_219_cast_fp16 = add(x = hidden_states_215_cast_fp16, y = hidden_states_217_cast_fp16)[name = string("hidden_states_219_cast_fp16")]; fp16 const_220_promoted_to_fp16 = const()[name = string("const_220_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7968_cast_fp16 = mul(x = hidden_states_219_cast_fp16, y = const_220_promoted_to_fp16)[name = string("op_7968_cast_fp16")]; int32 var_7966 = const()[name = string("op_7966"), val = int32(1)]; bool doubled_177_interleave_0 = const()[name = string("doubled_177_interleave_0"), val = bool(false)]; tensor doubled_177_cast_fp16 = concat(axis = var_7966, interleave = doubled_177_interleave_0, values = (hidden_states_219_cast_fp16, var_7968_cast_fp16))[name = string("doubled_177_cast_fp16")]; tensor out_89_axes_0 = const()[name = string("out_89_axes_0"), val = tensor([1])]; tensor out_89_gamma_0_to_fp16 = const()[name = string("out_89_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528786560)))]; fp16 var_7978_to_fp16 = const()[name = string("op_7978_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_89_cast_fp16 = layer_norm(axes = out_89_axes_0, epsilon = var_7978_to_fp16, gamma = out_89_gamma_0_to_fp16, x = doubled_177_cast_fp16)[name = string("out_89_cast_fp16")]; tensor var_7989_split_sizes_0 = const()[name = string("op_7989_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7989_axis_0 = const()[name = string("op_7989_axis_0"), val = int32(1)]; tensor var_7989_cast_fp16_0, tensor var_7989_cast_fp16_1 = split(axis = var_7989_axis_0, split_sizes = var_7989_split_sizes_0, x = out_89_cast_fp16)[name = string("op_7989_cast_fp16")]; tensor query_states_133_strides_0 = const()[name = string("query_states_133_strides_0"), val = tensor([1, 1])]; string query_states_133_pad_type_0 = const()[name = string("query_states_133_pad_type_0"), val = string("valid")]; tensor query_states_133_pad_0 = const()[name = string("query_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_133_dilations_0 = const()[name = string("query_states_133_dilations_0"), val = tensor([1, 1])]; int32 query_states_133_groups_0 = const()[name = string("query_states_133_groups_0"), val = int32(1)]; tensor query_states_133_cast_fp16 = conv(dilations = query_states_133_dilations_0, groups = query_states_133_groups_0, pad = query_states_133_pad_0, pad_type = query_states_133_pad_type_0, strides = query_states_133_strides_0, weight = layers_22_self_attn_q_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("query_states_133_cast_fp16")]; tensor key_states_221_strides_0 = const()[name = string("key_states_221_strides_0"), val = tensor([1, 1])]; string key_states_221_pad_type_0 = const()[name = string("key_states_221_pad_type_0"), val = string("valid")]; tensor key_states_221_pad_0 = const()[name = string("key_states_221_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_221_dilations_0 = const()[name = string("key_states_221_dilations_0"), val = tensor([1, 1])]; int32 key_states_221_groups_0 = const()[name = string("key_states_221_groups_0"), val = int32(1)]; tensor key_states_221_cast_fp16 = conv(dilations = key_states_221_dilations_0, groups = key_states_221_groups_0, pad = key_states_221_pad_0, pad_type = key_states_221_pad_type_0, strides = key_states_221_strides_0, weight = layers_22_self_attn_k_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("key_states_221_cast_fp16")]; tensor layers_22_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528794816)))]; tensor value_states_133_strides_0 = const()[name = string("value_states_133_strides_0"), val = tensor([1, 1])]; string value_states_133_pad_type_0 = const()[name = string("value_states_133_pad_type_0"), val = string("valid")]; tensor value_states_133_pad_0 = const()[name = string("value_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_133_dilations_0 = const()[name = string("value_states_133_dilations_0"), val = tensor([1, 1])]; int32 value_states_133_groups_0 = const()[name = string("value_states_133_groups_0"), val = int32(1)]; tensor value_states_133_cast_fp16 = conv(dilations = value_states_133_dilations_0, groups = value_states_133_groups_0, pad = value_states_133_pad_0, pad_type = value_states_133_pad_type_0, strides = value_states_133_strides_0, weight = layers_22_self_attn_v_proj_weight_to_fp16, x = var_7989_cast_fp16_0)[name = string("value_states_133_cast_fp16")]; tensor concat_264x = const()[name = string("concat_264x"), val = tensor([1, 16, 128, -1])]; tensor x_221_cast_fp16 = reshape(shape = concat_264x, x = query_states_133_cast_fp16)[name = string("x_221_cast_fp16")]; tensor concat_265x = const()[name = string("concat_265x"), val = tensor([1, 2, 128, -1])]; tensor var_8046_cast_fp16 = reshape(shape = concat_265x, x = key_states_221_cast_fp16)[name = string("op_8046_cast_fp16")]; tensor concat_266x = const()[name = string("concat_266x"), val = tensor([1, 2, 128, -1])]; tensor var_8053_cast_fp16 = reshape(shape = concat_266x, x = value_states_133_cast_fp16)[name = string("op_8053_cast_fp16")]; tensor var_8057_cast_fp16 = mul(x = x_221_cast_fp16, y = var_869_cast_fp16)[name = string("op_8057_cast_fp16")]; tensor var_8058_split_sizes_0 = const()[name = string("op_8058_split_sizes_0"), val = tensor([64, 64])]; int32 var_8058_axis_0 = const()[name = string("op_8058_axis_0"), val = int32(-2)]; tensor var_8058_cast_fp16_0, tensor var_8058_cast_fp16_1 = split(axis = var_8058_axis_0, split_sizes = var_8058_split_sizes_0, x = x_221_cast_fp16)[name = string("op_8058_cast_fp16")]; fp16 const_222_promoted_to_fp16 = const()[name = string("const_222_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8060_cast_fp16 = mul(x = var_8058_cast_fp16_1, y = const_222_promoted_to_fp16)[name = string("op_8060_cast_fp16")]; int32 var_8062 = const()[name = string("op_8062"), val = int32(-2)]; bool var_8063_interleave_0 = const()[name = string("op_8063_interleave_0"), val = bool(false)]; tensor var_8063_cast_fp16 = concat(axis = var_8062, interleave = var_8063_interleave_0, values = (var_8060_cast_fp16, var_8058_cast_fp16_0))[name = string("op_8063_cast_fp16")]; tensor var_8064_cast_fp16 = mul(x = var_8063_cast_fp16, y = var_878_cast_fp16)[name = string("op_8064_cast_fp16")]; tensor query_states_135_cast_fp16 = add(x = var_8057_cast_fp16, y = var_8064_cast_fp16)[name = string("query_states_135_cast_fp16")]; tensor var_8070_cast_fp16 = mul(x = var_8046_cast_fp16, y = var_869_cast_fp16)[name = string("op_8070_cast_fp16")]; tensor var_8071_split_sizes_0 = const()[name = string("op_8071_split_sizes_0"), val = tensor([64, 64])]; int32 var_8071_axis_0 = const()[name = string("op_8071_axis_0"), val = int32(-2)]; tensor var_8071_cast_fp16_0, tensor var_8071_cast_fp16_1 = split(axis = var_8071_axis_0, split_sizes = var_8071_split_sizes_0, x = var_8046_cast_fp16)[name = string("op_8071_cast_fp16")]; fp16 const_223_promoted_to_fp16 = const()[name = string("const_223_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8073_cast_fp16 = mul(x = var_8071_cast_fp16_1, y = const_223_promoted_to_fp16)[name = string("op_8073_cast_fp16")]; int32 var_8075 = const()[name = string("op_8075"), val = int32(-2)]; bool var_8076_interleave_0 = const()[name = string("op_8076_interleave_0"), val = bool(false)]; tensor var_8076_cast_fp16 = concat(axis = var_8075, interleave = var_8076_interleave_0, values = (var_8073_cast_fp16, var_8071_cast_fp16_0))[name = string("op_8076_cast_fp16")]; tensor var_8077_cast_fp16 = mul(x = var_8076_cast_fp16, y = var_878_cast_fp16)[name = string("op_8077_cast_fp16")]; tensor key_states_225_cast_fp16 = add(x = var_8070_cast_fp16, y = var_8077_cast_fp16)[name = string("key_states_225_cast_fp16")]; tensor expand_dims_264 = const()[name = string("expand_dims_264"), val = tensor([22])]; tensor expand_dims_265 = const()[name = string("expand_dims_265"), val = tensor([0])]; tensor expand_dims_267 = const()[name = string("expand_dims_267"), val = tensor([0])]; int32 concat_269_axis_0 = const()[name = string("concat_269_axis_0"), val = int32(0)]; bool concat_269_interleave_0 = const()[name = string("concat_269_interleave_0"), val = bool(false)]; tensor concat_269 = concat(axis = concat_269_axis_0, interleave = concat_269_interleave_0, values = (expand_dims_264, expand_dims_265, position_id, expand_dims_267))[name = string("concat_269")]; tensor expand_dims_268 = const()[name = string("expand_dims_268"), val = tensor([23])]; tensor concat_270_values1_0 = const()[name = string("concat_270_values1_0"), val = tensor([0])]; tensor concat_270_values3_0 = const()[name = string("concat_270_values3_0"), val = tensor([0])]; int32 concat_270_axis_0 = const()[name = string("concat_270_axis_0"), val = int32(0)]; bool concat_270_interleave_0 = const()[name = string("concat_270_interleave_0"), val = bool(false)]; tensor concat_270 = concat(axis = concat_270_axis_0, interleave = concat_270_interleave_0, values = (expand_dims_268, concat_270_values1_0, cache_position_end, concat_270_values3_0))[name = string("concat_270")]; tensor key_states_227_perm_0 = const()[name = string("key_states_227_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_23_stride_0 = const()[name = string("key_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_227_cast_fp16 = transpose(perm = key_states_227_perm_0, x = key_states_225_cast_fp16)[name = string("transpose_447")]; tensor key_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = key_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = key_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_23_squeeze_mask_0, stride = key_cache_internal_tensor_assign_23_stride_0, update = key_states_227_cast_fp16, x = coreml_update_state_322)[name = string("key_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_23_cast_fp16, input = key_cache)[name = string("coreml_update_state_324_write_state")]; tensor coreml_update_state_324 = read_state(input = key_cache)[name = string("coreml_update_state_324")]; tensor value_states_135_perm_0 = const()[name = string("value_states_135_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_23_stride_0 = const()[name = string("value_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_135_cast_fp16 = transpose(perm = value_states_135_perm_0, x = var_8053_cast_fp16)[name = string("transpose_446")]; tensor value_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = value_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = value_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_23_squeeze_mask_0, stride = value_cache_internal_tensor_assign_23_stride_0, update = value_states_135_cast_fp16, x = coreml_update_state_323)[name = string("value_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_23_cast_fp16, input = value_cache)[name = string("coreml_update_state_325_write_state")]; tensor coreml_update_state_325 = read_state(input = value_cache)[name = string("coreml_update_state_325")]; tensor var_8147_begin_0 = const()[name = string("op_8147_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8147_end_0 = const()[name = string("op_8147_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8147_end_mask_0 = const()[name = string("op_8147_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8147_cast_fp16 = slice_by_index(begin = var_8147_begin_0, end = var_8147_end_0, end_mask = var_8147_end_mask_0, x = coreml_update_state_324)[name = string("op_8147_cast_fp16")]; tensor tile_44 = const()[name = string("tile_44"), val = tensor([1, 1])]; int32 var_8150_axis_0 = const()[name = string("op_8150_axis_0"), val = int32(1)]; tensor var_8150_cast_fp16_0, tensor var_8150_cast_fp16_1 = split(axis = var_8150_axis_0, split_sizes = tile_44, x = var_8147_cast_fp16)[name = string("op_8150_cast_fp16")]; tensor var_8157_begin_0 = const()[name = string("op_8157_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8157_end_0 = const()[name = string("op_8157_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8157_end_mask_0 = const()[name = string("op_8157_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8157_cast_fp16 = slice_by_index(begin = var_8157_begin_0, end = var_8157_end_0, end_mask = var_8157_end_mask_0, x = coreml_update_state_325)[name = string("op_8157_cast_fp16")]; tensor tile_45 = const()[name = string("tile_45"), val = tensor([1, 1])]; int32 var_8160_axis_0 = const()[name = string("op_8160_axis_0"), val = int32(1)]; tensor var_8160_cast_fp16_0, tensor var_8160_cast_fp16_1 = split(axis = var_8160_axis_0, split_sizes = tile_45, x = var_8157_cast_fp16)[name = string("op_8160_cast_fp16")]; tensor var_8163_split_sizes_0 = const()[name = string("op_8163_split_sizes_0"), val = tensor([8, 8])]; int32 var_8163_axis_0 = const()[name = string("op_8163_axis_0"), val = int32(1)]; tensor var_8163_0, tensor var_8163_1 = split(axis = var_8163_axis_0, split_sizes = var_8163_split_sizes_0, x = query_states_135_cast_fp16)[name = string("op_8163")]; bool attn_weights_353_transpose_x_0 = const()[name = string("attn_weights_353_transpose_x_0"), val = bool(false)]; bool attn_weights_353_transpose_y_0 = const()[name = string("attn_weights_353_transpose_y_0"), val = bool(false)]; tensor attn_weights_353_cast_fp16 = matmul(transpose_x = attn_weights_353_transpose_x_0, transpose_y = attn_weights_353_transpose_y_0, x = var_8150_cast_fp16_0, y = var_8163_0)[name = string("attn_weights_353_cast_fp16")]; fp16 var_8166_to_fp16 = const()[name = string("op_8166_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_355_cast_fp16 = mul(x = attn_weights_353_cast_fp16, y = var_8166_to_fp16)[name = string("attn_weights_355_cast_fp16")]; tensor attn_weights_357_cast_fp16 = add(x = attn_weights_355_cast_fp16, y = attn_mask_1)[name = string("attn_weights_357_cast_fp16")]; int32 var_8170 = const()[name = string("op_8170"), val = int32(-2)]; tensor attn_weights_359_cast_fp16 = softmax(axis = var_8170, x = attn_weights_357_cast_fp16)[name = string("attn_weights_359_cast_fp16")]; bool var_8176_transpose_x_1 = const()[name = string("op_8176_transpose_x_1"), val = bool(true)]; bool var_8176_transpose_y_1 = const()[name = string("op_8176_transpose_y_1"), val = bool(false)]; tensor var_8176_cast_fp16 = matmul(transpose_x = var_8176_transpose_x_1, transpose_y = var_8176_transpose_y_1, x = attn_weights_359_cast_fp16, y = var_8160_cast_fp16_0)[name = string("op_8176_cast_fp16")]; bool attn_weights_361_transpose_x_0 = const()[name = string("attn_weights_361_transpose_x_0"), val = bool(false)]; bool attn_weights_361_transpose_y_0 = const()[name = string("attn_weights_361_transpose_y_0"), val = bool(false)]; tensor attn_weights_361_cast_fp16 = matmul(transpose_x = attn_weights_361_transpose_x_0, transpose_y = attn_weights_361_transpose_y_0, x = var_8150_cast_fp16_1, y = var_8163_1)[name = string("attn_weights_361_cast_fp16")]; fp16 var_8178_to_fp16 = const()[name = string("op_8178_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_363_cast_fp16 = mul(x = attn_weights_361_cast_fp16, y = var_8178_to_fp16)[name = string("attn_weights_363_cast_fp16")]; tensor attn_weights_365_cast_fp16 = add(x = attn_weights_363_cast_fp16, y = attn_mask_1)[name = string("attn_weights_365_cast_fp16")]; int32 var_8182 = const()[name = string("op_8182"), val = int32(-2)]; tensor attn_weights_367_cast_fp16 = softmax(axis = var_8182, x = attn_weights_365_cast_fp16)[name = string("attn_weights_367_cast_fp16")]; bool attn_output_177_transpose_x_1 = const()[name = string("attn_output_177_transpose_x_1"), val = bool(true)]; bool attn_output_177_transpose_y_1 = const()[name = string("attn_output_177_transpose_y_1"), val = bool(false)]; tensor attn_output_177_cast_fp16 = matmul(transpose_x = attn_output_177_transpose_x_1, transpose_y = attn_output_177_transpose_y_1, x = attn_weights_367_cast_fp16, y = var_8160_cast_fp16_1)[name = string("attn_output_177_cast_fp16")]; int32 var_8190 = const()[name = string("op_8190"), val = int32(1)]; bool attn_output_179_interleave_0 = const()[name = string("attn_output_179_interleave_0"), val = bool(false)]; tensor attn_output_179_cast_fp16 = concat(axis = var_8190, interleave = attn_output_179_interleave_0, values = (var_8176_cast_fp16, attn_output_177_cast_fp16))[name = string("attn_output_179_cast_fp16")]; tensor var_8194_perm_0 = const()[name = string("op_8194_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_275x = const()[name = string("concat_275x"), val = tensor([1, 2048, 1, -1])]; tensor var_8194_cast_fp16 = transpose(perm = var_8194_perm_0, x = attn_output_179_cast_fp16)[name = string("transpose_445")]; tensor attn_output_183_cast_fp16 = reshape(shape = concat_275x, x = var_8194_cast_fp16)[name = string("attn_output_183_cast_fp16")]; tensor layers_22_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1529843456)))]; tensor hidden_states_223_strides_0 = const()[name = string("hidden_states_223_strides_0"), val = tensor([1, 1])]; string hidden_states_223_pad_type_0 = const()[name = string("hidden_states_223_pad_type_0"), val = string("valid")]; tensor hidden_states_223_pad_0 = const()[name = string("hidden_states_223_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_223_dilations_0 = const()[name = string("hidden_states_223_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_223_groups_0 = const()[name = string("hidden_states_223_groups_0"), val = int32(1)]; tensor hidden_states_223_cast_fp16 = conv(dilations = hidden_states_223_dilations_0, groups = hidden_states_223_groups_0, pad = hidden_states_223_pad_0, pad_type = hidden_states_223_pad_type_0, strides = hidden_states_223_strides_0, weight = layers_22_self_attn_o_proj_weight_to_fp16, x = attn_output_183_cast_fp16)[name = string("hidden_states_223_cast_fp16")]; tensor hidden_states_225_cast_fp16 = add(x = hidden_states_219_cast_fp16, y = hidden_states_223_cast_fp16)[name = string("hidden_states_225_cast_fp16")]; fp16 const_228_promoted_to_fp16 = const()[name = string("const_228_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8227_cast_fp16 = mul(x = hidden_states_225_cast_fp16, y = const_228_promoted_to_fp16)[name = string("op_8227_cast_fp16")]; int32 var_8225 = const()[name = string("op_8225"), val = int32(1)]; bool doubled_181_interleave_0 = const()[name = string("doubled_181_interleave_0"), val = bool(false)]; tensor doubled_181_cast_fp16 = concat(axis = var_8225, interleave = doubled_181_interleave_0, values = (hidden_states_225_cast_fp16, var_8227_cast_fp16))[name = string("doubled_181_cast_fp16")]; tensor out_91_axes_0 = const()[name = string("out_91_axes_0"), val = tensor([1])]; tensor out_91_gamma_0_to_fp16 = const()[name = string("out_91_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538232128)))]; fp16 var_8237_to_fp16 = const()[name = string("op_8237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_91_cast_fp16 = layer_norm(axes = out_91_axes_0, epsilon = var_8237_to_fp16, gamma = out_91_gamma_0_to_fp16, x = doubled_181_cast_fp16)[name = string("out_91_cast_fp16")]; tensor var_8248_split_sizes_0 = const()[name = string("op_8248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8248_axis_0 = const()[name = string("op_8248_axis_0"), val = int32(1)]; tensor var_8248_cast_fp16_0, tensor var_8248_cast_fp16_1 = split(axis = var_8248_axis_0, split_sizes = var_8248_split_sizes_0, x = out_91_cast_fp16)[name = string("op_8248_cast_fp16")]; tensor input_45_strides_0 = const()[name = string("input_45_strides_0"), val = tensor([1, 1])]; string input_45_pad_type_0 = const()[name = string("input_45_pad_type_0"), val = string("valid")]; tensor input_45_pad_0 = const()[name = string("input_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_45_dilations_0 = const()[name = string("input_45_dilations_0"), val = tensor([1, 1])]; int32 input_45_groups_0 = const()[name = string("input_45_groups_0"), val = int32(1)]; tensor input_45_cast_fp16 = conv(dilations = input_45_dilations_0, groups = input_45_groups_0, pad = input_45_pad_0, pad_type = input_45_pad_type_0, strides = input_45_strides_0, weight = layers_22_mlp_gate_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("input_45_cast_fp16")]; tensor var_8265_cast_fp16 = silu(x = input_45_cast_fp16)[name = string("op_8265_cast_fp16")]; tensor var_8271_strides_0 = const()[name = string("op_8271_strides_0"), val = tensor([1, 1])]; string var_8271_pad_type_0 = const()[name = string("op_8271_pad_type_0"), val = string("valid")]; tensor var_8271_pad_0 = const()[name = string("op_8271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8271_dilations_0 = const()[name = string("op_8271_dilations_0"), val = tensor([1, 1])]; int32 var_8271_groups_0 = const()[name = string("op_8271_groups_0"), val = int32(1)]; tensor var_8271_cast_fp16 = conv(dilations = var_8271_dilations_0, groups = var_8271_groups_0, pad = var_8271_pad_0, pad_type = var_8271_pad_type_0, strides = var_8271_strides_0, weight = layers_22_mlp_up_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("op_8271_cast_fp16")]; tensor x_229_cast_fp16 = mul(x = var_8265_cast_fp16, y = var_8271_cast_fp16)[name = string("x_229_cast_fp16")]; tensor hidden_states_227_strides_0 = const()[name = string("hidden_states_227_strides_0"), val = tensor([1, 1])]; string hidden_states_227_pad_type_0 = const()[name = string("hidden_states_227_pad_type_0"), val = string("valid")]; tensor hidden_states_227_pad_0 = const()[name = string("hidden_states_227_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_227_dilations_0 = const()[name = string("hidden_states_227_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_227_groups_0 = const()[name = string("hidden_states_227_groups_0"), val = int32(1)]; tensor hidden_states_227_cast_fp16 = conv(dilations = hidden_states_227_dilations_0, groups = hidden_states_227_groups_0, pad = hidden_states_227_pad_0, pad_type = hidden_states_227_pad_type_0, strides = hidden_states_227_strides_0, weight = layers_22_mlp_down_proj_weight_cast_fp16, x = x_229_cast_fp16)[name = string("hidden_states_227_cast_fp16")]; tensor hidden_states_229_cast_fp16 = add(x = hidden_states_225_cast_fp16, y = hidden_states_227_cast_fp16)[name = string("hidden_states_229_cast_fp16")]; fp16 const_230_promoted_to_fp16 = const()[name = string("const_230_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8289_cast_fp16 = mul(x = hidden_states_229_cast_fp16, y = const_230_promoted_to_fp16)[name = string("op_8289_cast_fp16")]; int32 var_8287 = const()[name = string("op_8287"), val = int32(1)]; bool doubled_185_interleave_0 = const()[name = string("doubled_185_interleave_0"), val = bool(false)]; tensor doubled_185_cast_fp16 = concat(axis = var_8287, interleave = doubled_185_interleave_0, values = (hidden_states_229_cast_fp16, var_8289_cast_fp16))[name = string("doubled_185_cast_fp16")]; tensor out_93_axes_0 = const()[name = string("out_93_axes_0"), val = tensor([1])]; tensor out_93_gamma_0_to_fp16 = const()[name = string("out_93_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538240384)))]; fp16 var_8299_to_fp16 = const()[name = string("op_8299_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_93_cast_fp16 = layer_norm(axes = out_93_axes_0, epsilon = var_8299_to_fp16, gamma = out_93_gamma_0_to_fp16, x = doubled_185_cast_fp16)[name = string("out_93_cast_fp16")]; tensor var_8310_split_sizes_0 = const()[name = string("op_8310_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8310_axis_0 = const()[name = string("op_8310_axis_0"), val = int32(1)]; tensor var_8310_cast_fp16_0, tensor var_8310_cast_fp16_1 = split(axis = var_8310_axis_0, split_sizes = var_8310_split_sizes_0, x = out_93_cast_fp16)[name = string("op_8310_cast_fp16")]; tensor query_states_139_strides_0 = const()[name = string("query_states_139_strides_0"), val = tensor([1, 1])]; string query_states_139_pad_type_0 = const()[name = string("query_states_139_pad_type_0"), val = string("valid")]; tensor query_states_139_pad_0 = const()[name = string("query_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_139_dilations_0 = const()[name = string("query_states_139_dilations_0"), val = tensor([1, 1])]; int32 query_states_139_groups_0 = const()[name = string("query_states_139_groups_0"), val = int32(1)]; tensor query_states_139_cast_fp16 = conv(dilations = query_states_139_dilations_0, groups = query_states_139_groups_0, pad = query_states_139_pad_0, pad_type = query_states_139_pad_type_0, strides = query_states_139_strides_0, weight = layers_23_self_attn_q_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("query_states_139_cast_fp16")]; tensor key_states_231_strides_0 = const()[name = string("key_states_231_strides_0"), val = tensor([1, 1])]; string key_states_231_pad_type_0 = const()[name = string("key_states_231_pad_type_0"), val = string("valid")]; tensor key_states_231_pad_0 = const()[name = string("key_states_231_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_231_dilations_0 = const()[name = string("key_states_231_dilations_0"), val = tensor([1, 1])]; int32 key_states_231_groups_0 = const()[name = string("key_states_231_groups_0"), val = int32(1)]; tensor key_states_231_cast_fp16 = conv(dilations = key_states_231_dilations_0, groups = key_states_231_groups_0, pad = key_states_231_pad_0, pad_type = key_states_231_pad_type_0, strides = key_states_231_strides_0, weight = layers_23_self_attn_k_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("key_states_231_cast_fp16")]; tensor layers_23_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_23_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538248640)))]; tensor value_states_139_strides_0 = const()[name = string("value_states_139_strides_0"), val = tensor([1, 1])]; string value_states_139_pad_type_0 = const()[name = string("value_states_139_pad_type_0"), val = string("valid")]; tensor value_states_139_pad_0 = const()[name = string("value_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_139_dilations_0 = const()[name = string("value_states_139_dilations_0"), val = tensor([1, 1])]; int32 value_states_139_groups_0 = const()[name = string("value_states_139_groups_0"), val = int32(1)]; tensor value_states_139_cast_fp16 = conv(dilations = value_states_139_dilations_0, groups = value_states_139_groups_0, pad = value_states_139_pad_0, pad_type = value_states_139_pad_type_0, strides = value_states_139_strides_0, weight = layers_23_self_attn_v_proj_weight_to_fp16, x = var_8310_cast_fp16_0)[name = string("value_states_139_cast_fp16")]; tensor concat_276x = const()[name = string("concat_276x"), val = tensor([1, 16, 128, -1])]; tensor x_231_cast_fp16 = reshape(shape = concat_276x, x = query_states_139_cast_fp16)[name = string("x_231_cast_fp16")]; tensor concat_277x = const()[name = string("concat_277x"), val = tensor([1, 2, 128, -1])]; tensor var_8367_cast_fp16 = reshape(shape = concat_277x, x = key_states_231_cast_fp16)[name = string("op_8367_cast_fp16")]; tensor concat_278x = const()[name = string("concat_278x"), val = tensor([1, 2, 128, -1])]; tensor var_8374_cast_fp16 = reshape(shape = concat_278x, x = value_states_139_cast_fp16)[name = string("op_8374_cast_fp16")]; tensor var_8378_cast_fp16 = mul(x = x_231_cast_fp16, y = var_869_cast_fp16)[name = string("op_8378_cast_fp16")]; tensor var_8379_split_sizes_0 = const()[name = string("op_8379_split_sizes_0"), val = tensor([64, 64])]; int32 var_8379_axis_0 = const()[name = string("op_8379_axis_0"), val = int32(-2)]; tensor var_8379_cast_fp16_0, tensor var_8379_cast_fp16_1 = split(axis = var_8379_axis_0, split_sizes = var_8379_split_sizes_0, x = x_231_cast_fp16)[name = string("op_8379_cast_fp16")]; fp16 const_232_promoted_to_fp16 = const()[name = string("const_232_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8381_cast_fp16 = mul(x = var_8379_cast_fp16_1, y = const_232_promoted_to_fp16)[name = string("op_8381_cast_fp16")]; int32 var_8383 = const()[name = string("op_8383"), val = int32(-2)]; bool var_8384_interleave_0 = const()[name = string("op_8384_interleave_0"), val = bool(false)]; tensor var_8384_cast_fp16 = concat(axis = var_8383, interleave = var_8384_interleave_0, values = (var_8381_cast_fp16, var_8379_cast_fp16_0))[name = string("op_8384_cast_fp16")]; tensor var_8385_cast_fp16 = mul(x = var_8384_cast_fp16, y = var_878_cast_fp16)[name = string("op_8385_cast_fp16")]; tensor query_states_141_cast_fp16 = add(x = var_8378_cast_fp16, y = var_8385_cast_fp16)[name = string("query_states_141_cast_fp16")]; tensor var_8391_cast_fp16 = mul(x = var_8367_cast_fp16, y = var_869_cast_fp16)[name = string("op_8391_cast_fp16")]; tensor var_8392_split_sizes_0 = const()[name = string("op_8392_split_sizes_0"), val = tensor([64, 64])]; int32 var_8392_axis_0 = const()[name = string("op_8392_axis_0"), val = int32(-2)]; tensor var_8392_cast_fp16_0, tensor var_8392_cast_fp16_1 = split(axis = var_8392_axis_0, split_sizes = var_8392_split_sizes_0, x = var_8367_cast_fp16)[name = string("op_8392_cast_fp16")]; fp16 const_233_promoted_to_fp16 = const()[name = string("const_233_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8394_cast_fp16 = mul(x = var_8392_cast_fp16_1, y = const_233_promoted_to_fp16)[name = string("op_8394_cast_fp16")]; int32 var_8396 = const()[name = string("op_8396"), val = int32(-2)]; bool var_8397_interleave_0 = const()[name = string("op_8397_interleave_0"), val = bool(false)]; tensor var_8397_cast_fp16 = concat(axis = var_8396, interleave = var_8397_interleave_0, values = (var_8394_cast_fp16, var_8392_cast_fp16_0))[name = string("op_8397_cast_fp16")]; tensor var_8398_cast_fp16 = mul(x = var_8397_cast_fp16, y = var_878_cast_fp16)[name = string("op_8398_cast_fp16")]; tensor key_states_235_cast_fp16 = add(x = var_8391_cast_fp16, y = var_8398_cast_fp16)[name = string("key_states_235_cast_fp16")]; tensor expand_dims_276 = const()[name = string("expand_dims_276"), val = tensor([23])]; tensor expand_dims_277 = const()[name = string("expand_dims_277"), val = tensor([0])]; tensor expand_dims_279 = const()[name = string("expand_dims_279"), val = tensor([0])]; int32 concat_281_axis_0 = const()[name = string("concat_281_axis_0"), val = int32(0)]; bool concat_281_interleave_0 = const()[name = string("concat_281_interleave_0"), val = bool(false)]; tensor concat_281 = concat(axis = concat_281_axis_0, interleave = concat_281_interleave_0, values = (expand_dims_276, expand_dims_277, position_id, expand_dims_279))[name = string("concat_281")]; tensor expand_dims_280 = const()[name = string("expand_dims_280"), val = tensor([24])]; tensor concat_282_values1_0 = const()[name = string("concat_282_values1_0"), val = tensor([0])]; tensor concat_282_values3_0 = const()[name = string("concat_282_values3_0"), val = tensor([0])]; int32 concat_282_axis_0 = const()[name = string("concat_282_axis_0"), val = int32(0)]; bool concat_282_interleave_0 = const()[name = string("concat_282_interleave_0"), val = bool(false)]; tensor concat_282 = concat(axis = concat_282_axis_0, interleave = concat_282_interleave_0, values = (expand_dims_280, concat_282_values1_0, cache_position_end, concat_282_values3_0))[name = string("concat_282")]; tensor key_states_237_perm_0 = const()[name = string("key_states_237_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_24_stride_0 = const()[name = string("key_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_237_cast_fp16 = transpose(perm = key_states_237_perm_0, x = key_states_235_cast_fp16)[name = string("transpose_444")]; tensor key_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = key_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = key_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_24_squeeze_mask_0, stride = key_cache_internal_tensor_assign_24_stride_0, update = key_states_237_cast_fp16, x = coreml_update_state_324)[name = string("key_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_24_cast_fp16, input = key_cache)[name = string("coreml_update_state_326_write_state")]; tensor coreml_update_state_326 = read_state(input = key_cache)[name = string("coreml_update_state_326")]; tensor value_states_141_perm_0 = const()[name = string("value_states_141_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_24_stride_0 = const()[name = string("value_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_141_cast_fp16 = transpose(perm = value_states_141_perm_0, x = var_8374_cast_fp16)[name = string("transpose_443")]; tensor value_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = value_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = value_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_24_squeeze_mask_0, stride = value_cache_internal_tensor_assign_24_stride_0, update = value_states_141_cast_fp16, x = coreml_update_state_325)[name = string("value_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_24_cast_fp16, input = value_cache)[name = string("coreml_update_state_327_write_state")]; tensor coreml_update_state_327 = read_state(input = value_cache)[name = string("coreml_update_state_327")]; tensor var_8468_begin_0 = const()[name = string("op_8468_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8468_end_0 = const()[name = string("op_8468_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8468_end_mask_0 = const()[name = string("op_8468_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8468_cast_fp16 = slice_by_index(begin = var_8468_begin_0, end = var_8468_end_0, end_mask = var_8468_end_mask_0, x = coreml_update_state_326)[name = string("op_8468_cast_fp16")]; tensor tile_46 = const()[name = string("tile_46"), val = tensor([1, 1])]; int32 var_8471_axis_0 = const()[name = string("op_8471_axis_0"), val = int32(1)]; tensor var_8471_cast_fp16_0, tensor var_8471_cast_fp16_1 = split(axis = var_8471_axis_0, split_sizes = tile_46, x = var_8468_cast_fp16)[name = string("op_8471_cast_fp16")]; tensor var_8478_begin_0 = const()[name = string("op_8478_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8478_end_0 = const()[name = string("op_8478_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8478_end_mask_0 = const()[name = string("op_8478_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8478_cast_fp16 = slice_by_index(begin = var_8478_begin_0, end = var_8478_end_0, end_mask = var_8478_end_mask_0, x = coreml_update_state_327)[name = string("op_8478_cast_fp16")]; tensor tile_47 = const()[name = string("tile_47"), val = tensor([1, 1])]; int32 var_8481_axis_0 = const()[name = string("op_8481_axis_0"), val = int32(1)]; tensor var_8481_cast_fp16_0, tensor var_8481_cast_fp16_1 = split(axis = var_8481_axis_0, split_sizes = tile_47, x = var_8478_cast_fp16)[name = string("op_8481_cast_fp16")]; tensor var_8484_split_sizes_0 = const()[name = string("op_8484_split_sizes_0"), val = tensor([8, 8])]; int32 var_8484_axis_0 = const()[name = string("op_8484_axis_0"), val = int32(1)]; tensor var_8484_0, tensor var_8484_1 = split(axis = var_8484_axis_0, split_sizes = var_8484_split_sizes_0, x = query_states_141_cast_fp16)[name = string("op_8484")]; bool attn_weights_369_transpose_x_0 = const()[name = string("attn_weights_369_transpose_x_0"), val = bool(false)]; bool attn_weights_369_transpose_y_0 = const()[name = string("attn_weights_369_transpose_y_0"), val = bool(false)]; tensor attn_weights_369_cast_fp16 = matmul(transpose_x = attn_weights_369_transpose_x_0, transpose_y = attn_weights_369_transpose_y_0, x = var_8471_cast_fp16_0, y = var_8484_0)[name = string("attn_weights_369_cast_fp16")]; fp16 var_8487_to_fp16 = const()[name = string("op_8487_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_371_cast_fp16 = mul(x = attn_weights_369_cast_fp16, y = var_8487_to_fp16)[name = string("attn_weights_371_cast_fp16")]; tensor attn_weights_373_cast_fp16 = add(x = attn_weights_371_cast_fp16, y = attn_mask_1)[name = string("attn_weights_373_cast_fp16")]; int32 var_8491 = const()[name = string("op_8491"), val = int32(-2)]; tensor attn_weights_375_cast_fp16 = softmax(axis = var_8491, x = attn_weights_373_cast_fp16)[name = string("attn_weights_375_cast_fp16")]; bool var_8497_transpose_x_1 = const()[name = string("op_8497_transpose_x_1"), val = bool(true)]; bool var_8497_transpose_y_1 = const()[name = string("op_8497_transpose_y_1"), val = bool(false)]; tensor var_8497_cast_fp16 = matmul(transpose_x = var_8497_transpose_x_1, transpose_y = var_8497_transpose_y_1, x = attn_weights_375_cast_fp16, y = var_8481_cast_fp16_0)[name = string("op_8497_cast_fp16")]; bool attn_weights_377_transpose_x_0 = const()[name = string("attn_weights_377_transpose_x_0"), val = bool(false)]; bool attn_weights_377_transpose_y_0 = const()[name = string("attn_weights_377_transpose_y_0"), val = bool(false)]; tensor attn_weights_377_cast_fp16 = matmul(transpose_x = attn_weights_377_transpose_x_0, transpose_y = attn_weights_377_transpose_y_0, x = var_8471_cast_fp16_1, y = var_8484_1)[name = string("attn_weights_377_cast_fp16")]; fp16 var_8499_to_fp16 = const()[name = string("op_8499_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_379_cast_fp16 = mul(x = attn_weights_377_cast_fp16, y = var_8499_to_fp16)[name = string("attn_weights_379_cast_fp16")]; tensor attn_weights_381_cast_fp16 = add(x = attn_weights_379_cast_fp16, y = attn_mask_1)[name = string("attn_weights_381_cast_fp16")]; int32 var_8503 = const()[name = string("op_8503"), val = int32(-2)]; tensor attn_weights_383_cast_fp16 = softmax(axis = var_8503, x = attn_weights_381_cast_fp16)[name = string("attn_weights_383_cast_fp16")]; bool attn_output_185_transpose_x_1 = const()[name = string("attn_output_185_transpose_x_1"), val = bool(true)]; bool attn_output_185_transpose_y_1 = const()[name = string("attn_output_185_transpose_y_1"), val = bool(false)]; tensor attn_output_185_cast_fp16 = matmul(transpose_x = attn_output_185_transpose_x_1, transpose_y = attn_output_185_transpose_y_1, x = attn_weights_383_cast_fp16, y = var_8481_cast_fp16_1)[name = string("attn_output_185_cast_fp16")]; int32 var_8511 = const()[name = string("op_8511"), val = int32(1)]; bool attn_output_187_interleave_0 = const()[name = string("attn_output_187_interleave_0"), val = bool(false)]; tensor attn_output_187_cast_fp16 = concat(axis = var_8511, interleave = attn_output_187_interleave_0, values = (var_8497_cast_fp16, attn_output_185_cast_fp16))[name = string("attn_output_187_cast_fp16")]; tensor var_8515_perm_0 = const()[name = string("op_8515_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_287x = const()[name = string("concat_287x"), val = tensor([1, 2048, 1, -1])]; tensor var_8515_cast_fp16 = transpose(perm = var_8515_perm_0, x = attn_output_187_cast_fp16)[name = string("transpose_442")]; tensor attn_output_191_cast_fp16 = reshape(shape = concat_287x, x = var_8515_cast_fp16)[name = string("attn_output_191_cast_fp16")]; tensor hidden_states_233_strides_0 = const()[name = string("hidden_states_233_strides_0"), val = tensor([1, 1])]; string hidden_states_233_pad_type_0 = const()[name = string("hidden_states_233_pad_type_0"), val = string("valid")]; tensor hidden_states_233_pad_0 = const()[name = string("hidden_states_233_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_233_dilations_0 = const()[name = string("hidden_states_233_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_233_groups_0 = const()[name = string("hidden_states_233_groups_0"), val = int32(1)]; tensor hidden_states_233_cast_fp16 = conv(dilations = hidden_states_233_dilations_0, groups = hidden_states_233_groups_0, pad = hidden_states_233_pad_0, pad_type = hidden_states_233_pad_type_0, strides = hidden_states_233_strides_0, weight = layers_23_self_attn_o_proj_weight_cast_fp16, x = attn_output_191_cast_fp16)[name = string("hidden_states_233_cast_fp16")]; tensor hidden_states_235_cast_fp16 = add(x = hidden_states_229_cast_fp16, y = hidden_states_233_cast_fp16)[name = string("hidden_states_235_cast_fp16")]; fp16 const_238_promoted_to_fp16 = const()[name = string("const_238_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8548_cast_fp16 = mul(x = hidden_states_235_cast_fp16, y = const_238_promoted_to_fp16)[name = string("op_8548_cast_fp16")]; int32 var_8546 = const()[name = string("op_8546"), val = int32(1)]; bool doubled_189_interleave_0 = const()[name = string("doubled_189_interleave_0"), val = bool(false)]; tensor doubled_189_cast_fp16 = concat(axis = var_8546, interleave = doubled_189_interleave_0, values = (hidden_states_235_cast_fp16, var_8548_cast_fp16))[name = string("doubled_189_cast_fp16")]; tensor out_95_axes_0 = const()[name = string("out_95_axes_0"), val = tensor([1])]; tensor out_95_gamma_0_to_fp16 = const()[name = string("out_95_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539297280)))]; fp16 var_8558_to_fp16 = const()[name = string("op_8558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_95_cast_fp16 = layer_norm(axes = out_95_axes_0, epsilon = var_8558_to_fp16, gamma = out_95_gamma_0_to_fp16, x = doubled_189_cast_fp16)[name = string("out_95_cast_fp16")]; tensor var_8569_split_sizes_0 = const()[name = string("op_8569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8569_axis_0 = const()[name = string("op_8569_axis_0"), val = int32(1)]; tensor var_8569_cast_fp16_0, tensor var_8569_cast_fp16_1 = split(axis = var_8569_axis_0, split_sizes = var_8569_split_sizes_0, x = out_95_cast_fp16)[name = string("op_8569_cast_fp16")]; tensor input_47_strides_0 = const()[name = string("input_47_strides_0"), val = tensor([1, 1])]; string input_47_pad_type_0 = const()[name = string("input_47_pad_type_0"), val = string("valid")]; tensor input_47_pad_0 = const()[name = string("input_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_47_dilations_0 = const()[name = string("input_47_dilations_0"), val = tensor([1, 1])]; int32 input_47_groups_0 = const()[name = string("input_47_groups_0"), val = int32(1)]; tensor input_47_cast_fp16 = conv(dilations = input_47_dilations_0, groups = input_47_groups_0, pad = input_47_pad_0, pad_type = input_47_pad_type_0, strides = input_47_strides_0, weight = layers_23_mlp_gate_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("input_47_cast_fp16")]; tensor var_8586_cast_fp16 = silu(x = input_47_cast_fp16)[name = string("op_8586_cast_fp16")]; tensor var_8592_strides_0 = const()[name = string("op_8592_strides_0"), val = tensor([1, 1])]; string var_8592_pad_type_0 = const()[name = string("op_8592_pad_type_0"), val = string("valid")]; tensor var_8592_pad_0 = const()[name = string("op_8592_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8592_dilations_0 = const()[name = string("op_8592_dilations_0"), val = tensor([1, 1])]; int32 var_8592_groups_0 = const()[name = string("op_8592_groups_0"), val = int32(1)]; tensor var_8592_cast_fp16 = conv(dilations = var_8592_dilations_0, groups = var_8592_groups_0, pad = var_8592_pad_0, pad_type = var_8592_pad_type_0, strides = var_8592_strides_0, weight = layers_23_mlp_up_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("op_8592_cast_fp16")]; tensor x_239_cast_fp16 = mul(x = var_8586_cast_fp16, y = var_8592_cast_fp16)[name = string("x_239_cast_fp16")]; tensor hidden_states_237_strides_0 = const()[name = string("hidden_states_237_strides_0"), val = tensor([1, 1])]; string hidden_states_237_pad_type_0 = const()[name = string("hidden_states_237_pad_type_0"), val = string("valid")]; tensor hidden_states_237_pad_0 = const()[name = string("hidden_states_237_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_237_dilations_0 = const()[name = string("hidden_states_237_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_237_groups_0 = const()[name = string("hidden_states_237_groups_0"), val = int32(1)]; tensor hidden_states_237_cast_fp16 = conv(dilations = hidden_states_237_dilations_0, groups = hidden_states_237_groups_0, pad = hidden_states_237_pad_0, pad_type = hidden_states_237_pad_type_0, strides = hidden_states_237_strides_0, weight = layers_23_mlp_down_proj_weight_cast_fp16, x = x_239_cast_fp16)[name = string("hidden_states_237_cast_fp16")]; tensor hidden_states_239_cast_fp16 = add(x = hidden_states_235_cast_fp16, y = hidden_states_237_cast_fp16)[name = string("hidden_states_239_cast_fp16")]; fp16 const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8610_cast_fp16 = mul(x = hidden_states_239_cast_fp16, y = const_240_promoted_to_fp16)[name = string("op_8610_cast_fp16")]; int32 var_8608 = const()[name = string("op_8608"), val = int32(1)]; bool doubled_193_interleave_0 = const()[name = string("doubled_193_interleave_0"), val = bool(false)]; tensor doubled_193_cast_fp16 = concat(axis = var_8608, interleave = doubled_193_interleave_0, values = (hidden_states_239_cast_fp16, var_8610_cast_fp16))[name = string("doubled_193_cast_fp16")]; tensor out_97_axes_0 = const()[name = string("out_97_axes_0"), val = tensor([1])]; tensor out_97_gamma_0_to_fp16 = const()[name = string("out_97_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539305536)))]; fp16 var_8620_to_fp16 = const()[name = string("op_8620_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_97_cast_fp16 = layer_norm(axes = out_97_axes_0, epsilon = var_8620_to_fp16, gamma = out_97_gamma_0_to_fp16, x = doubled_193_cast_fp16)[name = string("out_97_cast_fp16")]; tensor var_8631_split_sizes_0 = const()[name = string("op_8631_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8631_axis_0 = const()[name = string("op_8631_axis_0"), val = int32(1)]; tensor var_8631_cast_fp16_0, tensor var_8631_cast_fp16_1 = split(axis = var_8631_axis_0, split_sizes = var_8631_split_sizes_0, x = out_97_cast_fp16)[name = string("op_8631_cast_fp16")]; tensor query_states_145_strides_0 = const()[name = string("query_states_145_strides_0"), val = tensor([1, 1])]; string query_states_145_pad_type_0 = const()[name = string("query_states_145_pad_type_0"), val = string("valid")]; tensor query_states_145_pad_0 = const()[name = string("query_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_145_dilations_0 = const()[name = string("query_states_145_dilations_0"), val = tensor([1, 1])]; int32 query_states_145_groups_0 = const()[name = string("query_states_145_groups_0"), val = int32(1)]; tensor query_states_145_cast_fp16 = conv(dilations = query_states_145_dilations_0, groups = query_states_145_groups_0, pad = query_states_145_pad_0, pad_type = query_states_145_pad_type_0, strides = query_states_145_strides_0, weight = layers_24_self_attn_q_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("query_states_145_cast_fp16")]; tensor key_states_241_strides_0 = const()[name = string("key_states_241_strides_0"), val = tensor([1, 1])]; string key_states_241_pad_type_0 = const()[name = string("key_states_241_pad_type_0"), val = string("valid")]; tensor key_states_241_pad_0 = const()[name = string("key_states_241_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_241_dilations_0 = const()[name = string("key_states_241_dilations_0"), val = tensor([1, 1])]; int32 key_states_241_groups_0 = const()[name = string("key_states_241_groups_0"), val = int32(1)]; tensor key_states_241_cast_fp16 = conv(dilations = key_states_241_dilations_0, groups = key_states_241_groups_0, pad = key_states_241_pad_0, pad_type = key_states_241_pad_type_0, strides = key_states_241_strides_0, weight = layers_24_self_attn_k_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("key_states_241_cast_fp16")]; tensor layers_24_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_24_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539313792)))]; tensor value_states_145_strides_0 = const()[name = string("value_states_145_strides_0"), val = tensor([1, 1])]; string value_states_145_pad_type_0 = const()[name = string("value_states_145_pad_type_0"), val = string("valid")]; tensor value_states_145_pad_0 = const()[name = string("value_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_145_dilations_0 = const()[name = string("value_states_145_dilations_0"), val = tensor([1, 1])]; int32 value_states_145_groups_0 = const()[name = string("value_states_145_groups_0"), val = int32(1)]; tensor value_states_145_cast_fp16 = conv(dilations = value_states_145_dilations_0, groups = value_states_145_groups_0, pad = value_states_145_pad_0, pad_type = value_states_145_pad_type_0, strides = value_states_145_strides_0, weight = layers_24_self_attn_v_proj_weight_to_fp16, x = var_8631_cast_fp16_0)[name = string("value_states_145_cast_fp16")]; tensor concat_288x = const()[name = string("concat_288x"), val = tensor([1, 16, 128, -1])]; tensor x_241_cast_fp16 = reshape(shape = concat_288x, x = query_states_145_cast_fp16)[name = string("x_241_cast_fp16")]; tensor concat_289x = const()[name = string("concat_289x"), val = tensor([1, 2, 128, -1])]; tensor var_8688_cast_fp16 = reshape(shape = concat_289x, x = key_states_241_cast_fp16)[name = string("op_8688_cast_fp16")]; tensor concat_290x = const()[name = string("concat_290x"), val = tensor([1, 2, 128, -1])]; tensor var_8695_cast_fp16 = reshape(shape = concat_290x, x = value_states_145_cast_fp16)[name = string("op_8695_cast_fp16")]; tensor var_8699_cast_fp16 = mul(x = x_241_cast_fp16, y = var_869_cast_fp16)[name = string("op_8699_cast_fp16")]; tensor var_8700_split_sizes_0 = const()[name = string("op_8700_split_sizes_0"), val = tensor([64, 64])]; int32 var_8700_axis_0 = const()[name = string("op_8700_axis_0"), val = int32(-2)]; tensor var_8700_cast_fp16_0, tensor var_8700_cast_fp16_1 = split(axis = var_8700_axis_0, split_sizes = var_8700_split_sizes_0, x = x_241_cast_fp16)[name = string("op_8700_cast_fp16")]; fp16 const_242_promoted_to_fp16 = const()[name = string("const_242_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8702_cast_fp16 = mul(x = var_8700_cast_fp16_1, y = const_242_promoted_to_fp16)[name = string("op_8702_cast_fp16")]; int32 var_8704 = const()[name = string("op_8704"), val = int32(-2)]; bool var_8705_interleave_0 = const()[name = string("op_8705_interleave_0"), val = bool(false)]; tensor var_8705_cast_fp16 = concat(axis = var_8704, interleave = var_8705_interleave_0, values = (var_8702_cast_fp16, var_8700_cast_fp16_0))[name = string("op_8705_cast_fp16")]; tensor var_8706_cast_fp16 = mul(x = var_8705_cast_fp16, y = var_878_cast_fp16)[name = string("op_8706_cast_fp16")]; tensor query_states_147_cast_fp16 = add(x = var_8699_cast_fp16, y = var_8706_cast_fp16)[name = string("query_states_147_cast_fp16")]; tensor var_8712_cast_fp16 = mul(x = var_8688_cast_fp16, y = var_869_cast_fp16)[name = string("op_8712_cast_fp16")]; tensor var_8713_split_sizes_0 = const()[name = string("op_8713_split_sizes_0"), val = tensor([64, 64])]; int32 var_8713_axis_0 = const()[name = string("op_8713_axis_0"), val = int32(-2)]; tensor var_8713_cast_fp16_0, tensor var_8713_cast_fp16_1 = split(axis = var_8713_axis_0, split_sizes = var_8713_split_sizes_0, x = var_8688_cast_fp16)[name = string("op_8713_cast_fp16")]; fp16 const_243_promoted_to_fp16 = const()[name = string("const_243_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8715_cast_fp16 = mul(x = var_8713_cast_fp16_1, y = const_243_promoted_to_fp16)[name = string("op_8715_cast_fp16")]; int32 var_8717 = const()[name = string("op_8717"), val = int32(-2)]; bool var_8718_interleave_0 = const()[name = string("op_8718_interleave_0"), val = bool(false)]; tensor var_8718_cast_fp16 = concat(axis = var_8717, interleave = var_8718_interleave_0, values = (var_8715_cast_fp16, var_8713_cast_fp16_0))[name = string("op_8718_cast_fp16")]; tensor var_8719_cast_fp16 = mul(x = var_8718_cast_fp16, y = var_878_cast_fp16)[name = string("op_8719_cast_fp16")]; tensor key_states_245_cast_fp16 = add(x = var_8712_cast_fp16, y = var_8719_cast_fp16)[name = string("key_states_245_cast_fp16")]; tensor expand_dims_288 = const()[name = string("expand_dims_288"), val = tensor([24])]; tensor expand_dims_289 = const()[name = string("expand_dims_289"), val = tensor([0])]; tensor expand_dims_291 = const()[name = string("expand_dims_291"), val = tensor([0])]; int32 concat_293_axis_0 = const()[name = string("concat_293_axis_0"), val = int32(0)]; bool concat_293_interleave_0 = const()[name = string("concat_293_interleave_0"), val = bool(false)]; tensor concat_293 = concat(axis = concat_293_axis_0, interleave = concat_293_interleave_0, values = (expand_dims_288, expand_dims_289, position_id, expand_dims_291))[name = string("concat_293")]; tensor expand_dims_292 = const()[name = string("expand_dims_292"), val = tensor([25])]; tensor concat_294_values1_0 = const()[name = string("concat_294_values1_0"), val = tensor([0])]; tensor concat_294_values3_0 = const()[name = string("concat_294_values3_0"), val = tensor([0])]; int32 concat_294_axis_0 = const()[name = string("concat_294_axis_0"), val = int32(0)]; bool concat_294_interleave_0 = const()[name = string("concat_294_interleave_0"), val = bool(false)]; tensor concat_294 = concat(axis = concat_294_axis_0, interleave = concat_294_interleave_0, values = (expand_dims_292, concat_294_values1_0, cache_position_end, concat_294_values3_0))[name = string("concat_294")]; tensor key_states_247_perm_0 = const()[name = string("key_states_247_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_25_stride_0 = const()[name = string("key_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_247_cast_fp16 = transpose(perm = key_states_247_perm_0, x = key_states_245_cast_fp16)[name = string("transpose_441")]; tensor key_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = key_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = key_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_25_squeeze_mask_0, stride = key_cache_internal_tensor_assign_25_stride_0, update = key_states_247_cast_fp16, x = coreml_update_state_326)[name = string("key_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_25_cast_fp16, input = key_cache)[name = string("coreml_update_state_328_write_state")]; tensor coreml_update_state_328 = read_state(input = key_cache)[name = string("coreml_update_state_328")]; tensor value_states_147_perm_0 = const()[name = string("value_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_25_stride_0 = const()[name = string("value_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_147_cast_fp16 = transpose(perm = value_states_147_perm_0, x = var_8695_cast_fp16)[name = string("transpose_440")]; tensor value_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = value_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = value_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_25_squeeze_mask_0, stride = value_cache_internal_tensor_assign_25_stride_0, update = value_states_147_cast_fp16, x = coreml_update_state_327)[name = string("value_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_25_cast_fp16, input = value_cache)[name = string("coreml_update_state_329_write_state")]; tensor coreml_update_state_329 = read_state(input = value_cache)[name = string("coreml_update_state_329")]; tensor var_8789_begin_0 = const()[name = string("op_8789_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8789_end_0 = const()[name = string("op_8789_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8789_end_mask_0 = const()[name = string("op_8789_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8789_cast_fp16 = slice_by_index(begin = var_8789_begin_0, end = var_8789_end_0, end_mask = var_8789_end_mask_0, x = coreml_update_state_328)[name = string("op_8789_cast_fp16")]; tensor tile_48 = const()[name = string("tile_48"), val = tensor([1, 1])]; int32 var_8792_axis_0 = const()[name = string("op_8792_axis_0"), val = int32(1)]; tensor var_8792_cast_fp16_0, tensor var_8792_cast_fp16_1 = split(axis = var_8792_axis_0, split_sizes = tile_48, x = var_8789_cast_fp16)[name = string("op_8792_cast_fp16")]; tensor var_8799_begin_0 = const()[name = string("op_8799_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8799_end_0 = const()[name = string("op_8799_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8799_end_mask_0 = const()[name = string("op_8799_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8799_cast_fp16 = slice_by_index(begin = var_8799_begin_0, end = var_8799_end_0, end_mask = var_8799_end_mask_0, x = coreml_update_state_329)[name = string("op_8799_cast_fp16")]; tensor tile_49 = const()[name = string("tile_49"), val = tensor([1, 1])]; int32 var_8802_axis_0 = const()[name = string("op_8802_axis_0"), val = int32(1)]; tensor var_8802_cast_fp16_0, tensor var_8802_cast_fp16_1 = split(axis = var_8802_axis_0, split_sizes = tile_49, x = var_8799_cast_fp16)[name = string("op_8802_cast_fp16")]; tensor var_8805_split_sizes_0 = const()[name = string("op_8805_split_sizes_0"), val = tensor([8, 8])]; int32 var_8805_axis_0 = const()[name = string("op_8805_axis_0"), val = int32(1)]; tensor var_8805_0, tensor var_8805_1 = split(axis = var_8805_axis_0, split_sizes = var_8805_split_sizes_0, x = query_states_147_cast_fp16)[name = string("op_8805")]; bool attn_weights_385_transpose_x_0 = const()[name = string("attn_weights_385_transpose_x_0"), val = bool(false)]; bool attn_weights_385_transpose_y_0 = const()[name = string("attn_weights_385_transpose_y_0"), val = bool(false)]; tensor attn_weights_385_cast_fp16 = matmul(transpose_x = attn_weights_385_transpose_x_0, transpose_y = attn_weights_385_transpose_y_0, x = var_8792_cast_fp16_0, y = var_8805_0)[name = string("attn_weights_385_cast_fp16")]; fp16 var_8808_to_fp16 = const()[name = string("op_8808_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_387_cast_fp16 = mul(x = attn_weights_385_cast_fp16, y = var_8808_to_fp16)[name = string("attn_weights_387_cast_fp16")]; tensor attn_weights_389_cast_fp16 = add(x = attn_weights_387_cast_fp16, y = attn_mask_1)[name = string("attn_weights_389_cast_fp16")]; int32 var_8812 = const()[name = string("op_8812"), val = int32(-2)]; tensor attn_weights_391_cast_fp16 = softmax(axis = var_8812, x = attn_weights_389_cast_fp16)[name = string("attn_weights_391_cast_fp16")]; bool var_8818_transpose_x_1 = const()[name = string("op_8818_transpose_x_1"), val = bool(true)]; bool var_8818_transpose_y_1 = const()[name = string("op_8818_transpose_y_1"), val = bool(false)]; tensor var_8818_cast_fp16 = matmul(transpose_x = var_8818_transpose_x_1, transpose_y = var_8818_transpose_y_1, x = attn_weights_391_cast_fp16, y = var_8802_cast_fp16_0)[name = string("op_8818_cast_fp16")]; bool attn_weights_393_transpose_x_0 = const()[name = string("attn_weights_393_transpose_x_0"), val = bool(false)]; bool attn_weights_393_transpose_y_0 = const()[name = string("attn_weights_393_transpose_y_0"), val = bool(false)]; tensor attn_weights_393_cast_fp16 = matmul(transpose_x = attn_weights_393_transpose_x_0, transpose_y = attn_weights_393_transpose_y_0, x = var_8792_cast_fp16_1, y = var_8805_1)[name = string("attn_weights_393_cast_fp16")]; fp16 var_8820_to_fp16 = const()[name = string("op_8820_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_395_cast_fp16 = mul(x = attn_weights_393_cast_fp16, y = var_8820_to_fp16)[name = string("attn_weights_395_cast_fp16")]; tensor attn_weights_397_cast_fp16 = add(x = attn_weights_395_cast_fp16, y = attn_mask_1)[name = string("attn_weights_397_cast_fp16")]; int32 var_8824 = const()[name = string("op_8824"), val = int32(-2)]; tensor attn_weights_399_cast_fp16 = softmax(axis = var_8824, x = attn_weights_397_cast_fp16)[name = string("attn_weights_399_cast_fp16")]; bool attn_output_193_transpose_x_1 = const()[name = string("attn_output_193_transpose_x_1"), val = bool(true)]; bool attn_output_193_transpose_y_1 = const()[name = string("attn_output_193_transpose_y_1"), val = bool(false)]; tensor attn_output_193_cast_fp16 = matmul(transpose_x = attn_output_193_transpose_x_1, transpose_y = attn_output_193_transpose_y_1, x = attn_weights_399_cast_fp16, y = var_8802_cast_fp16_1)[name = string("attn_output_193_cast_fp16")]; int32 var_8832 = const()[name = string("op_8832"), val = int32(1)]; bool attn_output_195_interleave_0 = const()[name = string("attn_output_195_interleave_0"), val = bool(false)]; tensor attn_output_195_cast_fp16 = concat(axis = var_8832, interleave = attn_output_195_interleave_0, values = (var_8818_cast_fp16, attn_output_193_cast_fp16))[name = string("attn_output_195_cast_fp16")]; tensor var_8836_perm_0 = const()[name = string("op_8836_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_299x = const()[name = string("concat_299x"), val = tensor([1, 2048, 1, -1])]; tensor var_8836_cast_fp16 = transpose(perm = var_8836_perm_0, x = attn_output_195_cast_fp16)[name = string("transpose_439")]; tensor attn_output_199_cast_fp16 = reshape(shape = concat_299x, x = var_8836_cast_fp16)[name = string("attn_output_199_cast_fp16")]; tensor hidden_states_243_strides_0 = const()[name = string("hidden_states_243_strides_0"), val = tensor([1, 1])]; string hidden_states_243_pad_type_0 = const()[name = string("hidden_states_243_pad_type_0"), val = string("valid")]; tensor hidden_states_243_pad_0 = const()[name = string("hidden_states_243_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_243_dilations_0 = const()[name = string("hidden_states_243_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_243_groups_0 = const()[name = string("hidden_states_243_groups_0"), val = int32(1)]; tensor hidden_states_243_cast_fp16 = conv(dilations = hidden_states_243_dilations_0, groups = hidden_states_243_groups_0, pad = hidden_states_243_pad_0, pad_type = hidden_states_243_pad_type_0, strides = hidden_states_243_strides_0, weight = layers_24_self_attn_o_proj_weight_cast_fp16, x = attn_output_199_cast_fp16)[name = string("hidden_states_243_cast_fp16")]; tensor hidden_states_245_cast_fp16 = add(x = hidden_states_239_cast_fp16, y = hidden_states_243_cast_fp16)[name = string("hidden_states_245_cast_fp16")]; fp16 const_248_promoted_to_fp16 = const()[name = string("const_248_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8869_cast_fp16 = mul(x = hidden_states_245_cast_fp16, y = const_248_promoted_to_fp16)[name = string("op_8869_cast_fp16")]; int32 var_8867 = const()[name = string("op_8867"), val = int32(1)]; bool doubled_197_interleave_0 = const()[name = string("doubled_197_interleave_0"), val = bool(false)]; tensor doubled_197_cast_fp16 = concat(axis = var_8867, interleave = doubled_197_interleave_0, values = (hidden_states_245_cast_fp16, var_8869_cast_fp16))[name = string("doubled_197_cast_fp16")]; tensor out_99_axes_0 = const()[name = string("out_99_axes_0"), val = tensor([1])]; tensor out_99_gamma_0_to_fp16 = const()[name = string("out_99_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540362432)))]; fp16 var_8879_to_fp16 = const()[name = string("op_8879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_99_cast_fp16 = layer_norm(axes = out_99_axes_0, epsilon = var_8879_to_fp16, gamma = out_99_gamma_0_to_fp16, x = doubled_197_cast_fp16)[name = string("out_99_cast_fp16")]; tensor var_8890_split_sizes_0 = const()[name = string("op_8890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8890_axis_0 = const()[name = string("op_8890_axis_0"), val = int32(1)]; tensor var_8890_cast_fp16_0, tensor var_8890_cast_fp16_1 = split(axis = var_8890_axis_0, split_sizes = var_8890_split_sizes_0, x = out_99_cast_fp16)[name = string("op_8890_cast_fp16")]; tensor input_49_strides_0 = const()[name = string("input_49_strides_0"), val = tensor([1, 1])]; string input_49_pad_type_0 = const()[name = string("input_49_pad_type_0"), val = string("valid")]; tensor input_49_pad_0 = const()[name = string("input_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_49_dilations_0 = const()[name = string("input_49_dilations_0"), val = tensor([1, 1])]; int32 input_49_groups_0 = const()[name = string("input_49_groups_0"), val = int32(1)]; tensor input_49_cast_fp16 = conv(dilations = input_49_dilations_0, groups = input_49_groups_0, pad = input_49_pad_0, pad_type = input_49_pad_type_0, strides = input_49_strides_0, weight = layers_24_mlp_gate_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("input_49_cast_fp16")]; tensor var_8907_cast_fp16 = silu(x = input_49_cast_fp16)[name = string("op_8907_cast_fp16")]; tensor var_8913_strides_0 = const()[name = string("op_8913_strides_0"), val = tensor([1, 1])]; string var_8913_pad_type_0 = const()[name = string("op_8913_pad_type_0"), val = string("valid")]; tensor var_8913_pad_0 = const()[name = string("op_8913_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8913_dilations_0 = const()[name = string("op_8913_dilations_0"), val = tensor([1, 1])]; int32 var_8913_groups_0 = const()[name = string("op_8913_groups_0"), val = int32(1)]; tensor var_8913_cast_fp16 = conv(dilations = var_8913_dilations_0, groups = var_8913_groups_0, pad = var_8913_pad_0, pad_type = var_8913_pad_type_0, strides = var_8913_strides_0, weight = layers_24_mlp_up_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("op_8913_cast_fp16")]; tensor x_249_cast_fp16 = mul(x = var_8907_cast_fp16, y = var_8913_cast_fp16)[name = string("x_249_cast_fp16")]; tensor hidden_states_247_strides_0 = const()[name = string("hidden_states_247_strides_0"), val = tensor([1, 1])]; string hidden_states_247_pad_type_0 = const()[name = string("hidden_states_247_pad_type_0"), val = string("valid")]; tensor hidden_states_247_pad_0 = const()[name = string("hidden_states_247_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_247_dilations_0 = const()[name = string("hidden_states_247_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_247_groups_0 = const()[name = string("hidden_states_247_groups_0"), val = int32(1)]; tensor hidden_states_247_cast_fp16 = conv(dilations = hidden_states_247_dilations_0, groups = hidden_states_247_groups_0, pad = hidden_states_247_pad_0, pad_type = hidden_states_247_pad_type_0, strides = hidden_states_247_strides_0, weight = layers_24_mlp_down_proj_weight_cast_fp16, x = x_249_cast_fp16)[name = string("hidden_states_247_cast_fp16")]; tensor hidden_states_249_cast_fp16 = add(x = hidden_states_245_cast_fp16, y = hidden_states_247_cast_fp16)[name = string("hidden_states_249_cast_fp16")]; fp16 const_250_promoted_to_fp16 = const()[name = string("const_250_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8931_cast_fp16 = mul(x = hidden_states_249_cast_fp16, y = const_250_promoted_to_fp16)[name = string("op_8931_cast_fp16")]; int32 var_8929 = const()[name = string("op_8929"), val = int32(1)]; bool doubled_201_interleave_0 = const()[name = string("doubled_201_interleave_0"), val = bool(false)]; tensor doubled_201_cast_fp16 = concat(axis = var_8929, interleave = doubled_201_interleave_0, values = (hidden_states_249_cast_fp16, var_8931_cast_fp16))[name = string("doubled_201_cast_fp16")]; tensor out_101_axes_0 = const()[name = string("out_101_axes_0"), val = tensor([1])]; tensor out_101_gamma_0_to_fp16 = const()[name = string("out_101_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540370688)))]; fp16 var_8941_to_fp16 = const()[name = string("op_8941_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_101_cast_fp16 = layer_norm(axes = out_101_axes_0, epsilon = var_8941_to_fp16, gamma = out_101_gamma_0_to_fp16, x = doubled_201_cast_fp16)[name = string("out_101_cast_fp16")]; tensor var_8952_split_sizes_0 = const()[name = string("op_8952_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8952_axis_0 = const()[name = string("op_8952_axis_0"), val = int32(1)]; tensor var_8952_cast_fp16_0, tensor var_8952_cast_fp16_1 = split(axis = var_8952_axis_0, split_sizes = var_8952_split_sizes_0, x = out_101_cast_fp16)[name = string("op_8952_cast_fp16")]; tensor query_states_151_strides_0 = const()[name = string("query_states_151_strides_0"), val = tensor([1, 1])]; string query_states_151_pad_type_0 = const()[name = string("query_states_151_pad_type_0"), val = string("valid")]; tensor query_states_151_pad_0 = const()[name = string("query_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_151_dilations_0 = const()[name = string("query_states_151_dilations_0"), val = tensor([1, 1])]; int32 query_states_151_groups_0 = const()[name = string("query_states_151_groups_0"), val = int32(1)]; tensor query_states_151_cast_fp16 = conv(dilations = query_states_151_dilations_0, groups = query_states_151_groups_0, pad = query_states_151_pad_0, pad_type = query_states_151_pad_type_0, strides = query_states_151_strides_0, weight = layers_25_self_attn_q_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("query_states_151_cast_fp16")]; tensor key_states_251_strides_0 = const()[name = string("key_states_251_strides_0"), val = tensor([1, 1])]; string key_states_251_pad_type_0 = const()[name = string("key_states_251_pad_type_0"), val = string("valid")]; tensor key_states_251_pad_0 = const()[name = string("key_states_251_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_251_dilations_0 = const()[name = string("key_states_251_dilations_0"), val = tensor([1, 1])]; int32 key_states_251_groups_0 = const()[name = string("key_states_251_groups_0"), val = int32(1)]; tensor key_states_251_cast_fp16 = conv(dilations = key_states_251_dilations_0, groups = key_states_251_groups_0, pad = key_states_251_pad_0, pad_type = key_states_251_pad_type_0, strides = key_states_251_strides_0, weight = layers_25_self_attn_k_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("key_states_251_cast_fp16")]; tensor layers_25_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_25_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540378944)))]; tensor value_states_151_strides_0 = const()[name = string("value_states_151_strides_0"), val = tensor([1, 1])]; string value_states_151_pad_type_0 = const()[name = string("value_states_151_pad_type_0"), val = string("valid")]; tensor value_states_151_pad_0 = const()[name = string("value_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_151_dilations_0 = const()[name = string("value_states_151_dilations_0"), val = tensor([1, 1])]; int32 value_states_151_groups_0 = const()[name = string("value_states_151_groups_0"), val = int32(1)]; tensor value_states_151_cast_fp16 = conv(dilations = value_states_151_dilations_0, groups = value_states_151_groups_0, pad = value_states_151_pad_0, pad_type = value_states_151_pad_type_0, strides = value_states_151_strides_0, weight = layers_25_self_attn_v_proj_weight_to_fp16, x = var_8952_cast_fp16_0)[name = string("value_states_151_cast_fp16")]; tensor concat_300x = const()[name = string("concat_300x"), val = tensor([1, 16, 128, -1])]; tensor x_251_cast_fp16 = reshape(shape = concat_300x, x = query_states_151_cast_fp16)[name = string("x_251_cast_fp16")]; tensor concat_301x = const()[name = string("concat_301x"), val = tensor([1, 2, 128, -1])]; tensor var_9009_cast_fp16 = reshape(shape = concat_301x, x = key_states_251_cast_fp16)[name = string("op_9009_cast_fp16")]; tensor concat_302x = const()[name = string("concat_302x"), val = tensor([1, 2, 128, -1])]; tensor var_9016_cast_fp16 = reshape(shape = concat_302x, x = value_states_151_cast_fp16)[name = string("op_9016_cast_fp16")]; tensor var_9020_cast_fp16 = mul(x = x_251_cast_fp16, y = var_869_cast_fp16)[name = string("op_9020_cast_fp16")]; tensor var_9021_split_sizes_0 = const()[name = string("op_9021_split_sizes_0"), val = tensor([64, 64])]; int32 var_9021_axis_0 = const()[name = string("op_9021_axis_0"), val = int32(-2)]; tensor var_9021_cast_fp16_0, tensor var_9021_cast_fp16_1 = split(axis = var_9021_axis_0, split_sizes = var_9021_split_sizes_0, x = x_251_cast_fp16)[name = string("op_9021_cast_fp16")]; fp16 const_252_promoted_to_fp16 = const()[name = string("const_252_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9023_cast_fp16 = mul(x = var_9021_cast_fp16_1, y = const_252_promoted_to_fp16)[name = string("op_9023_cast_fp16")]; int32 var_9025 = const()[name = string("op_9025"), val = int32(-2)]; bool var_9026_interleave_0 = const()[name = string("op_9026_interleave_0"), val = bool(false)]; tensor var_9026_cast_fp16 = concat(axis = var_9025, interleave = var_9026_interleave_0, values = (var_9023_cast_fp16, var_9021_cast_fp16_0))[name = string("op_9026_cast_fp16")]; tensor var_9027_cast_fp16 = mul(x = var_9026_cast_fp16, y = var_878_cast_fp16)[name = string("op_9027_cast_fp16")]; tensor query_states_153_cast_fp16 = add(x = var_9020_cast_fp16, y = var_9027_cast_fp16)[name = string("query_states_153_cast_fp16")]; tensor var_9033_cast_fp16 = mul(x = var_9009_cast_fp16, y = var_869_cast_fp16)[name = string("op_9033_cast_fp16")]; tensor var_9034_split_sizes_0 = const()[name = string("op_9034_split_sizes_0"), val = tensor([64, 64])]; int32 var_9034_axis_0 = const()[name = string("op_9034_axis_0"), val = int32(-2)]; tensor var_9034_cast_fp16_0, tensor var_9034_cast_fp16_1 = split(axis = var_9034_axis_0, split_sizes = var_9034_split_sizes_0, x = var_9009_cast_fp16)[name = string("op_9034_cast_fp16")]; fp16 const_253_promoted_to_fp16 = const()[name = string("const_253_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9036_cast_fp16 = mul(x = var_9034_cast_fp16_1, y = const_253_promoted_to_fp16)[name = string("op_9036_cast_fp16")]; int32 var_9038 = const()[name = string("op_9038"), val = int32(-2)]; bool var_9039_interleave_0 = const()[name = string("op_9039_interleave_0"), val = bool(false)]; tensor var_9039_cast_fp16 = concat(axis = var_9038, interleave = var_9039_interleave_0, values = (var_9036_cast_fp16, var_9034_cast_fp16_0))[name = string("op_9039_cast_fp16")]; tensor var_9040_cast_fp16 = mul(x = var_9039_cast_fp16, y = var_878_cast_fp16)[name = string("op_9040_cast_fp16")]; tensor key_states_255_cast_fp16 = add(x = var_9033_cast_fp16, y = var_9040_cast_fp16)[name = string("key_states_255_cast_fp16")]; tensor expand_dims_300 = const()[name = string("expand_dims_300"), val = tensor([25])]; tensor expand_dims_301 = const()[name = string("expand_dims_301"), val = tensor([0])]; tensor expand_dims_303 = const()[name = string("expand_dims_303"), val = tensor([0])]; int32 concat_305_axis_0 = const()[name = string("concat_305_axis_0"), val = int32(0)]; bool concat_305_interleave_0 = const()[name = string("concat_305_interleave_0"), val = bool(false)]; tensor concat_305 = concat(axis = concat_305_axis_0, interleave = concat_305_interleave_0, values = (expand_dims_300, expand_dims_301, position_id, expand_dims_303))[name = string("concat_305")]; tensor expand_dims_304 = const()[name = string("expand_dims_304"), val = tensor([26])]; tensor concat_306_values1_0 = const()[name = string("concat_306_values1_0"), val = tensor([0])]; tensor concat_306_values3_0 = const()[name = string("concat_306_values3_0"), val = tensor([0])]; int32 concat_306_axis_0 = const()[name = string("concat_306_axis_0"), val = int32(0)]; bool concat_306_interleave_0 = const()[name = string("concat_306_interleave_0"), val = bool(false)]; tensor concat_306 = concat(axis = concat_306_axis_0, interleave = concat_306_interleave_0, values = (expand_dims_304, concat_306_values1_0, cache_position_end, concat_306_values3_0))[name = string("concat_306")]; tensor key_states_257_perm_0 = const()[name = string("key_states_257_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_26_stride_0 = const()[name = string("key_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_257_cast_fp16 = transpose(perm = key_states_257_perm_0, x = key_states_255_cast_fp16)[name = string("transpose_438")]; tensor key_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = key_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = key_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_26_squeeze_mask_0, stride = key_cache_internal_tensor_assign_26_stride_0, update = key_states_257_cast_fp16, x = coreml_update_state_328)[name = string("key_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_26_cast_fp16, input = key_cache)[name = string("coreml_update_state_330_write_state")]; tensor coreml_update_state_330 = read_state(input = key_cache)[name = string("coreml_update_state_330")]; tensor value_states_153_perm_0 = const()[name = string("value_states_153_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_26_stride_0 = const()[name = string("value_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_153_cast_fp16 = transpose(perm = value_states_153_perm_0, x = var_9016_cast_fp16)[name = string("transpose_437")]; tensor value_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = value_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = value_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_26_squeeze_mask_0, stride = value_cache_internal_tensor_assign_26_stride_0, update = value_states_153_cast_fp16, x = coreml_update_state_329)[name = string("value_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_26_cast_fp16, input = value_cache)[name = string("coreml_update_state_331_write_state")]; tensor coreml_update_state_331 = read_state(input = value_cache)[name = string("coreml_update_state_331")]; tensor var_9110_begin_0 = const()[name = string("op_9110_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9110_end_0 = const()[name = string("op_9110_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9110_end_mask_0 = const()[name = string("op_9110_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9110_cast_fp16 = slice_by_index(begin = var_9110_begin_0, end = var_9110_end_0, end_mask = var_9110_end_mask_0, x = coreml_update_state_330)[name = string("op_9110_cast_fp16")]; tensor tile_50 = const()[name = string("tile_50"), val = tensor([1, 1])]; int32 var_9113_axis_0 = const()[name = string("op_9113_axis_0"), val = int32(1)]; tensor var_9113_cast_fp16_0, tensor var_9113_cast_fp16_1 = split(axis = var_9113_axis_0, split_sizes = tile_50, x = var_9110_cast_fp16)[name = string("op_9113_cast_fp16")]; tensor var_9120_begin_0 = const()[name = string("op_9120_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9120_end_0 = const()[name = string("op_9120_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9120_end_mask_0 = const()[name = string("op_9120_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9120_cast_fp16 = slice_by_index(begin = var_9120_begin_0, end = var_9120_end_0, end_mask = var_9120_end_mask_0, x = coreml_update_state_331)[name = string("op_9120_cast_fp16")]; tensor tile_51 = const()[name = string("tile_51"), val = tensor([1, 1])]; int32 var_9123_axis_0 = const()[name = string("op_9123_axis_0"), val = int32(1)]; tensor var_9123_cast_fp16_0, tensor var_9123_cast_fp16_1 = split(axis = var_9123_axis_0, split_sizes = tile_51, x = var_9120_cast_fp16)[name = string("op_9123_cast_fp16")]; tensor var_9126_split_sizes_0 = const()[name = string("op_9126_split_sizes_0"), val = tensor([8, 8])]; int32 var_9126_axis_0 = const()[name = string("op_9126_axis_0"), val = int32(1)]; tensor var_9126_0, tensor var_9126_1 = split(axis = var_9126_axis_0, split_sizes = var_9126_split_sizes_0, x = query_states_153_cast_fp16)[name = string("op_9126")]; bool attn_weights_401_transpose_x_0 = const()[name = string("attn_weights_401_transpose_x_0"), val = bool(false)]; bool attn_weights_401_transpose_y_0 = const()[name = string("attn_weights_401_transpose_y_0"), val = bool(false)]; tensor attn_weights_401_cast_fp16 = matmul(transpose_x = attn_weights_401_transpose_x_0, transpose_y = attn_weights_401_transpose_y_0, x = var_9113_cast_fp16_0, y = var_9126_0)[name = string("attn_weights_401_cast_fp16")]; fp16 var_9129_to_fp16 = const()[name = string("op_9129_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_403_cast_fp16 = mul(x = attn_weights_401_cast_fp16, y = var_9129_to_fp16)[name = string("attn_weights_403_cast_fp16")]; tensor attn_weights_405_cast_fp16 = add(x = attn_weights_403_cast_fp16, y = attn_mask_1)[name = string("attn_weights_405_cast_fp16")]; int32 var_9133 = const()[name = string("op_9133"), val = int32(-2)]; tensor attn_weights_407_cast_fp16 = softmax(axis = var_9133, x = attn_weights_405_cast_fp16)[name = string("attn_weights_407_cast_fp16")]; bool var_9139_transpose_x_1 = const()[name = string("op_9139_transpose_x_1"), val = bool(true)]; bool var_9139_transpose_y_1 = const()[name = string("op_9139_transpose_y_1"), val = bool(false)]; tensor var_9139_cast_fp16 = matmul(transpose_x = var_9139_transpose_x_1, transpose_y = var_9139_transpose_y_1, x = attn_weights_407_cast_fp16, y = var_9123_cast_fp16_0)[name = string("op_9139_cast_fp16")]; bool attn_weights_409_transpose_x_0 = const()[name = string("attn_weights_409_transpose_x_0"), val = bool(false)]; bool attn_weights_409_transpose_y_0 = const()[name = string("attn_weights_409_transpose_y_0"), val = bool(false)]; tensor attn_weights_409_cast_fp16 = matmul(transpose_x = attn_weights_409_transpose_x_0, transpose_y = attn_weights_409_transpose_y_0, x = var_9113_cast_fp16_1, y = var_9126_1)[name = string("attn_weights_409_cast_fp16")]; fp16 var_9141_to_fp16 = const()[name = string("op_9141_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_411_cast_fp16 = mul(x = attn_weights_409_cast_fp16, y = var_9141_to_fp16)[name = string("attn_weights_411_cast_fp16")]; tensor attn_weights_413_cast_fp16 = add(x = attn_weights_411_cast_fp16, y = attn_mask_1)[name = string("attn_weights_413_cast_fp16")]; int32 var_9145 = const()[name = string("op_9145"), val = int32(-2)]; tensor attn_weights_415_cast_fp16 = softmax(axis = var_9145, x = attn_weights_413_cast_fp16)[name = string("attn_weights_415_cast_fp16")]; bool attn_output_201_transpose_x_1 = const()[name = string("attn_output_201_transpose_x_1"), val = bool(true)]; bool attn_output_201_transpose_y_1 = const()[name = string("attn_output_201_transpose_y_1"), val = bool(false)]; tensor attn_output_201_cast_fp16 = matmul(transpose_x = attn_output_201_transpose_x_1, transpose_y = attn_output_201_transpose_y_1, x = attn_weights_415_cast_fp16, y = var_9123_cast_fp16_1)[name = string("attn_output_201_cast_fp16")]; int32 var_9153 = const()[name = string("op_9153"), val = int32(1)]; bool attn_output_203_interleave_0 = const()[name = string("attn_output_203_interleave_0"), val = bool(false)]; tensor attn_output_203_cast_fp16 = concat(axis = var_9153, interleave = attn_output_203_interleave_0, values = (var_9139_cast_fp16, attn_output_201_cast_fp16))[name = string("attn_output_203_cast_fp16")]; tensor var_9157_perm_0 = const()[name = string("op_9157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_311x = const()[name = string("concat_311x"), val = tensor([1, 2048, 1, -1])]; tensor var_9157_cast_fp16 = transpose(perm = var_9157_perm_0, x = attn_output_203_cast_fp16)[name = string("transpose_436")]; tensor attn_output_207_cast_fp16 = reshape(shape = concat_311x, x = var_9157_cast_fp16)[name = string("attn_output_207_cast_fp16")]; tensor hidden_states_253_strides_0 = const()[name = string("hidden_states_253_strides_0"), val = tensor([1, 1])]; string hidden_states_253_pad_type_0 = const()[name = string("hidden_states_253_pad_type_0"), val = string("valid")]; tensor hidden_states_253_pad_0 = const()[name = string("hidden_states_253_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_253_dilations_0 = const()[name = string("hidden_states_253_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_253_groups_0 = const()[name = string("hidden_states_253_groups_0"), val = int32(1)]; tensor hidden_states_253_cast_fp16 = conv(dilations = hidden_states_253_dilations_0, groups = hidden_states_253_groups_0, pad = hidden_states_253_pad_0, pad_type = hidden_states_253_pad_type_0, strides = hidden_states_253_strides_0, weight = layers_25_self_attn_o_proj_weight_cast_fp16, x = attn_output_207_cast_fp16)[name = string("hidden_states_253_cast_fp16")]; tensor hidden_states_255_cast_fp16 = add(x = hidden_states_249_cast_fp16, y = hidden_states_253_cast_fp16)[name = string("hidden_states_255_cast_fp16")]; fp16 const_258_promoted_to_fp16 = const()[name = string("const_258_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9190_cast_fp16 = mul(x = hidden_states_255_cast_fp16, y = const_258_promoted_to_fp16)[name = string("op_9190_cast_fp16")]; int32 var_9188 = const()[name = string("op_9188"), val = int32(1)]; bool doubled_205_interleave_0 = const()[name = string("doubled_205_interleave_0"), val = bool(false)]; tensor doubled_205_cast_fp16 = concat(axis = var_9188, interleave = doubled_205_interleave_0, values = (hidden_states_255_cast_fp16, var_9190_cast_fp16))[name = string("doubled_205_cast_fp16")]; tensor out_103_axes_0 = const()[name = string("out_103_axes_0"), val = tensor([1])]; tensor out_103_gamma_0_to_fp16 = const()[name = string("out_103_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541427584)))]; fp16 var_9200_to_fp16 = const()[name = string("op_9200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_103_cast_fp16 = layer_norm(axes = out_103_axes_0, epsilon = var_9200_to_fp16, gamma = out_103_gamma_0_to_fp16, x = doubled_205_cast_fp16)[name = string("out_103_cast_fp16")]; tensor var_9211_split_sizes_0 = const()[name = string("op_9211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9211_axis_0 = const()[name = string("op_9211_axis_0"), val = int32(1)]; tensor var_9211_cast_fp16_0, tensor var_9211_cast_fp16_1 = split(axis = var_9211_axis_0, split_sizes = var_9211_split_sizes_0, x = out_103_cast_fp16)[name = string("op_9211_cast_fp16")]; tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; tensor input_51_cast_fp16 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = layers_25_mlp_gate_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("input_51_cast_fp16")]; tensor var_9228_cast_fp16 = silu(x = input_51_cast_fp16)[name = string("op_9228_cast_fp16")]; tensor var_9234_strides_0 = const()[name = string("op_9234_strides_0"), val = tensor([1, 1])]; string var_9234_pad_type_0 = const()[name = string("op_9234_pad_type_0"), val = string("valid")]; tensor var_9234_pad_0 = const()[name = string("op_9234_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9234_dilations_0 = const()[name = string("op_9234_dilations_0"), val = tensor([1, 1])]; int32 var_9234_groups_0 = const()[name = string("op_9234_groups_0"), val = int32(1)]; tensor var_9234_cast_fp16 = conv(dilations = var_9234_dilations_0, groups = var_9234_groups_0, pad = var_9234_pad_0, pad_type = var_9234_pad_type_0, strides = var_9234_strides_0, weight = layers_25_mlp_up_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("op_9234_cast_fp16")]; tensor x_259_cast_fp16 = mul(x = var_9228_cast_fp16, y = var_9234_cast_fp16)[name = string("x_259_cast_fp16")]; tensor layers_25_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_25_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541435840)))]; tensor hidden_states_257_strides_0 = const()[name = string("hidden_states_257_strides_0"), val = tensor([1, 1])]; string hidden_states_257_pad_type_0 = const()[name = string("hidden_states_257_pad_type_0"), val = string("valid")]; tensor hidden_states_257_pad_0 = const()[name = string("hidden_states_257_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_257_dilations_0 = const()[name = string("hidden_states_257_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_257_groups_0 = const()[name = string("hidden_states_257_groups_0"), val = int32(1)]; tensor hidden_states_257_cast_fp16 = conv(dilations = hidden_states_257_dilations_0, groups = hidden_states_257_groups_0, pad = hidden_states_257_pad_0, pad_type = hidden_states_257_pad_type_0, strides = hidden_states_257_strides_0, weight = layers_25_mlp_down_proj_weight_to_fp16, x = x_259_cast_fp16)[name = string("hidden_states_257_cast_fp16")]; tensor hidden_states_259_cast_fp16 = add(x = hidden_states_255_cast_fp16, y = hidden_states_257_cast_fp16)[name = string("hidden_states_259_cast_fp16")]; fp16 const_260_promoted_to_fp16 = const()[name = string("const_260_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9252_cast_fp16 = mul(x = hidden_states_259_cast_fp16, y = const_260_promoted_to_fp16)[name = string("op_9252_cast_fp16")]; int32 var_9250 = const()[name = string("op_9250"), val = int32(1)]; bool doubled_209_interleave_0 = const()[name = string("doubled_209_interleave_0"), val = bool(false)]; tensor doubled_209_cast_fp16 = concat(axis = var_9250, interleave = doubled_209_interleave_0, values = (hidden_states_259_cast_fp16, var_9252_cast_fp16))[name = string("doubled_209_cast_fp16")]; tensor out_105_axes_0 = const()[name = string("out_105_axes_0"), val = tensor([1])]; tensor out_105_gamma_0_to_fp16 = const()[name = string("out_105_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566601728)))]; fp16 var_9262_to_fp16 = const()[name = string("op_9262_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_105_cast_fp16 = layer_norm(axes = out_105_axes_0, epsilon = var_9262_to_fp16, gamma = out_105_gamma_0_to_fp16, x = doubled_209_cast_fp16)[name = string("out_105_cast_fp16")]; tensor var_9273_split_sizes_0 = const()[name = string("op_9273_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9273_axis_0 = const()[name = string("op_9273_axis_0"), val = int32(1)]; tensor var_9273_cast_fp16_0, tensor var_9273_cast_fp16_1 = split(axis = var_9273_axis_0, split_sizes = var_9273_split_sizes_0, x = out_105_cast_fp16)[name = string("op_9273_cast_fp16")]; tensor query_states_157_strides_0 = const()[name = string("query_states_157_strides_0"), val = tensor([1, 1])]; string query_states_157_pad_type_0 = const()[name = string("query_states_157_pad_type_0"), val = string("valid")]; tensor query_states_157_pad_0 = const()[name = string("query_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_157_dilations_0 = const()[name = string("query_states_157_dilations_0"), val = tensor([1, 1])]; int32 query_states_157_groups_0 = const()[name = string("query_states_157_groups_0"), val = int32(1)]; tensor query_states_157_cast_fp16 = conv(dilations = query_states_157_dilations_0, groups = query_states_157_groups_0, pad = query_states_157_pad_0, pad_type = query_states_157_pad_type_0, strides = query_states_157_strides_0, weight = layers_26_self_attn_q_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("query_states_157_cast_fp16")]; tensor key_states_261_strides_0 = const()[name = string("key_states_261_strides_0"), val = tensor([1, 1])]; string key_states_261_pad_type_0 = const()[name = string("key_states_261_pad_type_0"), val = string("valid")]; tensor key_states_261_pad_0 = const()[name = string("key_states_261_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_261_dilations_0 = const()[name = string("key_states_261_dilations_0"), val = tensor([1, 1])]; int32 key_states_261_groups_0 = const()[name = string("key_states_261_groups_0"), val = int32(1)]; tensor key_states_261_cast_fp16 = conv(dilations = key_states_261_dilations_0, groups = key_states_261_groups_0, pad = key_states_261_pad_0, pad_type = key_states_261_pad_type_0, strides = key_states_261_strides_0, weight = layers_26_self_attn_k_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("key_states_261_cast_fp16")]; tensor layers_26_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_26_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566609984)))]; tensor value_states_157_strides_0 = const()[name = string("value_states_157_strides_0"), val = tensor([1, 1])]; string value_states_157_pad_type_0 = const()[name = string("value_states_157_pad_type_0"), val = string("valid")]; tensor value_states_157_pad_0 = const()[name = string("value_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_157_dilations_0 = const()[name = string("value_states_157_dilations_0"), val = tensor([1, 1])]; int32 value_states_157_groups_0 = const()[name = string("value_states_157_groups_0"), val = int32(1)]; tensor value_states_157_cast_fp16 = conv(dilations = value_states_157_dilations_0, groups = value_states_157_groups_0, pad = value_states_157_pad_0, pad_type = value_states_157_pad_type_0, strides = value_states_157_strides_0, weight = layers_26_self_attn_v_proj_weight_to_fp16, x = var_9273_cast_fp16_0)[name = string("value_states_157_cast_fp16")]; tensor concat_312x = const()[name = string("concat_312x"), val = tensor([1, 16, 128, -1])]; tensor x_261_cast_fp16 = reshape(shape = concat_312x, x = query_states_157_cast_fp16)[name = string("x_261_cast_fp16")]; tensor concat_313x = const()[name = string("concat_313x"), val = tensor([1, 2, 128, -1])]; tensor var_9330_cast_fp16 = reshape(shape = concat_313x, x = key_states_261_cast_fp16)[name = string("op_9330_cast_fp16")]; tensor concat_314x = const()[name = string("concat_314x"), val = tensor([1, 2, 128, -1])]; tensor var_9337_cast_fp16 = reshape(shape = concat_314x, x = value_states_157_cast_fp16)[name = string("op_9337_cast_fp16")]; tensor var_9341_cast_fp16 = mul(x = x_261_cast_fp16, y = var_869_cast_fp16)[name = string("op_9341_cast_fp16")]; tensor var_9342_split_sizes_0 = const()[name = string("op_9342_split_sizes_0"), val = tensor([64, 64])]; int32 var_9342_axis_0 = const()[name = string("op_9342_axis_0"), val = int32(-2)]; tensor var_9342_cast_fp16_0, tensor var_9342_cast_fp16_1 = split(axis = var_9342_axis_0, split_sizes = var_9342_split_sizes_0, x = x_261_cast_fp16)[name = string("op_9342_cast_fp16")]; fp16 const_262_promoted_to_fp16 = const()[name = string("const_262_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9344_cast_fp16 = mul(x = var_9342_cast_fp16_1, y = const_262_promoted_to_fp16)[name = string("op_9344_cast_fp16")]; int32 var_9346 = const()[name = string("op_9346"), val = int32(-2)]; bool var_9347_interleave_0 = const()[name = string("op_9347_interleave_0"), val = bool(false)]; tensor var_9347_cast_fp16 = concat(axis = var_9346, interleave = var_9347_interleave_0, values = (var_9344_cast_fp16, var_9342_cast_fp16_0))[name = string("op_9347_cast_fp16")]; tensor var_9348_cast_fp16 = mul(x = var_9347_cast_fp16, y = var_878_cast_fp16)[name = string("op_9348_cast_fp16")]; tensor query_states_159_cast_fp16 = add(x = var_9341_cast_fp16, y = var_9348_cast_fp16)[name = string("query_states_159_cast_fp16")]; tensor var_9354_cast_fp16 = mul(x = var_9330_cast_fp16, y = var_869_cast_fp16)[name = string("op_9354_cast_fp16")]; tensor var_9355_split_sizes_0 = const()[name = string("op_9355_split_sizes_0"), val = tensor([64, 64])]; int32 var_9355_axis_0 = const()[name = string("op_9355_axis_0"), val = int32(-2)]; tensor var_9355_cast_fp16_0, tensor var_9355_cast_fp16_1 = split(axis = var_9355_axis_0, split_sizes = var_9355_split_sizes_0, x = var_9330_cast_fp16)[name = string("op_9355_cast_fp16")]; fp16 const_263_promoted_to_fp16 = const()[name = string("const_263_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9357_cast_fp16 = mul(x = var_9355_cast_fp16_1, y = const_263_promoted_to_fp16)[name = string("op_9357_cast_fp16")]; int32 var_9359 = const()[name = string("op_9359"), val = int32(-2)]; bool var_9360_interleave_0 = const()[name = string("op_9360_interleave_0"), val = bool(false)]; tensor var_9360_cast_fp16 = concat(axis = var_9359, interleave = var_9360_interleave_0, values = (var_9357_cast_fp16, var_9355_cast_fp16_0))[name = string("op_9360_cast_fp16")]; tensor var_9361_cast_fp16 = mul(x = var_9360_cast_fp16, y = var_878_cast_fp16)[name = string("op_9361_cast_fp16")]; tensor key_states_265_cast_fp16 = add(x = var_9354_cast_fp16, y = var_9361_cast_fp16)[name = string("key_states_265_cast_fp16")]; tensor expand_dims_312 = const()[name = string("expand_dims_312"), val = tensor([26])]; tensor expand_dims_313 = const()[name = string("expand_dims_313"), val = tensor([0])]; tensor expand_dims_315 = const()[name = string("expand_dims_315"), val = tensor([0])]; int32 concat_317_axis_0 = const()[name = string("concat_317_axis_0"), val = int32(0)]; bool concat_317_interleave_0 = const()[name = string("concat_317_interleave_0"), val = bool(false)]; tensor concat_317 = concat(axis = concat_317_axis_0, interleave = concat_317_interleave_0, values = (expand_dims_312, expand_dims_313, position_id, expand_dims_315))[name = string("concat_317")]; tensor expand_dims_316 = const()[name = string("expand_dims_316"), val = tensor([27])]; tensor concat_318_values1_0 = const()[name = string("concat_318_values1_0"), val = tensor([0])]; tensor concat_318_values3_0 = const()[name = string("concat_318_values3_0"), val = tensor([0])]; int32 concat_318_axis_0 = const()[name = string("concat_318_axis_0"), val = int32(0)]; bool concat_318_interleave_0 = const()[name = string("concat_318_interleave_0"), val = bool(false)]; tensor concat_318 = concat(axis = concat_318_axis_0, interleave = concat_318_interleave_0, values = (expand_dims_316, concat_318_values1_0, cache_position_end, concat_318_values3_0))[name = string("concat_318")]; tensor key_states_267_perm_0 = const()[name = string("key_states_267_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_27_stride_0 = const()[name = string("key_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_267_cast_fp16 = transpose(perm = key_states_267_perm_0, x = key_states_265_cast_fp16)[name = string("transpose_435")]; tensor key_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = key_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = key_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_27_squeeze_mask_0, stride = key_cache_internal_tensor_assign_27_stride_0, update = key_states_267_cast_fp16, x = coreml_update_state_330)[name = string("key_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_27_cast_fp16, input = key_cache)[name = string("coreml_update_state_332_write_state")]; tensor coreml_update_state_332 = read_state(input = key_cache)[name = string("coreml_update_state_332")]; tensor value_states_159_perm_0 = const()[name = string("value_states_159_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_27_stride_0 = const()[name = string("value_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_159_cast_fp16 = transpose(perm = value_states_159_perm_0, x = var_9337_cast_fp16)[name = string("transpose_434")]; tensor value_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = value_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = value_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_27_squeeze_mask_0, stride = value_cache_internal_tensor_assign_27_stride_0, update = value_states_159_cast_fp16, x = coreml_update_state_331)[name = string("value_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_27_cast_fp16, input = value_cache)[name = string("coreml_update_state_333_write_state")]; tensor coreml_update_state_333 = read_state(input = value_cache)[name = string("coreml_update_state_333")]; tensor var_9431_begin_0 = const()[name = string("op_9431_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9431_end_0 = const()[name = string("op_9431_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9431_end_mask_0 = const()[name = string("op_9431_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9431_cast_fp16 = slice_by_index(begin = var_9431_begin_0, end = var_9431_end_0, end_mask = var_9431_end_mask_0, x = coreml_update_state_332)[name = string("op_9431_cast_fp16")]; tensor tile_52 = const()[name = string("tile_52"), val = tensor([1, 1])]; int32 var_9434_axis_0 = const()[name = string("op_9434_axis_0"), val = int32(1)]; tensor var_9434_cast_fp16_0, tensor var_9434_cast_fp16_1 = split(axis = var_9434_axis_0, split_sizes = tile_52, x = var_9431_cast_fp16)[name = string("op_9434_cast_fp16")]; tensor var_9441_begin_0 = const()[name = string("op_9441_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9441_end_0 = const()[name = string("op_9441_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9441_end_mask_0 = const()[name = string("op_9441_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9441_cast_fp16 = slice_by_index(begin = var_9441_begin_0, end = var_9441_end_0, end_mask = var_9441_end_mask_0, x = coreml_update_state_333)[name = string("op_9441_cast_fp16")]; tensor tile_53 = const()[name = string("tile_53"), val = tensor([1, 1])]; int32 var_9444_axis_0 = const()[name = string("op_9444_axis_0"), val = int32(1)]; tensor var_9444_cast_fp16_0, tensor var_9444_cast_fp16_1 = split(axis = var_9444_axis_0, split_sizes = tile_53, x = var_9441_cast_fp16)[name = string("op_9444_cast_fp16")]; tensor var_9447_split_sizes_0 = const()[name = string("op_9447_split_sizes_0"), val = tensor([8, 8])]; int32 var_9447_axis_0 = const()[name = string("op_9447_axis_0"), val = int32(1)]; tensor var_9447_0, tensor var_9447_1 = split(axis = var_9447_axis_0, split_sizes = var_9447_split_sizes_0, x = query_states_159_cast_fp16)[name = string("op_9447")]; bool attn_weights_417_transpose_x_0 = const()[name = string("attn_weights_417_transpose_x_0"), val = bool(false)]; bool attn_weights_417_transpose_y_0 = const()[name = string("attn_weights_417_transpose_y_0"), val = bool(false)]; tensor attn_weights_417_cast_fp16 = matmul(transpose_x = attn_weights_417_transpose_x_0, transpose_y = attn_weights_417_transpose_y_0, x = var_9434_cast_fp16_0, y = var_9447_0)[name = string("attn_weights_417_cast_fp16")]; fp16 var_9450_to_fp16 = const()[name = string("op_9450_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_419_cast_fp16 = mul(x = attn_weights_417_cast_fp16, y = var_9450_to_fp16)[name = string("attn_weights_419_cast_fp16")]; tensor attn_weights_421_cast_fp16 = add(x = attn_weights_419_cast_fp16, y = attn_mask_1)[name = string("attn_weights_421_cast_fp16")]; int32 var_9454 = const()[name = string("op_9454"), val = int32(-2)]; tensor attn_weights_423_cast_fp16 = softmax(axis = var_9454, x = attn_weights_421_cast_fp16)[name = string("attn_weights_423_cast_fp16")]; bool var_9460_transpose_x_1 = const()[name = string("op_9460_transpose_x_1"), val = bool(true)]; bool var_9460_transpose_y_1 = const()[name = string("op_9460_transpose_y_1"), val = bool(false)]; tensor var_9460_cast_fp16 = matmul(transpose_x = var_9460_transpose_x_1, transpose_y = var_9460_transpose_y_1, x = attn_weights_423_cast_fp16, y = var_9444_cast_fp16_0)[name = string("op_9460_cast_fp16")]; bool attn_weights_425_transpose_x_0 = const()[name = string("attn_weights_425_transpose_x_0"), val = bool(false)]; bool attn_weights_425_transpose_y_0 = const()[name = string("attn_weights_425_transpose_y_0"), val = bool(false)]; tensor attn_weights_425_cast_fp16 = matmul(transpose_x = attn_weights_425_transpose_x_0, transpose_y = attn_weights_425_transpose_y_0, x = var_9434_cast_fp16_1, y = var_9447_1)[name = string("attn_weights_425_cast_fp16")]; fp16 var_9462_to_fp16 = const()[name = string("op_9462_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_427_cast_fp16 = mul(x = attn_weights_425_cast_fp16, y = var_9462_to_fp16)[name = string("attn_weights_427_cast_fp16")]; tensor attn_weights_429_cast_fp16 = add(x = attn_weights_427_cast_fp16, y = attn_mask_1)[name = string("attn_weights_429_cast_fp16")]; int32 var_9466 = const()[name = string("op_9466"), val = int32(-2)]; tensor attn_weights_431_cast_fp16 = softmax(axis = var_9466, x = attn_weights_429_cast_fp16)[name = string("attn_weights_431_cast_fp16")]; bool attn_output_209_transpose_x_1 = const()[name = string("attn_output_209_transpose_x_1"), val = bool(true)]; bool attn_output_209_transpose_y_1 = const()[name = string("attn_output_209_transpose_y_1"), val = bool(false)]; tensor attn_output_209_cast_fp16 = matmul(transpose_x = attn_output_209_transpose_x_1, transpose_y = attn_output_209_transpose_y_1, x = attn_weights_431_cast_fp16, y = var_9444_cast_fp16_1)[name = string("attn_output_209_cast_fp16")]; int32 var_9474 = const()[name = string("op_9474"), val = int32(1)]; bool attn_output_211_interleave_0 = const()[name = string("attn_output_211_interleave_0"), val = bool(false)]; tensor attn_output_211_cast_fp16 = concat(axis = var_9474, interleave = attn_output_211_interleave_0, values = (var_9460_cast_fp16, attn_output_209_cast_fp16))[name = string("attn_output_211_cast_fp16")]; tensor var_9478_perm_0 = const()[name = string("op_9478_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_323x = const()[name = string("concat_323x"), val = tensor([1, 2048, 1, -1])]; tensor var_9478_cast_fp16 = transpose(perm = var_9478_perm_0, x = attn_output_211_cast_fp16)[name = string("transpose_433")]; tensor attn_output_215_cast_fp16 = reshape(shape = concat_323x, x = var_9478_cast_fp16)[name = string("attn_output_215_cast_fp16")]; tensor hidden_states_263_strides_0 = const()[name = string("hidden_states_263_strides_0"), val = tensor([1, 1])]; string hidden_states_263_pad_type_0 = const()[name = string("hidden_states_263_pad_type_0"), val = string("valid")]; tensor hidden_states_263_pad_0 = const()[name = string("hidden_states_263_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_263_dilations_0 = const()[name = string("hidden_states_263_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_263_groups_0 = const()[name = string("hidden_states_263_groups_0"), val = int32(1)]; tensor hidden_states_263_cast_fp16 = conv(dilations = hidden_states_263_dilations_0, groups = hidden_states_263_groups_0, pad = hidden_states_263_pad_0, pad_type = hidden_states_263_pad_type_0, strides = hidden_states_263_strides_0, weight = layers_26_self_attn_o_proj_weight_cast_fp16, x = attn_output_215_cast_fp16)[name = string("hidden_states_263_cast_fp16")]; tensor hidden_states_265_cast_fp16 = add(x = hidden_states_259_cast_fp16, y = hidden_states_263_cast_fp16)[name = string("hidden_states_265_cast_fp16")]; fp16 const_268_promoted_to_fp16 = const()[name = string("const_268_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9511_cast_fp16 = mul(x = hidden_states_265_cast_fp16, y = const_268_promoted_to_fp16)[name = string("op_9511_cast_fp16")]; int32 var_9509 = const()[name = string("op_9509"), val = int32(1)]; bool doubled_213_interleave_0 = const()[name = string("doubled_213_interleave_0"), val = bool(false)]; tensor doubled_213_cast_fp16 = concat(axis = var_9509, interleave = doubled_213_interleave_0, values = (hidden_states_265_cast_fp16, var_9511_cast_fp16))[name = string("doubled_213_cast_fp16")]; tensor out_107_axes_0 = const()[name = string("out_107_axes_0"), val = tensor([1])]; tensor out_107_gamma_0_to_fp16 = const()[name = string("out_107_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567658624)))]; fp16 var_9521_to_fp16 = const()[name = string("op_9521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_107_cast_fp16 = layer_norm(axes = out_107_axes_0, epsilon = var_9521_to_fp16, gamma = out_107_gamma_0_to_fp16, x = doubled_213_cast_fp16)[name = string("out_107_cast_fp16")]; tensor var_9532_split_sizes_0 = const()[name = string("op_9532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9532_axis_0 = const()[name = string("op_9532_axis_0"), val = int32(1)]; tensor var_9532_cast_fp16_0, tensor var_9532_cast_fp16_1 = split(axis = var_9532_axis_0, split_sizes = var_9532_split_sizes_0, x = out_107_cast_fp16)[name = string("op_9532_cast_fp16")]; tensor input_53_strides_0 = const()[name = string("input_53_strides_0"), val = tensor([1, 1])]; string input_53_pad_type_0 = const()[name = string("input_53_pad_type_0"), val = string("valid")]; tensor input_53_pad_0 = const()[name = string("input_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_53_dilations_0 = const()[name = string("input_53_dilations_0"), val = tensor([1, 1])]; int32 input_53_groups_0 = const()[name = string("input_53_groups_0"), val = int32(1)]; tensor input_53_cast_fp16 = conv(dilations = input_53_dilations_0, groups = input_53_groups_0, pad = input_53_pad_0, pad_type = input_53_pad_type_0, strides = input_53_strides_0, weight = layers_26_mlp_gate_proj_weight_cast_fp16, x = var_9532_cast_fp16_0)[name = string("input_53_cast_fp16")]; tensor var_9549_cast_fp16 = silu(x = input_53_cast_fp16)[name = string("op_9549_cast_fp16")]; tensor layers_26_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567666880)))]; tensor var_9555_strides_0 = const()[name = string("op_9555_strides_0"), val = tensor([1, 1])]; string var_9555_pad_type_0 = const()[name = string("op_9555_pad_type_0"), val = string("valid")]; tensor var_9555_pad_0 = const()[name = string("op_9555_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9555_dilations_0 = const()[name = string("op_9555_dilations_0"), val = tensor([1, 1])]; int32 var_9555_groups_0 = const()[name = string("op_9555_groups_0"), val = int32(1)]; tensor var_9555_cast_fp16 = conv(dilations = var_9555_dilations_0, groups = var_9555_groups_0, pad = var_9555_pad_0, pad_type = var_9555_pad_type_0, strides = var_9555_strides_0, weight = layers_26_mlp_up_proj_weight_to_fp16, x = var_9532_cast_fp16_0)[name = string("op_9555_cast_fp16")]; tensor x_269_cast_fp16 = mul(x = var_9549_cast_fp16, y = var_9555_cast_fp16)[name = string("x_269_cast_fp16")]; tensor layers_26_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1592832768)))]; tensor hidden_states_267_strides_0 = const()[name = string("hidden_states_267_strides_0"), val = tensor([1, 1])]; string hidden_states_267_pad_type_0 = const()[name = string("hidden_states_267_pad_type_0"), val = string("valid")]; tensor hidden_states_267_pad_0 = const()[name = string("hidden_states_267_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_267_dilations_0 = const()[name = string("hidden_states_267_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_267_groups_0 = const()[name = string("hidden_states_267_groups_0"), val = int32(1)]; tensor hidden_states_267_cast_fp16 = conv(dilations = hidden_states_267_dilations_0, groups = hidden_states_267_groups_0, pad = hidden_states_267_pad_0, pad_type = hidden_states_267_pad_type_0, strides = hidden_states_267_strides_0, weight = layers_26_mlp_down_proj_weight_to_fp16, x = x_269_cast_fp16)[name = string("hidden_states_267_cast_fp16")]; tensor hidden_states_269_cast_fp16 = add(x = hidden_states_265_cast_fp16, y = hidden_states_267_cast_fp16)[name = string("hidden_states_269_cast_fp16")]; fp16 const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9573_cast_fp16 = mul(x = hidden_states_269_cast_fp16, y = const_270_promoted_to_fp16)[name = string("op_9573_cast_fp16")]; int32 var_9571 = const()[name = string("op_9571"), val = int32(1)]; bool doubled_217_interleave_0 = const()[name = string("doubled_217_interleave_0"), val = bool(false)]; tensor doubled_217_cast_fp16 = concat(axis = var_9571, interleave = doubled_217_interleave_0, values = (hidden_states_269_cast_fp16, var_9573_cast_fp16))[name = string("doubled_217_cast_fp16")]; tensor out_109_axes_0 = const()[name = string("out_109_axes_0"), val = tensor([1])]; tensor out_109_gamma_0_to_fp16 = const()[name = string("out_109_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1617998656)))]; fp16 var_9583_to_fp16 = const()[name = string("op_9583_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_109_cast_fp16 = layer_norm(axes = out_109_axes_0, epsilon = var_9583_to_fp16, gamma = out_109_gamma_0_to_fp16, x = doubled_217_cast_fp16)[name = string("out_109_cast_fp16")]; tensor var_9594_split_sizes_0 = const()[name = string("op_9594_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9594_axis_0 = const()[name = string("op_9594_axis_0"), val = int32(1)]; tensor var_9594_cast_fp16_0, tensor var_9594_cast_fp16_1 = split(axis = var_9594_axis_0, split_sizes = var_9594_split_sizes_0, x = out_109_cast_fp16)[name = string("op_9594_cast_fp16")]; tensor layers_27_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1618006912)))]; tensor query_states_163_strides_0 = const()[name = string("query_states_163_strides_0"), val = tensor([1, 1])]; string query_states_163_pad_type_0 = const()[name = string("query_states_163_pad_type_0"), val = string("valid")]; tensor query_states_163_pad_0 = const()[name = string("query_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_163_dilations_0 = const()[name = string("query_states_163_dilations_0"), val = tensor([1, 1])]; int32 query_states_163_groups_0 = const()[name = string("query_states_163_groups_0"), val = int32(1)]; tensor query_states_163_cast_fp16 = conv(dilations = query_states_163_dilations_0, groups = query_states_163_groups_0, pad = query_states_163_pad_0, pad_type = query_states_163_pad_type_0, strides = query_states_163_strides_0, weight = layers_27_self_attn_q_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("query_states_163_cast_fp16")]; tensor layers_27_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1626395584)))]; tensor key_states_271_strides_0 = const()[name = string("key_states_271_strides_0"), val = tensor([1, 1])]; string key_states_271_pad_type_0 = const()[name = string("key_states_271_pad_type_0"), val = string("valid")]; tensor key_states_271_pad_0 = const()[name = string("key_states_271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_271_dilations_0 = const()[name = string("key_states_271_dilations_0"), val = tensor([1, 1])]; int32 key_states_271_groups_0 = const()[name = string("key_states_271_groups_0"), val = int32(1)]; tensor key_states_271_cast_fp16 = conv(dilations = key_states_271_dilations_0, groups = key_states_271_groups_0, pad = key_states_271_pad_0, pad_type = key_states_271_pad_type_0, strides = key_states_271_strides_0, weight = layers_27_self_attn_k_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("key_states_271_cast_fp16")]; tensor layers_27_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1627444224)))]; tensor value_states_163_strides_0 = const()[name = string("value_states_163_strides_0"), val = tensor([1, 1])]; string value_states_163_pad_type_0 = const()[name = string("value_states_163_pad_type_0"), val = string("valid")]; tensor value_states_163_pad_0 = const()[name = string("value_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_163_dilations_0 = const()[name = string("value_states_163_dilations_0"), val = tensor([1, 1])]; int32 value_states_163_groups_0 = const()[name = string("value_states_163_groups_0"), val = int32(1)]; tensor value_states_163_cast_fp16 = conv(dilations = value_states_163_dilations_0, groups = value_states_163_groups_0, pad = value_states_163_pad_0, pad_type = value_states_163_pad_type_0, strides = value_states_163_strides_0, weight = layers_27_self_attn_v_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("value_states_163_cast_fp16")]; tensor concat_324x = const()[name = string("concat_324x"), val = tensor([1, 16, 128, -1])]; tensor x_271_cast_fp16 = reshape(shape = concat_324x, x = query_states_163_cast_fp16)[name = string("x_271_cast_fp16")]; tensor concat_325x = const()[name = string("concat_325x"), val = tensor([1, 2, 128, -1])]; tensor var_9651_cast_fp16 = reshape(shape = concat_325x, x = key_states_271_cast_fp16)[name = string("op_9651_cast_fp16")]; tensor concat_326x = const()[name = string("concat_326x"), val = tensor([1, 2, 128, -1])]; tensor var_9658_cast_fp16 = reshape(shape = concat_326x, x = value_states_163_cast_fp16)[name = string("op_9658_cast_fp16")]; tensor var_9662_cast_fp16 = mul(x = x_271_cast_fp16, y = var_869_cast_fp16)[name = string("op_9662_cast_fp16")]; tensor var_9663_split_sizes_0 = const()[name = string("op_9663_split_sizes_0"), val = tensor([64, 64])]; int32 var_9663_axis_0 = const()[name = string("op_9663_axis_0"), val = int32(-2)]; tensor var_9663_cast_fp16_0, tensor var_9663_cast_fp16_1 = split(axis = var_9663_axis_0, split_sizes = var_9663_split_sizes_0, x = x_271_cast_fp16)[name = string("op_9663_cast_fp16")]; fp16 const_272_promoted_to_fp16 = const()[name = string("const_272_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9665_cast_fp16 = mul(x = var_9663_cast_fp16_1, y = const_272_promoted_to_fp16)[name = string("op_9665_cast_fp16")]; int32 var_9667 = const()[name = string("op_9667"), val = int32(-2)]; bool var_9668_interleave_0 = const()[name = string("op_9668_interleave_0"), val = bool(false)]; tensor var_9668_cast_fp16 = concat(axis = var_9667, interleave = var_9668_interleave_0, values = (var_9665_cast_fp16, var_9663_cast_fp16_0))[name = string("op_9668_cast_fp16")]; tensor var_9669_cast_fp16 = mul(x = var_9668_cast_fp16, y = var_878_cast_fp16)[name = string("op_9669_cast_fp16")]; tensor query_states_165_cast_fp16 = add(x = var_9662_cast_fp16, y = var_9669_cast_fp16)[name = string("query_states_165_cast_fp16")]; tensor var_9675_cast_fp16 = mul(x = var_9651_cast_fp16, y = var_869_cast_fp16)[name = string("op_9675_cast_fp16")]; tensor var_9676_split_sizes_0 = const()[name = string("op_9676_split_sizes_0"), val = tensor([64, 64])]; int32 var_9676_axis_0 = const()[name = string("op_9676_axis_0"), val = int32(-2)]; tensor var_9676_cast_fp16_0, tensor var_9676_cast_fp16_1 = split(axis = var_9676_axis_0, split_sizes = var_9676_split_sizes_0, x = var_9651_cast_fp16)[name = string("op_9676_cast_fp16")]; fp16 const_273_promoted_to_fp16 = const()[name = string("const_273_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9678_cast_fp16 = mul(x = var_9676_cast_fp16_1, y = const_273_promoted_to_fp16)[name = string("op_9678_cast_fp16")]; int32 var_9680 = const()[name = string("op_9680"), val = int32(-2)]; bool var_9681_interleave_0 = const()[name = string("op_9681_interleave_0"), val = bool(false)]; tensor var_9681_cast_fp16 = concat(axis = var_9680, interleave = var_9681_interleave_0, values = (var_9678_cast_fp16, var_9676_cast_fp16_0))[name = string("op_9681_cast_fp16")]; tensor var_9682_cast_fp16 = mul(x = var_9681_cast_fp16, y = var_878_cast_fp16)[name = string("op_9682_cast_fp16")]; tensor key_states_275_cast_fp16 = add(x = var_9675_cast_fp16, y = var_9682_cast_fp16)[name = string("key_states_275_cast_fp16")]; tensor expand_dims_324 = const()[name = string("expand_dims_324"), val = tensor([27])]; tensor expand_dims_325 = const()[name = string("expand_dims_325"), val = tensor([0])]; tensor expand_dims_327 = const()[name = string("expand_dims_327"), val = tensor([0])]; int32 concat_329_axis_0 = const()[name = string("concat_329_axis_0"), val = int32(0)]; bool concat_329_interleave_0 = const()[name = string("concat_329_interleave_0"), val = bool(false)]; tensor concat_329 = concat(axis = concat_329_axis_0, interleave = concat_329_interleave_0, values = (expand_dims_324, expand_dims_325, position_id, expand_dims_327))[name = string("concat_329")]; tensor expand_dims_328 = const()[name = string("expand_dims_328"), val = tensor([28])]; tensor concat_330_values1_0 = const()[name = string("concat_330_values1_0"), val = tensor([0])]; tensor concat_330_values3_0 = const()[name = string("concat_330_values3_0"), val = tensor([0])]; int32 concat_330_axis_0 = const()[name = string("concat_330_axis_0"), val = int32(0)]; bool concat_330_interleave_0 = const()[name = string("concat_330_interleave_0"), val = bool(false)]; tensor concat_330 = concat(axis = concat_330_axis_0, interleave = concat_330_interleave_0, values = (expand_dims_328, concat_330_values1_0, cache_position_end, concat_330_values3_0))[name = string("concat_330")]; tensor key_states_277_perm_0 = const()[name = string("key_states_277_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_28_stride_0 = const()[name = string("key_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_277_cast_fp16 = transpose(perm = key_states_277_perm_0, x = key_states_275_cast_fp16)[name = string("transpose_432")]; tensor key_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = key_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = key_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_28_squeeze_mask_0, stride = key_cache_internal_tensor_assign_28_stride_0, update = key_states_277_cast_fp16, x = coreml_update_state_332)[name = string("key_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_28_cast_fp16, input = key_cache)[name = string("coreml_update_state_334_write_state")]; tensor coreml_update_state_334 = read_state(input = key_cache)[name = string("coreml_update_state_334")]; tensor value_states_165_perm_0 = const()[name = string("value_states_165_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_28_stride_0 = const()[name = string("value_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_165_cast_fp16 = transpose(perm = value_states_165_perm_0, x = var_9658_cast_fp16)[name = string("transpose_431")]; tensor value_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = value_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = value_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_28_squeeze_mask_0, stride = value_cache_internal_tensor_assign_28_stride_0, update = value_states_165_cast_fp16, x = coreml_update_state_333)[name = string("value_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_28_cast_fp16, input = value_cache)[name = string("coreml_update_state_335_write_state")]; tensor coreml_update_state_335 = read_state(input = value_cache)[name = string("coreml_update_state_335")]; tensor var_9752_begin_0 = const()[name = string("op_9752_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9752_end_0 = const()[name = string("op_9752_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9752_end_mask_0 = const()[name = string("op_9752_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9752_cast_fp16 = slice_by_index(begin = var_9752_begin_0, end = var_9752_end_0, end_mask = var_9752_end_mask_0, x = coreml_update_state_334)[name = string("op_9752_cast_fp16")]; tensor tile_54 = const()[name = string("tile_54"), val = tensor([1, 1])]; int32 var_9755_axis_0 = const()[name = string("op_9755_axis_0"), val = int32(1)]; tensor var_9755_cast_fp16_0, tensor var_9755_cast_fp16_1 = split(axis = var_9755_axis_0, split_sizes = tile_54, x = var_9752_cast_fp16)[name = string("op_9755_cast_fp16")]; tensor var_9762_begin_0 = const()[name = string("op_9762_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9762_end_0 = const()[name = string("op_9762_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9762_end_mask_0 = const()[name = string("op_9762_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9762_cast_fp16 = slice_by_index(begin = var_9762_begin_0, end = var_9762_end_0, end_mask = var_9762_end_mask_0, x = coreml_update_state_335)[name = string("op_9762_cast_fp16")]; tensor tile_55 = const()[name = string("tile_55"), val = tensor([1, 1])]; int32 var_9765_axis_0 = const()[name = string("op_9765_axis_0"), val = int32(1)]; tensor var_9765_cast_fp16_0, tensor var_9765_cast_fp16_1 = split(axis = var_9765_axis_0, split_sizes = tile_55, x = var_9762_cast_fp16)[name = string("op_9765_cast_fp16")]; tensor var_9768_split_sizes_0 = const()[name = string("op_9768_split_sizes_0"), val = tensor([8, 8])]; int32 var_9768_axis_0 = const()[name = string("op_9768_axis_0"), val = int32(1)]; tensor var_9768_0, tensor var_9768_1 = split(axis = var_9768_axis_0, split_sizes = var_9768_split_sizes_0, x = query_states_165_cast_fp16)[name = string("op_9768")]; bool attn_weights_433_transpose_x_0 = const()[name = string("attn_weights_433_transpose_x_0"), val = bool(false)]; bool attn_weights_433_transpose_y_0 = const()[name = string("attn_weights_433_transpose_y_0"), val = bool(false)]; tensor attn_weights_433_cast_fp16 = matmul(transpose_x = attn_weights_433_transpose_x_0, transpose_y = attn_weights_433_transpose_y_0, x = var_9755_cast_fp16_0, y = var_9768_0)[name = string("attn_weights_433_cast_fp16")]; fp16 var_9771_to_fp16 = const()[name = string("op_9771_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_435_cast_fp16 = mul(x = attn_weights_433_cast_fp16, y = var_9771_to_fp16)[name = string("attn_weights_435_cast_fp16")]; tensor attn_weights_437_cast_fp16 = add(x = attn_weights_435_cast_fp16, y = attn_mask_1)[name = string("attn_weights_437_cast_fp16")]; int32 var_9775 = const()[name = string("op_9775"), val = int32(-2)]; tensor attn_weights_439_cast_fp16 = softmax(axis = var_9775, x = attn_weights_437_cast_fp16)[name = string("attn_weights_439_cast_fp16")]; bool var_9781_transpose_x_1 = const()[name = string("op_9781_transpose_x_1"), val = bool(true)]; bool var_9781_transpose_y_1 = const()[name = string("op_9781_transpose_y_1"), val = bool(false)]; tensor var_9781_cast_fp16 = matmul(transpose_x = var_9781_transpose_x_1, transpose_y = var_9781_transpose_y_1, x = attn_weights_439_cast_fp16, y = var_9765_cast_fp16_0)[name = string("op_9781_cast_fp16")]; bool attn_weights_441_transpose_x_0 = const()[name = string("attn_weights_441_transpose_x_0"), val = bool(false)]; bool attn_weights_441_transpose_y_0 = const()[name = string("attn_weights_441_transpose_y_0"), val = bool(false)]; tensor attn_weights_441_cast_fp16 = matmul(transpose_x = attn_weights_441_transpose_x_0, transpose_y = attn_weights_441_transpose_y_0, x = var_9755_cast_fp16_1, y = var_9768_1)[name = string("attn_weights_441_cast_fp16")]; fp16 var_9783_to_fp16 = const()[name = string("op_9783_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_443_cast_fp16 = mul(x = attn_weights_441_cast_fp16, y = var_9783_to_fp16)[name = string("attn_weights_443_cast_fp16")]; tensor attn_weights_445_cast_fp16 = add(x = attn_weights_443_cast_fp16, y = attn_mask_1)[name = string("attn_weights_445_cast_fp16")]; int32 var_9787 = const()[name = string("op_9787"), val = int32(-2)]; tensor attn_weights_cast_fp16 = softmax(axis = var_9787, x = attn_weights_445_cast_fp16)[name = string("attn_weights_cast_fp16")]; bool attn_output_217_transpose_x_1 = const()[name = string("attn_output_217_transpose_x_1"), val = bool(true)]; bool attn_output_217_transpose_y_1 = const()[name = string("attn_output_217_transpose_y_1"), val = bool(false)]; tensor attn_output_217_cast_fp16 = matmul(transpose_x = attn_output_217_transpose_x_1, transpose_y = attn_output_217_transpose_y_1, x = attn_weights_cast_fp16, y = var_9765_cast_fp16_1)[name = string("attn_output_217_cast_fp16")]; int32 var_9795 = const()[name = string("op_9795"), val = int32(1)]; bool attn_output_219_interleave_0 = const()[name = string("attn_output_219_interleave_0"), val = bool(false)]; tensor attn_output_219_cast_fp16 = concat(axis = var_9795, interleave = attn_output_219_interleave_0, values = (var_9781_cast_fp16, attn_output_217_cast_fp16))[name = string("attn_output_219_cast_fp16")]; tensor var_9799_perm_0 = const()[name = string("op_9799_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_335x = const()[name = string("concat_335x"), val = tensor([1, 2048, 1, -1])]; tensor var_9799_cast_fp16 = transpose(perm = var_9799_perm_0, x = attn_output_219_cast_fp16)[name = string("transpose_430")]; tensor attn_output_cast_fp16 = reshape(shape = concat_335x, x = var_9799_cast_fp16)[name = string("attn_output_cast_fp16")]; tensor layers_27_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1628492864)))]; tensor hidden_states_273_strides_0 = const()[name = string("hidden_states_273_strides_0"), val = tensor([1, 1])]; string hidden_states_273_pad_type_0 = const()[name = string("hidden_states_273_pad_type_0"), val = string("valid")]; tensor hidden_states_273_pad_0 = const()[name = string("hidden_states_273_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_273_dilations_0 = const()[name = string("hidden_states_273_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_273_groups_0 = const()[name = string("hidden_states_273_groups_0"), val = int32(1)]; tensor hidden_states_273_cast_fp16 = conv(dilations = hidden_states_273_dilations_0, groups = hidden_states_273_groups_0, pad = hidden_states_273_pad_0, pad_type = hidden_states_273_pad_type_0, strides = hidden_states_273_strides_0, weight = layers_27_self_attn_o_proj_weight_to_fp16, x = attn_output_cast_fp16)[name = string("hidden_states_273_cast_fp16")]; tensor hidden_states_275_cast_fp16 = add(x = hidden_states_269_cast_fp16, y = hidden_states_273_cast_fp16)[name = string("hidden_states_275_cast_fp16")]; fp16 const_278_promoted_to_fp16 = const()[name = string("const_278_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9832_cast_fp16 = mul(x = hidden_states_275_cast_fp16, y = const_278_promoted_to_fp16)[name = string("op_9832_cast_fp16")]; int32 var_9830 = const()[name = string("op_9830"), val = int32(1)]; bool doubled_221_interleave_0 = const()[name = string("doubled_221_interleave_0"), val = bool(false)]; tensor doubled_221_cast_fp16 = concat(axis = var_9830, interleave = doubled_221_interleave_0, values = (hidden_states_275_cast_fp16, var_9832_cast_fp16))[name = string("doubled_221_cast_fp16")]; tensor out_111_axes_0 = const()[name = string("out_111_axes_0"), val = tensor([1])]; tensor out_111_gamma_0_to_fp16 = const()[name = string("out_111_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636881536)))]; fp16 var_9842_to_fp16 = const()[name = string("op_9842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_111_cast_fp16 = layer_norm(axes = out_111_axes_0, epsilon = var_9842_to_fp16, gamma = out_111_gamma_0_to_fp16, x = doubled_221_cast_fp16)[name = string("out_111_cast_fp16")]; tensor var_9853_split_sizes_0 = const()[name = string("op_9853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9853_axis_0 = const()[name = string("op_9853_axis_0"), val = int32(1)]; tensor var_9853_cast_fp16_0, tensor var_9853_cast_fp16_1 = split(axis = var_9853_axis_0, split_sizes = var_9853_split_sizes_0, x = out_111_cast_fp16)[name = string("op_9853_cast_fp16")]; tensor layers_27_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636889792)))]; tensor input_strides_0 = const()[name = string("input_strides_0"), val = tensor([1, 1])]; string input_pad_type_0 = const()[name = string("input_pad_type_0"), val = string("valid")]; tensor input_pad_0 = const()[name = string("input_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_dilations_0 = const()[name = string("input_dilations_0"), val = tensor([1, 1])]; int32 input_groups_0 = const()[name = string("input_groups_0"), val = int32(1)]; tensor input_cast_fp16 = conv(dilations = input_dilations_0, groups = input_groups_0, pad = input_pad_0, pad_type = input_pad_type_0, strides = input_strides_0, weight = layers_27_mlp_gate_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("input_cast_fp16")]; tensor var_9870_cast_fp16 = silu(x = input_cast_fp16)[name = string("op_9870_cast_fp16")]; tensor layers_27_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1662055680)))]; tensor var_9876_strides_0 = const()[name = string("op_9876_strides_0"), val = tensor([1, 1])]; string var_9876_pad_type_0 = const()[name = string("op_9876_pad_type_0"), val = string("valid")]; tensor var_9876_pad_0 = const()[name = string("op_9876_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9876_dilations_0 = const()[name = string("op_9876_dilations_0"), val = tensor([1, 1])]; int32 var_9876_groups_0 = const()[name = string("op_9876_groups_0"), val = int32(1)]; tensor var_9876_cast_fp16 = conv(dilations = var_9876_dilations_0, groups = var_9876_groups_0, pad = var_9876_pad_0, pad_type = var_9876_pad_type_0, strides = var_9876_strides_0, weight = layers_27_mlp_up_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("op_9876_cast_fp16")]; tensor x_cast_fp16 = mul(x = var_9870_cast_fp16, y = var_9876_cast_fp16)[name = string("x_cast_fp16")]; tensor layers_27_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1687221568)))]; tensor hidden_states_277_strides_0 = const()[name = string("hidden_states_277_strides_0"), val = tensor([1, 1])]; string hidden_states_277_pad_type_0 = const()[name = string("hidden_states_277_pad_type_0"), val = string("valid")]; tensor hidden_states_277_pad_0 = const()[name = string("hidden_states_277_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_277_dilations_0 = const()[name = string("hidden_states_277_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_277_groups_0 = const()[name = string("hidden_states_277_groups_0"), val = int32(1)]; tensor hidden_states_277_cast_fp16 = conv(dilations = hidden_states_277_dilations_0, groups = hidden_states_277_groups_0, pad = hidden_states_277_pad_0, pad_type = hidden_states_277_pad_type_0, strides = hidden_states_277_strides_0, weight = layers_27_mlp_down_proj_weight_to_fp16, x = x_cast_fp16)[name = string("hidden_states_277_cast_fp16")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_275_cast_fp16, y = hidden_states_277_cast_fp16)[name = string("hidden_states_cast_fp16")]; fp16 const_280_promoted_to_fp16 = const()[name = string("const_280_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9894_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_280_promoted_to_fp16)[name = string("op_9894_cast_fp16")]; int32 var_9892 = const()[name = string("op_9892"), val = int32(1)]; bool doubled_225_interleave_0 = const()[name = string("doubled_225_interleave_0"), val = bool(false)]; tensor doubled_225_cast_fp16 = concat(axis = var_9892, interleave = doubled_225_interleave_0, values = (hidden_states_cast_fp16, var_9894_cast_fp16))[name = string("doubled_225_cast_fp16")]; tensor out_axes_0 = const()[name = string("out_axes_0"), val = tensor([1])]; tensor out_gamma_0_to_fp16 = const()[name = string("out_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1712387456)))]; fp16 var_9904_to_fp16 = const()[name = string("op_9904_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_cast_fp16 = layer_norm(axes = out_axes_0, epsilon = var_9904_to_fp16, gamma = out_gamma_0_to_fp16, x = doubled_225_cast_fp16)[name = string("out_cast_fp16")]; tensor var_9915_split_sizes_0 = const()[name = string("op_9915_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9915_axis_0 = const()[name = string("op_9915_axis_0"), val = int32(1)]; tensor hidden_states, tensor var_9915_cast_fp16_1 = split(axis = var_9915_axis_0, split_sizes = var_9915_split_sizes_0, x = out_cast_fp16)[name = string("op_9915_cast_fp16")]; } -> (hidden_states); func length_8(tensor inputs_embeds, state> key_cache, tensor position_id, tensor position_index_seed, state> value_cache) { tensor layers_1_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524992))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524416))))[name = string("layers_1_self_attn_v_proj_weight_cast_fp16")]; tensor layers_1_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(525312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13120640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13108288))))[name = string("layers_1_mlp_up_proj_weight_cast_fp16")]; tensor layers_2_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13126848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651776))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651200))))[name = string("layers_2_self_attn_v_proj_weight_cast_fp16")]; tensor layers_2_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13652096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26247424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26235072))))[name = string("layers_2_mlp_up_proj_weight_cast_fp16")]; tensor layers_3_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26253632))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778560))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26777984))))[name = string("layers_3_self_attn_v_proj_weight_cast_fp16")]; tensor layers_3_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30977408))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30973248))))[name = string("layers_3_self_attn_o_proj_weight_cast_fp16")]; tensor layers_3_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30979520))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43566656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43562496))))[name = string("layers_3_mlp_down_proj_weight_cast_fp16")]; tensor layers_4_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43568768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093696))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093120))))[name = string("layers_4_self_attn_v_proj_weight_cast_fp16")]; tensor layers_4_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44094016))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48292544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48288384))))[name = string("layers_4_self_attn_o_proj_weight_cast_fp16")]; tensor layers_4_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48294656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60877632))))[name = string("layers_4_mlp_gate_proj_weight_cast_fp16")]; tensor layers_4_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60896192))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73491520))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73479168))))[name = string("layers_4_mlp_up_proj_weight_cast_fp16")]; tensor layers_4_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73497728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86084864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86080704))))[name = string("layers_4_mlp_down_proj_weight_cast_fp16")]; tensor layers_5_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86086976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611328))))[name = string("layers_5_self_attn_v_proj_weight_cast_fp16")]; tensor layers_5_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86612224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90810752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90806592))))[name = string("layers_5_self_attn_o_proj_weight_cast_fp16")]; tensor layers_5_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90812864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103408192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103395840))))[name = string("layers_5_mlp_up_proj_weight_cast_fp16")]; tensor layers_5_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103414400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116001536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115997376))))[name = string("layers_5_mlp_down_proj_weight_cast_fp16")]; tensor layers_6_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116003648))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528576))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528000))))[name = string("layers_6_self_attn_v_proj_weight_cast_fp16")]; tensor layers_6_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528896))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120727424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120723264))))[name = string("layers_6_self_attn_o_proj_weight_cast_fp16")]; tensor layers_6_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120729536))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133324864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133312512))))[name = string("layers_6_mlp_gate_proj_weight_cast_fp16")]; tensor layers_6_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133331072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145926400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145914048))))[name = string("layers_6_mlp_up_proj_weight_cast_fp16")]; tensor layers_6_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145932608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158519744))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158515584))))[name = string("layers_6_mlp_down_proj_weight_cast_fp16")]; tensor layers_7_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158521856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046208))))[name = string("layers_7_self_attn_v_proj_weight_cast_fp16")]; tensor layers_7_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159047104))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163245632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163241472))))[name = string("layers_7_self_attn_o_proj_weight_cast_fp16")]; tensor layers_7_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163247744))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175843072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175830720))))[name = string("layers_7_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175849280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374208))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176373632))))[name = string("layers_8_self_attn_v_proj_weight_cast_fp16")]; tensor layers_8_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374528))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180573056))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180568896))))[name = string("layers_8_self_attn_o_proj_weight_cast_fp16")]; tensor layers_8_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180575168))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193170496))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193158144))))[name = string("layers_8_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193176704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205772032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205759680))))[name = string("layers_8_mlp_up_proj_weight_cast_fp16")]; tensor layers_8_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205778240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218365376))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218361216))))[name = string("layers_8_mlp_down_proj_weight_cast_fp16")]; tensor layers_9_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218367488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218891840))))[name = string("layers_9_self_attn_v_proj_weight_cast_fp16")]; tensor layers_9_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223091264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223087104))))[name = string("layers_9_self_attn_o_proj_weight_cast_fp16")]; tensor layers_9_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223093376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235688704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235676352))))[name = string("layers_9_mlp_gate_proj_weight_cast_fp16")]; tensor layers_9_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235694912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248290240))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248277888))))[name = string("layers_9_mlp_up_proj_weight_cast_fp16")]; tensor layers_9_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248296448))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260883584))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260879424))))[name = string("layers_9_mlp_down_proj_weight_cast_fp16")]; tensor layers_10_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260885696))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410048))))[name = string("layers_10_self_attn_v_proj_weight_cast_fp16")]; tensor layers_10_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410944))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265609472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265605312))))[name = string("layers_10_self_attn_o_proj_weight_cast_fp16")]; tensor layers_10_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265611584))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278206912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278194560))))[name = string("layers_10_mlp_gate_proj_weight_cast_fp16")]; tensor layers_10_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278213120))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290808448))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290796096))))[name = string("layers_10_mlp_up_proj_weight_cast_fp16")]; tensor layers_10_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290814656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303401792))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303397632))))[name = string("layers_10_mlp_down_proj_weight_cast_fp16")]; tensor layers_11_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303403904))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307602432))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307598272))))[name = string("layers_11_self_attn_q_proj_weight_cast_fp16")]; tensor layers_11_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307604544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308128896))))[name = string("layers_11_self_attn_k_proj_weight_cast_fp16")]; tensor layers_11_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129792))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654720))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654144))))[name = string("layers_11_self_attn_v_proj_weight_cast_fp16")]; tensor layers_11_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308655040))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312853568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312849408))))[name = string("layers_11_self_attn_o_proj_weight_cast_fp16")]; tensor layers_11_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312855680))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325451008))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325438656))))[name = string("layers_11_mlp_gate_proj_weight_cast_fp16")]; tensor layers_11_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325457216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338052544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338040192))))[name = string("layers_11_mlp_up_proj_weight_cast_fp16")]; tensor layers_11_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338058752))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350645888))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350641728))))[name = string("layers_11_mlp_down_proj_weight_cast_fp16")]; tensor layers_12_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350648000))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354846528))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354842368))))[name = string("layers_12_self_attn_q_proj_weight_cast_fp16")]; tensor layers_12_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354848640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355372992))))[name = string("layers_12_self_attn_k_proj_weight_cast_fp16")]; tensor layers_12_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373888))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898816))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898240))))[name = string("layers_12_self_attn_v_proj_weight_cast_fp16")]; tensor layers_12_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355899136))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360097664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360093504))))[name = string("layers_12_self_attn_o_proj_weight_cast_fp16")]; tensor layers_12_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360099776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372695104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372682752))))[name = string("layers_12_mlp_gate_proj_weight_cast_fp16")]; tensor layers_12_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372701312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385296640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385284288))))[name = string("layers_12_mlp_up_proj_weight_cast_fp16")]; tensor layers_12_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385302848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397885824))))[name = string("layers_12_mlp_down_proj_weight_cast_fp16")]; tensor layers_13_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397892096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402090624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402086464))))[name = string("layers_13_self_attn_q_proj_weight_cast_fp16")]; tensor layers_13_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402092736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617088))))[name = string("layers_13_self_attn_k_proj_weight_cast_fp16")]; tensor layers_13_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617984))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142336))))[name = string("layers_13_self_attn_v_proj_weight_cast_fp16")]; tensor layers_13_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403143232))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407341760))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407337600))))[name = string("layers_13_self_attn_o_proj_weight_cast_fp16")]; tensor layers_13_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407343872))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419939200))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419926848))))[name = string("layers_13_mlp_gate_proj_weight_cast_fp16")]; tensor layers_13_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419945408))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432532544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432528384))))[name = string("layers_13_mlp_down_proj_weight_cast_fp16")]; tensor layers_14_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432534656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436733184))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436729024))))[name = string("layers_14_self_attn_q_proj_weight_cast_fp16")]; tensor layers_14_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436735296))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437259648))))[name = string("layers_14_self_attn_v_proj_weight_cast_fp16")]; tensor layers_14_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441459072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441454912))))[name = string("layers_14_self_attn_o_proj_weight_cast_fp16")]; tensor layers_14_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441461184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454056512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454044160))))[name = string("layers_14_mlp_gate_proj_weight_cast_fp16")]; tensor layers_14_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454062720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466658048))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466645696))))[name = string("layers_14_mlp_up_proj_weight_cast_fp16")]; tensor layers_14_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466664256))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479251392))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479247232))))[name = string("layers_14_mlp_down_proj_weight_cast_fp16")]; tensor layers_15_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479253504))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483452032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483447872))))[name = string("layers_15_self_attn_q_proj_weight_cast_fp16")]; tensor layers_15_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483454144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483978496))))[name = string("layers_15_self_attn_k_proj_weight_cast_fp16")]; tensor layers_15_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504320))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484503744))))[name = string("layers_15_self_attn_v_proj_weight_cast_fp16")]; tensor layers_15_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488703168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488699008))))[name = string("layers_15_self_attn_o_proj_weight_cast_fp16")]; tensor layers_15_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488705280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501300608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501288256))))[name = string("layers_15_mlp_gate_proj_weight_cast_fp16")]; tensor layers_15_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501306816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513902144))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513889792))))[name = string("layers_15_mlp_up_proj_weight_cast_fp16")]; tensor layers_15_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513908352))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526495488))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526491328))))[name = string("layers_15_mlp_down_proj_weight_cast_fp16")]; tensor layers_16_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526497600))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530696128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530691968))))[name = string("layers_16_self_attn_q_proj_weight_cast_fp16")]; tensor layers_16_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530698240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531222592))))[name = string("layers_16_self_attn_k_proj_weight_cast_fp16")]; tensor layers_16_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531747840))))[name = string("layers_16_self_attn_v_proj_weight_cast_fp16")]; tensor layers_16_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535947264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535943104))))[name = string("layers_16_self_attn_o_proj_weight_cast_fp16")]; tensor layers_16_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535949376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548536512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548532352))))[name = string("layers_16_mlp_down_proj_weight_cast_fp16")]; tensor layers_17_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548538624))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552737152))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552732992))))[name = string("layers_17_self_attn_q_proj_weight_cast_fp16")]; tensor layers_17_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552739264))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553263616))))[name = string("layers_17_self_attn_k_proj_weight_cast_fp16")]; tensor layers_17_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264512))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553788864))))[name = string("layers_17_self_attn_v_proj_weight_cast_fp16")]; tensor layers_17_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789760))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557988288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557984128))))[name = string("layers_17_self_attn_o_proj_weight_cast_fp16")]; tensor layers_17_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557990400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570585728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570573376))))[name = string("layers_17_mlp_gate_proj_weight_cast_fp16")]; tensor layers_17_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570591936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583187264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583174912))))[name = string("layers_17_mlp_up_proj_weight_cast_fp16")]; tensor layers_17_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583193472))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595780608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595776448))))[name = string("layers_17_mlp_down_proj_weight_cast_fp16")]; tensor layers_18_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595782720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599981248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599977088))))[name = string("layers_18_self_attn_q_proj_weight_cast_fp16")]; tensor layers_18_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599983360))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600507712))))[name = string("layers_18_self_attn_k_proj_weight_cast_fp16")]; tensor layers_18_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601032960))))[name = string("layers_18_self_attn_v_proj_weight_cast_fp16")]; tensor layers_18_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605232384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605228224))))[name = string("layers_18_self_attn_o_proj_weight_cast_fp16")]; tensor layers_18_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605234496))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617829824))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617817472))))[name = string("layers_18_mlp_gate_proj_weight_cast_fp16")]; tensor layers_18_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617836032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630431360))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630419008))))[name = string("layers_18_mlp_up_proj_weight_cast_fp16")]; tensor layers_18_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630437568))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643024704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643020544))))[name = string("layers_18_mlp_down_proj_weight_cast_fp16")]; tensor layers_19_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643026816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647225344))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647221184))))[name = string("layers_19_self_attn_q_proj_weight_cast_fp16")]; tensor layers_19_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647227456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647751808))))[name = string("layers_19_self_attn_k_proj_weight_cast_fp16")]; tensor layers_19_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660348032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660335680))))[name = string("layers_19_mlp_gate_proj_weight_cast_fp16")]; tensor layers_19_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660354240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672949568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672937216))))[name = string("layers_19_mlp_up_proj_weight_cast_fp16")]; tensor layers_19_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672955776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685542912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685538752))))[name = string("layers_19_mlp_down_proj_weight_cast_fp16")]; tensor layers_20_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685545024))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689743552))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689739392))))[name = string("layers_20_self_attn_q_proj_weight_cast_fp16")]; tensor layers_20_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689745664))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270592))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270016))))[name = string("layers_20_self_attn_k_proj_weight_cast_fp16")]; tensor layers_20_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694469440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694465280))))[name = string("layers_20_self_attn_o_proj_weight_cast_fp16")]; tensor layers_20_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694471552))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707066880))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707054528))))[name = string("layers_20_mlp_gate_proj_weight_cast_fp16")]; tensor layers_20_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707073088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719660224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719656064))))[name = string("layers_20_mlp_down_proj_weight_cast_fp16")]; tensor layers_21_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719662336))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723860864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723856704))))[name = string("layers_21_self_attn_q_proj_weight_cast_fp16")]; tensor layers_21_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723862976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387328))))[name = string("layers_21_self_attn_k_proj_weight_cast_fp16")]; tensor layers_21_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724388224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728586752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728582592))))[name = string("layers_21_self_attn_o_proj_weight_cast_fp16")]; tensor layers_21_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728588864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741184192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741171840))))[name = string("layers_21_mlp_gate_proj_weight_cast_fp16")]; tensor layers_21_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741190400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753785728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753773376))))[name = string("layers_21_mlp_up_proj_weight_cast_fp16")]; tensor layers_21_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753791936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766379072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766374912))))[name = string("layers_21_mlp_down_proj_weight_cast_fp16")]; tensor layers_22_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766381184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770579712))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770575552))))[name = string("layers_22_self_attn_q_proj_weight_cast_fp16")]; tensor layers_22_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770581824))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106176))))[name = string("layers_22_self_attn_k_proj_weight_cast_fp16")]; tensor layers_22_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771107072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783702400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783690048))))[name = string("layers_22_mlp_gate_proj_weight_cast_fp16")]; tensor layers_22_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783708608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796303936))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796291584))))[name = string("layers_22_mlp_up_proj_weight_cast_fp16")]; tensor layers_22_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796310144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808897280))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808893120))))[name = string("layers_22_mlp_down_proj_weight_cast_fp16")]; tensor layers_23_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808899392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813097920))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813093760))))[name = string("layers_23_self_attn_q_proj_weight_cast_fp16")]; tensor layers_23_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813100032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624960))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624384))))[name = string("layers_23_self_attn_k_proj_weight_cast_fp16")]; tensor layers_23_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813625280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817823808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817819648))))[name = string("layers_23_self_attn_o_proj_weight_cast_fp16")]; tensor layers_23_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817825920))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830421248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830408896))))[name = string("layers_23_mlp_gate_proj_weight_cast_fp16")]; tensor layers_23_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830427456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843022784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843010432))))[name = string("layers_23_mlp_up_proj_weight_cast_fp16")]; tensor layers_23_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843028992))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855616128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855611968))))[name = string("layers_23_mlp_down_proj_weight_cast_fp16")]; tensor layers_24_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855618240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859816768))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859812608))))[name = string("layers_24_self_attn_q_proj_weight_cast_fp16")]; tensor layers_24_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859818880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343232))))[name = string("layers_24_self_attn_k_proj_weight_cast_fp16")]; tensor layers_24_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860344128))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864542656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864538496))))[name = string("layers_24_self_attn_o_proj_weight_cast_fp16")]; tensor layers_24_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864544768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877140096))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877127744))))[name = string("layers_24_mlp_gate_proj_weight_cast_fp16")]; tensor layers_24_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877146304))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889741632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889729280))))[name = string("layers_24_mlp_up_proj_weight_cast_fp16")]; tensor layers_24_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889747840))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902334976))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902330816))))[name = string("layers_24_mlp_down_proj_weight_cast_fp16")]; tensor layers_25_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902337088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906535616))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906531456))))[name = string("layers_25_self_attn_q_proj_weight_cast_fp16")]; tensor layers_25_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906537728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062080))))[name = string("layers_25_self_attn_k_proj_weight_cast_fp16")]; tensor layers_25_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911261504))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911257344))))[name = string("layers_25_self_attn_o_proj_weight_cast_fp16")]; tensor layers_25_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911263616))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923858944))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923846592))))[name = string("layers_25_mlp_gate_proj_weight_cast_fp16")]; tensor layers_25_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923865152))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936460480))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936448128))))[name = string("layers_25_mlp_up_proj_weight_cast_fp16")]; tensor layers_26_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936466688))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940665216))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940661056))))[name = string("layers_26_self_attn_q_proj_weight_cast_fp16")]; tensor layers_26_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940667328))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192256))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941191680))))[name = string("layers_26_self_attn_k_proj_weight_cast_fp16")]; tensor layers_26_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192576))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945391104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945386944))))[name = string("layers_26_self_attn_o_proj_weight_cast_fp16")]; tensor layers_26_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945393216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957988544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957976192))))[name = string("layers_26_mlp_gate_proj_weight_cast_fp16")]; int32 var_765 = const()[name = string("op_765"), val = int32(0)]; tensor var_766 = mul(x = position_index_seed, y = var_765)[name = string("op_766")]; int32 var_768 = const()[name = string("op_768"), val = int32(1)]; tensor ones = add(x = var_766, y = var_768)[name = string("ones")]; int32 var_770 = const()[name = string("op_770"), val = int32(0)]; bool var_772_exclusive_0 = const()[name = string("op_772_exclusive_0"), val = bool(false)]; bool var_772_reverse_0 = const()[name = string("op_772_reverse_0"), val = bool(false)]; tensor var_772 = cumsum(axis = var_770, exclusive = var_772_exclusive_0, reverse = var_772_reverse_0, x = ones)[name = string("op_772")]; int32 var_774 = const()[name = string("op_774"), val = int32(1)]; tensor position_offsets = sub(x = var_772, y = var_774)[name = string("position_offsets")]; tensor position_ids_1 = add(x = position_offsets, y = position_id)[name = string("position_ids_1")]; bool var_784_keep_dims_0 = const()[name = string("op_784_keep_dims_0"), val = bool(false)]; int32 var_784 = reduce_sum(keep_dims = var_784_keep_dims_0, x = ones)[name = string("op_784")]; int32 var_786 = const()[name = string("op_786"), val = int32(1)]; int32 offset = sub(x = var_784, y = var_786)[name = string("offset")]; tensor var_789 = add(x = position_id, y = offset)[name = string("op_789")]; int32 var_791 = const()[name = string("op_791"), val = int32(1)]; tensor cache_position_end = add(x = var_789, y = var_791)[name = string("cache_position_end")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor greater_equal_0 = greater_equal(x = position_ids_1, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(32768)]; tensor add_0 = add(x = position_ids_1, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = position_ids_1, b = add_0, cond = greater_equal_0)[name = string("select_0")]; tensor rope_emb_cos_cached_to_fp16 = const()[name = string("rope_emb_cos_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957994752)))]; int32 cos_1_batch_dims_0 = const()[name = string("cos_1_batch_dims_0"), val = int32(0)]; bool cos_1_validate_indices_0 = const()[name = string("cos_1_validate_indices_0"), val = bool(false)]; int32 greater_equal_4_y_0 = const()[name = string("greater_equal_4_y_0"), val = int32(0)]; tensor greater_equal_4 = greater_equal(x = select_0, y = greater_equal_4_y_0)[name = string("greater_equal_4")]; int32 slice_by_index_4 = const()[name = string("slice_by_index_4"), val = int32(32768)]; tensor add_4 = add(x = select_0, y = slice_by_index_4)[name = string("add_4")]; tensor select_4 = select(a = select_0, b = add_4, cond = greater_equal_4)[name = string("select_4")]; int32 cos_1_cast_fp16_axis_2 = const()[name = string("cos_1_cast_fp16_axis_2"), val = int32(0)]; tensor cos_1_cast_fp16 = gather(axis = cos_1_cast_fp16_axis_2, batch_dims = cos_1_batch_dims_0, indices = select_4, validate_indices = cos_1_validate_indices_0, x = rope_emb_cos_cached_to_fp16)[name = string("cos_1_cast_fp16")]; tensor rope_emb_sin_cached_to_fp16 = const()[name = string("rope_emb_sin_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(966383424)))]; int32 sin_1_batch_dims_0 = const()[name = string("sin_1_batch_dims_0"), val = int32(0)]; bool sin_1_validate_indices_0 = const()[name = string("sin_1_validate_indices_0"), val = bool(false)]; int32 sin_1_cast_fp16_axis_2 = const()[name = string("sin_1_cast_fp16_axis_2"), val = int32(0)]; tensor sin_1_cast_fp16 = gather(axis = sin_1_cast_fp16_axis_2, batch_dims = sin_1_batch_dims_0, indices = select_4, validate_indices = sin_1_validate_indices_0, x = rope_emb_sin_cached_to_fp16)[name = string("sin_1_cast_fp16")]; tensor var_865_perm_0 = const()[name = string("op_865_perm_0"), val = tensor([-1, -2])]; tensor var_867_axes_0 = const()[name = string("op_867_axes_0"), val = tensor([0])]; tensor var_865_cast_fp16 = transpose(perm = var_865_perm_0, x = cos_1_cast_fp16)[name = string("transpose_257")]; tensor var_867_cast_fp16 = expand_dims(axes = var_867_axes_0, x = var_865_cast_fp16)[name = string("op_867_cast_fp16")]; tensor var_869_axes_0 = const()[name = string("op_869_axes_0"), val = tensor([0])]; tensor var_869_cast_fp16 = expand_dims(axes = var_869_axes_0, x = var_867_cast_fp16)[name = string("op_869_cast_fp16")]; tensor var_874_perm_0 = const()[name = string("op_874_perm_0"), val = tensor([-1, -2])]; tensor var_876_axes_0 = const()[name = string("op_876_axes_0"), val = tensor([0])]; tensor var_874_cast_fp16 = transpose(perm = var_874_perm_0, x = sin_1_cast_fp16)[name = string("transpose_256")]; tensor var_876_cast_fp16 = expand_dims(axes = var_876_axes_0, x = var_874_cast_fp16)[name = string("op_876_cast_fp16")]; tensor var_878_axes_0 = const()[name = string("op_878_axes_0"), val = tensor([0])]; tensor var_878_cast_fp16 = expand_dims(axes = var_878_axes_0, x = var_876_cast_fp16)[name = string("op_878_cast_fp16")]; string position_ids_1_to_uint16_dtype_0 = const()[name = string("position_ids_1_to_uint16_dtype_0"), val = string("uint16")]; tensor causal_mask = const()[name = string("causal_mask"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(974772096)))]; int32 mask_axis_0 = const()[name = string("mask_axis_0"), val = int32(1)]; int32 mask_batch_dims_0 = const()[name = string("mask_batch_dims_0"), val = int32(0)]; bool mask_validate_indices_0 = const()[name = string("mask_validate_indices_0"), val = bool(false)]; tensor position_ids_1_to_uint16 = cast(dtype = position_ids_1_to_uint16_dtype_0, x = position_ids_1)[name = string("cast_5")]; tensor mask_cast_uint16 = gather(axis = mask_axis_0, batch_dims = mask_batch_dims_0, indices = position_ids_1_to_uint16, validate_indices = mask_validate_indices_0, x = causal_mask)[name = string("mask_cast_uint16")]; tensor var_895_axes_0 = const()[name = string("op_895_axes_0"), val = tensor([0])]; tensor var_895 = expand_dims(axes = var_895_axes_0, x = mask_cast_uint16)[name = string("op_895")]; tensor attn_mask_1_axes_0 = const()[name = string("attn_mask_1_axes_0"), val = tensor([0])]; tensor attn_mask_1 = expand_dims(axes = attn_mask_1_axes_0, x = var_895)[name = string("attn_mask_1")]; string inputs_embeds_to_fp16_dtype_0 = const()[name = string("inputs_embeds_to_fp16_dtype_0"), val = string("fp16")]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor inputs_embeds_to_fp16 = cast(dtype = inputs_embeds_to_fp16_dtype_0, x = inputs_embeds)[name = string("cast_4")]; tensor var_906_cast_fp16 = mul(x = inputs_embeds_to_fp16, y = const_0_promoted_to_fp16)[name = string("op_906_cast_fp16")]; int32 var_904 = const()[name = string("op_904"), val = int32(1)]; bool doubled_1_interleave_0 = const()[name = string("doubled_1_interleave_0"), val = bool(false)]; tensor doubled_1_cast_fp16 = concat(axis = var_904, interleave = doubled_1_interleave_0, values = (inputs_embeds_to_fp16, var_906_cast_fp16))[name = string("doubled_1_cast_fp16")]; tensor out_1_axes_0 = const()[name = string("out_1_axes_0"), val = tensor([1])]; tensor out_1_gamma_0_to_fp16 = const()[name = string("out_1_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983160768)))]; fp16 var_916_to_fp16 = const()[name = string("op_916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_1_cast_fp16 = layer_norm(axes = out_1_axes_0, epsilon = var_916_to_fp16, gamma = out_1_gamma_0_to_fp16, x = doubled_1_cast_fp16)[name = string("out_1_cast_fp16")]; tensor var_927_split_sizes_0 = const()[name = string("op_927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_927_axis_0 = const()[name = string("op_927_axis_0"), val = int32(1)]; tensor var_927_cast_fp16_0, tensor var_927_cast_fp16_1 = split(axis = var_927_axis_0, split_sizes = var_927_split_sizes_0, x = out_1_cast_fp16)[name = string("op_927_cast_fp16")]; tensor layers_0_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983169024)))]; tensor query_states_1_strides_0 = const()[name = string("query_states_1_strides_0"), val = tensor([1, 1])]; string query_states_1_pad_type_0 = const()[name = string("query_states_1_pad_type_0"), val = string("valid")]; tensor query_states_1_pad_0 = const()[name = string("query_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_1_dilations_0 = const()[name = string("query_states_1_dilations_0"), val = tensor([1, 1])]; int32 query_states_1_groups_0 = const()[name = string("query_states_1_groups_0"), val = int32(1)]; tensor query_states_1_cast_fp16 = conv(dilations = query_states_1_dilations_0, groups = query_states_1_groups_0, pad = query_states_1_pad_0, pad_type = query_states_1_pad_type_0, strides = query_states_1_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("query_states_1_cast_fp16")]; tensor layers_0_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(991557696)))]; tensor key_states_1_strides_0 = const()[name = string("key_states_1_strides_0"), val = tensor([1, 1])]; string key_states_1_pad_type_0 = const()[name = string("key_states_1_pad_type_0"), val = string("valid")]; tensor key_states_1_pad_0 = const()[name = string("key_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_1_dilations_0 = const()[name = string("key_states_1_dilations_0"), val = tensor([1, 1])]; int32 key_states_1_groups_0 = const()[name = string("key_states_1_groups_0"), val = int32(1)]; tensor key_states_1_cast_fp16 = conv(dilations = key_states_1_dilations_0, groups = key_states_1_groups_0, pad = key_states_1_pad_0, pad_type = key_states_1_pad_type_0, strides = key_states_1_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("key_states_1_cast_fp16")]; tensor layers_0_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(992606336)))]; tensor value_states_1_strides_0 = const()[name = string("value_states_1_strides_0"), val = tensor([1, 1])]; string value_states_1_pad_type_0 = const()[name = string("value_states_1_pad_type_0"), val = string("valid")]; tensor value_states_1_pad_0 = const()[name = string("value_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_1_dilations_0 = const()[name = string("value_states_1_dilations_0"), val = tensor([1, 1])]; int32 value_states_1_groups_0 = const()[name = string("value_states_1_groups_0"), val = int32(1)]; tensor value_states_1_cast_fp16 = conv(dilations = value_states_1_dilations_0, groups = value_states_1_groups_0, pad = value_states_1_pad_0, pad_type = value_states_1_pad_type_0, strides = value_states_1_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("value_states_1_cast_fp16")]; tensor concat_0x = const()[name = string("concat_0x"), val = tensor([1, 16, 128, -1])]; tensor x_1_cast_fp16 = reshape(shape = concat_0x, x = query_states_1_cast_fp16)[name = string("x_1_cast_fp16")]; tensor concat_1x = const()[name = string("concat_1x"), val = tensor([1, 2, 128, -1])]; tensor var_984_cast_fp16 = reshape(shape = concat_1x, x = key_states_1_cast_fp16)[name = string("op_984_cast_fp16")]; tensor concat_2x = const()[name = string("concat_2x"), val = tensor([1, 2, 128, -1])]; tensor var_991_cast_fp16 = reshape(shape = concat_2x, x = value_states_1_cast_fp16)[name = string("op_991_cast_fp16")]; tensor var_995_cast_fp16 = mul(x = x_1_cast_fp16, y = var_869_cast_fp16)[name = string("op_995_cast_fp16")]; tensor var_996_split_sizes_0 = const()[name = string("op_996_split_sizes_0"), val = tensor([64, 64])]; int32 var_996_axis_0 = const()[name = string("op_996_axis_0"), val = int32(-2)]; tensor var_996_cast_fp16_0, tensor var_996_cast_fp16_1 = split(axis = var_996_axis_0, split_sizes = var_996_split_sizes_0, x = x_1_cast_fp16)[name = string("op_996_cast_fp16")]; fp16 const_2_promoted_to_fp16 = const()[name = string("const_2_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_998_cast_fp16 = mul(x = var_996_cast_fp16_1, y = const_2_promoted_to_fp16)[name = string("op_998_cast_fp16")]; int32 var_1000 = const()[name = string("op_1000"), val = int32(-2)]; bool var_1001_interleave_0 = const()[name = string("op_1001_interleave_0"), val = bool(false)]; tensor var_1001_cast_fp16 = concat(axis = var_1000, interleave = var_1001_interleave_0, values = (var_998_cast_fp16, var_996_cast_fp16_0))[name = string("op_1001_cast_fp16")]; tensor var_1002_cast_fp16 = mul(x = var_1001_cast_fp16, y = var_878_cast_fp16)[name = string("op_1002_cast_fp16")]; tensor query_states_3_cast_fp16 = add(x = var_995_cast_fp16, y = var_1002_cast_fp16)[name = string("query_states_3_cast_fp16")]; tensor var_1008_cast_fp16 = mul(x = var_984_cast_fp16, y = var_869_cast_fp16)[name = string("op_1008_cast_fp16")]; tensor var_1009_split_sizes_0 = const()[name = string("op_1009_split_sizes_0"), val = tensor([64, 64])]; int32 var_1009_axis_0 = const()[name = string("op_1009_axis_0"), val = int32(-2)]; tensor var_1009_cast_fp16_0, tensor var_1009_cast_fp16_1 = split(axis = var_1009_axis_0, split_sizes = var_1009_split_sizes_0, x = var_984_cast_fp16)[name = string("op_1009_cast_fp16")]; fp16 const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1011_cast_fp16 = mul(x = var_1009_cast_fp16_1, y = const_3_promoted_to_fp16)[name = string("op_1011_cast_fp16")]; int32 var_1013 = const()[name = string("op_1013"), val = int32(-2)]; bool var_1014_interleave_0 = const()[name = string("op_1014_interleave_0"), val = bool(false)]; tensor var_1014_cast_fp16 = concat(axis = var_1013, interleave = var_1014_interleave_0, values = (var_1011_cast_fp16, var_1009_cast_fp16_0))[name = string("op_1014_cast_fp16")]; tensor var_1015_cast_fp16 = mul(x = var_1014_cast_fp16, y = var_878_cast_fp16)[name = string("op_1015_cast_fp16")]; tensor key_states_5_cast_fp16 = add(x = var_1008_cast_fp16, y = var_1015_cast_fp16)[name = string("key_states_5_cast_fp16")]; tensor read_state_0 = read_state(input = key_cache)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; int32 concat_5_axis_0 = const()[name = string("concat_5_axis_0"), val = int32(0)]; bool concat_5_interleave_0 = const()[name = string("concat_5_interleave_0"), val = bool(false)]; tensor concat_5 = concat(axis = concat_5_axis_0, interleave = concat_5_interleave_0, values = (expand_dims_0, expand_dims_1, position_id, expand_dims_3))[name = string("concat_5")]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; tensor concat_6_values1_0 = const()[name = string("concat_6_values1_0"), val = tensor([0])]; tensor concat_6_values3_0 = const()[name = string("concat_6_values3_0"), val = tensor([0])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_4, concat_6_values1_0, cache_position_end, concat_6_values3_0))[name = string("concat_6")]; tensor key_states_7_perm_0 = const()[name = string("key_states_7_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_1_stride_0 = const()[name = string("key_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_7_cast_fp16 = transpose(perm = key_states_7_perm_0, x = key_states_5_cast_fp16)[name = string("transpose_255")]; tensor key_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = key_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = key_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_1_squeeze_mask_0, stride = key_cache_internal_tensor_assign_1_stride_0, update = key_states_7_cast_fp16, x = read_state_0)[name = string("key_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_1_cast_fp16, input = key_cache)[name = string("coreml_update_state_112_write_state")]; tensor coreml_update_state_112 = read_state(input = key_cache)[name = string("coreml_update_state_112")]; tensor read_state_1 = read_state(input = value_cache)[name = string("read_state_1")]; tensor value_states_3_perm_0 = const()[name = string("value_states_3_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_1_stride_0 = const()[name = string("value_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_3_cast_fp16 = transpose(perm = value_states_3_perm_0, x = var_991_cast_fp16)[name = string("transpose_254")]; tensor value_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = value_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = value_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_1_squeeze_mask_0, stride = value_cache_internal_tensor_assign_1_stride_0, update = value_states_3_cast_fp16, x = read_state_1)[name = string("value_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_1_cast_fp16, input = value_cache)[name = string("coreml_update_state_113_write_state")]; tensor coreml_update_state_113 = read_state(input = value_cache)[name = string("coreml_update_state_113")]; tensor var_1085_begin_0 = const()[name = string("op_1085_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1085_end_0 = const()[name = string("op_1085_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1085_end_mask_0 = const()[name = string("op_1085_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1085_cast_fp16 = slice_by_index(begin = var_1085_begin_0, end = var_1085_end_0, end_mask = var_1085_end_mask_0, x = coreml_update_state_112)[name = string("op_1085_cast_fp16")]; tensor tile_0 = const()[name = string("tile_0"), val = tensor([1, 1])]; int32 var_1088_axis_0 = const()[name = string("op_1088_axis_0"), val = int32(1)]; tensor var_1088_cast_fp16_0, tensor var_1088_cast_fp16_1 = split(axis = var_1088_axis_0, split_sizes = tile_0, x = var_1085_cast_fp16)[name = string("op_1088_cast_fp16")]; tensor var_1095_begin_0 = const()[name = string("op_1095_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1095_end_0 = const()[name = string("op_1095_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1095_end_mask_0 = const()[name = string("op_1095_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1095_cast_fp16 = slice_by_index(begin = var_1095_begin_0, end = var_1095_end_0, end_mask = var_1095_end_mask_0, x = coreml_update_state_113)[name = string("op_1095_cast_fp16")]; tensor tile_1 = const()[name = string("tile_1"), val = tensor([1, 1])]; int32 var_1098_axis_0 = const()[name = string("op_1098_axis_0"), val = int32(1)]; tensor var_1098_cast_fp16_0, tensor var_1098_cast_fp16_1 = split(axis = var_1098_axis_0, split_sizes = tile_1, x = var_1095_cast_fp16)[name = string("op_1098_cast_fp16")]; tensor var_1101_split_sizes_0 = const()[name = string("op_1101_split_sizes_0"), val = tensor([8, 8])]; int32 var_1101_axis_0 = const()[name = string("op_1101_axis_0"), val = int32(1)]; tensor var_1101_0, tensor var_1101_1 = split(axis = var_1101_axis_0, split_sizes = var_1101_split_sizes_0, x = query_states_3_cast_fp16)[name = string("op_1101")]; bool attn_weights_1_transpose_x_0 = const()[name = string("attn_weights_1_transpose_x_0"), val = bool(false)]; bool attn_weights_1_transpose_y_0 = const()[name = string("attn_weights_1_transpose_y_0"), val = bool(false)]; tensor attn_weights_1_cast_fp16 = matmul(transpose_x = attn_weights_1_transpose_x_0, transpose_y = attn_weights_1_transpose_y_0, x = var_1088_cast_fp16_0, y = var_1101_0)[name = string("attn_weights_1_cast_fp16")]; fp16 var_1104_to_fp16 = const()[name = string("op_1104_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_3_cast_fp16 = mul(x = attn_weights_1_cast_fp16, y = var_1104_to_fp16)[name = string("attn_weights_3_cast_fp16")]; tensor attn_weights_5_cast_fp16 = add(x = attn_weights_3_cast_fp16, y = attn_mask_1)[name = string("attn_weights_5_cast_fp16")]; int32 var_1108 = const()[name = string("op_1108"), val = int32(-2)]; tensor attn_weights_7_cast_fp16 = softmax(axis = var_1108, x = attn_weights_5_cast_fp16)[name = string("attn_weights_7_cast_fp16")]; bool var_1114_transpose_x_1 = const()[name = string("op_1114_transpose_x_1"), val = bool(true)]; bool var_1114_transpose_y_1 = const()[name = string("op_1114_transpose_y_1"), val = bool(false)]; tensor var_1114_cast_fp16 = matmul(transpose_x = var_1114_transpose_x_1, transpose_y = var_1114_transpose_y_1, x = attn_weights_7_cast_fp16, y = var_1098_cast_fp16_0)[name = string("op_1114_cast_fp16")]; bool attn_weights_9_transpose_x_0 = const()[name = string("attn_weights_9_transpose_x_0"), val = bool(false)]; bool attn_weights_9_transpose_y_0 = const()[name = string("attn_weights_9_transpose_y_0"), val = bool(false)]; tensor attn_weights_9_cast_fp16 = matmul(transpose_x = attn_weights_9_transpose_x_0, transpose_y = attn_weights_9_transpose_y_0, x = var_1088_cast_fp16_1, y = var_1101_1)[name = string("attn_weights_9_cast_fp16")]; fp16 var_1116_to_fp16 = const()[name = string("op_1116_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_11_cast_fp16 = mul(x = attn_weights_9_cast_fp16, y = var_1116_to_fp16)[name = string("attn_weights_11_cast_fp16")]; tensor attn_weights_13_cast_fp16 = add(x = attn_weights_11_cast_fp16, y = attn_mask_1)[name = string("attn_weights_13_cast_fp16")]; int32 var_1120 = const()[name = string("op_1120"), val = int32(-2)]; tensor attn_weights_15_cast_fp16 = softmax(axis = var_1120, x = attn_weights_13_cast_fp16)[name = string("attn_weights_15_cast_fp16")]; bool attn_output_1_transpose_x_1 = const()[name = string("attn_output_1_transpose_x_1"), val = bool(true)]; bool attn_output_1_transpose_y_1 = const()[name = string("attn_output_1_transpose_y_1"), val = bool(false)]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_1, transpose_y = attn_output_1_transpose_y_1, x = attn_weights_15_cast_fp16, y = var_1098_cast_fp16_1)[name = string("attn_output_1_cast_fp16")]; int32 var_1128 = const()[name = string("op_1128"), val = int32(1)]; bool attn_output_3_interleave_0 = const()[name = string("attn_output_3_interleave_0"), val = bool(false)]; tensor attn_output_3_cast_fp16 = concat(axis = var_1128, interleave = attn_output_3_interleave_0, values = (var_1114_cast_fp16, attn_output_1_cast_fp16))[name = string("attn_output_3_cast_fp16")]; tensor var_1132_perm_0 = const()[name = string("op_1132_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_11x = const()[name = string("concat_11x"), val = tensor([1, 2048, 1, -1])]; tensor var_1132_cast_fp16 = transpose(perm = var_1132_perm_0, x = attn_output_3_cast_fp16)[name = string("transpose_253")]; tensor attn_output_7_cast_fp16 = reshape(shape = concat_11x, x = var_1132_cast_fp16)[name = string("attn_output_7_cast_fp16")]; tensor layers_0_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(993654976)))]; tensor hidden_states_3_strides_0 = const()[name = string("hidden_states_3_strides_0"), val = tensor([1, 1])]; string hidden_states_3_pad_type_0 = const()[name = string("hidden_states_3_pad_type_0"), val = string("valid")]; tensor hidden_states_3_pad_0 = const()[name = string("hidden_states_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_3_dilations_0 = const()[name = string("hidden_states_3_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_3_groups_0 = const()[name = string("hidden_states_3_groups_0"), val = int32(1)]; tensor hidden_states_3_cast_fp16 = conv(dilations = hidden_states_3_dilations_0, groups = hidden_states_3_groups_0, pad = hidden_states_3_pad_0, pad_type = hidden_states_3_pad_type_0, strides = hidden_states_3_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16, x = attn_output_7_cast_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor hidden_states_5_cast_fp16 = add(x = inputs_embeds_to_fp16, y = hidden_states_3_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1165_cast_fp16 = mul(x = hidden_states_5_cast_fp16, y = const_8_promoted_to_fp16)[name = string("op_1165_cast_fp16")]; int32 var_1163 = const()[name = string("op_1163"), val = int32(1)]; bool doubled_5_interleave_0 = const()[name = string("doubled_5_interleave_0"), val = bool(false)]; tensor doubled_5_cast_fp16 = concat(axis = var_1163, interleave = doubled_5_interleave_0, values = (hidden_states_5_cast_fp16, var_1165_cast_fp16))[name = string("doubled_5_cast_fp16")]; tensor out_3_axes_0 = const()[name = string("out_3_axes_0"), val = tensor([1])]; tensor out_3_gamma_0_to_fp16 = const()[name = string("out_3_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002043648)))]; fp16 var_1175_to_fp16 = const()[name = string("op_1175_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_3_cast_fp16 = layer_norm(axes = out_3_axes_0, epsilon = var_1175_to_fp16, gamma = out_3_gamma_0_to_fp16, x = doubled_5_cast_fp16)[name = string("out_3_cast_fp16")]; tensor var_1186_split_sizes_0 = const()[name = string("op_1186_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1186_axis_0 = const()[name = string("op_1186_axis_0"), val = int32(1)]; tensor var_1186_cast_fp16_0, tensor var_1186_cast_fp16_1 = split(axis = var_1186_axis_0, split_sizes = var_1186_split_sizes_0, x = out_3_cast_fp16)[name = string("op_1186_cast_fp16")]; tensor layers_0_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002051904)))]; tensor input_1_strides_0 = const()[name = string("input_1_strides_0"), val = tensor([1, 1])]; string input_1_pad_type_0 = const()[name = string("input_1_pad_type_0"), val = string("valid")]; tensor input_1_pad_0 = const()[name = string("input_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_1_dilations_0 = const()[name = string("input_1_dilations_0"), val = tensor([1, 1])]; int32 input_1_groups_0 = const()[name = string("input_1_groups_0"), val = int32(1)]; tensor input_1_cast_fp16 = conv(dilations = input_1_dilations_0, groups = input_1_groups_0, pad = input_1_pad_0, pad_type = input_1_pad_type_0, strides = input_1_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("input_1_cast_fp16")]; tensor var_1203_cast_fp16 = silu(x = input_1_cast_fp16)[name = string("op_1203_cast_fp16")]; tensor layers_0_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1027217792)))]; tensor var_1209_strides_0 = const()[name = string("op_1209_strides_0"), val = tensor([1, 1])]; string var_1209_pad_type_0 = const()[name = string("op_1209_pad_type_0"), val = string("valid")]; tensor var_1209_pad_0 = const()[name = string("op_1209_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1209_dilations_0 = const()[name = string("op_1209_dilations_0"), val = tensor([1, 1])]; int32 var_1209_groups_0 = const()[name = string("op_1209_groups_0"), val = int32(1)]; tensor var_1209_cast_fp16 = conv(dilations = var_1209_dilations_0, groups = var_1209_groups_0, pad = var_1209_pad_0, pad_type = var_1209_pad_type_0, strides = var_1209_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("op_1209_cast_fp16")]; tensor x_9_cast_fp16 = mul(x = var_1203_cast_fp16, y = var_1209_cast_fp16)[name = string("x_9_cast_fp16")]; tensor layers_0_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1052383680)))]; tensor hidden_states_7_strides_0 = const()[name = string("hidden_states_7_strides_0"), val = tensor([1, 1])]; string hidden_states_7_pad_type_0 = const()[name = string("hidden_states_7_pad_type_0"), val = string("valid")]; tensor hidden_states_7_pad_0 = const()[name = string("hidden_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_7_dilations_0 = const()[name = string("hidden_states_7_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_7_groups_0 = const()[name = string("hidden_states_7_groups_0"), val = int32(1)]; tensor hidden_states_7_cast_fp16 = conv(dilations = hidden_states_7_dilations_0, groups = hidden_states_7_groups_0, pad = hidden_states_7_pad_0, pad_type = hidden_states_7_pad_type_0, strides = hidden_states_7_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16, x = x_9_cast_fp16)[name = string("hidden_states_7_cast_fp16")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = hidden_states_7_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; fp16 const_10_promoted_to_fp16 = const()[name = string("const_10_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1227_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_10_promoted_to_fp16)[name = string("op_1227_cast_fp16")]; int32 var_1225 = const()[name = string("op_1225"), val = int32(1)]; bool doubled_9_interleave_0 = const()[name = string("doubled_9_interleave_0"), val = bool(false)]; tensor doubled_9_cast_fp16 = concat(axis = var_1225, interleave = doubled_9_interleave_0, values = (hidden_states_9_cast_fp16, var_1227_cast_fp16))[name = string("doubled_9_cast_fp16")]; tensor out_5_axes_0 = const()[name = string("out_5_axes_0"), val = tensor([1])]; tensor out_5_gamma_0_to_fp16 = const()[name = string("out_5_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077549568)))]; fp16 var_1237_to_fp16 = const()[name = string("op_1237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_5_cast_fp16 = layer_norm(axes = out_5_axes_0, epsilon = var_1237_to_fp16, gamma = out_5_gamma_0_to_fp16, x = doubled_9_cast_fp16)[name = string("out_5_cast_fp16")]; tensor var_1248_split_sizes_0 = const()[name = string("op_1248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1248_axis_0 = const()[name = string("op_1248_axis_0"), val = int32(1)]; tensor var_1248_cast_fp16_0, tensor var_1248_cast_fp16_1 = split(axis = var_1248_axis_0, split_sizes = var_1248_split_sizes_0, x = out_5_cast_fp16)[name = string("op_1248_cast_fp16")]; tensor layers_1_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077557824)))]; tensor query_states_7_strides_0 = const()[name = string("query_states_7_strides_0"), val = tensor([1, 1])]; string query_states_7_pad_type_0 = const()[name = string("query_states_7_pad_type_0"), val = string("valid")]; tensor query_states_7_pad_0 = const()[name = string("query_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_7_dilations_0 = const()[name = string("query_states_7_dilations_0"), val = tensor([1, 1])]; int32 query_states_7_groups_0 = const()[name = string("query_states_7_groups_0"), val = int32(1)]; tensor query_states_7_cast_fp16 = conv(dilations = query_states_7_dilations_0, groups = query_states_7_groups_0, pad = query_states_7_pad_0, pad_type = query_states_7_pad_type_0, strides = query_states_7_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("query_states_7_cast_fp16")]; tensor layers_1_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1085946496)))]; tensor key_states_11_strides_0 = const()[name = string("key_states_11_strides_0"), val = tensor([1, 1])]; string key_states_11_pad_type_0 = const()[name = string("key_states_11_pad_type_0"), val = string("valid")]; tensor key_states_11_pad_0 = const()[name = string("key_states_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_11_dilations_0 = const()[name = string("key_states_11_dilations_0"), val = tensor([1, 1])]; int32 key_states_11_groups_0 = const()[name = string("key_states_11_groups_0"), val = int32(1)]; tensor key_states_11_cast_fp16 = conv(dilations = key_states_11_dilations_0, groups = key_states_11_groups_0, pad = key_states_11_pad_0, pad_type = key_states_11_pad_type_0, strides = key_states_11_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("key_states_11_cast_fp16")]; tensor value_states_7_strides_0 = const()[name = string("value_states_7_strides_0"), val = tensor([1, 1])]; string value_states_7_pad_type_0 = const()[name = string("value_states_7_pad_type_0"), val = string("valid")]; tensor value_states_7_pad_0 = const()[name = string("value_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_7_dilations_0 = const()[name = string("value_states_7_dilations_0"), val = tensor([1, 1])]; int32 value_states_7_groups_0 = const()[name = string("value_states_7_groups_0"), val = int32(1)]; tensor value_states_7_cast_fp16 = conv(dilations = value_states_7_dilations_0, groups = value_states_7_groups_0, pad = value_states_7_pad_0, pad_type = value_states_7_pad_type_0, strides = value_states_7_strides_0, weight = layers_1_self_attn_v_proj_weight_cast_fp16, x = var_1248_cast_fp16_0)[name = string("value_states_7_cast_fp16")]; tensor concat_12x = const()[name = string("concat_12x"), val = tensor([1, 16, 128, -1])]; tensor x_11_cast_fp16 = reshape(shape = concat_12x, x = query_states_7_cast_fp16)[name = string("x_11_cast_fp16")]; tensor concat_13x = const()[name = string("concat_13x"), val = tensor([1, 2, 128, -1])]; tensor var_1305_cast_fp16 = reshape(shape = concat_13x, x = key_states_11_cast_fp16)[name = string("op_1305_cast_fp16")]; tensor concat_14x = const()[name = string("concat_14x"), val = tensor([1, 2, 128, -1])]; tensor var_1312_cast_fp16 = reshape(shape = concat_14x, x = value_states_7_cast_fp16)[name = string("op_1312_cast_fp16")]; tensor var_1316_cast_fp16 = mul(x = x_11_cast_fp16, y = var_869_cast_fp16)[name = string("op_1316_cast_fp16")]; tensor var_1317_split_sizes_0 = const()[name = string("op_1317_split_sizes_0"), val = tensor([64, 64])]; int32 var_1317_axis_0 = const()[name = string("op_1317_axis_0"), val = int32(-2)]; tensor var_1317_cast_fp16_0, tensor var_1317_cast_fp16_1 = split(axis = var_1317_axis_0, split_sizes = var_1317_split_sizes_0, x = x_11_cast_fp16)[name = string("op_1317_cast_fp16")]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1319_cast_fp16 = mul(x = var_1317_cast_fp16_1, y = const_12_promoted_to_fp16)[name = string("op_1319_cast_fp16")]; int32 var_1321 = const()[name = string("op_1321"), val = int32(-2)]; bool var_1322_interleave_0 = const()[name = string("op_1322_interleave_0"), val = bool(false)]; tensor var_1322_cast_fp16 = concat(axis = var_1321, interleave = var_1322_interleave_0, values = (var_1319_cast_fp16, var_1317_cast_fp16_0))[name = string("op_1322_cast_fp16")]; tensor var_1323_cast_fp16 = mul(x = var_1322_cast_fp16, y = var_878_cast_fp16)[name = string("op_1323_cast_fp16")]; tensor query_states_9_cast_fp16 = add(x = var_1316_cast_fp16, y = var_1323_cast_fp16)[name = string("query_states_9_cast_fp16")]; tensor var_1329_cast_fp16 = mul(x = var_1305_cast_fp16, y = var_869_cast_fp16)[name = string("op_1329_cast_fp16")]; tensor var_1330_split_sizes_0 = const()[name = string("op_1330_split_sizes_0"), val = tensor([64, 64])]; int32 var_1330_axis_0 = const()[name = string("op_1330_axis_0"), val = int32(-2)]; tensor var_1330_cast_fp16_0, tensor var_1330_cast_fp16_1 = split(axis = var_1330_axis_0, split_sizes = var_1330_split_sizes_0, x = var_1305_cast_fp16)[name = string("op_1330_cast_fp16")]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1332_cast_fp16 = mul(x = var_1330_cast_fp16_1, y = const_13_promoted_to_fp16)[name = string("op_1332_cast_fp16")]; int32 var_1334 = const()[name = string("op_1334"), val = int32(-2)]; bool var_1335_interleave_0 = const()[name = string("op_1335_interleave_0"), val = bool(false)]; tensor var_1335_cast_fp16 = concat(axis = var_1334, interleave = var_1335_interleave_0, values = (var_1332_cast_fp16, var_1330_cast_fp16_0))[name = string("op_1335_cast_fp16")]; tensor var_1336_cast_fp16 = mul(x = var_1335_cast_fp16, y = var_878_cast_fp16)[name = string("op_1336_cast_fp16")]; tensor key_states_15_cast_fp16 = add(x = var_1329_cast_fp16, y = var_1336_cast_fp16)[name = string("key_states_15_cast_fp16")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; int32 concat_17_axis_0 = const()[name = string("concat_17_axis_0"), val = int32(0)]; bool concat_17_interleave_0 = const()[name = string("concat_17_interleave_0"), val = bool(false)]; tensor concat_17 = concat(axis = concat_17_axis_0, interleave = concat_17_interleave_0, values = (expand_dims_12, expand_dims_13, position_id, expand_dims_15))[name = string("concat_17")]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; tensor concat_18_values1_0 = const()[name = string("concat_18_values1_0"), val = tensor([0])]; tensor concat_18_values3_0 = const()[name = string("concat_18_values3_0"), val = tensor([0])]; int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_16, concat_18_values1_0, cache_position_end, concat_18_values3_0))[name = string("concat_18")]; tensor key_states_17_perm_0 = const()[name = string("key_states_17_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_2_stride_0 = const()[name = string("key_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_17_cast_fp16 = transpose(perm = key_states_17_perm_0, x = key_states_15_cast_fp16)[name = string("transpose_252")]; tensor key_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = key_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = key_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_2_squeeze_mask_0, stride = key_cache_internal_tensor_assign_2_stride_0, update = key_states_17_cast_fp16, x = coreml_update_state_112)[name = string("key_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_2_cast_fp16, input = key_cache)[name = string("coreml_update_state_114_write_state")]; tensor coreml_update_state_114 = read_state(input = key_cache)[name = string("coreml_update_state_114")]; tensor value_states_9_perm_0 = const()[name = string("value_states_9_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_2_stride_0 = const()[name = string("value_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_9_cast_fp16 = transpose(perm = value_states_9_perm_0, x = var_1312_cast_fp16)[name = string("transpose_251")]; tensor value_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = value_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = value_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_2_squeeze_mask_0, stride = value_cache_internal_tensor_assign_2_stride_0, update = value_states_9_cast_fp16, x = coreml_update_state_113)[name = string("value_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_2_cast_fp16, input = value_cache)[name = string("coreml_update_state_115_write_state")]; tensor coreml_update_state_115 = read_state(input = value_cache)[name = string("coreml_update_state_115")]; tensor var_1406_begin_0 = const()[name = string("op_1406_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1406_end_0 = const()[name = string("op_1406_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1406_end_mask_0 = const()[name = string("op_1406_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1406_cast_fp16 = slice_by_index(begin = var_1406_begin_0, end = var_1406_end_0, end_mask = var_1406_end_mask_0, x = coreml_update_state_114)[name = string("op_1406_cast_fp16")]; tensor tile_2 = const()[name = string("tile_2"), val = tensor([1, 1])]; int32 var_1409_axis_0 = const()[name = string("op_1409_axis_0"), val = int32(1)]; tensor var_1409_cast_fp16_0, tensor var_1409_cast_fp16_1 = split(axis = var_1409_axis_0, split_sizes = tile_2, x = var_1406_cast_fp16)[name = string("op_1409_cast_fp16")]; tensor var_1416_begin_0 = const()[name = string("op_1416_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1416_end_0 = const()[name = string("op_1416_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1416_end_mask_0 = const()[name = string("op_1416_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1416_cast_fp16 = slice_by_index(begin = var_1416_begin_0, end = var_1416_end_0, end_mask = var_1416_end_mask_0, x = coreml_update_state_115)[name = string("op_1416_cast_fp16")]; tensor tile_3 = const()[name = string("tile_3"), val = tensor([1, 1])]; int32 var_1419_axis_0 = const()[name = string("op_1419_axis_0"), val = int32(1)]; tensor var_1419_cast_fp16_0, tensor var_1419_cast_fp16_1 = split(axis = var_1419_axis_0, split_sizes = tile_3, x = var_1416_cast_fp16)[name = string("op_1419_cast_fp16")]; tensor var_1422_split_sizes_0 = const()[name = string("op_1422_split_sizes_0"), val = tensor([8, 8])]; int32 var_1422_axis_0 = const()[name = string("op_1422_axis_0"), val = int32(1)]; tensor var_1422_0, tensor var_1422_1 = split(axis = var_1422_axis_0, split_sizes = var_1422_split_sizes_0, x = query_states_9_cast_fp16)[name = string("op_1422")]; bool attn_weights_17_transpose_x_0 = const()[name = string("attn_weights_17_transpose_x_0"), val = bool(false)]; bool attn_weights_17_transpose_y_0 = const()[name = string("attn_weights_17_transpose_y_0"), val = bool(false)]; tensor attn_weights_17_cast_fp16 = matmul(transpose_x = attn_weights_17_transpose_x_0, transpose_y = attn_weights_17_transpose_y_0, x = var_1409_cast_fp16_0, y = var_1422_0)[name = string("attn_weights_17_cast_fp16")]; fp16 var_1425_to_fp16 = const()[name = string("op_1425_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_19_cast_fp16 = mul(x = attn_weights_17_cast_fp16, y = var_1425_to_fp16)[name = string("attn_weights_19_cast_fp16")]; tensor attn_weights_21_cast_fp16 = add(x = attn_weights_19_cast_fp16, y = attn_mask_1)[name = string("attn_weights_21_cast_fp16")]; int32 var_1429 = const()[name = string("op_1429"), val = int32(-2)]; tensor attn_weights_23_cast_fp16 = softmax(axis = var_1429, x = attn_weights_21_cast_fp16)[name = string("attn_weights_23_cast_fp16")]; bool var_1435_transpose_x_1 = const()[name = string("op_1435_transpose_x_1"), val = bool(true)]; bool var_1435_transpose_y_1 = const()[name = string("op_1435_transpose_y_1"), val = bool(false)]; tensor var_1435_cast_fp16 = matmul(transpose_x = var_1435_transpose_x_1, transpose_y = var_1435_transpose_y_1, x = attn_weights_23_cast_fp16, y = var_1419_cast_fp16_0)[name = string("op_1435_cast_fp16")]; bool attn_weights_25_transpose_x_0 = const()[name = string("attn_weights_25_transpose_x_0"), val = bool(false)]; bool attn_weights_25_transpose_y_0 = const()[name = string("attn_weights_25_transpose_y_0"), val = bool(false)]; tensor attn_weights_25_cast_fp16 = matmul(transpose_x = attn_weights_25_transpose_x_0, transpose_y = attn_weights_25_transpose_y_0, x = var_1409_cast_fp16_1, y = var_1422_1)[name = string("attn_weights_25_cast_fp16")]; fp16 var_1437_to_fp16 = const()[name = string("op_1437_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_27_cast_fp16 = mul(x = attn_weights_25_cast_fp16, y = var_1437_to_fp16)[name = string("attn_weights_27_cast_fp16")]; tensor attn_weights_29_cast_fp16 = add(x = attn_weights_27_cast_fp16, y = attn_mask_1)[name = string("attn_weights_29_cast_fp16")]; int32 var_1441 = const()[name = string("op_1441"), val = int32(-2)]; tensor attn_weights_31_cast_fp16 = softmax(axis = var_1441, x = attn_weights_29_cast_fp16)[name = string("attn_weights_31_cast_fp16")]; bool attn_output_9_transpose_x_1 = const()[name = string("attn_output_9_transpose_x_1"), val = bool(true)]; bool attn_output_9_transpose_y_1 = const()[name = string("attn_output_9_transpose_y_1"), val = bool(false)]; tensor attn_output_9_cast_fp16 = matmul(transpose_x = attn_output_9_transpose_x_1, transpose_y = attn_output_9_transpose_y_1, x = attn_weights_31_cast_fp16, y = var_1419_cast_fp16_1)[name = string("attn_output_9_cast_fp16")]; int32 var_1449 = const()[name = string("op_1449"), val = int32(1)]; bool attn_output_11_interleave_0 = const()[name = string("attn_output_11_interleave_0"), val = bool(false)]; tensor attn_output_11_cast_fp16 = concat(axis = var_1449, interleave = attn_output_11_interleave_0, values = (var_1435_cast_fp16, attn_output_9_cast_fp16))[name = string("attn_output_11_cast_fp16")]; tensor var_1453_perm_0 = const()[name = string("op_1453_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_23x = const()[name = string("concat_23x"), val = tensor([1, 2048, 1, -1])]; tensor var_1453_cast_fp16 = transpose(perm = var_1453_perm_0, x = attn_output_11_cast_fp16)[name = string("transpose_250")]; tensor attn_output_15_cast_fp16 = reshape(shape = concat_23x, x = var_1453_cast_fp16)[name = string("attn_output_15_cast_fp16")]; tensor layers_1_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086995136)))]; tensor hidden_states_13_strides_0 = const()[name = string("hidden_states_13_strides_0"), val = tensor([1, 1])]; string hidden_states_13_pad_type_0 = const()[name = string("hidden_states_13_pad_type_0"), val = string("valid")]; tensor hidden_states_13_pad_0 = const()[name = string("hidden_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_13_dilations_0 = const()[name = string("hidden_states_13_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_13_groups_0 = const()[name = string("hidden_states_13_groups_0"), val = int32(1)]; tensor hidden_states_13_cast_fp16 = conv(dilations = hidden_states_13_dilations_0, groups = hidden_states_13_groups_0, pad = hidden_states_13_pad_0, pad_type = hidden_states_13_pad_type_0, strides = hidden_states_13_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16, x = attn_output_15_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; tensor hidden_states_15_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = hidden_states_13_cast_fp16)[name = string("hidden_states_15_cast_fp16")]; fp16 const_18_promoted_to_fp16 = const()[name = string("const_18_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1486_cast_fp16 = mul(x = hidden_states_15_cast_fp16, y = const_18_promoted_to_fp16)[name = string("op_1486_cast_fp16")]; int32 var_1484 = const()[name = string("op_1484"), val = int32(1)]; bool doubled_13_interleave_0 = const()[name = string("doubled_13_interleave_0"), val = bool(false)]; tensor doubled_13_cast_fp16 = concat(axis = var_1484, interleave = doubled_13_interleave_0, values = (hidden_states_15_cast_fp16, var_1486_cast_fp16))[name = string("doubled_13_cast_fp16")]; tensor out_7_axes_0 = const()[name = string("out_7_axes_0"), val = tensor([1])]; tensor out_7_gamma_0_to_fp16 = const()[name = string("out_7_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095383808)))]; fp16 var_1496_to_fp16 = const()[name = string("op_1496_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_7_cast_fp16 = layer_norm(axes = out_7_axes_0, epsilon = var_1496_to_fp16, gamma = out_7_gamma_0_to_fp16, x = doubled_13_cast_fp16)[name = string("out_7_cast_fp16")]; tensor var_1507_split_sizes_0 = const()[name = string("op_1507_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1507_axis_0 = const()[name = string("op_1507_axis_0"), val = int32(1)]; tensor var_1507_cast_fp16_0, tensor var_1507_cast_fp16_1 = split(axis = var_1507_axis_0, split_sizes = var_1507_split_sizes_0, x = out_7_cast_fp16)[name = string("op_1507_cast_fp16")]; tensor layers_1_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095392064)))]; tensor input_3_strides_0 = const()[name = string("input_3_strides_0"), val = tensor([1, 1])]; string input_3_pad_type_0 = const()[name = string("input_3_pad_type_0"), val = string("valid")]; tensor input_3_pad_0 = const()[name = string("input_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_3_dilations_0 = const()[name = string("input_3_dilations_0"), val = tensor([1, 1])]; int32 input_3_groups_0 = const()[name = string("input_3_groups_0"), val = int32(1)]; tensor input_3_cast_fp16 = conv(dilations = input_3_dilations_0, groups = input_3_groups_0, pad = input_3_pad_0, pad_type = input_3_pad_type_0, strides = input_3_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16, x = var_1507_cast_fp16_0)[name = string("input_3_cast_fp16")]; tensor var_1524_cast_fp16 = silu(x = input_3_cast_fp16)[name = string("op_1524_cast_fp16")]; tensor var_1530_strides_0 = const()[name = string("op_1530_strides_0"), val = tensor([1, 1])]; string var_1530_pad_type_0 = const()[name = string("op_1530_pad_type_0"), val = string("valid")]; tensor var_1530_pad_0 = const()[name = string("op_1530_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1530_dilations_0 = const()[name = string("op_1530_dilations_0"), val = tensor([1, 1])]; int32 var_1530_groups_0 = const()[name = string("op_1530_groups_0"), val = int32(1)]; tensor var_1530_cast_fp16 = conv(dilations = var_1530_dilations_0, groups = var_1530_groups_0, pad = var_1530_pad_0, pad_type = var_1530_pad_type_0, strides = var_1530_strides_0, weight = layers_1_mlp_up_proj_weight_cast_fp16, x = var_1507_cast_fp16_0)[name = string("op_1530_cast_fp16")]; tensor x_19_cast_fp16 = mul(x = var_1524_cast_fp16, y = var_1530_cast_fp16)[name = string("x_19_cast_fp16")]; tensor layers_1_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1120557952)))]; tensor hidden_states_17_strides_0 = const()[name = string("hidden_states_17_strides_0"), val = tensor([1, 1])]; string hidden_states_17_pad_type_0 = const()[name = string("hidden_states_17_pad_type_0"), val = string("valid")]; tensor hidden_states_17_pad_0 = const()[name = string("hidden_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_17_dilations_0 = const()[name = string("hidden_states_17_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_17_groups_0 = const()[name = string("hidden_states_17_groups_0"), val = int32(1)]; tensor hidden_states_17_cast_fp16 = conv(dilations = hidden_states_17_dilations_0, groups = hidden_states_17_groups_0, pad = hidden_states_17_pad_0, pad_type = hidden_states_17_pad_type_0, strides = hidden_states_17_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16, x = x_19_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; tensor hidden_states_19_cast_fp16 = add(x = hidden_states_15_cast_fp16, y = hidden_states_17_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1548_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_1548_cast_fp16")]; int32 var_1546 = const()[name = string("op_1546"), val = int32(1)]; bool doubled_17_interleave_0 = const()[name = string("doubled_17_interleave_0"), val = bool(false)]; tensor doubled_17_cast_fp16 = concat(axis = var_1546, interleave = doubled_17_interleave_0, values = (hidden_states_19_cast_fp16, var_1548_cast_fp16))[name = string("doubled_17_cast_fp16")]; tensor out_9_axes_0 = const()[name = string("out_9_axes_0"), val = tensor([1])]; tensor out_9_gamma_0_to_fp16 = const()[name = string("out_9_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145723840)))]; fp16 var_1558_to_fp16 = const()[name = string("op_1558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_9_cast_fp16 = layer_norm(axes = out_9_axes_0, epsilon = var_1558_to_fp16, gamma = out_9_gamma_0_to_fp16, x = doubled_17_cast_fp16)[name = string("out_9_cast_fp16")]; tensor var_1569_split_sizes_0 = const()[name = string("op_1569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1569_axis_0 = const()[name = string("op_1569_axis_0"), val = int32(1)]; tensor var_1569_cast_fp16_0, tensor var_1569_cast_fp16_1 = split(axis = var_1569_axis_0, split_sizes = var_1569_split_sizes_0, x = out_9_cast_fp16)[name = string("op_1569_cast_fp16")]; tensor layers_2_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145732096)))]; tensor query_states_13_strides_0 = const()[name = string("query_states_13_strides_0"), val = tensor([1, 1])]; string query_states_13_pad_type_0 = const()[name = string("query_states_13_pad_type_0"), val = string("valid")]; tensor query_states_13_pad_0 = const()[name = string("query_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_13_dilations_0 = const()[name = string("query_states_13_dilations_0"), val = tensor([1, 1])]; int32 query_states_13_groups_0 = const()[name = string("query_states_13_groups_0"), val = int32(1)]; tensor query_states_13_cast_fp16 = conv(dilations = query_states_13_dilations_0, groups = query_states_13_groups_0, pad = query_states_13_pad_0, pad_type = query_states_13_pad_type_0, strides = query_states_13_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("query_states_13_cast_fp16")]; tensor layers_2_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1154120768)))]; tensor key_states_21_strides_0 = const()[name = string("key_states_21_strides_0"), val = tensor([1, 1])]; string key_states_21_pad_type_0 = const()[name = string("key_states_21_pad_type_0"), val = string("valid")]; tensor key_states_21_pad_0 = const()[name = string("key_states_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_21_dilations_0 = const()[name = string("key_states_21_dilations_0"), val = tensor([1, 1])]; int32 key_states_21_groups_0 = const()[name = string("key_states_21_groups_0"), val = int32(1)]; tensor key_states_21_cast_fp16 = conv(dilations = key_states_21_dilations_0, groups = key_states_21_groups_0, pad = key_states_21_pad_0, pad_type = key_states_21_pad_type_0, strides = key_states_21_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("key_states_21_cast_fp16")]; tensor value_states_13_strides_0 = const()[name = string("value_states_13_strides_0"), val = tensor([1, 1])]; string value_states_13_pad_type_0 = const()[name = string("value_states_13_pad_type_0"), val = string("valid")]; tensor value_states_13_pad_0 = const()[name = string("value_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_13_dilations_0 = const()[name = string("value_states_13_dilations_0"), val = tensor([1, 1])]; int32 value_states_13_groups_0 = const()[name = string("value_states_13_groups_0"), val = int32(1)]; tensor value_states_13_cast_fp16 = conv(dilations = value_states_13_dilations_0, groups = value_states_13_groups_0, pad = value_states_13_pad_0, pad_type = value_states_13_pad_type_0, strides = value_states_13_strides_0, weight = layers_2_self_attn_v_proj_weight_cast_fp16, x = var_1569_cast_fp16_0)[name = string("value_states_13_cast_fp16")]; tensor concat_24x = const()[name = string("concat_24x"), val = tensor([1, 16, 128, -1])]; tensor x_21_cast_fp16 = reshape(shape = concat_24x, x = query_states_13_cast_fp16)[name = string("x_21_cast_fp16")]; tensor concat_25x = const()[name = string("concat_25x"), val = tensor([1, 2, 128, -1])]; tensor var_1626_cast_fp16 = reshape(shape = concat_25x, x = key_states_21_cast_fp16)[name = string("op_1626_cast_fp16")]; tensor concat_26x = const()[name = string("concat_26x"), val = tensor([1, 2, 128, -1])]; tensor var_1633_cast_fp16 = reshape(shape = concat_26x, x = value_states_13_cast_fp16)[name = string("op_1633_cast_fp16")]; tensor var_1637_cast_fp16 = mul(x = x_21_cast_fp16, y = var_869_cast_fp16)[name = string("op_1637_cast_fp16")]; tensor var_1638_split_sizes_0 = const()[name = string("op_1638_split_sizes_0"), val = tensor([64, 64])]; int32 var_1638_axis_0 = const()[name = string("op_1638_axis_0"), val = int32(-2)]; tensor var_1638_cast_fp16_0, tensor var_1638_cast_fp16_1 = split(axis = var_1638_axis_0, split_sizes = var_1638_split_sizes_0, x = x_21_cast_fp16)[name = string("op_1638_cast_fp16")]; fp16 const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1640_cast_fp16 = mul(x = var_1638_cast_fp16_1, y = const_22_promoted_to_fp16)[name = string("op_1640_cast_fp16")]; int32 var_1642 = const()[name = string("op_1642"), val = int32(-2)]; bool var_1643_interleave_0 = const()[name = string("op_1643_interleave_0"), val = bool(false)]; tensor var_1643_cast_fp16 = concat(axis = var_1642, interleave = var_1643_interleave_0, values = (var_1640_cast_fp16, var_1638_cast_fp16_0))[name = string("op_1643_cast_fp16")]; tensor var_1644_cast_fp16 = mul(x = var_1643_cast_fp16, y = var_878_cast_fp16)[name = string("op_1644_cast_fp16")]; tensor query_states_15_cast_fp16 = add(x = var_1637_cast_fp16, y = var_1644_cast_fp16)[name = string("query_states_15_cast_fp16")]; tensor var_1650_cast_fp16 = mul(x = var_1626_cast_fp16, y = var_869_cast_fp16)[name = string("op_1650_cast_fp16")]; tensor var_1651_split_sizes_0 = const()[name = string("op_1651_split_sizes_0"), val = tensor([64, 64])]; int32 var_1651_axis_0 = const()[name = string("op_1651_axis_0"), val = int32(-2)]; tensor var_1651_cast_fp16_0, tensor var_1651_cast_fp16_1 = split(axis = var_1651_axis_0, split_sizes = var_1651_split_sizes_0, x = var_1626_cast_fp16)[name = string("op_1651_cast_fp16")]; fp16 const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1653_cast_fp16 = mul(x = var_1651_cast_fp16_1, y = const_23_promoted_to_fp16)[name = string("op_1653_cast_fp16")]; int32 var_1655 = const()[name = string("op_1655"), val = int32(-2)]; bool var_1656_interleave_0 = const()[name = string("op_1656_interleave_0"), val = bool(false)]; tensor var_1656_cast_fp16 = concat(axis = var_1655, interleave = var_1656_interleave_0, values = (var_1653_cast_fp16, var_1651_cast_fp16_0))[name = string("op_1656_cast_fp16")]; tensor var_1657_cast_fp16 = mul(x = var_1656_cast_fp16, y = var_878_cast_fp16)[name = string("op_1657_cast_fp16")]; tensor key_states_25_cast_fp16 = add(x = var_1650_cast_fp16, y = var_1657_cast_fp16)[name = string("key_states_25_cast_fp16")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; int32 concat_29_axis_0 = const()[name = string("concat_29_axis_0"), val = int32(0)]; bool concat_29_interleave_0 = const()[name = string("concat_29_interleave_0"), val = bool(false)]; tensor concat_29 = concat(axis = concat_29_axis_0, interleave = concat_29_interleave_0, values = (expand_dims_24, expand_dims_25, position_id, expand_dims_27))[name = string("concat_29")]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; tensor concat_30_values1_0 = const()[name = string("concat_30_values1_0"), val = tensor([0])]; tensor concat_30_values3_0 = const()[name = string("concat_30_values3_0"), val = tensor([0])]; int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_28, concat_30_values1_0, cache_position_end, concat_30_values3_0))[name = string("concat_30")]; tensor key_states_27_perm_0 = const()[name = string("key_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_3_stride_0 = const()[name = string("key_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_27_cast_fp16 = transpose(perm = key_states_27_perm_0, x = key_states_25_cast_fp16)[name = string("transpose_249")]; tensor key_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = key_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = key_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_3_squeeze_mask_0, stride = key_cache_internal_tensor_assign_3_stride_0, update = key_states_27_cast_fp16, x = coreml_update_state_114)[name = string("key_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_3_cast_fp16, input = key_cache)[name = string("coreml_update_state_116_write_state")]; tensor coreml_update_state_116 = read_state(input = key_cache)[name = string("coreml_update_state_116")]; tensor value_states_15_perm_0 = const()[name = string("value_states_15_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_3_stride_0 = const()[name = string("value_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_15_cast_fp16 = transpose(perm = value_states_15_perm_0, x = var_1633_cast_fp16)[name = string("transpose_248")]; tensor value_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = value_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = value_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_3_squeeze_mask_0, stride = value_cache_internal_tensor_assign_3_stride_0, update = value_states_15_cast_fp16, x = coreml_update_state_115)[name = string("value_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_3_cast_fp16, input = value_cache)[name = string("coreml_update_state_117_write_state")]; tensor coreml_update_state_117 = read_state(input = value_cache)[name = string("coreml_update_state_117")]; tensor var_1727_begin_0 = const()[name = string("op_1727_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1727_end_0 = const()[name = string("op_1727_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1727_end_mask_0 = const()[name = string("op_1727_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1727_cast_fp16 = slice_by_index(begin = var_1727_begin_0, end = var_1727_end_0, end_mask = var_1727_end_mask_0, x = coreml_update_state_116)[name = string("op_1727_cast_fp16")]; tensor tile_4 = const()[name = string("tile_4"), val = tensor([1, 1])]; int32 var_1730_axis_0 = const()[name = string("op_1730_axis_0"), val = int32(1)]; tensor var_1730_cast_fp16_0, tensor var_1730_cast_fp16_1 = split(axis = var_1730_axis_0, split_sizes = tile_4, x = var_1727_cast_fp16)[name = string("op_1730_cast_fp16")]; tensor var_1737_begin_0 = const()[name = string("op_1737_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1737_end_0 = const()[name = string("op_1737_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1737_end_mask_0 = const()[name = string("op_1737_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1737_cast_fp16 = slice_by_index(begin = var_1737_begin_0, end = var_1737_end_0, end_mask = var_1737_end_mask_0, x = coreml_update_state_117)[name = string("op_1737_cast_fp16")]; tensor tile_5 = const()[name = string("tile_5"), val = tensor([1, 1])]; int32 var_1740_axis_0 = const()[name = string("op_1740_axis_0"), val = int32(1)]; tensor var_1740_cast_fp16_0, tensor var_1740_cast_fp16_1 = split(axis = var_1740_axis_0, split_sizes = tile_5, x = var_1737_cast_fp16)[name = string("op_1740_cast_fp16")]; tensor var_1743_split_sizes_0 = const()[name = string("op_1743_split_sizes_0"), val = tensor([8, 8])]; int32 var_1743_axis_0 = const()[name = string("op_1743_axis_0"), val = int32(1)]; tensor var_1743_0, tensor var_1743_1 = split(axis = var_1743_axis_0, split_sizes = var_1743_split_sizes_0, x = query_states_15_cast_fp16)[name = string("op_1743")]; bool attn_weights_33_transpose_x_0 = const()[name = string("attn_weights_33_transpose_x_0"), val = bool(false)]; bool attn_weights_33_transpose_y_0 = const()[name = string("attn_weights_33_transpose_y_0"), val = bool(false)]; tensor attn_weights_33_cast_fp16 = matmul(transpose_x = attn_weights_33_transpose_x_0, transpose_y = attn_weights_33_transpose_y_0, x = var_1730_cast_fp16_0, y = var_1743_0)[name = string("attn_weights_33_cast_fp16")]; fp16 var_1746_to_fp16 = const()[name = string("op_1746_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_35_cast_fp16 = mul(x = attn_weights_33_cast_fp16, y = var_1746_to_fp16)[name = string("attn_weights_35_cast_fp16")]; tensor attn_weights_37_cast_fp16 = add(x = attn_weights_35_cast_fp16, y = attn_mask_1)[name = string("attn_weights_37_cast_fp16")]; int32 var_1750 = const()[name = string("op_1750"), val = int32(-2)]; tensor attn_weights_39_cast_fp16 = softmax(axis = var_1750, x = attn_weights_37_cast_fp16)[name = string("attn_weights_39_cast_fp16")]; bool var_1756_transpose_x_1 = const()[name = string("op_1756_transpose_x_1"), val = bool(true)]; bool var_1756_transpose_y_1 = const()[name = string("op_1756_transpose_y_1"), val = bool(false)]; tensor var_1756_cast_fp16 = matmul(transpose_x = var_1756_transpose_x_1, transpose_y = var_1756_transpose_y_1, x = attn_weights_39_cast_fp16, y = var_1740_cast_fp16_0)[name = string("op_1756_cast_fp16")]; bool attn_weights_41_transpose_x_0 = const()[name = string("attn_weights_41_transpose_x_0"), val = bool(false)]; bool attn_weights_41_transpose_y_0 = const()[name = string("attn_weights_41_transpose_y_0"), val = bool(false)]; tensor attn_weights_41_cast_fp16 = matmul(transpose_x = attn_weights_41_transpose_x_0, transpose_y = attn_weights_41_transpose_y_0, x = var_1730_cast_fp16_1, y = var_1743_1)[name = string("attn_weights_41_cast_fp16")]; fp16 var_1758_to_fp16 = const()[name = string("op_1758_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_43_cast_fp16 = mul(x = attn_weights_41_cast_fp16, y = var_1758_to_fp16)[name = string("attn_weights_43_cast_fp16")]; tensor attn_weights_45_cast_fp16 = add(x = attn_weights_43_cast_fp16, y = attn_mask_1)[name = string("attn_weights_45_cast_fp16")]; int32 var_1762 = const()[name = string("op_1762"), val = int32(-2)]; tensor attn_weights_47_cast_fp16 = softmax(axis = var_1762, x = attn_weights_45_cast_fp16)[name = string("attn_weights_47_cast_fp16")]; bool attn_output_17_transpose_x_1 = const()[name = string("attn_output_17_transpose_x_1"), val = bool(true)]; bool attn_output_17_transpose_y_1 = const()[name = string("attn_output_17_transpose_y_1"), val = bool(false)]; tensor attn_output_17_cast_fp16 = matmul(transpose_x = attn_output_17_transpose_x_1, transpose_y = attn_output_17_transpose_y_1, x = attn_weights_47_cast_fp16, y = var_1740_cast_fp16_1)[name = string("attn_output_17_cast_fp16")]; int32 var_1770 = const()[name = string("op_1770"), val = int32(1)]; bool attn_output_19_interleave_0 = const()[name = string("attn_output_19_interleave_0"), val = bool(false)]; tensor attn_output_19_cast_fp16 = concat(axis = var_1770, interleave = attn_output_19_interleave_0, values = (var_1756_cast_fp16, attn_output_17_cast_fp16))[name = string("attn_output_19_cast_fp16")]; tensor var_1774_perm_0 = const()[name = string("op_1774_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_35x = const()[name = string("concat_35x"), val = tensor([1, 2048, 1, -1])]; tensor var_1774_cast_fp16 = transpose(perm = var_1774_perm_0, x = attn_output_19_cast_fp16)[name = string("transpose_247")]; tensor attn_output_23_cast_fp16 = reshape(shape = concat_35x, x = var_1774_cast_fp16)[name = string("attn_output_23_cast_fp16")]; tensor layers_2_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1155169408)))]; tensor hidden_states_23_strides_0 = const()[name = string("hidden_states_23_strides_0"), val = tensor([1, 1])]; string hidden_states_23_pad_type_0 = const()[name = string("hidden_states_23_pad_type_0"), val = string("valid")]; tensor hidden_states_23_pad_0 = const()[name = string("hidden_states_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_23_dilations_0 = const()[name = string("hidden_states_23_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_23_groups_0 = const()[name = string("hidden_states_23_groups_0"), val = int32(1)]; tensor hidden_states_23_cast_fp16 = conv(dilations = hidden_states_23_dilations_0, groups = hidden_states_23_groups_0, pad = hidden_states_23_pad_0, pad_type = hidden_states_23_pad_type_0, strides = hidden_states_23_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16, x = attn_output_23_cast_fp16)[name = string("hidden_states_23_cast_fp16")]; tensor hidden_states_25_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = hidden_states_23_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1807_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_1807_cast_fp16")]; int32 var_1805 = const()[name = string("op_1805"), val = int32(1)]; bool doubled_21_interleave_0 = const()[name = string("doubled_21_interleave_0"), val = bool(false)]; tensor doubled_21_cast_fp16 = concat(axis = var_1805, interleave = doubled_21_interleave_0, values = (hidden_states_25_cast_fp16, var_1807_cast_fp16))[name = string("doubled_21_cast_fp16")]; tensor out_11_axes_0 = const()[name = string("out_11_axes_0"), val = tensor([1])]; tensor out_11_gamma_0_to_fp16 = const()[name = string("out_11_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163558080)))]; fp16 var_1817_to_fp16 = const()[name = string("op_1817_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_11_cast_fp16 = layer_norm(axes = out_11_axes_0, epsilon = var_1817_to_fp16, gamma = out_11_gamma_0_to_fp16, x = doubled_21_cast_fp16)[name = string("out_11_cast_fp16")]; tensor var_1828_split_sizes_0 = const()[name = string("op_1828_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1828_axis_0 = const()[name = string("op_1828_axis_0"), val = int32(1)]; tensor var_1828_cast_fp16_0, tensor var_1828_cast_fp16_1 = split(axis = var_1828_axis_0, split_sizes = var_1828_split_sizes_0, x = out_11_cast_fp16)[name = string("op_1828_cast_fp16")]; tensor layers_2_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163566336)))]; tensor input_5_strides_0 = const()[name = string("input_5_strides_0"), val = tensor([1, 1])]; string input_5_pad_type_0 = const()[name = string("input_5_pad_type_0"), val = string("valid")]; tensor input_5_pad_0 = const()[name = string("input_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_5_dilations_0 = const()[name = string("input_5_dilations_0"), val = tensor([1, 1])]; int32 input_5_groups_0 = const()[name = string("input_5_groups_0"), val = int32(1)]; tensor input_5_cast_fp16 = conv(dilations = input_5_dilations_0, groups = input_5_groups_0, pad = input_5_pad_0, pad_type = input_5_pad_type_0, strides = input_5_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16, x = var_1828_cast_fp16_0)[name = string("input_5_cast_fp16")]; tensor var_1845_cast_fp16 = silu(x = input_5_cast_fp16)[name = string("op_1845_cast_fp16")]; tensor var_1851_strides_0 = const()[name = string("op_1851_strides_0"), val = tensor([1, 1])]; string var_1851_pad_type_0 = const()[name = string("op_1851_pad_type_0"), val = string("valid")]; tensor var_1851_pad_0 = const()[name = string("op_1851_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1851_dilations_0 = const()[name = string("op_1851_dilations_0"), val = tensor([1, 1])]; int32 var_1851_groups_0 = const()[name = string("op_1851_groups_0"), val = int32(1)]; tensor var_1851_cast_fp16 = conv(dilations = var_1851_dilations_0, groups = var_1851_groups_0, pad = var_1851_pad_0, pad_type = var_1851_pad_type_0, strides = var_1851_strides_0, weight = layers_2_mlp_up_proj_weight_cast_fp16, x = var_1828_cast_fp16_0)[name = string("op_1851_cast_fp16")]; tensor x_29_cast_fp16 = mul(x = var_1845_cast_fp16, y = var_1851_cast_fp16)[name = string("x_29_cast_fp16")]; tensor layers_2_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1188732224)))]; tensor hidden_states_27_strides_0 = const()[name = string("hidden_states_27_strides_0"), val = tensor([1, 1])]; string hidden_states_27_pad_type_0 = const()[name = string("hidden_states_27_pad_type_0"), val = string("valid")]; tensor hidden_states_27_pad_0 = const()[name = string("hidden_states_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_27_dilations_0 = const()[name = string("hidden_states_27_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_27_groups_0 = const()[name = string("hidden_states_27_groups_0"), val = int32(1)]; tensor hidden_states_27_cast_fp16 = conv(dilations = hidden_states_27_dilations_0, groups = hidden_states_27_groups_0, pad = hidden_states_27_pad_0, pad_type = hidden_states_27_pad_type_0, strides = hidden_states_27_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16, x = x_29_cast_fp16)[name = string("hidden_states_27_cast_fp16")]; tensor hidden_states_29_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = hidden_states_27_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; fp16 const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1869_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_30_promoted_to_fp16)[name = string("op_1869_cast_fp16")]; int32 var_1867 = const()[name = string("op_1867"), val = int32(1)]; bool doubled_25_interleave_0 = const()[name = string("doubled_25_interleave_0"), val = bool(false)]; tensor doubled_25_cast_fp16 = concat(axis = var_1867, interleave = doubled_25_interleave_0, values = (hidden_states_29_cast_fp16, var_1869_cast_fp16))[name = string("doubled_25_cast_fp16")]; tensor out_13_axes_0 = const()[name = string("out_13_axes_0"), val = tensor([1])]; tensor out_13_gamma_0_to_fp16 = const()[name = string("out_13_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213898112)))]; fp16 var_1879_to_fp16 = const()[name = string("op_1879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_13_cast_fp16 = layer_norm(axes = out_13_axes_0, epsilon = var_1879_to_fp16, gamma = out_13_gamma_0_to_fp16, x = doubled_25_cast_fp16)[name = string("out_13_cast_fp16")]; tensor var_1890_split_sizes_0 = const()[name = string("op_1890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1890_axis_0 = const()[name = string("op_1890_axis_0"), val = int32(1)]; tensor var_1890_cast_fp16_0, tensor var_1890_cast_fp16_1 = split(axis = var_1890_axis_0, split_sizes = var_1890_split_sizes_0, x = out_13_cast_fp16)[name = string("op_1890_cast_fp16")]; tensor layers_3_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213906368)))]; tensor query_states_19_strides_0 = const()[name = string("query_states_19_strides_0"), val = tensor([1, 1])]; string query_states_19_pad_type_0 = const()[name = string("query_states_19_pad_type_0"), val = string("valid")]; tensor query_states_19_pad_0 = const()[name = string("query_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_19_dilations_0 = const()[name = string("query_states_19_dilations_0"), val = tensor([1, 1])]; int32 query_states_19_groups_0 = const()[name = string("query_states_19_groups_0"), val = int32(1)]; tensor query_states_19_cast_fp16 = conv(dilations = query_states_19_dilations_0, groups = query_states_19_groups_0, pad = query_states_19_pad_0, pad_type = query_states_19_pad_type_0, strides = query_states_19_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("query_states_19_cast_fp16")]; tensor layers_3_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1222295040)))]; tensor key_states_31_strides_0 = const()[name = string("key_states_31_strides_0"), val = tensor([1, 1])]; string key_states_31_pad_type_0 = const()[name = string("key_states_31_pad_type_0"), val = string("valid")]; tensor key_states_31_pad_0 = const()[name = string("key_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_31_dilations_0 = const()[name = string("key_states_31_dilations_0"), val = tensor([1, 1])]; int32 key_states_31_groups_0 = const()[name = string("key_states_31_groups_0"), val = int32(1)]; tensor key_states_31_cast_fp16 = conv(dilations = key_states_31_dilations_0, groups = key_states_31_groups_0, pad = key_states_31_pad_0, pad_type = key_states_31_pad_type_0, strides = key_states_31_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("key_states_31_cast_fp16")]; tensor value_states_19_strides_0 = const()[name = string("value_states_19_strides_0"), val = tensor([1, 1])]; string value_states_19_pad_type_0 = const()[name = string("value_states_19_pad_type_0"), val = string("valid")]; tensor value_states_19_pad_0 = const()[name = string("value_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_19_dilations_0 = const()[name = string("value_states_19_dilations_0"), val = tensor([1, 1])]; int32 value_states_19_groups_0 = const()[name = string("value_states_19_groups_0"), val = int32(1)]; tensor value_states_19_cast_fp16 = conv(dilations = value_states_19_dilations_0, groups = value_states_19_groups_0, pad = value_states_19_pad_0, pad_type = value_states_19_pad_type_0, strides = value_states_19_strides_0, weight = layers_3_self_attn_v_proj_weight_cast_fp16, x = var_1890_cast_fp16_0)[name = string("value_states_19_cast_fp16")]; tensor concat_36x = const()[name = string("concat_36x"), val = tensor([1, 16, 128, -1])]; tensor x_31_cast_fp16 = reshape(shape = concat_36x, x = query_states_19_cast_fp16)[name = string("x_31_cast_fp16")]; tensor concat_37x = const()[name = string("concat_37x"), val = tensor([1, 2, 128, -1])]; tensor var_1947_cast_fp16 = reshape(shape = concat_37x, x = key_states_31_cast_fp16)[name = string("op_1947_cast_fp16")]; tensor concat_38x = const()[name = string("concat_38x"), val = tensor([1, 2, 128, -1])]; tensor var_1954_cast_fp16 = reshape(shape = concat_38x, x = value_states_19_cast_fp16)[name = string("op_1954_cast_fp16")]; tensor var_1958_cast_fp16 = mul(x = x_31_cast_fp16, y = var_869_cast_fp16)[name = string("op_1958_cast_fp16")]; tensor var_1959_split_sizes_0 = const()[name = string("op_1959_split_sizes_0"), val = tensor([64, 64])]; int32 var_1959_axis_0 = const()[name = string("op_1959_axis_0"), val = int32(-2)]; tensor var_1959_cast_fp16_0, tensor var_1959_cast_fp16_1 = split(axis = var_1959_axis_0, split_sizes = var_1959_split_sizes_0, x = x_31_cast_fp16)[name = string("op_1959_cast_fp16")]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1961_cast_fp16 = mul(x = var_1959_cast_fp16_1, y = const_32_promoted_to_fp16)[name = string("op_1961_cast_fp16")]; int32 var_1963 = const()[name = string("op_1963"), val = int32(-2)]; bool var_1964_interleave_0 = const()[name = string("op_1964_interleave_0"), val = bool(false)]; tensor var_1964_cast_fp16 = concat(axis = var_1963, interleave = var_1964_interleave_0, values = (var_1961_cast_fp16, var_1959_cast_fp16_0))[name = string("op_1964_cast_fp16")]; tensor var_1965_cast_fp16 = mul(x = var_1964_cast_fp16, y = var_878_cast_fp16)[name = string("op_1965_cast_fp16")]; tensor query_states_21_cast_fp16 = add(x = var_1958_cast_fp16, y = var_1965_cast_fp16)[name = string("query_states_21_cast_fp16")]; tensor var_1971_cast_fp16 = mul(x = var_1947_cast_fp16, y = var_869_cast_fp16)[name = string("op_1971_cast_fp16")]; tensor var_1972_split_sizes_0 = const()[name = string("op_1972_split_sizes_0"), val = tensor([64, 64])]; int32 var_1972_axis_0 = const()[name = string("op_1972_axis_0"), val = int32(-2)]; tensor var_1972_cast_fp16_0, tensor var_1972_cast_fp16_1 = split(axis = var_1972_axis_0, split_sizes = var_1972_split_sizes_0, x = var_1947_cast_fp16)[name = string("op_1972_cast_fp16")]; fp16 const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1974_cast_fp16 = mul(x = var_1972_cast_fp16_1, y = const_33_promoted_to_fp16)[name = string("op_1974_cast_fp16")]; int32 var_1976 = const()[name = string("op_1976"), val = int32(-2)]; bool var_1977_interleave_0 = const()[name = string("op_1977_interleave_0"), val = bool(false)]; tensor var_1977_cast_fp16 = concat(axis = var_1976, interleave = var_1977_interleave_0, values = (var_1974_cast_fp16, var_1972_cast_fp16_0))[name = string("op_1977_cast_fp16")]; tensor var_1978_cast_fp16 = mul(x = var_1977_cast_fp16, y = var_878_cast_fp16)[name = string("op_1978_cast_fp16")]; tensor key_states_35_cast_fp16 = add(x = var_1971_cast_fp16, y = var_1978_cast_fp16)[name = string("key_states_35_cast_fp16")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; int32 concat_41_axis_0 = const()[name = string("concat_41_axis_0"), val = int32(0)]; bool concat_41_interleave_0 = const()[name = string("concat_41_interleave_0"), val = bool(false)]; tensor concat_41 = concat(axis = concat_41_axis_0, interleave = concat_41_interleave_0, values = (expand_dims_36, expand_dims_37, position_id, expand_dims_39))[name = string("concat_41")]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; tensor concat_42_values1_0 = const()[name = string("concat_42_values1_0"), val = tensor([0])]; tensor concat_42_values3_0 = const()[name = string("concat_42_values3_0"), val = tensor([0])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_40, concat_42_values1_0, cache_position_end, concat_42_values3_0))[name = string("concat_42")]; tensor key_states_37_perm_0 = const()[name = string("key_states_37_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_4_stride_0 = const()[name = string("key_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_37_cast_fp16 = transpose(perm = key_states_37_perm_0, x = key_states_35_cast_fp16)[name = string("transpose_246")]; tensor key_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = key_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = key_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_4_squeeze_mask_0, stride = key_cache_internal_tensor_assign_4_stride_0, update = key_states_37_cast_fp16, x = coreml_update_state_116)[name = string("key_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_4_cast_fp16, input = key_cache)[name = string("coreml_update_state_118_write_state")]; tensor coreml_update_state_118 = read_state(input = key_cache)[name = string("coreml_update_state_118")]; tensor value_states_21_perm_0 = const()[name = string("value_states_21_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_4_stride_0 = const()[name = string("value_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_21_cast_fp16 = transpose(perm = value_states_21_perm_0, x = var_1954_cast_fp16)[name = string("transpose_245")]; tensor value_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = value_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = value_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_4_squeeze_mask_0, stride = value_cache_internal_tensor_assign_4_stride_0, update = value_states_21_cast_fp16, x = coreml_update_state_117)[name = string("value_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_4_cast_fp16, input = value_cache)[name = string("coreml_update_state_119_write_state")]; tensor coreml_update_state_119 = read_state(input = value_cache)[name = string("coreml_update_state_119")]; tensor var_2048_begin_0 = const()[name = string("op_2048_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2048_end_0 = const()[name = string("op_2048_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2048_end_mask_0 = const()[name = string("op_2048_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2048_cast_fp16 = slice_by_index(begin = var_2048_begin_0, end = var_2048_end_0, end_mask = var_2048_end_mask_0, x = coreml_update_state_118)[name = string("op_2048_cast_fp16")]; tensor tile_6 = const()[name = string("tile_6"), val = tensor([1, 1])]; int32 var_2051_axis_0 = const()[name = string("op_2051_axis_0"), val = int32(1)]; tensor var_2051_cast_fp16_0, tensor var_2051_cast_fp16_1 = split(axis = var_2051_axis_0, split_sizes = tile_6, x = var_2048_cast_fp16)[name = string("op_2051_cast_fp16")]; tensor var_2058_begin_0 = const()[name = string("op_2058_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2058_end_0 = const()[name = string("op_2058_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2058_end_mask_0 = const()[name = string("op_2058_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2058_cast_fp16 = slice_by_index(begin = var_2058_begin_0, end = var_2058_end_0, end_mask = var_2058_end_mask_0, x = coreml_update_state_119)[name = string("op_2058_cast_fp16")]; tensor tile_7 = const()[name = string("tile_7"), val = tensor([1, 1])]; int32 var_2061_axis_0 = const()[name = string("op_2061_axis_0"), val = int32(1)]; tensor var_2061_cast_fp16_0, tensor var_2061_cast_fp16_1 = split(axis = var_2061_axis_0, split_sizes = tile_7, x = var_2058_cast_fp16)[name = string("op_2061_cast_fp16")]; tensor var_2064_split_sizes_0 = const()[name = string("op_2064_split_sizes_0"), val = tensor([8, 8])]; int32 var_2064_axis_0 = const()[name = string("op_2064_axis_0"), val = int32(1)]; tensor var_2064_0, tensor var_2064_1 = split(axis = var_2064_axis_0, split_sizes = var_2064_split_sizes_0, x = query_states_21_cast_fp16)[name = string("op_2064")]; bool attn_weights_49_transpose_x_0 = const()[name = string("attn_weights_49_transpose_x_0"), val = bool(false)]; bool attn_weights_49_transpose_y_0 = const()[name = string("attn_weights_49_transpose_y_0"), val = bool(false)]; tensor attn_weights_49_cast_fp16 = matmul(transpose_x = attn_weights_49_transpose_x_0, transpose_y = attn_weights_49_transpose_y_0, x = var_2051_cast_fp16_0, y = var_2064_0)[name = string("attn_weights_49_cast_fp16")]; fp16 var_2067_to_fp16 = const()[name = string("op_2067_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_51_cast_fp16 = mul(x = attn_weights_49_cast_fp16, y = var_2067_to_fp16)[name = string("attn_weights_51_cast_fp16")]; tensor attn_weights_53_cast_fp16 = add(x = attn_weights_51_cast_fp16, y = attn_mask_1)[name = string("attn_weights_53_cast_fp16")]; int32 var_2071 = const()[name = string("op_2071"), val = int32(-2)]; tensor attn_weights_55_cast_fp16 = softmax(axis = var_2071, x = attn_weights_53_cast_fp16)[name = string("attn_weights_55_cast_fp16")]; bool var_2077_transpose_x_1 = const()[name = string("op_2077_transpose_x_1"), val = bool(true)]; bool var_2077_transpose_y_1 = const()[name = string("op_2077_transpose_y_1"), val = bool(false)]; tensor var_2077_cast_fp16 = matmul(transpose_x = var_2077_transpose_x_1, transpose_y = var_2077_transpose_y_1, x = attn_weights_55_cast_fp16, y = var_2061_cast_fp16_0)[name = string("op_2077_cast_fp16")]; bool attn_weights_57_transpose_x_0 = const()[name = string("attn_weights_57_transpose_x_0"), val = bool(false)]; bool attn_weights_57_transpose_y_0 = const()[name = string("attn_weights_57_transpose_y_0"), val = bool(false)]; tensor attn_weights_57_cast_fp16 = matmul(transpose_x = attn_weights_57_transpose_x_0, transpose_y = attn_weights_57_transpose_y_0, x = var_2051_cast_fp16_1, y = var_2064_1)[name = string("attn_weights_57_cast_fp16")]; fp16 var_2079_to_fp16 = const()[name = string("op_2079_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_59_cast_fp16 = mul(x = attn_weights_57_cast_fp16, y = var_2079_to_fp16)[name = string("attn_weights_59_cast_fp16")]; tensor attn_weights_61_cast_fp16 = add(x = attn_weights_59_cast_fp16, y = attn_mask_1)[name = string("attn_weights_61_cast_fp16")]; int32 var_2083 = const()[name = string("op_2083"), val = int32(-2)]; tensor attn_weights_63_cast_fp16 = softmax(axis = var_2083, x = attn_weights_61_cast_fp16)[name = string("attn_weights_63_cast_fp16")]; bool attn_output_25_transpose_x_1 = const()[name = string("attn_output_25_transpose_x_1"), val = bool(true)]; bool attn_output_25_transpose_y_1 = const()[name = string("attn_output_25_transpose_y_1"), val = bool(false)]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_1, transpose_y = attn_output_25_transpose_y_1, x = attn_weights_63_cast_fp16, y = var_2061_cast_fp16_1)[name = string("attn_output_25_cast_fp16")]; int32 var_2091 = const()[name = string("op_2091"), val = int32(1)]; bool attn_output_27_interleave_0 = const()[name = string("attn_output_27_interleave_0"), val = bool(false)]; tensor attn_output_27_cast_fp16 = concat(axis = var_2091, interleave = attn_output_27_interleave_0, values = (var_2077_cast_fp16, attn_output_25_cast_fp16))[name = string("attn_output_27_cast_fp16")]; tensor var_2095_perm_0 = const()[name = string("op_2095_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_47x = const()[name = string("concat_47x"), val = tensor([1, 2048, 1, -1])]; tensor var_2095_cast_fp16 = transpose(perm = var_2095_perm_0, x = attn_output_27_cast_fp16)[name = string("transpose_244")]; tensor attn_output_31_cast_fp16 = reshape(shape = concat_47x, x = var_2095_cast_fp16)[name = string("attn_output_31_cast_fp16")]; tensor hidden_states_33_strides_0 = const()[name = string("hidden_states_33_strides_0"), val = tensor([1, 1])]; string hidden_states_33_pad_type_0 = const()[name = string("hidden_states_33_pad_type_0"), val = string("valid")]; tensor hidden_states_33_pad_0 = const()[name = string("hidden_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_33_dilations_0 = const()[name = string("hidden_states_33_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_33_groups_0 = const()[name = string("hidden_states_33_groups_0"), val = int32(1)]; tensor hidden_states_33_cast_fp16 = conv(dilations = hidden_states_33_dilations_0, groups = hidden_states_33_groups_0, pad = hidden_states_33_pad_0, pad_type = hidden_states_33_pad_type_0, strides = hidden_states_33_strides_0, weight = layers_3_self_attn_o_proj_weight_cast_fp16, x = attn_output_31_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; tensor hidden_states_35_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = hidden_states_33_cast_fp16)[name = string("hidden_states_35_cast_fp16")]; fp16 const_38_promoted_to_fp16 = const()[name = string("const_38_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2128_cast_fp16 = mul(x = hidden_states_35_cast_fp16, y = const_38_promoted_to_fp16)[name = string("op_2128_cast_fp16")]; int32 var_2126 = const()[name = string("op_2126"), val = int32(1)]; bool doubled_29_interleave_0 = const()[name = string("doubled_29_interleave_0"), val = bool(false)]; tensor doubled_29_cast_fp16 = concat(axis = var_2126, interleave = doubled_29_interleave_0, values = (hidden_states_35_cast_fp16, var_2128_cast_fp16))[name = string("doubled_29_cast_fp16")]; tensor out_15_axes_0 = const()[name = string("out_15_axes_0"), val = tensor([1])]; tensor out_15_gamma_0_to_fp16 = const()[name = string("out_15_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223343680)))]; fp16 var_2138_to_fp16 = const()[name = string("op_2138_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_15_cast_fp16 = layer_norm(axes = out_15_axes_0, epsilon = var_2138_to_fp16, gamma = out_15_gamma_0_to_fp16, x = doubled_29_cast_fp16)[name = string("out_15_cast_fp16")]; tensor var_2149_split_sizes_0 = const()[name = string("op_2149_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2149_axis_0 = const()[name = string("op_2149_axis_0"), val = int32(1)]; tensor var_2149_cast_fp16_0, tensor var_2149_cast_fp16_1 = split(axis = var_2149_axis_0, split_sizes = var_2149_split_sizes_0, x = out_15_cast_fp16)[name = string("op_2149_cast_fp16")]; tensor layers_3_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223351936)))]; tensor input_7_strides_0 = const()[name = string("input_7_strides_0"), val = tensor([1, 1])]; string input_7_pad_type_0 = const()[name = string("input_7_pad_type_0"), val = string("valid")]; tensor input_7_pad_0 = const()[name = string("input_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_7_dilations_0 = const()[name = string("input_7_dilations_0"), val = tensor([1, 1])]; int32 input_7_groups_0 = const()[name = string("input_7_groups_0"), val = int32(1)]; tensor input_7_cast_fp16 = conv(dilations = input_7_dilations_0, groups = input_7_groups_0, pad = input_7_pad_0, pad_type = input_7_pad_type_0, strides = input_7_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("input_7_cast_fp16")]; tensor var_2166_cast_fp16 = silu(x = input_7_cast_fp16)[name = string("op_2166_cast_fp16")]; tensor layers_3_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1248517824)))]; tensor var_2172_strides_0 = const()[name = string("op_2172_strides_0"), val = tensor([1, 1])]; string var_2172_pad_type_0 = const()[name = string("op_2172_pad_type_0"), val = string("valid")]; tensor var_2172_pad_0 = const()[name = string("op_2172_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2172_dilations_0 = const()[name = string("op_2172_dilations_0"), val = tensor([1, 1])]; int32 var_2172_groups_0 = const()[name = string("op_2172_groups_0"), val = int32(1)]; tensor var_2172_cast_fp16 = conv(dilations = var_2172_dilations_0, groups = var_2172_groups_0, pad = var_2172_pad_0, pad_type = var_2172_pad_type_0, strides = var_2172_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("op_2172_cast_fp16")]; tensor x_39_cast_fp16 = mul(x = var_2166_cast_fp16, y = var_2172_cast_fp16)[name = string("x_39_cast_fp16")]; tensor hidden_states_37_strides_0 = const()[name = string("hidden_states_37_strides_0"), val = tensor([1, 1])]; string hidden_states_37_pad_type_0 = const()[name = string("hidden_states_37_pad_type_0"), val = string("valid")]; tensor hidden_states_37_pad_0 = const()[name = string("hidden_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_37_dilations_0 = const()[name = string("hidden_states_37_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_37_groups_0 = const()[name = string("hidden_states_37_groups_0"), val = int32(1)]; tensor hidden_states_37_cast_fp16 = conv(dilations = hidden_states_37_dilations_0, groups = hidden_states_37_groups_0, pad = hidden_states_37_pad_0, pad_type = hidden_states_37_pad_type_0, strides = hidden_states_37_strides_0, weight = layers_3_mlp_down_proj_weight_cast_fp16, x = x_39_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; tensor hidden_states_39_cast_fp16 = add(x = hidden_states_35_cast_fp16, y = hidden_states_37_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; fp16 const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2190_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_40_promoted_to_fp16)[name = string("op_2190_cast_fp16")]; int32 var_2188 = const()[name = string("op_2188"), val = int32(1)]; bool doubled_33_interleave_0 = const()[name = string("doubled_33_interleave_0"), val = bool(false)]; tensor doubled_33_cast_fp16 = concat(axis = var_2188, interleave = doubled_33_interleave_0, values = (hidden_states_39_cast_fp16, var_2190_cast_fp16))[name = string("doubled_33_cast_fp16")]; tensor out_17_axes_0 = const()[name = string("out_17_axes_0"), val = tensor([1])]; tensor out_17_gamma_0_to_fp16 = const()[name = string("out_17_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273683712)))]; fp16 var_2200_to_fp16 = const()[name = string("op_2200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_17_cast_fp16 = layer_norm(axes = out_17_axes_0, epsilon = var_2200_to_fp16, gamma = out_17_gamma_0_to_fp16, x = doubled_33_cast_fp16)[name = string("out_17_cast_fp16")]; tensor var_2211_split_sizes_0 = const()[name = string("op_2211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2211_axis_0 = const()[name = string("op_2211_axis_0"), val = int32(1)]; tensor var_2211_cast_fp16_0, tensor var_2211_cast_fp16_1 = split(axis = var_2211_axis_0, split_sizes = var_2211_split_sizes_0, x = out_17_cast_fp16)[name = string("op_2211_cast_fp16")]; tensor layers_4_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273691968)))]; tensor query_states_25_strides_0 = const()[name = string("query_states_25_strides_0"), val = tensor([1, 1])]; string query_states_25_pad_type_0 = const()[name = string("query_states_25_pad_type_0"), val = string("valid")]; tensor query_states_25_pad_0 = const()[name = string("query_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_25_dilations_0 = const()[name = string("query_states_25_dilations_0"), val = tensor([1, 1])]; int32 query_states_25_groups_0 = const()[name = string("query_states_25_groups_0"), val = int32(1)]; tensor query_states_25_cast_fp16 = conv(dilations = query_states_25_dilations_0, groups = query_states_25_groups_0, pad = query_states_25_pad_0, pad_type = query_states_25_pad_type_0, strides = query_states_25_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("query_states_25_cast_fp16")]; tensor layers_4_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1282080640)))]; tensor key_states_41_strides_0 = const()[name = string("key_states_41_strides_0"), val = tensor([1, 1])]; string key_states_41_pad_type_0 = const()[name = string("key_states_41_pad_type_0"), val = string("valid")]; tensor key_states_41_pad_0 = const()[name = string("key_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_41_dilations_0 = const()[name = string("key_states_41_dilations_0"), val = tensor([1, 1])]; int32 key_states_41_groups_0 = const()[name = string("key_states_41_groups_0"), val = int32(1)]; tensor key_states_41_cast_fp16 = conv(dilations = key_states_41_dilations_0, groups = key_states_41_groups_0, pad = key_states_41_pad_0, pad_type = key_states_41_pad_type_0, strides = key_states_41_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("key_states_41_cast_fp16")]; tensor value_states_25_strides_0 = const()[name = string("value_states_25_strides_0"), val = tensor([1, 1])]; string value_states_25_pad_type_0 = const()[name = string("value_states_25_pad_type_0"), val = string("valid")]; tensor value_states_25_pad_0 = const()[name = string("value_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_25_dilations_0 = const()[name = string("value_states_25_dilations_0"), val = tensor([1, 1])]; int32 value_states_25_groups_0 = const()[name = string("value_states_25_groups_0"), val = int32(1)]; tensor value_states_25_cast_fp16 = conv(dilations = value_states_25_dilations_0, groups = value_states_25_groups_0, pad = value_states_25_pad_0, pad_type = value_states_25_pad_type_0, strides = value_states_25_strides_0, weight = layers_4_self_attn_v_proj_weight_cast_fp16, x = var_2211_cast_fp16_0)[name = string("value_states_25_cast_fp16")]; tensor concat_48x = const()[name = string("concat_48x"), val = tensor([1, 16, 128, -1])]; tensor x_41_cast_fp16 = reshape(shape = concat_48x, x = query_states_25_cast_fp16)[name = string("x_41_cast_fp16")]; tensor concat_49x = const()[name = string("concat_49x"), val = tensor([1, 2, 128, -1])]; tensor var_2268_cast_fp16 = reshape(shape = concat_49x, x = key_states_41_cast_fp16)[name = string("op_2268_cast_fp16")]; tensor concat_50x = const()[name = string("concat_50x"), val = tensor([1, 2, 128, -1])]; tensor var_2275_cast_fp16 = reshape(shape = concat_50x, x = value_states_25_cast_fp16)[name = string("op_2275_cast_fp16")]; tensor var_2279_cast_fp16 = mul(x = x_41_cast_fp16, y = var_869_cast_fp16)[name = string("op_2279_cast_fp16")]; tensor var_2280_split_sizes_0 = const()[name = string("op_2280_split_sizes_0"), val = tensor([64, 64])]; int32 var_2280_axis_0 = const()[name = string("op_2280_axis_0"), val = int32(-2)]; tensor var_2280_cast_fp16_0, tensor var_2280_cast_fp16_1 = split(axis = var_2280_axis_0, split_sizes = var_2280_split_sizes_0, x = x_41_cast_fp16)[name = string("op_2280_cast_fp16")]; fp16 const_42_promoted_to_fp16 = const()[name = string("const_42_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2282_cast_fp16 = mul(x = var_2280_cast_fp16_1, y = const_42_promoted_to_fp16)[name = string("op_2282_cast_fp16")]; int32 var_2284 = const()[name = string("op_2284"), val = int32(-2)]; bool var_2285_interleave_0 = const()[name = string("op_2285_interleave_0"), val = bool(false)]; tensor var_2285_cast_fp16 = concat(axis = var_2284, interleave = var_2285_interleave_0, values = (var_2282_cast_fp16, var_2280_cast_fp16_0))[name = string("op_2285_cast_fp16")]; tensor var_2286_cast_fp16 = mul(x = var_2285_cast_fp16, y = var_878_cast_fp16)[name = string("op_2286_cast_fp16")]; tensor query_states_27_cast_fp16 = add(x = var_2279_cast_fp16, y = var_2286_cast_fp16)[name = string("query_states_27_cast_fp16")]; tensor var_2292_cast_fp16 = mul(x = var_2268_cast_fp16, y = var_869_cast_fp16)[name = string("op_2292_cast_fp16")]; tensor var_2293_split_sizes_0 = const()[name = string("op_2293_split_sizes_0"), val = tensor([64, 64])]; int32 var_2293_axis_0 = const()[name = string("op_2293_axis_0"), val = int32(-2)]; tensor var_2293_cast_fp16_0, tensor var_2293_cast_fp16_1 = split(axis = var_2293_axis_0, split_sizes = var_2293_split_sizes_0, x = var_2268_cast_fp16)[name = string("op_2293_cast_fp16")]; fp16 const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2295_cast_fp16 = mul(x = var_2293_cast_fp16_1, y = const_43_promoted_to_fp16)[name = string("op_2295_cast_fp16")]; int32 var_2297 = const()[name = string("op_2297"), val = int32(-2)]; bool var_2298_interleave_0 = const()[name = string("op_2298_interleave_0"), val = bool(false)]; tensor var_2298_cast_fp16 = concat(axis = var_2297, interleave = var_2298_interleave_0, values = (var_2295_cast_fp16, var_2293_cast_fp16_0))[name = string("op_2298_cast_fp16")]; tensor var_2299_cast_fp16 = mul(x = var_2298_cast_fp16, y = var_878_cast_fp16)[name = string("op_2299_cast_fp16")]; tensor key_states_45_cast_fp16 = add(x = var_2292_cast_fp16, y = var_2299_cast_fp16)[name = string("key_states_45_cast_fp16")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; int32 concat_53_axis_0 = const()[name = string("concat_53_axis_0"), val = int32(0)]; bool concat_53_interleave_0 = const()[name = string("concat_53_interleave_0"), val = bool(false)]; tensor concat_53 = concat(axis = concat_53_axis_0, interleave = concat_53_interleave_0, values = (expand_dims_48, expand_dims_49, position_id, expand_dims_51))[name = string("concat_53")]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; tensor concat_54_values1_0 = const()[name = string("concat_54_values1_0"), val = tensor([0])]; tensor concat_54_values3_0 = const()[name = string("concat_54_values3_0"), val = tensor([0])]; int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_52, concat_54_values1_0, cache_position_end, concat_54_values3_0))[name = string("concat_54")]; tensor key_states_47_perm_0 = const()[name = string("key_states_47_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_5_stride_0 = const()[name = string("key_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_47_cast_fp16 = transpose(perm = key_states_47_perm_0, x = key_states_45_cast_fp16)[name = string("transpose_243")]; tensor key_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = key_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = key_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_5_squeeze_mask_0, stride = key_cache_internal_tensor_assign_5_stride_0, update = key_states_47_cast_fp16, x = coreml_update_state_118)[name = string("key_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_5_cast_fp16, input = key_cache)[name = string("coreml_update_state_120_write_state")]; tensor coreml_update_state_120 = read_state(input = key_cache)[name = string("coreml_update_state_120")]; tensor value_states_27_perm_0 = const()[name = string("value_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_5_stride_0 = const()[name = string("value_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_27_cast_fp16 = transpose(perm = value_states_27_perm_0, x = var_2275_cast_fp16)[name = string("transpose_242")]; tensor value_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = value_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = value_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_5_squeeze_mask_0, stride = value_cache_internal_tensor_assign_5_stride_0, update = value_states_27_cast_fp16, x = coreml_update_state_119)[name = string("value_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_5_cast_fp16, input = value_cache)[name = string("coreml_update_state_121_write_state")]; tensor coreml_update_state_121 = read_state(input = value_cache)[name = string("coreml_update_state_121")]; tensor var_2369_begin_0 = const()[name = string("op_2369_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2369_end_0 = const()[name = string("op_2369_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2369_end_mask_0 = const()[name = string("op_2369_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2369_cast_fp16 = slice_by_index(begin = var_2369_begin_0, end = var_2369_end_0, end_mask = var_2369_end_mask_0, x = coreml_update_state_120)[name = string("op_2369_cast_fp16")]; tensor tile_8 = const()[name = string("tile_8"), val = tensor([1, 1])]; int32 var_2372_axis_0 = const()[name = string("op_2372_axis_0"), val = int32(1)]; tensor var_2372_cast_fp16_0, tensor var_2372_cast_fp16_1 = split(axis = var_2372_axis_0, split_sizes = tile_8, x = var_2369_cast_fp16)[name = string("op_2372_cast_fp16")]; tensor var_2379_begin_0 = const()[name = string("op_2379_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2379_end_0 = const()[name = string("op_2379_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2379_end_mask_0 = const()[name = string("op_2379_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2379_cast_fp16 = slice_by_index(begin = var_2379_begin_0, end = var_2379_end_0, end_mask = var_2379_end_mask_0, x = coreml_update_state_121)[name = string("op_2379_cast_fp16")]; tensor tile_9 = const()[name = string("tile_9"), val = tensor([1, 1])]; int32 var_2382_axis_0 = const()[name = string("op_2382_axis_0"), val = int32(1)]; tensor var_2382_cast_fp16_0, tensor var_2382_cast_fp16_1 = split(axis = var_2382_axis_0, split_sizes = tile_9, x = var_2379_cast_fp16)[name = string("op_2382_cast_fp16")]; tensor var_2385_split_sizes_0 = const()[name = string("op_2385_split_sizes_0"), val = tensor([8, 8])]; int32 var_2385_axis_0 = const()[name = string("op_2385_axis_0"), val = int32(1)]; tensor var_2385_0, tensor var_2385_1 = split(axis = var_2385_axis_0, split_sizes = var_2385_split_sizes_0, x = query_states_27_cast_fp16)[name = string("op_2385")]; bool attn_weights_65_transpose_x_0 = const()[name = string("attn_weights_65_transpose_x_0"), val = bool(false)]; bool attn_weights_65_transpose_y_0 = const()[name = string("attn_weights_65_transpose_y_0"), val = bool(false)]; tensor attn_weights_65_cast_fp16 = matmul(transpose_x = attn_weights_65_transpose_x_0, transpose_y = attn_weights_65_transpose_y_0, x = var_2372_cast_fp16_0, y = var_2385_0)[name = string("attn_weights_65_cast_fp16")]; fp16 var_2388_to_fp16 = const()[name = string("op_2388_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_67_cast_fp16 = mul(x = attn_weights_65_cast_fp16, y = var_2388_to_fp16)[name = string("attn_weights_67_cast_fp16")]; tensor attn_weights_69_cast_fp16 = add(x = attn_weights_67_cast_fp16, y = attn_mask_1)[name = string("attn_weights_69_cast_fp16")]; int32 var_2392 = const()[name = string("op_2392"), val = int32(-2)]; tensor attn_weights_71_cast_fp16 = softmax(axis = var_2392, x = attn_weights_69_cast_fp16)[name = string("attn_weights_71_cast_fp16")]; bool var_2398_transpose_x_1 = const()[name = string("op_2398_transpose_x_1"), val = bool(true)]; bool var_2398_transpose_y_1 = const()[name = string("op_2398_transpose_y_1"), val = bool(false)]; tensor var_2398_cast_fp16 = matmul(transpose_x = var_2398_transpose_x_1, transpose_y = var_2398_transpose_y_1, x = attn_weights_71_cast_fp16, y = var_2382_cast_fp16_0)[name = string("op_2398_cast_fp16")]; bool attn_weights_73_transpose_x_0 = const()[name = string("attn_weights_73_transpose_x_0"), val = bool(false)]; bool attn_weights_73_transpose_y_0 = const()[name = string("attn_weights_73_transpose_y_0"), val = bool(false)]; tensor attn_weights_73_cast_fp16 = matmul(transpose_x = attn_weights_73_transpose_x_0, transpose_y = attn_weights_73_transpose_y_0, x = var_2372_cast_fp16_1, y = var_2385_1)[name = string("attn_weights_73_cast_fp16")]; fp16 var_2400_to_fp16 = const()[name = string("op_2400_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_75_cast_fp16 = mul(x = attn_weights_73_cast_fp16, y = var_2400_to_fp16)[name = string("attn_weights_75_cast_fp16")]; tensor attn_weights_77_cast_fp16 = add(x = attn_weights_75_cast_fp16, y = attn_mask_1)[name = string("attn_weights_77_cast_fp16")]; int32 var_2404 = const()[name = string("op_2404"), val = int32(-2)]; tensor attn_weights_79_cast_fp16 = softmax(axis = var_2404, x = attn_weights_77_cast_fp16)[name = string("attn_weights_79_cast_fp16")]; bool attn_output_33_transpose_x_1 = const()[name = string("attn_output_33_transpose_x_1"), val = bool(true)]; bool attn_output_33_transpose_y_1 = const()[name = string("attn_output_33_transpose_y_1"), val = bool(false)]; tensor attn_output_33_cast_fp16 = matmul(transpose_x = attn_output_33_transpose_x_1, transpose_y = attn_output_33_transpose_y_1, x = attn_weights_79_cast_fp16, y = var_2382_cast_fp16_1)[name = string("attn_output_33_cast_fp16")]; int32 var_2412 = const()[name = string("op_2412"), val = int32(1)]; bool attn_output_35_interleave_0 = const()[name = string("attn_output_35_interleave_0"), val = bool(false)]; tensor attn_output_35_cast_fp16 = concat(axis = var_2412, interleave = attn_output_35_interleave_0, values = (var_2398_cast_fp16, attn_output_33_cast_fp16))[name = string("attn_output_35_cast_fp16")]; tensor var_2416_perm_0 = const()[name = string("op_2416_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_59x = const()[name = string("concat_59x"), val = tensor([1, 2048, 1, -1])]; tensor var_2416_cast_fp16 = transpose(perm = var_2416_perm_0, x = attn_output_35_cast_fp16)[name = string("transpose_241")]; tensor attn_output_39_cast_fp16 = reshape(shape = concat_59x, x = var_2416_cast_fp16)[name = string("attn_output_39_cast_fp16")]; tensor hidden_states_43_strides_0 = const()[name = string("hidden_states_43_strides_0"), val = tensor([1, 1])]; string hidden_states_43_pad_type_0 = const()[name = string("hidden_states_43_pad_type_0"), val = string("valid")]; tensor hidden_states_43_pad_0 = const()[name = string("hidden_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_43_dilations_0 = const()[name = string("hidden_states_43_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_43_groups_0 = const()[name = string("hidden_states_43_groups_0"), val = int32(1)]; tensor hidden_states_43_cast_fp16 = conv(dilations = hidden_states_43_dilations_0, groups = hidden_states_43_groups_0, pad = hidden_states_43_pad_0, pad_type = hidden_states_43_pad_type_0, strides = hidden_states_43_strides_0, weight = layers_4_self_attn_o_proj_weight_cast_fp16, x = attn_output_39_cast_fp16)[name = string("hidden_states_43_cast_fp16")]; tensor hidden_states_45_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = hidden_states_43_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; fp16 const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2449_cast_fp16 = mul(x = hidden_states_45_cast_fp16, y = const_48_promoted_to_fp16)[name = string("op_2449_cast_fp16")]; int32 var_2447 = const()[name = string("op_2447"), val = int32(1)]; bool doubled_37_interleave_0 = const()[name = string("doubled_37_interleave_0"), val = bool(false)]; tensor doubled_37_cast_fp16 = concat(axis = var_2447, interleave = doubled_37_interleave_0, values = (hidden_states_45_cast_fp16, var_2449_cast_fp16))[name = string("doubled_37_cast_fp16")]; tensor out_19_axes_0 = const()[name = string("out_19_axes_0"), val = tensor([1])]; tensor out_19_gamma_0_to_fp16 = const()[name = string("out_19_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283129280)))]; fp16 var_2459_to_fp16 = const()[name = string("op_2459_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_19_cast_fp16 = layer_norm(axes = out_19_axes_0, epsilon = var_2459_to_fp16, gamma = out_19_gamma_0_to_fp16, x = doubled_37_cast_fp16)[name = string("out_19_cast_fp16")]; tensor var_2470_split_sizes_0 = const()[name = string("op_2470_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2470_axis_0 = const()[name = string("op_2470_axis_0"), val = int32(1)]; tensor var_2470_cast_fp16_0, tensor var_2470_cast_fp16_1 = split(axis = var_2470_axis_0, split_sizes = var_2470_split_sizes_0, x = out_19_cast_fp16)[name = string("op_2470_cast_fp16")]; tensor input_9_strides_0 = const()[name = string("input_9_strides_0"), val = tensor([1, 1])]; string input_9_pad_type_0 = const()[name = string("input_9_pad_type_0"), val = string("valid")]; tensor input_9_pad_0 = const()[name = string("input_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_9_dilations_0 = const()[name = string("input_9_dilations_0"), val = tensor([1, 1])]; int32 input_9_groups_0 = const()[name = string("input_9_groups_0"), val = int32(1)]; tensor input_9_cast_fp16 = conv(dilations = input_9_dilations_0, groups = input_9_groups_0, pad = input_9_pad_0, pad_type = input_9_pad_type_0, strides = input_9_strides_0, weight = layers_4_mlp_gate_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("input_9_cast_fp16")]; tensor var_2487_cast_fp16 = silu(x = input_9_cast_fp16)[name = string("op_2487_cast_fp16")]; tensor var_2493_strides_0 = const()[name = string("op_2493_strides_0"), val = tensor([1, 1])]; string var_2493_pad_type_0 = const()[name = string("op_2493_pad_type_0"), val = string("valid")]; tensor var_2493_pad_0 = const()[name = string("op_2493_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2493_dilations_0 = const()[name = string("op_2493_dilations_0"), val = tensor([1, 1])]; int32 var_2493_groups_0 = const()[name = string("op_2493_groups_0"), val = int32(1)]; tensor var_2493_cast_fp16 = conv(dilations = var_2493_dilations_0, groups = var_2493_groups_0, pad = var_2493_pad_0, pad_type = var_2493_pad_type_0, strides = var_2493_strides_0, weight = layers_4_mlp_up_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("op_2493_cast_fp16")]; tensor x_49_cast_fp16 = mul(x = var_2487_cast_fp16, y = var_2493_cast_fp16)[name = string("x_49_cast_fp16")]; tensor hidden_states_47_strides_0 = const()[name = string("hidden_states_47_strides_0"), val = tensor([1, 1])]; string hidden_states_47_pad_type_0 = const()[name = string("hidden_states_47_pad_type_0"), val = string("valid")]; tensor hidden_states_47_pad_0 = const()[name = string("hidden_states_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_47_dilations_0 = const()[name = string("hidden_states_47_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_47_groups_0 = const()[name = string("hidden_states_47_groups_0"), val = int32(1)]; tensor hidden_states_47_cast_fp16 = conv(dilations = hidden_states_47_dilations_0, groups = hidden_states_47_groups_0, pad = hidden_states_47_pad_0, pad_type = hidden_states_47_pad_type_0, strides = hidden_states_47_strides_0, weight = layers_4_mlp_down_proj_weight_cast_fp16, x = x_49_cast_fp16)[name = string("hidden_states_47_cast_fp16")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = hidden_states_47_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; fp16 const_50_promoted_to_fp16 = const()[name = string("const_50_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2511_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_50_promoted_to_fp16)[name = string("op_2511_cast_fp16")]; int32 var_2509 = const()[name = string("op_2509"), val = int32(1)]; bool doubled_41_interleave_0 = const()[name = string("doubled_41_interleave_0"), val = bool(false)]; tensor doubled_41_cast_fp16 = concat(axis = var_2509, interleave = doubled_41_interleave_0, values = (hidden_states_49_cast_fp16, var_2511_cast_fp16))[name = string("doubled_41_cast_fp16")]; tensor out_21_axes_0 = const()[name = string("out_21_axes_0"), val = tensor([1])]; tensor out_21_gamma_0_to_fp16 = const()[name = string("out_21_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283137536)))]; fp16 var_2521_to_fp16 = const()[name = string("op_2521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_21_cast_fp16 = layer_norm(axes = out_21_axes_0, epsilon = var_2521_to_fp16, gamma = out_21_gamma_0_to_fp16, x = doubled_41_cast_fp16)[name = string("out_21_cast_fp16")]; tensor var_2532_split_sizes_0 = const()[name = string("op_2532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2532_axis_0 = const()[name = string("op_2532_axis_0"), val = int32(1)]; tensor var_2532_cast_fp16_0, tensor var_2532_cast_fp16_1 = split(axis = var_2532_axis_0, split_sizes = var_2532_split_sizes_0, x = out_21_cast_fp16)[name = string("op_2532_cast_fp16")]; tensor layers_5_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283145792)))]; tensor query_states_31_strides_0 = const()[name = string("query_states_31_strides_0"), val = tensor([1, 1])]; string query_states_31_pad_type_0 = const()[name = string("query_states_31_pad_type_0"), val = string("valid")]; tensor query_states_31_pad_0 = const()[name = string("query_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_31_dilations_0 = const()[name = string("query_states_31_dilations_0"), val = tensor([1, 1])]; int32 query_states_31_groups_0 = const()[name = string("query_states_31_groups_0"), val = int32(1)]; tensor query_states_31_cast_fp16 = conv(dilations = query_states_31_dilations_0, groups = query_states_31_groups_0, pad = query_states_31_pad_0, pad_type = query_states_31_pad_type_0, strides = query_states_31_strides_0, weight = layers_5_self_attn_q_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("query_states_31_cast_fp16")]; tensor layers_5_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1291534464)))]; tensor key_states_51_strides_0 = const()[name = string("key_states_51_strides_0"), val = tensor([1, 1])]; string key_states_51_pad_type_0 = const()[name = string("key_states_51_pad_type_0"), val = string("valid")]; tensor key_states_51_pad_0 = const()[name = string("key_states_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_51_dilations_0 = const()[name = string("key_states_51_dilations_0"), val = tensor([1, 1])]; int32 key_states_51_groups_0 = const()[name = string("key_states_51_groups_0"), val = int32(1)]; tensor key_states_51_cast_fp16 = conv(dilations = key_states_51_dilations_0, groups = key_states_51_groups_0, pad = key_states_51_pad_0, pad_type = key_states_51_pad_type_0, strides = key_states_51_strides_0, weight = layers_5_self_attn_k_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("key_states_51_cast_fp16")]; tensor value_states_31_strides_0 = const()[name = string("value_states_31_strides_0"), val = tensor([1, 1])]; string value_states_31_pad_type_0 = const()[name = string("value_states_31_pad_type_0"), val = string("valid")]; tensor value_states_31_pad_0 = const()[name = string("value_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_31_dilations_0 = const()[name = string("value_states_31_dilations_0"), val = tensor([1, 1])]; int32 value_states_31_groups_0 = const()[name = string("value_states_31_groups_0"), val = int32(1)]; tensor value_states_31_cast_fp16 = conv(dilations = value_states_31_dilations_0, groups = value_states_31_groups_0, pad = value_states_31_pad_0, pad_type = value_states_31_pad_type_0, strides = value_states_31_strides_0, weight = layers_5_self_attn_v_proj_weight_cast_fp16, x = var_2532_cast_fp16_0)[name = string("value_states_31_cast_fp16")]; tensor concat_60x = const()[name = string("concat_60x"), val = tensor([1, 16, 128, -1])]; tensor x_51_cast_fp16 = reshape(shape = concat_60x, x = query_states_31_cast_fp16)[name = string("x_51_cast_fp16")]; tensor concat_61x = const()[name = string("concat_61x"), val = tensor([1, 2, 128, -1])]; tensor var_2589_cast_fp16 = reshape(shape = concat_61x, x = key_states_51_cast_fp16)[name = string("op_2589_cast_fp16")]; tensor concat_62x = const()[name = string("concat_62x"), val = tensor([1, 2, 128, -1])]; tensor var_2596_cast_fp16 = reshape(shape = concat_62x, x = value_states_31_cast_fp16)[name = string("op_2596_cast_fp16")]; tensor var_2600_cast_fp16 = mul(x = x_51_cast_fp16, y = var_869_cast_fp16)[name = string("op_2600_cast_fp16")]; tensor var_2601_split_sizes_0 = const()[name = string("op_2601_split_sizes_0"), val = tensor([64, 64])]; int32 var_2601_axis_0 = const()[name = string("op_2601_axis_0"), val = int32(-2)]; tensor var_2601_cast_fp16_0, tensor var_2601_cast_fp16_1 = split(axis = var_2601_axis_0, split_sizes = var_2601_split_sizes_0, x = x_51_cast_fp16)[name = string("op_2601_cast_fp16")]; fp16 const_52_promoted_to_fp16 = const()[name = string("const_52_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2603_cast_fp16 = mul(x = var_2601_cast_fp16_1, y = const_52_promoted_to_fp16)[name = string("op_2603_cast_fp16")]; int32 var_2605 = const()[name = string("op_2605"), val = int32(-2)]; bool var_2606_interleave_0 = const()[name = string("op_2606_interleave_0"), val = bool(false)]; tensor var_2606_cast_fp16 = concat(axis = var_2605, interleave = var_2606_interleave_0, values = (var_2603_cast_fp16, var_2601_cast_fp16_0))[name = string("op_2606_cast_fp16")]; tensor var_2607_cast_fp16 = mul(x = var_2606_cast_fp16, y = var_878_cast_fp16)[name = string("op_2607_cast_fp16")]; tensor query_states_33_cast_fp16 = add(x = var_2600_cast_fp16, y = var_2607_cast_fp16)[name = string("query_states_33_cast_fp16")]; tensor var_2613_cast_fp16 = mul(x = var_2589_cast_fp16, y = var_869_cast_fp16)[name = string("op_2613_cast_fp16")]; tensor var_2614_split_sizes_0 = const()[name = string("op_2614_split_sizes_0"), val = tensor([64, 64])]; int32 var_2614_axis_0 = const()[name = string("op_2614_axis_0"), val = int32(-2)]; tensor var_2614_cast_fp16_0, tensor var_2614_cast_fp16_1 = split(axis = var_2614_axis_0, split_sizes = var_2614_split_sizes_0, x = var_2589_cast_fp16)[name = string("op_2614_cast_fp16")]; fp16 const_53_promoted_to_fp16 = const()[name = string("const_53_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2616_cast_fp16 = mul(x = var_2614_cast_fp16_1, y = const_53_promoted_to_fp16)[name = string("op_2616_cast_fp16")]; int32 var_2618 = const()[name = string("op_2618"), val = int32(-2)]; bool var_2619_interleave_0 = const()[name = string("op_2619_interleave_0"), val = bool(false)]; tensor var_2619_cast_fp16 = concat(axis = var_2618, interleave = var_2619_interleave_0, values = (var_2616_cast_fp16, var_2614_cast_fp16_0))[name = string("op_2619_cast_fp16")]; tensor var_2620_cast_fp16 = mul(x = var_2619_cast_fp16, y = var_878_cast_fp16)[name = string("op_2620_cast_fp16")]; tensor key_states_55_cast_fp16 = add(x = var_2613_cast_fp16, y = var_2620_cast_fp16)[name = string("key_states_55_cast_fp16")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; int32 concat_65_axis_0 = const()[name = string("concat_65_axis_0"), val = int32(0)]; bool concat_65_interleave_0 = const()[name = string("concat_65_interleave_0"), val = bool(false)]; tensor concat_65 = concat(axis = concat_65_axis_0, interleave = concat_65_interleave_0, values = (expand_dims_60, expand_dims_61, position_id, expand_dims_63))[name = string("concat_65")]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; tensor concat_66_values1_0 = const()[name = string("concat_66_values1_0"), val = tensor([0])]; tensor concat_66_values3_0 = const()[name = string("concat_66_values3_0"), val = tensor([0])]; int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_64, concat_66_values1_0, cache_position_end, concat_66_values3_0))[name = string("concat_66")]; tensor key_states_57_perm_0 = const()[name = string("key_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_6_stride_0 = const()[name = string("key_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_57_cast_fp16 = transpose(perm = key_states_57_perm_0, x = key_states_55_cast_fp16)[name = string("transpose_240")]; tensor key_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = key_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = key_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_6_squeeze_mask_0, stride = key_cache_internal_tensor_assign_6_stride_0, update = key_states_57_cast_fp16, x = coreml_update_state_120)[name = string("key_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_6_cast_fp16, input = key_cache)[name = string("coreml_update_state_122_write_state")]; tensor coreml_update_state_122 = read_state(input = key_cache)[name = string("coreml_update_state_122")]; tensor value_states_33_perm_0 = const()[name = string("value_states_33_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_6_stride_0 = const()[name = string("value_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_33_cast_fp16 = transpose(perm = value_states_33_perm_0, x = var_2596_cast_fp16)[name = string("transpose_239")]; tensor value_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = value_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = value_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_6_squeeze_mask_0, stride = value_cache_internal_tensor_assign_6_stride_0, update = value_states_33_cast_fp16, x = coreml_update_state_121)[name = string("value_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_6_cast_fp16, input = value_cache)[name = string("coreml_update_state_123_write_state")]; tensor coreml_update_state_123 = read_state(input = value_cache)[name = string("coreml_update_state_123")]; tensor var_2690_begin_0 = const()[name = string("op_2690_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2690_end_0 = const()[name = string("op_2690_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2690_end_mask_0 = const()[name = string("op_2690_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2690_cast_fp16 = slice_by_index(begin = var_2690_begin_0, end = var_2690_end_0, end_mask = var_2690_end_mask_0, x = coreml_update_state_122)[name = string("op_2690_cast_fp16")]; tensor tile_10 = const()[name = string("tile_10"), val = tensor([1, 1])]; int32 var_2693_axis_0 = const()[name = string("op_2693_axis_0"), val = int32(1)]; tensor var_2693_cast_fp16_0, tensor var_2693_cast_fp16_1 = split(axis = var_2693_axis_0, split_sizes = tile_10, x = var_2690_cast_fp16)[name = string("op_2693_cast_fp16")]; tensor var_2700_begin_0 = const()[name = string("op_2700_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2700_end_0 = const()[name = string("op_2700_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2700_end_mask_0 = const()[name = string("op_2700_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2700_cast_fp16 = slice_by_index(begin = var_2700_begin_0, end = var_2700_end_0, end_mask = var_2700_end_mask_0, x = coreml_update_state_123)[name = string("op_2700_cast_fp16")]; tensor tile_11 = const()[name = string("tile_11"), val = tensor([1, 1])]; int32 var_2703_axis_0 = const()[name = string("op_2703_axis_0"), val = int32(1)]; tensor var_2703_cast_fp16_0, tensor var_2703_cast_fp16_1 = split(axis = var_2703_axis_0, split_sizes = tile_11, x = var_2700_cast_fp16)[name = string("op_2703_cast_fp16")]; tensor var_2706_split_sizes_0 = const()[name = string("op_2706_split_sizes_0"), val = tensor([8, 8])]; int32 var_2706_axis_0 = const()[name = string("op_2706_axis_0"), val = int32(1)]; tensor var_2706_0, tensor var_2706_1 = split(axis = var_2706_axis_0, split_sizes = var_2706_split_sizes_0, x = query_states_33_cast_fp16)[name = string("op_2706")]; bool attn_weights_81_transpose_x_0 = const()[name = string("attn_weights_81_transpose_x_0"), val = bool(false)]; bool attn_weights_81_transpose_y_0 = const()[name = string("attn_weights_81_transpose_y_0"), val = bool(false)]; tensor attn_weights_81_cast_fp16 = matmul(transpose_x = attn_weights_81_transpose_x_0, transpose_y = attn_weights_81_transpose_y_0, x = var_2693_cast_fp16_0, y = var_2706_0)[name = string("attn_weights_81_cast_fp16")]; fp16 var_2709_to_fp16 = const()[name = string("op_2709_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_83_cast_fp16 = mul(x = attn_weights_81_cast_fp16, y = var_2709_to_fp16)[name = string("attn_weights_83_cast_fp16")]; tensor attn_weights_85_cast_fp16 = add(x = attn_weights_83_cast_fp16, y = attn_mask_1)[name = string("attn_weights_85_cast_fp16")]; int32 var_2713 = const()[name = string("op_2713"), val = int32(-2)]; tensor attn_weights_87_cast_fp16 = softmax(axis = var_2713, x = attn_weights_85_cast_fp16)[name = string("attn_weights_87_cast_fp16")]; bool var_2719_transpose_x_1 = const()[name = string("op_2719_transpose_x_1"), val = bool(true)]; bool var_2719_transpose_y_1 = const()[name = string("op_2719_transpose_y_1"), val = bool(false)]; tensor var_2719_cast_fp16 = matmul(transpose_x = var_2719_transpose_x_1, transpose_y = var_2719_transpose_y_1, x = attn_weights_87_cast_fp16, y = var_2703_cast_fp16_0)[name = string("op_2719_cast_fp16")]; bool attn_weights_89_transpose_x_0 = const()[name = string("attn_weights_89_transpose_x_0"), val = bool(false)]; bool attn_weights_89_transpose_y_0 = const()[name = string("attn_weights_89_transpose_y_0"), val = bool(false)]; tensor attn_weights_89_cast_fp16 = matmul(transpose_x = attn_weights_89_transpose_x_0, transpose_y = attn_weights_89_transpose_y_0, x = var_2693_cast_fp16_1, y = var_2706_1)[name = string("attn_weights_89_cast_fp16")]; fp16 var_2721_to_fp16 = const()[name = string("op_2721_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_91_cast_fp16 = mul(x = attn_weights_89_cast_fp16, y = var_2721_to_fp16)[name = string("attn_weights_91_cast_fp16")]; tensor attn_weights_93_cast_fp16 = add(x = attn_weights_91_cast_fp16, y = attn_mask_1)[name = string("attn_weights_93_cast_fp16")]; int32 var_2725 = const()[name = string("op_2725"), val = int32(-2)]; tensor attn_weights_95_cast_fp16 = softmax(axis = var_2725, x = attn_weights_93_cast_fp16)[name = string("attn_weights_95_cast_fp16")]; bool attn_output_41_transpose_x_1 = const()[name = string("attn_output_41_transpose_x_1"), val = bool(true)]; bool attn_output_41_transpose_y_1 = const()[name = string("attn_output_41_transpose_y_1"), val = bool(false)]; tensor attn_output_41_cast_fp16 = matmul(transpose_x = attn_output_41_transpose_x_1, transpose_y = attn_output_41_transpose_y_1, x = attn_weights_95_cast_fp16, y = var_2703_cast_fp16_1)[name = string("attn_output_41_cast_fp16")]; int32 var_2733 = const()[name = string("op_2733"), val = int32(1)]; bool attn_output_43_interleave_0 = const()[name = string("attn_output_43_interleave_0"), val = bool(false)]; tensor attn_output_43_cast_fp16 = concat(axis = var_2733, interleave = attn_output_43_interleave_0, values = (var_2719_cast_fp16, attn_output_41_cast_fp16))[name = string("attn_output_43_cast_fp16")]; tensor var_2737_perm_0 = const()[name = string("op_2737_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_71x = const()[name = string("concat_71x"), val = tensor([1, 2048, 1, -1])]; tensor var_2737_cast_fp16 = transpose(perm = var_2737_perm_0, x = attn_output_43_cast_fp16)[name = string("transpose_238")]; tensor attn_output_47_cast_fp16 = reshape(shape = concat_71x, x = var_2737_cast_fp16)[name = string("attn_output_47_cast_fp16")]; tensor hidden_states_53_strides_0 = const()[name = string("hidden_states_53_strides_0"), val = tensor([1, 1])]; string hidden_states_53_pad_type_0 = const()[name = string("hidden_states_53_pad_type_0"), val = string("valid")]; tensor hidden_states_53_pad_0 = const()[name = string("hidden_states_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_53_dilations_0 = const()[name = string("hidden_states_53_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_53_groups_0 = const()[name = string("hidden_states_53_groups_0"), val = int32(1)]; tensor hidden_states_53_cast_fp16 = conv(dilations = hidden_states_53_dilations_0, groups = hidden_states_53_groups_0, pad = hidden_states_53_pad_0, pad_type = hidden_states_53_pad_type_0, strides = hidden_states_53_strides_0, weight = layers_5_self_attn_o_proj_weight_cast_fp16, x = attn_output_47_cast_fp16)[name = string("hidden_states_53_cast_fp16")]; tensor hidden_states_55_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = hidden_states_53_cast_fp16)[name = string("hidden_states_55_cast_fp16")]; fp16 const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2770_cast_fp16 = mul(x = hidden_states_55_cast_fp16, y = const_58_promoted_to_fp16)[name = string("op_2770_cast_fp16")]; int32 var_2768 = const()[name = string("op_2768"), val = int32(1)]; bool doubled_45_interleave_0 = const()[name = string("doubled_45_interleave_0"), val = bool(false)]; tensor doubled_45_cast_fp16 = concat(axis = var_2768, interleave = doubled_45_interleave_0, values = (hidden_states_55_cast_fp16, var_2770_cast_fp16))[name = string("doubled_45_cast_fp16")]; tensor out_23_axes_0 = const()[name = string("out_23_axes_0"), val = tensor([1])]; tensor out_23_gamma_0_to_fp16 = const()[name = string("out_23_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292583104)))]; fp16 var_2780_to_fp16 = const()[name = string("op_2780_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_23_cast_fp16 = layer_norm(axes = out_23_axes_0, epsilon = var_2780_to_fp16, gamma = out_23_gamma_0_to_fp16, x = doubled_45_cast_fp16)[name = string("out_23_cast_fp16")]; tensor var_2791_split_sizes_0 = const()[name = string("op_2791_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2791_axis_0 = const()[name = string("op_2791_axis_0"), val = int32(1)]; tensor var_2791_cast_fp16_0, tensor var_2791_cast_fp16_1 = split(axis = var_2791_axis_0, split_sizes = var_2791_split_sizes_0, x = out_23_cast_fp16)[name = string("op_2791_cast_fp16")]; tensor layers_5_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_5_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292591360)))]; tensor input_11_strides_0 = const()[name = string("input_11_strides_0"), val = tensor([1, 1])]; string input_11_pad_type_0 = const()[name = string("input_11_pad_type_0"), val = string("valid")]; tensor input_11_pad_0 = const()[name = string("input_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_11_dilations_0 = const()[name = string("input_11_dilations_0"), val = tensor([1, 1])]; int32 input_11_groups_0 = const()[name = string("input_11_groups_0"), val = int32(1)]; tensor input_11_cast_fp16 = conv(dilations = input_11_dilations_0, groups = input_11_groups_0, pad = input_11_pad_0, pad_type = input_11_pad_type_0, strides = input_11_strides_0, weight = layers_5_mlp_gate_proj_weight_to_fp16, x = var_2791_cast_fp16_0)[name = string("input_11_cast_fp16")]; tensor var_2808_cast_fp16 = silu(x = input_11_cast_fp16)[name = string("op_2808_cast_fp16")]; tensor var_2814_strides_0 = const()[name = string("op_2814_strides_0"), val = tensor([1, 1])]; string var_2814_pad_type_0 = const()[name = string("op_2814_pad_type_0"), val = string("valid")]; tensor var_2814_pad_0 = const()[name = string("op_2814_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2814_dilations_0 = const()[name = string("op_2814_dilations_0"), val = tensor([1, 1])]; int32 var_2814_groups_0 = const()[name = string("op_2814_groups_0"), val = int32(1)]; tensor var_2814_cast_fp16 = conv(dilations = var_2814_dilations_0, groups = var_2814_groups_0, pad = var_2814_pad_0, pad_type = var_2814_pad_type_0, strides = var_2814_strides_0, weight = layers_5_mlp_up_proj_weight_cast_fp16, x = var_2791_cast_fp16_0)[name = string("op_2814_cast_fp16")]; tensor x_59_cast_fp16 = mul(x = var_2808_cast_fp16, y = var_2814_cast_fp16)[name = string("x_59_cast_fp16")]; tensor hidden_states_57_strides_0 = const()[name = string("hidden_states_57_strides_0"), val = tensor([1, 1])]; string hidden_states_57_pad_type_0 = const()[name = string("hidden_states_57_pad_type_0"), val = string("valid")]; tensor hidden_states_57_pad_0 = const()[name = string("hidden_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_57_dilations_0 = const()[name = string("hidden_states_57_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_57_groups_0 = const()[name = string("hidden_states_57_groups_0"), val = int32(1)]; tensor hidden_states_57_cast_fp16 = conv(dilations = hidden_states_57_dilations_0, groups = hidden_states_57_groups_0, pad = hidden_states_57_pad_0, pad_type = hidden_states_57_pad_type_0, strides = hidden_states_57_strides_0, weight = layers_5_mlp_down_proj_weight_cast_fp16, x = x_59_cast_fp16)[name = string("hidden_states_57_cast_fp16")]; tensor hidden_states_59_cast_fp16 = add(x = hidden_states_55_cast_fp16, y = hidden_states_57_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2832_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_2832_cast_fp16")]; int32 var_2830 = const()[name = string("op_2830"), val = int32(1)]; bool doubled_49_interleave_0 = const()[name = string("doubled_49_interleave_0"), val = bool(false)]; tensor doubled_49_cast_fp16 = concat(axis = var_2830, interleave = doubled_49_interleave_0, values = (hidden_states_59_cast_fp16, var_2832_cast_fp16))[name = string("doubled_49_cast_fp16")]; tensor out_25_axes_0 = const()[name = string("out_25_axes_0"), val = tensor([1])]; tensor out_25_gamma_0_to_fp16 = const()[name = string("out_25_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317757248)))]; fp16 var_2842_to_fp16 = const()[name = string("op_2842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_25_cast_fp16 = layer_norm(axes = out_25_axes_0, epsilon = var_2842_to_fp16, gamma = out_25_gamma_0_to_fp16, x = doubled_49_cast_fp16)[name = string("out_25_cast_fp16")]; tensor var_2853_split_sizes_0 = const()[name = string("op_2853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2853_axis_0 = const()[name = string("op_2853_axis_0"), val = int32(1)]; tensor var_2853_cast_fp16_0, tensor var_2853_cast_fp16_1 = split(axis = var_2853_axis_0, split_sizes = var_2853_split_sizes_0, x = out_25_cast_fp16)[name = string("op_2853_cast_fp16")]; tensor layers_6_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317765504)))]; tensor query_states_37_strides_0 = const()[name = string("query_states_37_strides_0"), val = tensor([1, 1])]; string query_states_37_pad_type_0 = const()[name = string("query_states_37_pad_type_0"), val = string("valid")]; tensor query_states_37_pad_0 = const()[name = string("query_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_37_dilations_0 = const()[name = string("query_states_37_dilations_0"), val = tensor([1, 1])]; int32 query_states_37_groups_0 = const()[name = string("query_states_37_groups_0"), val = int32(1)]; tensor query_states_37_cast_fp16 = conv(dilations = query_states_37_dilations_0, groups = query_states_37_groups_0, pad = query_states_37_pad_0, pad_type = query_states_37_pad_type_0, strides = query_states_37_strides_0, weight = layers_6_self_attn_q_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("query_states_37_cast_fp16")]; tensor layers_6_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1326154176)))]; tensor key_states_61_strides_0 = const()[name = string("key_states_61_strides_0"), val = tensor([1, 1])]; string key_states_61_pad_type_0 = const()[name = string("key_states_61_pad_type_0"), val = string("valid")]; tensor key_states_61_pad_0 = const()[name = string("key_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_61_dilations_0 = const()[name = string("key_states_61_dilations_0"), val = tensor([1, 1])]; int32 key_states_61_groups_0 = const()[name = string("key_states_61_groups_0"), val = int32(1)]; tensor key_states_61_cast_fp16 = conv(dilations = key_states_61_dilations_0, groups = key_states_61_groups_0, pad = key_states_61_pad_0, pad_type = key_states_61_pad_type_0, strides = key_states_61_strides_0, weight = layers_6_self_attn_k_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("key_states_61_cast_fp16")]; tensor value_states_37_strides_0 = const()[name = string("value_states_37_strides_0"), val = tensor([1, 1])]; string value_states_37_pad_type_0 = const()[name = string("value_states_37_pad_type_0"), val = string("valid")]; tensor value_states_37_pad_0 = const()[name = string("value_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_37_dilations_0 = const()[name = string("value_states_37_dilations_0"), val = tensor([1, 1])]; int32 value_states_37_groups_0 = const()[name = string("value_states_37_groups_0"), val = int32(1)]; tensor value_states_37_cast_fp16 = conv(dilations = value_states_37_dilations_0, groups = value_states_37_groups_0, pad = value_states_37_pad_0, pad_type = value_states_37_pad_type_0, strides = value_states_37_strides_0, weight = layers_6_self_attn_v_proj_weight_cast_fp16, x = var_2853_cast_fp16_0)[name = string("value_states_37_cast_fp16")]; tensor concat_72x = const()[name = string("concat_72x"), val = tensor([1, 16, 128, -1])]; tensor x_61_cast_fp16 = reshape(shape = concat_72x, x = query_states_37_cast_fp16)[name = string("x_61_cast_fp16")]; tensor concat_73x = const()[name = string("concat_73x"), val = tensor([1, 2, 128, -1])]; tensor var_2910_cast_fp16 = reshape(shape = concat_73x, x = key_states_61_cast_fp16)[name = string("op_2910_cast_fp16")]; tensor concat_74x = const()[name = string("concat_74x"), val = tensor([1, 2, 128, -1])]; tensor var_2917_cast_fp16 = reshape(shape = concat_74x, x = value_states_37_cast_fp16)[name = string("op_2917_cast_fp16")]; tensor var_2921_cast_fp16 = mul(x = x_61_cast_fp16, y = var_869_cast_fp16)[name = string("op_2921_cast_fp16")]; tensor var_2922_split_sizes_0 = const()[name = string("op_2922_split_sizes_0"), val = tensor([64, 64])]; int32 var_2922_axis_0 = const()[name = string("op_2922_axis_0"), val = int32(-2)]; tensor var_2922_cast_fp16_0, tensor var_2922_cast_fp16_1 = split(axis = var_2922_axis_0, split_sizes = var_2922_split_sizes_0, x = x_61_cast_fp16)[name = string("op_2922_cast_fp16")]; fp16 const_62_promoted_to_fp16 = const()[name = string("const_62_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2924_cast_fp16 = mul(x = var_2922_cast_fp16_1, y = const_62_promoted_to_fp16)[name = string("op_2924_cast_fp16")]; int32 var_2926 = const()[name = string("op_2926"), val = int32(-2)]; bool var_2927_interleave_0 = const()[name = string("op_2927_interleave_0"), val = bool(false)]; tensor var_2927_cast_fp16 = concat(axis = var_2926, interleave = var_2927_interleave_0, values = (var_2924_cast_fp16, var_2922_cast_fp16_0))[name = string("op_2927_cast_fp16")]; tensor var_2928_cast_fp16 = mul(x = var_2927_cast_fp16, y = var_878_cast_fp16)[name = string("op_2928_cast_fp16")]; tensor query_states_39_cast_fp16 = add(x = var_2921_cast_fp16, y = var_2928_cast_fp16)[name = string("query_states_39_cast_fp16")]; tensor var_2934_cast_fp16 = mul(x = var_2910_cast_fp16, y = var_869_cast_fp16)[name = string("op_2934_cast_fp16")]; tensor var_2935_split_sizes_0 = const()[name = string("op_2935_split_sizes_0"), val = tensor([64, 64])]; int32 var_2935_axis_0 = const()[name = string("op_2935_axis_0"), val = int32(-2)]; tensor var_2935_cast_fp16_0, tensor var_2935_cast_fp16_1 = split(axis = var_2935_axis_0, split_sizes = var_2935_split_sizes_0, x = var_2910_cast_fp16)[name = string("op_2935_cast_fp16")]; fp16 const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2937_cast_fp16 = mul(x = var_2935_cast_fp16_1, y = const_63_promoted_to_fp16)[name = string("op_2937_cast_fp16")]; int32 var_2939 = const()[name = string("op_2939"), val = int32(-2)]; bool var_2940_interleave_0 = const()[name = string("op_2940_interleave_0"), val = bool(false)]; tensor var_2940_cast_fp16 = concat(axis = var_2939, interleave = var_2940_interleave_0, values = (var_2937_cast_fp16, var_2935_cast_fp16_0))[name = string("op_2940_cast_fp16")]; tensor var_2941_cast_fp16 = mul(x = var_2940_cast_fp16, y = var_878_cast_fp16)[name = string("op_2941_cast_fp16")]; tensor key_states_65_cast_fp16 = add(x = var_2934_cast_fp16, y = var_2941_cast_fp16)[name = string("key_states_65_cast_fp16")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; int32 concat_77_axis_0 = const()[name = string("concat_77_axis_0"), val = int32(0)]; bool concat_77_interleave_0 = const()[name = string("concat_77_interleave_0"), val = bool(false)]; tensor concat_77 = concat(axis = concat_77_axis_0, interleave = concat_77_interleave_0, values = (expand_dims_72, expand_dims_73, position_id, expand_dims_75))[name = string("concat_77")]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; tensor concat_78_values1_0 = const()[name = string("concat_78_values1_0"), val = tensor([0])]; tensor concat_78_values3_0 = const()[name = string("concat_78_values3_0"), val = tensor([0])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_76, concat_78_values1_0, cache_position_end, concat_78_values3_0))[name = string("concat_78")]; tensor key_states_67_perm_0 = const()[name = string("key_states_67_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_7_stride_0 = const()[name = string("key_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_67_cast_fp16 = transpose(perm = key_states_67_perm_0, x = key_states_65_cast_fp16)[name = string("transpose_237")]; tensor key_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = key_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = key_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_7_squeeze_mask_0, stride = key_cache_internal_tensor_assign_7_stride_0, update = key_states_67_cast_fp16, x = coreml_update_state_122)[name = string("key_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_7_cast_fp16, input = key_cache)[name = string("coreml_update_state_124_write_state")]; tensor coreml_update_state_124 = read_state(input = key_cache)[name = string("coreml_update_state_124")]; tensor value_states_39_perm_0 = const()[name = string("value_states_39_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_7_stride_0 = const()[name = string("value_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_39_cast_fp16 = transpose(perm = value_states_39_perm_0, x = var_2917_cast_fp16)[name = string("transpose_236")]; tensor value_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = value_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = value_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_7_squeeze_mask_0, stride = value_cache_internal_tensor_assign_7_stride_0, update = value_states_39_cast_fp16, x = coreml_update_state_123)[name = string("value_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_7_cast_fp16, input = value_cache)[name = string("coreml_update_state_125_write_state")]; tensor coreml_update_state_125 = read_state(input = value_cache)[name = string("coreml_update_state_125")]; tensor var_3011_begin_0 = const()[name = string("op_3011_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3011_end_0 = const()[name = string("op_3011_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3011_end_mask_0 = const()[name = string("op_3011_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3011_cast_fp16 = slice_by_index(begin = var_3011_begin_0, end = var_3011_end_0, end_mask = var_3011_end_mask_0, x = coreml_update_state_124)[name = string("op_3011_cast_fp16")]; tensor tile_12 = const()[name = string("tile_12"), val = tensor([1, 1])]; int32 var_3014_axis_0 = const()[name = string("op_3014_axis_0"), val = int32(1)]; tensor var_3014_cast_fp16_0, tensor var_3014_cast_fp16_1 = split(axis = var_3014_axis_0, split_sizes = tile_12, x = var_3011_cast_fp16)[name = string("op_3014_cast_fp16")]; tensor var_3021_begin_0 = const()[name = string("op_3021_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3021_end_0 = const()[name = string("op_3021_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3021_end_mask_0 = const()[name = string("op_3021_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3021_cast_fp16 = slice_by_index(begin = var_3021_begin_0, end = var_3021_end_0, end_mask = var_3021_end_mask_0, x = coreml_update_state_125)[name = string("op_3021_cast_fp16")]; tensor tile_13 = const()[name = string("tile_13"), val = tensor([1, 1])]; int32 var_3024_axis_0 = const()[name = string("op_3024_axis_0"), val = int32(1)]; tensor var_3024_cast_fp16_0, tensor var_3024_cast_fp16_1 = split(axis = var_3024_axis_0, split_sizes = tile_13, x = var_3021_cast_fp16)[name = string("op_3024_cast_fp16")]; tensor var_3027_split_sizes_0 = const()[name = string("op_3027_split_sizes_0"), val = tensor([8, 8])]; int32 var_3027_axis_0 = const()[name = string("op_3027_axis_0"), val = int32(1)]; tensor var_3027_0, tensor var_3027_1 = split(axis = var_3027_axis_0, split_sizes = var_3027_split_sizes_0, x = query_states_39_cast_fp16)[name = string("op_3027")]; bool attn_weights_97_transpose_x_0 = const()[name = string("attn_weights_97_transpose_x_0"), val = bool(false)]; bool attn_weights_97_transpose_y_0 = const()[name = string("attn_weights_97_transpose_y_0"), val = bool(false)]; tensor attn_weights_97_cast_fp16 = matmul(transpose_x = attn_weights_97_transpose_x_0, transpose_y = attn_weights_97_transpose_y_0, x = var_3014_cast_fp16_0, y = var_3027_0)[name = string("attn_weights_97_cast_fp16")]; fp16 var_3030_to_fp16 = const()[name = string("op_3030_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_99_cast_fp16 = mul(x = attn_weights_97_cast_fp16, y = var_3030_to_fp16)[name = string("attn_weights_99_cast_fp16")]; tensor attn_weights_101_cast_fp16 = add(x = attn_weights_99_cast_fp16, y = attn_mask_1)[name = string("attn_weights_101_cast_fp16")]; int32 var_3034 = const()[name = string("op_3034"), val = int32(-2)]; tensor attn_weights_103_cast_fp16 = softmax(axis = var_3034, x = attn_weights_101_cast_fp16)[name = string("attn_weights_103_cast_fp16")]; bool var_3040_transpose_x_1 = const()[name = string("op_3040_transpose_x_1"), val = bool(true)]; bool var_3040_transpose_y_1 = const()[name = string("op_3040_transpose_y_1"), val = bool(false)]; tensor var_3040_cast_fp16 = matmul(transpose_x = var_3040_transpose_x_1, transpose_y = var_3040_transpose_y_1, x = attn_weights_103_cast_fp16, y = var_3024_cast_fp16_0)[name = string("op_3040_cast_fp16")]; bool attn_weights_105_transpose_x_0 = const()[name = string("attn_weights_105_transpose_x_0"), val = bool(false)]; bool attn_weights_105_transpose_y_0 = const()[name = string("attn_weights_105_transpose_y_0"), val = bool(false)]; tensor attn_weights_105_cast_fp16 = matmul(transpose_x = attn_weights_105_transpose_x_0, transpose_y = attn_weights_105_transpose_y_0, x = var_3014_cast_fp16_1, y = var_3027_1)[name = string("attn_weights_105_cast_fp16")]; fp16 var_3042_to_fp16 = const()[name = string("op_3042_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_107_cast_fp16 = mul(x = attn_weights_105_cast_fp16, y = var_3042_to_fp16)[name = string("attn_weights_107_cast_fp16")]; tensor attn_weights_109_cast_fp16 = add(x = attn_weights_107_cast_fp16, y = attn_mask_1)[name = string("attn_weights_109_cast_fp16")]; int32 var_3046 = const()[name = string("op_3046"), val = int32(-2)]; tensor attn_weights_111_cast_fp16 = softmax(axis = var_3046, x = attn_weights_109_cast_fp16)[name = string("attn_weights_111_cast_fp16")]; bool attn_output_49_transpose_x_1 = const()[name = string("attn_output_49_transpose_x_1"), val = bool(true)]; bool attn_output_49_transpose_y_1 = const()[name = string("attn_output_49_transpose_y_1"), val = bool(false)]; tensor attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_1, transpose_y = attn_output_49_transpose_y_1, x = attn_weights_111_cast_fp16, y = var_3024_cast_fp16_1)[name = string("attn_output_49_cast_fp16")]; int32 var_3054 = const()[name = string("op_3054"), val = int32(1)]; bool attn_output_51_interleave_0 = const()[name = string("attn_output_51_interleave_0"), val = bool(false)]; tensor attn_output_51_cast_fp16 = concat(axis = var_3054, interleave = attn_output_51_interleave_0, values = (var_3040_cast_fp16, attn_output_49_cast_fp16))[name = string("attn_output_51_cast_fp16")]; tensor var_3058_perm_0 = const()[name = string("op_3058_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_83x = const()[name = string("concat_83x"), val = tensor([1, 2048, 1, -1])]; tensor var_3058_cast_fp16 = transpose(perm = var_3058_perm_0, x = attn_output_51_cast_fp16)[name = string("transpose_235")]; tensor attn_output_55_cast_fp16 = reshape(shape = concat_83x, x = var_3058_cast_fp16)[name = string("attn_output_55_cast_fp16")]; tensor hidden_states_63_strides_0 = const()[name = string("hidden_states_63_strides_0"), val = tensor([1, 1])]; string hidden_states_63_pad_type_0 = const()[name = string("hidden_states_63_pad_type_0"), val = string("valid")]; tensor hidden_states_63_pad_0 = const()[name = string("hidden_states_63_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_63_dilations_0 = const()[name = string("hidden_states_63_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_63_groups_0 = const()[name = string("hidden_states_63_groups_0"), val = int32(1)]; tensor hidden_states_63_cast_fp16 = conv(dilations = hidden_states_63_dilations_0, groups = hidden_states_63_groups_0, pad = hidden_states_63_pad_0, pad_type = hidden_states_63_pad_type_0, strides = hidden_states_63_strides_0, weight = layers_6_self_attn_o_proj_weight_cast_fp16, x = attn_output_55_cast_fp16)[name = string("hidden_states_63_cast_fp16")]; tensor hidden_states_65_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = hidden_states_63_cast_fp16)[name = string("hidden_states_65_cast_fp16")]; fp16 const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3091_cast_fp16 = mul(x = hidden_states_65_cast_fp16, y = const_68_promoted_to_fp16)[name = string("op_3091_cast_fp16")]; int32 var_3089 = const()[name = string("op_3089"), val = int32(1)]; bool doubled_53_interleave_0 = const()[name = string("doubled_53_interleave_0"), val = bool(false)]; tensor doubled_53_cast_fp16 = concat(axis = var_3089, interleave = doubled_53_interleave_0, values = (hidden_states_65_cast_fp16, var_3091_cast_fp16))[name = string("doubled_53_cast_fp16")]; tensor out_27_axes_0 = const()[name = string("out_27_axes_0"), val = tensor([1])]; tensor out_27_gamma_0_to_fp16 = const()[name = string("out_27_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327202816)))]; fp16 var_3101_to_fp16 = const()[name = string("op_3101_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_27_cast_fp16 = layer_norm(axes = out_27_axes_0, epsilon = var_3101_to_fp16, gamma = out_27_gamma_0_to_fp16, x = doubled_53_cast_fp16)[name = string("out_27_cast_fp16")]; tensor var_3112_split_sizes_0 = const()[name = string("op_3112_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3112_axis_0 = const()[name = string("op_3112_axis_0"), val = int32(1)]; tensor var_3112_cast_fp16_0, tensor var_3112_cast_fp16_1 = split(axis = var_3112_axis_0, split_sizes = var_3112_split_sizes_0, x = out_27_cast_fp16)[name = string("op_3112_cast_fp16")]; tensor input_13_strides_0 = const()[name = string("input_13_strides_0"), val = tensor([1, 1])]; string input_13_pad_type_0 = const()[name = string("input_13_pad_type_0"), val = string("valid")]; tensor input_13_pad_0 = const()[name = string("input_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_13_dilations_0 = const()[name = string("input_13_dilations_0"), val = tensor([1, 1])]; int32 input_13_groups_0 = const()[name = string("input_13_groups_0"), val = int32(1)]; tensor input_13_cast_fp16 = conv(dilations = input_13_dilations_0, groups = input_13_groups_0, pad = input_13_pad_0, pad_type = input_13_pad_type_0, strides = input_13_strides_0, weight = layers_6_mlp_gate_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("input_13_cast_fp16")]; tensor var_3129_cast_fp16 = silu(x = input_13_cast_fp16)[name = string("op_3129_cast_fp16")]; tensor var_3135_strides_0 = const()[name = string("op_3135_strides_0"), val = tensor([1, 1])]; string var_3135_pad_type_0 = const()[name = string("op_3135_pad_type_0"), val = string("valid")]; tensor var_3135_pad_0 = const()[name = string("op_3135_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3135_dilations_0 = const()[name = string("op_3135_dilations_0"), val = tensor([1, 1])]; int32 var_3135_groups_0 = const()[name = string("op_3135_groups_0"), val = int32(1)]; tensor var_3135_cast_fp16 = conv(dilations = var_3135_dilations_0, groups = var_3135_groups_0, pad = var_3135_pad_0, pad_type = var_3135_pad_type_0, strides = var_3135_strides_0, weight = layers_6_mlp_up_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("op_3135_cast_fp16")]; tensor x_69_cast_fp16 = mul(x = var_3129_cast_fp16, y = var_3135_cast_fp16)[name = string("x_69_cast_fp16")]; tensor hidden_states_67_strides_0 = const()[name = string("hidden_states_67_strides_0"), val = tensor([1, 1])]; string hidden_states_67_pad_type_0 = const()[name = string("hidden_states_67_pad_type_0"), val = string("valid")]; tensor hidden_states_67_pad_0 = const()[name = string("hidden_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_67_dilations_0 = const()[name = string("hidden_states_67_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_67_groups_0 = const()[name = string("hidden_states_67_groups_0"), val = int32(1)]; tensor hidden_states_67_cast_fp16 = conv(dilations = hidden_states_67_dilations_0, groups = hidden_states_67_groups_0, pad = hidden_states_67_pad_0, pad_type = hidden_states_67_pad_type_0, strides = hidden_states_67_strides_0, weight = layers_6_mlp_down_proj_weight_cast_fp16, x = x_69_cast_fp16)[name = string("hidden_states_67_cast_fp16")]; tensor hidden_states_69_cast_fp16 = add(x = hidden_states_65_cast_fp16, y = hidden_states_67_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3153_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_70_promoted_to_fp16)[name = string("op_3153_cast_fp16")]; int32 var_3151 = const()[name = string("op_3151"), val = int32(1)]; bool doubled_57_interleave_0 = const()[name = string("doubled_57_interleave_0"), val = bool(false)]; tensor doubled_57_cast_fp16 = concat(axis = var_3151, interleave = doubled_57_interleave_0, values = (hidden_states_69_cast_fp16, var_3153_cast_fp16))[name = string("doubled_57_cast_fp16")]; tensor out_29_axes_0 = const()[name = string("out_29_axes_0"), val = tensor([1])]; tensor out_29_gamma_0_to_fp16 = const()[name = string("out_29_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327211072)))]; fp16 var_3163_to_fp16 = const()[name = string("op_3163_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_29_cast_fp16 = layer_norm(axes = out_29_axes_0, epsilon = var_3163_to_fp16, gamma = out_29_gamma_0_to_fp16, x = doubled_57_cast_fp16)[name = string("out_29_cast_fp16")]; tensor var_3174_split_sizes_0 = const()[name = string("op_3174_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3174_axis_0 = const()[name = string("op_3174_axis_0"), val = int32(1)]; tensor var_3174_cast_fp16_0, tensor var_3174_cast_fp16_1 = split(axis = var_3174_axis_0, split_sizes = var_3174_split_sizes_0, x = out_29_cast_fp16)[name = string("op_3174_cast_fp16")]; tensor layers_7_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327219328)))]; tensor query_states_43_strides_0 = const()[name = string("query_states_43_strides_0"), val = tensor([1, 1])]; string query_states_43_pad_type_0 = const()[name = string("query_states_43_pad_type_0"), val = string("valid")]; tensor query_states_43_pad_0 = const()[name = string("query_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_43_dilations_0 = const()[name = string("query_states_43_dilations_0"), val = tensor([1, 1])]; int32 query_states_43_groups_0 = const()[name = string("query_states_43_groups_0"), val = int32(1)]; tensor query_states_43_cast_fp16 = conv(dilations = query_states_43_dilations_0, groups = query_states_43_groups_0, pad = query_states_43_pad_0, pad_type = query_states_43_pad_type_0, strides = query_states_43_strides_0, weight = layers_7_self_attn_q_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("query_states_43_cast_fp16")]; tensor layers_7_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1335608000)))]; tensor key_states_71_strides_0 = const()[name = string("key_states_71_strides_0"), val = tensor([1, 1])]; string key_states_71_pad_type_0 = const()[name = string("key_states_71_pad_type_0"), val = string("valid")]; tensor key_states_71_pad_0 = const()[name = string("key_states_71_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_71_dilations_0 = const()[name = string("key_states_71_dilations_0"), val = tensor([1, 1])]; int32 key_states_71_groups_0 = const()[name = string("key_states_71_groups_0"), val = int32(1)]; tensor key_states_71_cast_fp16 = conv(dilations = key_states_71_dilations_0, groups = key_states_71_groups_0, pad = key_states_71_pad_0, pad_type = key_states_71_pad_type_0, strides = key_states_71_strides_0, weight = layers_7_self_attn_k_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("key_states_71_cast_fp16")]; tensor value_states_43_strides_0 = const()[name = string("value_states_43_strides_0"), val = tensor([1, 1])]; string value_states_43_pad_type_0 = const()[name = string("value_states_43_pad_type_0"), val = string("valid")]; tensor value_states_43_pad_0 = const()[name = string("value_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_43_dilations_0 = const()[name = string("value_states_43_dilations_0"), val = tensor([1, 1])]; int32 value_states_43_groups_0 = const()[name = string("value_states_43_groups_0"), val = int32(1)]; tensor value_states_43_cast_fp16 = conv(dilations = value_states_43_dilations_0, groups = value_states_43_groups_0, pad = value_states_43_pad_0, pad_type = value_states_43_pad_type_0, strides = value_states_43_strides_0, weight = layers_7_self_attn_v_proj_weight_cast_fp16, x = var_3174_cast_fp16_0)[name = string("value_states_43_cast_fp16")]; tensor concat_84x = const()[name = string("concat_84x"), val = tensor([1, 16, 128, -1])]; tensor x_71_cast_fp16 = reshape(shape = concat_84x, x = query_states_43_cast_fp16)[name = string("x_71_cast_fp16")]; tensor concat_85x = const()[name = string("concat_85x"), val = tensor([1, 2, 128, -1])]; tensor var_3231_cast_fp16 = reshape(shape = concat_85x, x = key_states_71_cast_fp16)[name = string("op_3231_cast_fp16")]; tensor concat_86x = const()[name = string("concat_86x"), val = tensor([1, 2, 128, -1])]; tensor var_3238_cast_fp16 = reshape(shape = concat_86x, x = value_states_43_cast_fp16)[name = string("op_3238_cast_fp16")]; tensor var_3242_cast_fp16 = mul(x = x_71_cast_fp16, y = var_869_cast_fp16)[name = string("op_3242_cast_fp16")]; tensor var_3243_split_sizes_0 = const()[name = string("op_3243_split_sizes_0"), val = tensor([64, 64])]; int32 var_3243_axis_0 = const()[name = string("op_3243_axis_0"), val = int32(-2)]; tensor var_3243_cast_fp16_0, tensor var_3243_cast_fp16_1 = split(axis = var_3243_axis_0, split_sizes = var_3243_split_sizes_0, x = x_71_cast_fp16)[name = string("op_3243_cast_fp16")]; fp16 const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3245_cast_fp16 = mul(x = var_3243_cast_fp16_1, y = const_72_promoted_to_fp16)[name = string("op_3245_cast_fp16")]; int32 var_3247 = const()[name = string("op_3247"), val = int32(-2)]; bool var_3248_interleave_0 = const()[name = string("op_3248_interleave_0"), val = bool(false)]; tensor var_3248_cast_fp16 = concat(axis = var_3247, interleave = var_3248_interleave_0, values = (var_3245_cast_fp16, var_3243_cast_fp16_0))[name = string("op_3248_cast_fp16")]; tensor var_3249_cast_fp16 = mul(x = var_3248_cast_fp16, y = var_878_cast_fp16)[name = string("op_3249_cast_fp16")]; tensor query_states_45_cast_fp16 = add(x = var_3242_cast_fp16, y = var_3249_cast_fp16)[name = string("query_states_45_cast_fp16")]; tensor var_3255_cast_fp16 = mul(x = var_3231_cast_fp16, y = var_869_cast_fp16)[name = string("op_3255_cast_fp16")]; tensor var_3256_split_sizes_0 = const()[name = string("op_3256_split_sizes_0"), val = tensor([64, 64])]; int32 var_3256_axis_0 = const()[name = string("op_3256_axis_0"), val = int32(-2)]; tensor var_3256_cast_fp16_0, tensor var_3256_cast_fp16_1 = split(axis = var_3256_axis_0, split_sizes = var_3256_split_sizes_0, x = var_3231_cast_fp16)[name = string("op_3256_cast_fp16")]; fp16 const_73_promoted_to_fp16 = const()[name = string("const_73_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3258_cast_fp16 = mul(x = var_3256_cast_fp16_1, y = const_73_promoted_to_fp16)[name = string("op_3258_cast_fp16")]; int32 var_3260 = const()[name = string("op_3260"), val = int32(-2)]; bool var_3261_interleave_0 = const()[name = string("op_3261_interleave_0"), val = bool(false)]; tensor var_3261_cast_fp16 = concat(axis = var_3260, interleave = var_3261_interleave_0, values = (var_3258_cast_fp16, var_3256_cast_fp16_0))[name = string("op_3261_cast_fp16")]; tensor var_3262_cast_fp16 = mul(x = var_3261_cast_fp16, y = var_878_cast_fp16)[name = string("op_3262_cast_fp16")]; tensor key_states_75_cast_fp16 = add(x = var_3255_cast_fp16, y = var_3262_cast_fp16)[name = string("key_states_75_cast_fp16")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; int32 concat_89_axis_0 = const()[name = string("concat_89_axis_0"), val = int32(0)]; bool concat_89_interleave_0 = const()[name = string("concat_89_interleave_0"), val = bool(false)]; tensor concat_89 = concat(axis = concat_89_axis_0, interleave = concat_89_interleave_0, values = (expand_dims_84, expand_dims_85, position_id, expand_dims_87))[name = string("concat_89")]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; tensor concat_90_values1_0 = const()[name = string("concat_90_values1_0"), val = tensor([0])]; tensor concat_90_values3_0 = const()[name = string("concat_90_values3_0"), val = tensor([0])]; int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_88, concat_90_values1_0, cache_position_end, concat_90_values3_0))[name = string("concat_90")]; tensor key_states_77_perm_0 = const()[name = string("key_states_77_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_8_stride_0 = const()[name = string("key_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_77_cast_fp16 = transpose(perm = key_states_77_perm_0, x = key_states_75_cast_fp16)[name = string("transpose_234")]; tensor key_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = key_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = key_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_8_squeeze_mask_0, stride = key_cache_internal_tensor_assign_8_stride_0, update = key_states_77_cast_fp16, x = coreml_update_state_124)[name = string("key_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_8_cast_fp16, input = key_cache)[name = string("coreml_update_state_126_write_state")]; tensor coreml_update_state_126 = read_state(input = key_cache)[name = string("coreml_update_state_126")]; tensor value_states_45_perm_0 = const()[name = string("value_states_45_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_8_stride_0 = const()[name = string("value_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_45_cast_fp16 = transpose(perm = value_states_45_perm_0, x = var_3238_cast_fp16)[name = string("transpose_233")]; tensor value_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = value_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = value_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_8_squeeze_mask_0, stride = value_cache_internal_tensor_assign_8_stride_0, update = value_states_45_cast_fp16, x = coreml_update_state_125)[name = string("value_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_8_cast_fp16, input = value_cache)[name = string("coreml_update_state_127_write_state")]; tensor coreml_update_state_127 = read_state(input = value_cache)[name = string("coreml_update_state_127")]; tensor var_3332_begin_0 = const()[name = string("op_3332_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3332_end_0 = const()[name = string("op_3332_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3332_end_mask_0 = const()[name = string("op_3332_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3332_cast_fp16 = slice_by_index(begin = var_3332_begin_0, end = var_3332_end_0, end_mask = var_3332_end_mask_0, x = coreml_update_state_126)[name = string("op_3332_cast_fp16")]; tensor tile_14 = const()[name = string("tile_14"), val = tensor([1, 1])]; int32 var_3335_axis_0 = const()[name = string("op_3335_axis_0"), val = int32(1)]; tensor var_3335_cast_fp16_0, tensor var_3335_cast_fp16_1 = split(axis = var_3335_axis_0, split_sizes = tile_14, x = var_3332_cast_fp16)[name = string("op_3335_cast_fp16")]; tensor var_3342_begin_0 = const()[name = string("op_3342_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3342_end_0 = const()[name = string("op_3342_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3342_end_mask_0 = const()[name = string("op_3342_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3342_cast_fp16 = slice_by_index(begin = var_3342_begin_0, end = var_3342_end_0, end_mask = var_3342_end_mask_0, x = coreml_update_state_127)[name = string("op_3342_cast_fp16")]; tensor tile_15 = const()[name = string("tile_15"), val = tensor([1, 1])]; int32 var_3345_axis_0 = const()[name = string("op_3345_axis_0"), val = int32(1)]; tensor var_3345_cast_fp16_0, tensor var_3345_cast_fp16_1 = split(axis = var_3345_axis_0, split_sizes = tile_15, x = var_3342_cast_fp16)[name = string("op_3345_cast_fp16")]; tensor var_3348_split_sizes_0 = const()[name = string("op_3348_split_sizes_0"), val = tensor([8, 8])]; int32 var_3348_axis_0 = const()[name = string("op_3348_axis_0"), val = int32(1)]; tensor var_3348_0, tensor var_3348_1 = split(axis = var_3348_axis_0, split_sizes = var_3348_split_sizes_0, x = query_states_45_cast_fp16)[name = string("op_3348")]; bool attn_weights_113_transpose_x_0 = const()[name = string("attn_weights_113_transpose_x_0"), val = bool(false)]; bool attn_weights_113_transpose_y_0 = const()[name = string("attn_weights_113_transpose_y_0"), val = bool(false)]; tensor attn_weights_113_cast_fp16 = matmul(transpose_x = attn_weights_113_transpose_x_0, transpose_y = attn_weights_113_transpose_y_0, x = var_3335_cast_fp16_0, y = var_3348_0)[name = string("attn_weights_113_cast_fp16")]; fp16 var_3351_to_fp16 = const()[name = string("op_3351_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_115_cast_fp16 = mul(x = attn_weights_113_cast_fp16, y = var_3351_to_fp16)[name = string("attn_weights_115_cast_fp16")]; tensor attn_weights_117_cast_fp16 = add(x = attn_weights_115_cast_fp16, y = attn_mask_1)[name = string("attn_weights_117_cast_fp16")]; int32 var_3355 = const()[name = string("op_3355"), val = int32(-2)]; tensor attn_weights_119_cast_fp16 = softmax(axis = var_3355, x = attn_weights_117_cast_fp16)[name = string("attn_weights_119_cast_fp16")]; bool var_3361_transpose_x_1 = const()[name = string("op_3361_transpose_x_1"), val = bool(true)]; bool var_3361_transpose_y_1 = const()[name = string("op_3361_transpose_y_1"), val = bool(false)]; tensor var_3361_cast_fp16 = matmul(transpose_x = var_3361_transpose_x_1, transpose_y = var_3361_transpose_y_1, x = attn_weights_119_cast_fp16, y = var_3345_cast_fp16_0)[name = string("op_3361_cast_fp16")]; bool attn_weights_121_transpose_x_0 = const()[name = string("attn_weights_121_transpose_x_0"), val = bool(false)]; bool attn_weights_121_transpose_y_0 = const()[name = string("attn_weights_121_transpose_y_0"), val = bool(false)]; tensor attn_weights_121_cast_fp16 = matmul(transpose_x = attn_weights_121_transpose_x_0, transpose_y = attn_weights_121_transpose_y_0, x = var_3335_cast_fp16_1, y = var_3348_1)[name = string("attn_weights_121_cast_fp16")]; fp16 var_3363_to_fp16 = const()[name = string("op_3363_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_123_cast_fp16 = mul(x = attn_weights_121_cast_fp16, y = var_3363_to_fp16)[name = string("attn_weights_123_cast_fp16")]; tensor attn_weights_125_cast_fp16 = add(x = attn_weights_123_cast_fp16, y = attn_mask_1)[name = string("attn_weights_125_cast_fp16")]; int32 var_3367 = const()[name = string("op_3367"), val = int32(-2)]; tensor attn_weights_127_cast_fp16 = softmax(axis = var_3367, x = attn_weights_125_cast_fp16)[name = string("attn_weights_127_cast_fp16")]; bool attn_output_57_transpose_x_1 = const()[name = string("attn_output_57_transpose_x_1"), val = bool(true)]; bool attn_output_57_transpose_y_1 = const()[name = string("attn_output_57_transpose_y_1"), val = bool(false)]; tensor attn_output_57_cast_fp16 = matmul(transpose_x = attn_output_57_transpose_x_1, transpose_y = attn_output_57_transpose_y_1, x = attn_weights_127_cast_fp16, y = var_3345_cast_fp16_1)[name = string("attn_output_57_cast_fp16")]; int32 var_3375 = const()[name = string("op_3375"), val = int32(1)]; bool attn_output_59_interleave_0 = const()[name = string("attn_output_59_interleave_0"), val = bool(false)]; tensor attn_output_59_cast_fp16 = concat(axis = var_3375, interleave = attn_output_59_interleave_0, values = (var_3361_cast_fp16, attn_output_57_cast_fp16))[name = string("attn_output_59_cast_fp16")]; tensor var_3379_perm_0 = const()[name = string("op_3379_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_95x = const()[name = string("concat_95x"), val = tensor([1, 2048, 1, -1])]; tensor var_3379_cast_fp16 = transpose(perm = var_3379_perm_0, x = attn_output_59_cast_fp16)[name = string("transpose_232")]; tensor attn_output_63_cast_fp16 = reshape(shape = concat_95x, x = var_3379_cast_fp16)[name = string("attn_output_63_cast_fp16")]; tensor hidden_states_73_strides_0 = const()[name = string("hidden_states_73_strides_0"), val = tensor([1, 1])]; string hidden_states_73_pad_type_0 = const()[name = string("hidden_states_73_pad_type_0"), val = string("valid")]; tensor hidden_states_73_pad_0 = const()[name = string("hidden_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_73_dilations_0 = const()[name = string("hidden_states_73_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_73_groups_0 = const()[name = string("hidden_states_73_groups_0"), val = int32(1)]; tensor hidden_states_73_cast_fp16 = conv(dilations = hidden_states_73_dilations_0, groups = hidden_states_73_groups_0, pad = hidden_states_73_pad_0, pad_type = hidden_states_73_pad_type_0, strides = hidden_states_73_strides_0, weight = layers_7_self_attn_o_proj_weight_cast_fp16, x = attn_output_63_cast_fp16)[name = string("hidden_states_73_cast_fp16")]; tensor hidden_states_75_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = hidden_states_73_cast_fp16)[name = string("hidden_states_75_cast_fp16")]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3412_cast_fp16 = mul(x = hidden_states_75_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_3412_cast_fp16")]; int32 var_3410 = const()[name = string("op_3410"), val = int32(1)]; bool doubled_61_interleave_0 = const()[name = string("doubled_61_interleave_0"), val = bool(false)]; tensor doubled_61_cast_fp16 = concat(axis = var_3410, interleave = doubled_61_interleave_0, values = (hidden_states_75_cast_fp16, var_3412_cast_fp16))[name = string("doubled_61_cast_fp16")]; tensor out_31_axes_0 = const()[name = string("out_31_axes_0"), val = tensor([1])]; tensor out_31_gamma_0_to_fp16 = const()[name = string("out_31_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336656640)))]; fp16 var_3422_to_fp16 = const()[name = string("op_3422_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_31_cast_fp16 = layer_norm(axes = out_31_axes_0, epsilon = var_3422_to_fp16, gamma = out_31_gamma_0_to_fp16, x = doubled_61_cast_fp16)[name = string("out_31_cast_fp16")]; tensor var_3433_split_sizes_0 = const()[name = string("op_3433_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3433_axis_0 = const()[name = string("op_3433_axis_0"), val = int32(1)]; tensor var_3433_cast_fp16_0, tensor var_3433_cast_fp16_1 = split(axis = var_3433_axis_0, split_sizes = var_3433_split_sizes_0, x = out_31_cast_fp16)[name = string("op_3433_cast_fp16")]; tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; tensor input_15_cast_fp16 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = layers_7_mlp_gate_proj_weight_cast_fp16, x = var_3433_cast_fp16_0)[name = string("input_15_cast_fp16")]; tensor var_3450_cast_fp16 = silu(x = input_15_cast_fp16)[name = string("op_3450_cast_fp16")]; tensor layers_7_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336664896)))]; tensor var_3456_strides_0 = const()[name = string("op_3456_strides_0"), val = tensor([1, 1])]; string var_3456_pad_type_0 = const()[name = string("op_3456_pad_type_0"), val = string("valid")]; tensor var_3456_pad_0 = const()[name = string("op_3456_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3456_dilations_0 = const()[name = string("op_3456_dilations_0"), val = tensor([1, 1])]; int32 var_3456_groups_0 = const()[name = string("op_3456_groups_0"), val = int32(1)]; tensor var_3456_cast_fp16 = conv(dilations = var_3456_dilations_0, groups = var_3456_groups_0, pad = var_3456_pad_0, pad_type = var_3456_pad_type_0, strides = var_3456_strides_0, weight = layers_7_mlp_up_proj_weight_to_fp16, x = var_3433_cast_fp16_0)[name = string("op_3456_cast_fp16")]; tensor x_79_cast_fp16 = mul(x = var_3450_cast_fp16, y = var_3456_cast_fp16)[name = string("x_79_cast_fp16")]; tensor layers_7_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1361830784)))]; tensor hidden_states_77_strides_0 = const()[name = string("hidden_states_77_strides_0"), val = tensor([1, 1])]; string hidden_states_77_pad_type_0 = const()[name = string("hidden_states_77_pad_type_0"), val = string("valid")]; tensor hidden_states_77_pad_0 = const()[name = string("hidden_states_77_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_77_dilations_0 = const()[name = string("hidden_states_77_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_77_groups_0 = const()[name = string("hidden_states_77_groups_0"), val = int32(1)]; tensor hidden_states_77_cast_fp16 = conv(dilations = hidden_states_77_dilations_0, groups = hidden_states_77_groups_0, pad = hidden_states_77_pad_0, pad_type = hidden_states_77_pad_type_0, strides = hidden_states_77_strides_0, weight = layers_7_mlp_down_proj_weight_to_fp16, x = x_79_cast_fp16)[name = string("hidden_states_77_cast_fp16")]; tensor hidden_states_79_cast_fp16 = add(x = hidden_states_75_cast_fp16, y = hidden_states_77_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3474_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_80_promoted_to_fp16)[name = string("op_3474_cast_fp16")]; int32 var_3472 = const()[name = string("op_3472"), val = int32(1)]; bool doubled_65_interleave_0 = const()[name = string("doubled_65_interleave_0"), val = bool(false)]; tensor doubled_65_cast_fp16 = concat(axis = var_3472, interleave = doubled_65_interleave_0, values = (hidden_states_79_cast_fp16, var_3474_cast_fp16))[name = string("doubled_65_cast_fp16")]; tensor out_33_axes_0 = const()[name = string("out_33_axes_0"), val = tensor([1])]; tensor out_33_gamma_0_to_fp16 = const()[name = string("out_33_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1386996672)))]; fp16 var_3484_to_fp16 = const()[name = string("op_3484_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_33_cast_fp16 = layer_norm(axes = out_33_axes_0, epsilon = var_3484_to_fp16, gamma = out_33_gamma_0_to_fp16, x = doubled_65_cast_fp16)[name = string("out_33_cast_fp16")]; tensor var_3495_split_sizes_0 = const()[name = string("op_3495_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3495_axis_0 = const()[name = string("op_3495_axis_0"), val = int32(1)]; tensor var_3495_cast_fp16_0, tensor var_3495_cast_fp16_1 = split(axis = var_3495_axis_0, split_sizes = var_3495_split_sizes_0, x = out_33_cast_fp16)[name = string("op_3495_cast_fp16")]; tensor layers_8_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1387004928)))]; tensor query_states_49_strides_0 = const()[name = string("query_states_49_strides_0"), val = tensor([1, 1])]; string query_states_49_pad_type_0 = const()[name = string("query_states_49_pad_type_0"), val = string("valid")]; tensor query_states_49_pad_0 = const()[name = string("query_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_49_dilations_0 = const()[name = string("query_states_49_dilations_0"), val = tensor([1, 1])]; int32 query_states_49_groups_0 = const()[name = string("query_states_49_groups_0"), val = int32(1)]; tensor query_states_49_cast_fp16 = conv(dilations = query_states_49_dilations_0, groups = query_states_49_groups_0, pad = query_states_49_pad_0, pad_type = query_states_49_pad_type_0, strides = query_states_49_strides_0, weight = layers_8_self_attn_q_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("query_states_49_cast_fp16")]; tensor layers_8_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1395393600)))]; tensor key_states_81_strides_0 = const()[name = string("key_states_81_strides_0"), val = tensor([1, 1])]; string key_states_81_pad_type_0 = const()[name = string("key_states_81_pad_type_0"), val = string("valid")]; tensor key_states_81_pad_0 = const()[name = string("key_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_81_dilations_0 = const()[name = string("key_states_81_dilations_0"), val = tensor([1, 1])]; int32 key_states_81_groups_0 = const()[name = string("key_states_81_groups_0"), val = int32(1)]; tensor key_states_81_cast_fp16 = conv(dilations = key_states_81_dilations_0, groups = key_states_81_groups_0, pad = key_states_81_pad_0, pad_type = key_states_81_pad_type_0, strides = key_states_81_strides_0, weight = layers_8_self_attn_k_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("key_states_81_cast_fp16")]; tensor value_states_49_strides_0 = const()[name = string("value_states_49_strides_0"), val = tensor([1, 1])]; string value_states_49_pad_type_0 = const()[name = string("value_states_49_pad_type_0"), val = string("valid")]; tensor value_states_49_pad_0 = const()[name = string("value_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_49_dilations_0 = const()[name = string("value_states_49_dilations_0"), val = tensor([1, 1])]; int32 value_states_49_groups_0 = const()[name = string("value_states_49_groups_0"), val = int32(1)]; tensor value_states_49_cast_fp16 = conv(dilations = value_states_49_dilations_0, groups = value_states_49_groups_0, pad = value_states_49_pad_0, pad_type = value_states_49_pad_type_0, strides = value_states_49_strides_0, weight = layers_8_self_attn_v_proj_weight_cast_fp16, x = var_3495_cast_fp16_0)[name = string("value_states_49_cast_fp16")]; tensor concat_96x = const()[name = string("concat_96x"), val = tensor([1, 16, 128, -1])]; tensor x_81_cast_fp16 = reshape(shape = concat_96x, x = query_states_49_cast_fp16)[name = string("x_81_cast_fp16")]; tensor concat_97x = const()[name = string("concat_97x"), val = tensor([1, 2, 128, -1])]; tensor var_3552_cast_fp16 = reshape(shape = concat_97x, x = key_states_81_cast_fp16)[name = string("op_3552_cast_fp16")]; tensor concat_98x = const()[name = string("concat_98x"), val = tensor([1, 2, 128, -1])]; tensor var_3559_cast_fp16 = reshape(shape = concat_98x, x = value_states_49_cast_fp16)[name = string("op_3559_cast_fp16")]; tensor var_3563_cast_fp16 = mul(x = x_81_cast_fp16, y = var_869_cast_fp16)[name = string("op_3563_cast_fp16")]; tensor var_3564_split_sizes_0 = const()[name = string("op_3564_split_sizes_0"), val = tensor([64, 64])]; int32 var_3564_axis_0 = const()[name = string("op_3564_axis_0"), val = int32(-2)]; tensor var_3564_cast_fp16_0, tensor var_3564_cast_fp16_1 = split(axis = var_3564_axis_0, split_sizes = var_3564_split_sizes_0, x = x_81_cast_fp16)[name = string("op_3564_cast_fp16")]; fp16 const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3566_cast_fp16 = mul(x = var_3564_cast_fp16_1, y = const_82_promoted_to_fp16)[name = string("op_3566_cast_fp16")]; int32 var_3568 = const()[name = string("op_3568"), val = int32(-2)]; bool var_3569_interleave_0 = const()[name = string("op_3569_interleave_0"), val = bool(false)]; tensor var_3569_cast_fp16 = concat(axis = var_3568, interleave = var_3569_interleave_0, values = (var_3566_cast_fp16, var_3564_cast_fp16_0))[name = string("op_3569_cast_fp16")]; tensor var_3570_cast_fp16 = mul(x = var_3569_cast_fp16, y = var_878_cast_fp16)[name = string("op_3570_cast_fp16")]; tensor query_states_51_cast_fp16 = add(x = var_3563_cast_fp16, y = var_3570_cast_fp16)[name = string("query_states_51_cast_fp16")]; tensor var_3576_cast_fp16 = mul(x = var_3552_cast_fp16, y = var_869_cast_fp16)[name = string("op_3576_cast_fp16")]; tensor var_3577_split_sizes_0 = const()[name = string("op_3577_split_sizes_0"), val = tensor([64, 64])]; int32 var_3577_axis_0 = const()[name = string("op_3577_axis_0"), val = int32(-2)]; tensor var_3577_cast_fp16_0, tensor var_3577_cast_fp16_1 = split(axis = var_3577_axis_0, split_sizes = var_3577_split_sizes_0, x = var_3552_cast_fp16)[name = string("op_3577_cast_fp16")]; fp16 const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3579_cast_fp16 = mul(x = var_3577_cast_fp16_1, y = const_83_promoted_to_fp16)[name = string("op_3579_cast_fp16")]; int32 var_3581 = const()[name = string("op_3581"), val = int32(-2)]; bool var_3582_interleave_0 = const()[name = string("op_3582_interleave_0"), val = bool(false)]; tensor var_3582_cast_fp16 = concat(axis = var_3581, interleave = var_3582_interleave_0, values = (var_3579_cast_fp16, var_3577_cast_fp16_0))[name = string("op_3582_cast_fp16")]; tensor var_3583_cast_fp16 = mul(x = var_3582_cast_fp16, y = var_878_cast_fp16)[name = string("op_3583_cast_fp16")]; tensor key_states_85_cast_fp16 = add(x = var_3576_cast_fp16, y = var_3583_cast_fp16)[name = string("key_states_85_cast_fp16")]; tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; int32 concat_101_axis_0 = const()[name = string("concat_101_axis_0"), val = int32(0)]; bool concat_101_interleave_0 = const()[name = string("concat_101_interleave_0"), val = bool(false)]; tensor concat_101 = concat(axis = concat_101_axis_0, interleave = concat_101_interleave_0, values = (expand_dims_96, expand_dims_97, position_id, expand_dims_99))[name = string("concat_101")]; tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; tensor concat_102_values1_0 = const()[name = string("concat_102_values1_0"), val = tensor([0])]; tensor concat_102_values3_0 = const()[name = string("concat_102_values3_0"), val = tensor([0])]; int32 concat_102_axis_0 = const()[name = string("concat_102_axis_0"), val = int32(0)]; bool concat_102_interleave_0 = const()[name = string("concat_102_interleave_0"), val = bool(false)]; tensor concat_102 = concat(axis = concat_102_axis_0, interleave = concat_102_interleave_0, values = (expand_dims_100, concat_102_values1_0, cache_position_end, concat_102_values3_0))[name = string("concat_102")]; tensor key_states_87_perm_0 = const()[name = string("key_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_9_stride_0 = const()[name = string("key_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_87_cast_fp16 = transpose(perm = key_states_87_perm_0, x = key_states_85_cast_fp16)[name = string("transpose_231")]; tensor key_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = key_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = key_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_9_squeeze_mask_0, stride = key_cache_internal_tensor_assign_9_stride_0, update = key_states_87_cast_fp16, x = coreml_update_state_126)[name = string("key_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_9_cast_fp16, input = key_cache)[name = string("coreml_update_state_128_write_state")]; tensor coreml_update_state_128 = read_state(input = key_cache)[name = string("coreml_update_state_128")]; tensor value_states_51_perm_0 = const()[name = string("value_states_51_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_9_stride_0 = const()[name = string("value_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_51_cast_fp16 = transpose(perm = value_states_51_perm_0, x = var_3559_cast_fp16)[name = string("transpose_230")]; tensor value_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = value_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = value_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_9_squeeze_mask_0, stride = value_cache_internal_tensor_assign_9_stride_0, update = value_states_51_cast_fp16, x = coreml_update_state_127)[name = string("value_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_9_cast_fp16, input = value_cache)[name = string("coreml_update_state_129_write_state")]; tensor coreml_update_state_129 = read_state(input = value_cache)[name = string("coreml_update_state_129")]; tensor var_3653_begin_0 = const()[name = string("op_3653_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3653_end_0 = const()[name = string("op_3653_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3653_end_mask_0 = const()[name = string("op_3653_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3653_cast_fp16 = slice_by_index(begin = var_3653_begin_0, end = var_3653_end_0, end_mask = var_3653_end_mask_0, x = coreml_update_state_128)[name = string("op_3653_cast_fp16")]; tensor tile_16 = const()[name = string("tile_16"), val = tensor([1, 1])]; int32 var_3656_axis_0 = const()[name = string("op_3656_axis_0"), val = int32(1)]; tensor var_3656_cast_fp16_0, tensor var_3656_cast_fp16_1 = split(axis = var_3656_axis_0, split_sizes = tile_16, x = var_3653_cast_fp16)[name = string("op_3656_cast_fp16")]; tensor var_3663_begin_0 = const()[name = string("op_3663_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3663_end_0 = const()[name = string("op_3663_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3663_end_mask_0 = const()[name = string("op_3663_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3663_cast_fp16 = slice_by_index(begin = var_3663_begin_0, end = var_3663_end_0, end_mask = var_3663_end_mask_0, x = coreml_update_state_129)[name = string("op_3663_cast_fp16")]; tensor tile_17 = const()[name = string("tile_17"), val = tensor([1, 1])]; int32 var_3666_axis_0 = const()[name = string("op_3666_axis_0"), val = int32(1)]; tensor var_3666_cast_fp16_0, tensor var_3666_cast_fp16_1 = split(axis = var_3666_axis_0, split_sizes = tile_17, x = var_3663_cast_fp16)[name = string("op_3666_cast_fp16")]; tensor var_3669_split_sizes_0 = const()[name = string("op_3669_split_sizes_0"), val = tensor([8, 8])]; int32 var_3669_axis_0 = const()[name = string("op_3669_axis_0"), val = int32(1)]; tensor var_3669_0, tensor var_3669_1 = split(axis = var_3669_axis_0, split_sizes = var_3669_split_sizes_0, x = query_states_51_cast_fp16)[name = string("op_3669")]; bool attn_weights_129_transpose_x_0 = const()[name = string("attn_weights_129_transpose_x_0"), val = bool(false)]; bool attn_weights_129_transpose_y_0 = const()[name = string("attn_weights_129_transpose_y_0"), val = bool(false)]; tensor attn_weights_129_cast_fp16 = matmul(transpose_x = attn_weights_129_transpose_x_0, transpose_y = attn_weights_129_transpose_y_0, x = var_3656_cast_fp16_0, y = var_3669_0)[name = string("attn_weights_129_cast_fp16")]; fp16 var_3672_to_fp16 = const()[name = string("op_3672_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_131_cast_fp16 = mul(x = attn_weights_129_cast_fp16, y = var_3672_to_fp16)[name = string("attn_weights_131_cast_fp16")]; tensor attn_weights_133_cast_fp16 = add(x = attn_weights_131_cast_fp16, y = attn_mask_1)[name = string("attn_weights_133_cast_fp16")]; int32 var_3676 = const()[name = string("op_3676"), val = int32(-2)]; tensor attn_weights_135_cast_fp16 = softmax(axis = var_3676, x = attn_weights_133_cast_fp16)[name = string("attn_weights_135_cast_fp16")]; bool var_3682_transpose_x_1 = const()[name = string("op_3682_transpose_x_1"), val = bool(true)]; bool var_3682_transpose_y_1 = const()[name = string("op_3682_transpose_y_1"), val = bool(false)]; tensor var_3682_cast_fp16 = matmul(transpose_x = var_3682_transpose_x_1, transpose_y = var_3682_transpose_y_1, x = attn_weights_135_cast_fp16, y = var_3666_cast_fp16_0)[name = string("op_3682_cast_fp16")]; bool attn_weights_137_transpose_x_0 = const()[name = string("attn_weights_137_transpose_x_0"), val = bool(false)]; bool attn_weights_137_transpose_y_0 = const()[name = string("attn_weights_137_transpose_y_0"), val = bool(false)]; tensor attn_weights_137_cast_fp16 = matmul(transpose_x = attn_weights_137_transpose_x_0, transpose_y = attn_weights_137_transpose_y_0, x = var_3656_cast_fp16_1, y = var_3669_1)[name = string("attn_weights_137_cast_fp16")]; fp16 var_3684_to_fp16 = const()[name = string("op_3684_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_139_cast_fp16 = mul(x = attn_weights_137_cast_fp16, y = var_3684_to_fp16)[name = string("attn_weights_139_cast_fp16")]; tensor attn_weights_141_cast_fp16 = add(x = attn_weights_139_cast_fp16, y = attn_mask_1)[name = string("attn_weights_141_cast_fp16")]; int32 var_3688 = const()[name = string("op_3688"), val = int32(-2)]; tensor attn_weights_143_cast_fp16 = softmax(axis = var_3688, x = attn_weights_141_cast_fp16)[name = string("attn_weights_143_cast_fp16")]; bool attn_output_65_transpose_x_1 = const()[name = string("attn_output_65_transpose_x_1"), val = bool(true)]; bool attn_output_65_transpose_y_1 = const()[name = string("attn_output_65_transpose_y_1"), val = bool(false)]; tensor attn_output_65_cast_fp16 = matmul(transpose_x = attn_output_65_transpose_x_1, transpose_y = attn_output_65_transpose_y_1, x = attn_weights_143_cast_fp16, y = var_3666_cast_fp16_1)[name = string("attn_output_65_cast_fp16")]; int32 var_3696 = const()[name = string("op_3696"), val = int32(1)]; bool attn_output_67_interleave_0 = const()[name = string("attn_output_67_interleave_0"), val = bool(false)]; tensor attn_output_67_cast_fp16 = concat(axis = var_3696, interleave = attn_output_67_interleave_0, values = (var_3682_cast_fp16, attn_output_65_cast_fp16))[name = string("attn_output_67_cast_fp16")]; tensor var_3700_perm_0 = const()[name = string("op_3700_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_107x = const()[name = string("concat_107x"), val = tensor([1, 2048, 1, -1])]; tensor var_3700_cast_fp16 = transpose(perm = var_3700_perm_0, x = attn_output_67_cast_fp16)[name = string("transpose_229")]; tensor attn_output_71_cast_fp16 = reshape(shape = concat_107x, x = var_3700_cast_fp16)[name = string("attn_output_71_cast_fp16")]; tensor hidden_states_83_strides_0 = const()[name = string("hidden_states_83_strides_0"), val = tensor([1, 1])]; string hidden_states_83_pad_type_0 = const()[name = string("hidden_states_83_pad_type_0"), val = string("valid")]; tensor hidden_states_83_pad_0 = const()[name = string("hidden_states_83_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_83_dilations_0 = const()[name = string("hidden_states_83_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_83_groups_0 = const()[name = string("hidden_states_83_groups_0"), val = int32(1)]; tensor hidden_states_83_cast_fp16 = conv(dilations = hidden_states_83_dilations_0, groups = hidden_states_83_groups_0, pad = hidden_states_83_pad_0, pad_type = hidden_states_83_pad_type_0, strides = hidden_states_83_strides_0, weight = layers_8_self_attn_o_proj_weight_cast_fp16, x = attn_output_71_cast_fp16)[name = string("hidden_states_83_cast_fp16")]; tensor hidden_states_85_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = hidden_states_83_cast_fp16)[name = string("hidden_states_85_cast_fp16")]; fp16 const_88_promoted_to_fp16 = const()[name = string("const_88_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3733_cast_fp16 = mul(x = hidden_states_85_cast_fp16, y = const_88_promoted_to_fp16)[name = string("op_3733_cast_fp16")]; int32 var_3731 = const()[name = string("op_3731"), val = int32(1)]; bool doubled_69_interleave_0 = const()[name = string("doubled_69_interleave_0"), val = bool(false)]; tensor doubled_69_cast_fp16 = concat(axis = var_3731, interleave = doubled_69_interleave_0, values = (hidden_states_85_cast_fp16, var_3733_cast_fp16))[name = string("doubled_69_cast_fp16")]; tensor out_35_axes_0 = const()[name = string("out_35_axes_0"), val = tensor([1])]; tensor out_35_gamma_0_to_fp16 = const()[name = string("out_35_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396442240)))]; fp16 var_3743_to_fp16 = const()[name = string("op_3743_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_35_cast_fp16 = layer_norm(axes = out_35_axes_0, epsilon = var_3743_to_fp16, gamma = out_35_gamma_0_to_fp16, x = doubled_69_cast_fp16)[name = string("out_35_cast_fp16")]; tensor var_3754_split_sizes_0 = const()[name = string("op_3754_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3754_axis_0 = const()[name = string("op_3754_axis_0"), val = int32(1)]; tensor var_3754_cast_fp16_0, tensor var_3754_cast_fp16_1 = split(axis = var_3754_axis_0, split_sizes = var_3754_split_sizes_0, x = out_35_cast_fp16)[name = string("op_3754_cast_fp16")]; tensor input_17_strides_0 = const()[name = string("input_17_strides_0"), val = tensor([1, 1])]; string input_17_pad_type_0 = const()[name = string("input_17_pad_type_0"), val = string("valid")]; tensor input_17_pad_0 = const()[name = string("input_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_17_dilations_0 = const()[name = string("input_17_dilations_0"), val = tensor([1, 1])]; int32 input_17_groups_0 = const()[name = string("input_17_groups_0"), val = int32(1)]; tensor input_17_cast_fp16 = conv(dilations = input_17_dilations_0, groups = input_17_groups_0, pad = input_17_pad_0, pad_type = input_17_pad_type_0, strides = input_17_strides_0, weight = layers_8_mlp_gate_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("input_17_cast_fp16")]; tensor var_3771_cast_fp16 = silu(x = input_17_cast_fp16)[name = string("op_3771_cast_fp16")]; tensor var_3777_strides_0 = const()[name = string("op_3777_strides_0"), val = tensor([1, 1])]; string var_3777_pad_type_0 = const()[name = string("op_3777_pad_type_0"), val = string("valid")]; tensor var_3777_pad_0 = const()[name = string("op_3777_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3777_dilations_0 = const()[name = string("op_3777_dilations_0"), val = tensor([1, 1])]; int32 var_3777_groups_0 = const()[name = string("op_3777_groups_0"), val = int32(1)]; tensor var_3777_cast_fp16 = conv(dilations = var_3777_dilations_0, groups = var_3777_groups_0, pad = var_3777_pad_0, pad_type = var_3777_pad_type_0, strides = var_3777_strides_0, weight = layers_8_mlp_up_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("op_3777_cast_fp16")]; tensor x_89_cast_fp16 = mul(x = var_3771_cast_fp16, y = var_3777_cast_fp16)[name = string("x_89_cast_fp16")]; tensor hidden_states_87_strides_0 = const()[name = string("hidden_states_87_strides_0"), val = tensor([1, 1])]; string hidden_states_87_pad_type_0 = const()[name = string("hidden_states_87_pad_type_0"), val = string("valid")]; tensor hidden_states_87_pad_0 = const()[name = string("hidden_states_87_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_87_dilations_0 = const()[name = string("hidden_states_87_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_87_groups_0 = const()[name = string("hidden_states_87_groups_0"), val = int32(1)]; tensor hidden_states_87_cast_fp16 = conv(dilations = hidden_states_87_dilations_0, groups = hidden_states_87_groups_0, pad = hidden_states_87_pad_0, pad_type = hidden_states_87_pad_type_0, strides = hidden_states_87_strides_0, weight = layers_8_mlp_down_proj_weight_cast_fp16, x = x_89_cast_fp16)[name = string("hidden_states_87_cast_fp16")]; tensor hidden_states_89_cast_fp16 = add(x = hidden_states_85_cast_fp16, y = hidden_states_87_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3795_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_3795_cast_fp16")]; int32 var_3793 = const()[name = string("op_3793"), val = int32(1)]; bool doubled_73_interleave_0 = const()[name = string("doubled_73_interleave_0"), val = bool(false)]; tensor doubled_73_cast_fp16 = concat(axis = var_3793, interleave = doubled_73_interleave_0, values = (hidden_states_89_cast_fp16, var_3795_cast_fp16))[name = string("doubled_73_cast_fp16")]; tensor out_37_axes_0 = const()[name = string("out_37_axes_0"), val = tensor([1])]; tensor out_37_gamma_0_to_fp16 = const()[name = string("out_37_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396450496)))]; fp16 var_3805_to_fp16 = const()[name = string("op_3805_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_37_cast_fp16 = layer_norm(axes = out_37_axes_0, epsilon = var_3805_to_fp16, gamma = out_37_gamma_0_to_fp16, x = doubled_73_cast_fp16)[name = string("out_37_cast_fp16")]; tensor var_3816_split_sizes_0 = const()[name = string("op_3816_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3816_axis_0 = const()[name = string("op_3816_axis_0"), val = int32(1)]; tensor var_3816_cast_fp16_0, tensor var_3816_cast_fp16_1 = split(axis = var_3816_axis_0, split_sizes = var_3816_split_sizes_0, x = out_37_cast_fp16)[name = string("op_3816_cast_fp16")]; tensor layers_9_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396458752)))]; tensor query_states_55_strides_0 = const()[name = string("query_states_55_strides_0"), val = tensor([1, 1])]; string query_states_55_pad_type_0 = const()[name = string("query_states_55_pad_type_0"), val = string("valid")]; tensor query_states_55_pad_0 = const()[name = string("query_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_55_dilations_0 = const()[name = string("query_states_55_dilations_0"), val = tensor([1, 1])]; int32 query_states_55_groups_0 = const()[name = string("query_states_55_groups_0"), val = int32(1)]; tensor query_states_55_cast_fp16 = conv(dilations = query_states_55_dilations_0, groups = query_states_55_groups_0, pad = query_states_55_pad_0, pad_type = query_states_55_pad_type_0, strides = query_states_55_strides_0, weight = layers_9_self_attn_q_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("query_states_55_cast_fp16")]; tensor layers_9_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1404847424)))]; tensor key_states_91_strides_0 = const()[name = string("key_states_91_strides_0"), val = tensor([1, 1])]; string key_states_91_pad_type_0 = const()[name = string("key_states_91_pad_type_0"), val = string("valid")]; tensor key_states_91_pad_0 = const()[name = string("key_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_91_dilations_0 = const()[name = string("key_states_91_dilations_0"), val = tensor([1, 1])]; int32 key_states_91_groups_0 = const()[name = string("key_states_91_groups_0"), val = int32(1)]; tensor key_states_91_cast_fp16 = conv(dilations = key_states_91_dilations_0, groups = key_states_91_groups_0, pad = key_states_91_pad_0, pad_type = key_states_91_pad_type_0, strides = key_states_91_strides_0, weight = layers_9_self_attn_k_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("key_states_91_cast_fp16")]; tensor value_states_55_strides_0 = const()[name = string("value_states_55_strides_0"), val = tensor([1, 1])]; string value_states_55_pad_type_0 = const()[name = string("value_states_55_pad_type_0"), val = string("valid")]; tensor value_states_55_pad_0 = const()[name = string("value_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_55_dilations_0 = const()[name = string("value_states_55_dilations_0"), val = tensor([1, 1])]; int32 value_states_55_groups_0 = const()[name = string("value_states_55_groups_0"), val = int32(1)]; tensor value_states_55_cast_fp16 = conv(dilations = value_states_55_dilations_0, groups = value_states_55_groups_0, pad = value_states_55_pad_0, pad_type = value_states_55_pad_type_0, strides = value_states_55_strides_0, weight = layers_9_self_attn_v_proj_weight_cast_fp16, x = var_3816_cast_fp16_0)[name = string("value_states_55_cast_fp16")]; tensor concat_108x = const()[name = string("concat_108x"), val = tensor([1, 16, 128, -1])]; tensor x_91_cast_fp16 = reshape(shape = concat_108x, x = query_states_55_cast_fp16)[name = string("x_91_cast_fp16")]; tensor concat_109x = const()[name = string("concat_109x"), val = tensor([1, 2, 128, -1])]; tensor var_3873_cast_fp16 = reshape(shape = concat_109x, x = key_states_91_cast_fp16)[name = string("op_3873_cast_fp16")]; tensor concat_110x = const()[name = string("concat_110x"), val = tensor([1, 2, 128, -1])]; tensor var_3880_cast_fp16 = reshape(shape = concat_110x, x = value_states_55_cast_fp16)[name = string("op_3880_cast_fp16")]; tensor var_3884_cast_fp16 = mul(x = x_91_cast_fp16, y = var_869_cast_fp16)[name = string("op_3884_cast_fp16")]; tensor var_3885_split_sizes_0 = const()[name = string("op_3885_split_sizes_0"), val = tensor([64, 64])]; int32 var_3885_axis_0 = const()[name = string("op_3885_axis_0"), val = int32(-2)]; tensor var_3885_cast_fp16_0, tensor var_3885_cast_fp16_1 = split(axis = var_3885_axis_0, split_sizes = var_3885_split_sizes_0, x = x_91_cast_fp16)[name = string("op_3885_cast_fp16")]; fp16 const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3887_cast_fp16 = mul(x = var_3885_cast_fp16_1, y = const_92_promoted_to_fp16)[name = string("op_3887_cast_fp16")]; int32 var_3889 = const()[name = string("op_3889"), val = int32(-2)]; bool var_3890_interleave_0 = const()[name = string("op_3890_interleave_0"), val = bool(false)]; tensor var_3890_cast_fp16 = concat(axis = var_3889, interleave = var_3890_interleave_0, values = (var_3887_cast_fp16, var_3885_cast_fp16_0))[name = string("op_3890_cast_fp16")]; tensor var_3891_cast_fp16 = mul(x = var_3890_cast_fp16, y = var_878_cast_fp16)[name = string("op_3891_cast_fp16")]; tensor query_states_57_cast_fp16 = add(x = var_3884_cast_fp16, y = var_3891_cast_fp16)[name = string("query_states_57_cast_fp16")]; tensor var_3897_cast_fp16 = mul(x = var_3873_cast_fp16, y = var_869_cast_fp16)[name = string("op_3897_cast_fp16")]; tensor var_3898_split_sizes_0 = const()[name = string("op_3898_split_sizes_0"), val = tensor([64, 64])]; int32 var_3898_axis_0 = const()[name = string("op_3898_axis_0"), val = int32(-2)]; tensor var_3898_cast_fp16_0, tensor var_3898_cast_fp16_1 = split(axis = var_3898_axis_0, split_sizes = var_3898_split_sizes_0, x = var_3873_cast_fp16)[name = string("op_3898_cast_fp16")]; fp16 const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3900_cast_fp16 = mul(x = var_3898_cast_fp16_1, y = const_93_promoted_to_fp16)[name = string("op_3900_cast_fp16")]; int32 var_3902 = const()[name = string("op_3902"), val = int32(-2)]; bool var_3903_interleave_0 = const()[name = string("op_3903_interleave_0"), val = bool(false)]; tensor var_3903_cast_fp16 = concat(axis = var_3902, interleave = var_3903_interleave_0, values = (var_3900_cast_fp16, var_3898_cast_fp16_0))[name = string("op_3903_cast_fp16")]; tensor var_3904_cast_fp16 = mul(x = var_3903_cast_fp16, y = var_878_cast_fp16)[name = string("op_3904_cast_fp16")]; tensor key_states_95_cast_fp16 = add(x = var_3897_cast_fp16, y = var_3904_cast_fp16)[name = string("key_states_95_cast_fp16")]; tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; int32 concat_113_axis_0 = const()[name = string("concat_113_axis_0"), val = int32(0)]; bool concat_113_interleave_0 = const()[name = string("concat_113_interleave_0"), val = bool(false)]; tensor concat_113 = concat(axis = concat_113_axis_0, interleave = concat_113_interleave_0, values = (expand_dims_108, expand_dims_109, position_id, expand_dims_111))[name = string("concat_113")]; tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; tensor concat_114_values1_0 = const()[name = string("concat_114_values1_0"), val = tensor([0])]; tensor concat_114_values3_0 = const()[name = string("concat_114_values3_0"), val = tensor([0])]; int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_112, concat_114_values1_0, cache_position_end, concat_114_values3_0))[name = string("concat_114")]; tensor key_states_97_perm_0 = const()[name = string("key_states_97_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_10_stride_0 = const()[name = string("key_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_97_cast_fp16 = transpose(perm = key_states_97_perm_0, x = key_states_95_cast_fp16)[name = string("transpose_228")]; tensor key_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = key_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = key_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_10_squeeze_mask_0, stride = key_cache_internal_tensor_assign_10_stride_0, update = key_states_97_cast_fp16, x = coreml_update_state_128)[name = string("key_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_10_cast_fp16, input = key_cache)[name = string("coreml_update_state_130_write_state")]; tensor coreml_update_state_130 = read_state(input = key_cache)[name = string("coreml_update_state_130")]; tensor value_states_57_perm_0 = const()[name = string("value_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_10_stride_0 = const()[name = string("value_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_57_cast_fp16 = transpose(perm = value_states_57_perm_0, x = var_3880_cast_fp16)[name = string("transpose_227")]; tensor value_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = value_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = value_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_10_squeeze_mask_0, stride = value_cache_internal_tensor_assign_10_stride_0, update = value_states_57_cast_fp16, x = coreml_update_state_129)[name = string("value_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_10_cast_fp16, input = value_cache)[name = string("coreml_update_state_131_write_state")]; tensor coreml_update_state_131 = read_state(input = value_cache)[name = string("coreml_update_state_131")]; tensor var_3974_begin_0 = const()[name = string("op_3974_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3974_end_0 = const()[name = string("op_3974_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3974_end_mask_0 = const()[name = string("op_3974_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3974_cast_fp16 = slice_by_index(begin = var_3974_begin_0, end = var_3974_end_0, end_mask = var_3974_end_mask_0, x = coreml_update_state_130)[name = string("op_3974_cast_fp16")]; tensor tile_18 = const()[name = string("tile_18"), val = tensor([1, 1])]; int32 var_3977_axis_0 = const()[name = string("op_3977_axis_0"), val = int32(1)]; tensor var_3977_cast_fp16_0, tensor var_3977_cast_fp16_1 = split(axis = var_3977_axis_0, split_sizes = tile_18, x = var_3974_cast_fp16)[name = string("op_3977_cast_fp16")]; tensor var_3984_begin_0 = const()[name = string("op_3984_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3984_end_0 = const()[name = string("op_3984_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3984_end_mask_0 = const()[name = string("op_3984_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3984_cast_fp16 = slice_by_index(begin = var_3984_begin_0, end = var_3984_end_0, end_mask = var_3984_end_mask_0, x = coreml_update_state_131)[name = string("op_3984_cast_fp16")]; tensor tile_19 = const()[name = string("tile_19"), val = tensor([1, 1])]; int32 var_3987_axis_0 = const()[name = string("op_3987_axis_0"), val = int32(1)]; tensor var_3987_cast_fp16_0, tensor var_3987_cast_fp16_1 = split(axis = var_3987_axis_0, split_sizes = tile_19, x = var_3984_cast_fp16)[name = string("op_3987_cast_fp16")]; tensor var_3990_split_sizes_0 = const()[name = string("op_3990_split_sizes_0"), val = tensor([8, 8])]; int32 var_3990_axis_0 = const()[name = string("op_3990_axis_0"), val = int32(1)]; tensor var_3990_0, tensor var_3990_1 = split(axis = var_3990_axis_0, split_sizes = var_3990_split_sizes_0, x = query_states_57_cast_fp16)[name = string("op_3990")]; bool attn_weights_145_transpose_x_0 = const()[name = string("attn_weights_145_transpose_x_0"), val = bool(false)]; bool attn_weights_145_transpose_y_0 = const()[name = string("attn_weights_145_transpose_y_0"), val = bool(false)]; tensor attn_weights_145_cast_fp16 = matmul(transpose_x = attn_weights_145_transpose_x_0, transpose_y = attn_weights_145_transpose_y_0, x = var_3977_cast_fp16_0, y = var_3990_0)[name = string("attn_weights_145_cast_fp16")]; fp16 var_3993_to_fp16 = const()[name = string("op_3993_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_147_cast_fp16 = mul(x = attn_weights_145_cast_fp16, y = var_3993_to_fp16)[name = string("attn_weights_147_cast_fp16")]; tensor attn_weights_149_cast_fp16 = add(x = attn_weights_147_cast_fp16, y = attn_mask_1)[name = string("attn_weights_149_cast_fp16")]; int32 var_3997 = const()[name = string("op_3997"), val = int32(-2)]; tensor attn_weights_151_cast_fp16 = softmax(axis = var_3997, x = attn_weights_149_cast_fp16)[name = string("attn_weights_151_cast_fp16")]; bool var_4003_transpose_x_1 = const()[name = string("op_4003_transpose_x_1"), val = bool(true)]; bool var_4003_transpose_y_1 = const()[name = string("op_4003_transpose_y_1"), val = bool(false)]; tensor var_4003_cast_fp16 = matmul(transpose_x = var_4003_transpose_x_1, transpose_y = var_4003_transpose_y_1, x = attn_weights_151_cast_fp16, y = var_3987_cast_fp16_0)[name = string("op_4003_cast_fp16")]; bool attn_weights_153_transpose_x_0 = const()[name = string("attn_weights_153_transpose_x_0"), val = bool(false)]; bool attn_weights_153_transpose_y_0 = const()[name = string("attn_weights_153_transpose_y_0"), val = bool(false)]; tensor attn_weights_153_cast_fp16 = matmul(transpose_x = attn_weights_153_transpose_x_0, transpose_y = attn_weights_153_transpose_y_0, x = var_3977_cast_fp16_1, y = var_3990_1)[name = string("attn_weights_153_cast_fp16")]; fp16 var_4005_to_fp16 = const()[name = string("op_4005_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_155_cast_fp16 = mul(x = attn_weights_153_cast_fp16, y = var_4005_to_fp16)[name = string("attn_weights_155_cast_fp16")]; tensor attn_weights_157_cast_fp16 = add(x = attn_weights_155_cast_fp16, y = attn_mask_1)[name = string("attn_weights_157_cast_fp16")]; int32 var_4009 = const()[name = string("op_4009"), val = int32(-2)]; tensor attn_weights_159_cast_fp16 = softmax(axis = var_4009, x = attn_weights_157_cast_fp16)[name = string("attn_weights_159_cast_fp16")]; bool attn_output_73_transpose_x_1 = const()[name = string("attn_output_73_transpose_x_1"), val = bool(true)]; bool attn_output_73_transpose_y_1 = const()[name = string("attn_output_73_transpose_y_1"), val = bool(false)]; tensor attn_output_73_cast_fp16 = matmul(transpose_x = attn_output_73_transpose_x_1, transpose_y = attn_output_73_transpose_y_1, x = attn_weights_159_cast_fp16, y = var_3987_cast_fp16_1)[name = string("attn_output_73_cast_fp16")]; int32 var_4017 = const()[name = string("op_4017"), val = int32(1)]; bool attn_output_75_interleave_0 = const()[name = string("attn_output_75_interleave_0"), val = bool(false)]; tensor attn_output_75_cast_fp16 = concat(axis = var_4017, interleave = attn_output_75_interleave_0, values = (var_4003_cast_fp16, attn_output_73_cast_fp16))[name = string("attn_output_75_cast_fp16")]; tensor var_4021_perm_0 = const()[name = string("op_4021_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_119x = const()[name = string("concat_119x"), val = tensor([1, 2048, 1, -1])]; tensor var_4021_cast_fp16 = transpose(perm = var_4021_perm_0, x = attn_output_75_cast_fp16)[name = string("transpose_226")]; tensor attn_output_79_cast_fp16 = reshape(shape = concat_119x, x = var_4021_cast_fp16)[name = string("attn_output_79_cast_fp16")]; tensor hidden_states_93_strides_0 = const()[name = string("hidden_states_93_strides_0"), val = tensor([1, 1])]; string hidden_states_93_pad_type_0 = const()[name = string("hidden_states_93_pad_type_0"), val = string("valid")]; tensor hidden_states_93_pad_0 = const()[name = string("hidden_states_93_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_93_dilations_0 = const()[name = string("hidden_states_93_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_93_groups_0 = const()[name = string("hidden_states_93_groups_0"), val = int32(1)]; tensor hidden_states_93_cast_fp16 = conv(dilations = hidden_states_93_dilations_0, groups = hidden_states_93_groups_0, pad = hidden_states_93_pad_0, pad_type = hidden_states_93_pad_type_0, strides = hidden_states_93_strides_0, weight = layers_9_self_attn_o_proj_weight_cast_fp16, x = attn_output_79_cast_fp16)[name = string("hidden_states_93_cast_fp16")]; tensor hidden_states_95_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = hidden_states_93_cast_fp16)[name = string("hidden_states_95_cast_fp16")]; fp16 const_98_promoted_to_fp16 = const()[name = string("const_98_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4054_cast_fp16 = mul(x = hidden_states_95_cast_fp16, y = const_98_promoted_to_fp16)[name = string("op_4054_cast_fp16")]; int32 var_4052 = const()[name = string("op_4052"), val = int32(1)]; bool doubled_77_interleave_0 = const()[name = string("doubled_77_interleave_0"), val = bool(false)]; tensor doubled_77_cast_fp16 = concat(axis = var_4052, interleave = doubled_77_interleave_0, values = (hidden_states_95_cast_fp16, var_4054_cast_fp16))[name = string("doubled_77_cast_fp16")]; tensor out_39_axes_0 = const()[name = string("out_39_axes_0"), val = tensor([1])]; tensor out_39_gamma_0_to_fp16 = const()[name = string("out_39_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405896064)))]; fp16 var_4064_to_fp16 = const()[name = string("op_4064_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_39_cast_fp16 = layer_norm(axes = out_39_axes_0, epsilon = var_4064_to_fp16, gamma = out_39_gamma_0_to_fp16, x = doubled_77_cast_fp16)[name = string("out_39_cast_fp16")]; tensor var_4075_split_sizes_0 = const()[name = string("op_4075_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4075_axis_0 = const()[name = string("op_4075_axis_0"), val = int32(1)]; tensor var_4075_cast_fp16_0, tensor var_4075_cast_fp16_1 = split(axis = var_4075_axis_0, split_sizes = var_4075_split_sizes_0, x = out_39_cast_fp16)[name = string("op_4075_cast_fp16")]; tensor input_19_strides_0 = const()[name = string("input_19_strides_0"), val = tensor([1, 1])]; string input_19_pad_type_0 = const()[name = string("input_19_pad_type_0"), val = string("valid")]; tensor input_19_pad_0 = const()[name = string("input_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_19_dilations_0 = const()[name = string("input_19_dilations_0"), val = tensor([1, 1])]; int32 input_19_groups_0 = const()[name = string("input_19_groups_0"), val = int32(1)]; tensor input_19_cast_fp16 = conv(dilations = input_19_dilations_0, groups = input_19_groups_0, pad = input_19_pad_0, pad_type = input_19_pad_type_0, strides = input_19_strides_0, weight = layers_9_mlp_gate_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("input_19_cast_fp16")]; tensor var_4092_cast_fp16 = silu(x = input_19_cast_fp16)[name = string("op_4092_cast_fp16")]; tensor var_4098_strides_0 = const()[name = string("op_4098_strides_0"), val = tensor([1, 1])]; string var_4098_pad_type_0 = const()[name = string("op_4098_pad_type_0"), val = string("valid")]; tensor var_4098_pad_0 = const()[name = string("op_4098_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4098_dilations_0 = const()[name = string("op_4098_dilations_0"), val = tensor([1, 1])]; int32 var_4098_groups_0 = const()[name = string("op_4098_groups_0"), val = int32(1)]; tensor var_4098_cast_fp16 = conv(dilations = var_4098_dilations_0, groups = var_4098_groups_0, pad = var_4098_pad_0, pad_type = var_4098_pad_type_0, strides = var_4098_strides_0, weight = layers_9_mlp_up_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("op_4098_cast_fp16")]; tensor x_99_cast_fp16 = mul(x = var_4092_cast_fp16, y = var_4098_cast_fp16)[name = string("x_99_cast_fp16")]; tensor hidden_states_97_strides_0 = const()[name = string("hidden_states_97_strides_0"), val = tensor([1, 1])]; string hidden_states_97_pad_type_0 = const()[name = string("hidden_states_97_pad_type_0"), val = string("valid")]; tensor hidden_states_97_pad_0 = const()[name = string("hidden_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_97_dilations_0 = const()[name = string("hidden_states_97_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_97_groups_0 = const()[name = string("hidden_states_97_groups_0"), val = int32(1)]; tensor hidden_states_97_cast_fp16 = conv(dilations = hidden_states_97_dilations_0, groups = hidden_states_97_groups_0, pad = hidden_states_97_pad_0, pad_type = hidden_states_97_pad_type_0, strides = hidden_states_97_strides_0, weight = layers_9_mlp_down_proj_weight_cast_fp16, x = x_99_cast_fp16)[name = string("hidden_states_97_cast_fp16")]; tensor hidden_states_99_cast_fp16 = add(x = hidden_states_95_cast_fp16, y = hidden_states_97_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; fp16 const_100_promoted_to_fp16 = const()[name = string("const_100_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4116_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_100_promoted_to_fp16)[name = string("op_4116_cast_fp16")]; int32 var_4114 = const()[name = string("op_4114"), val = int32(1)]; bool doubled_81_interleave_0 = const()[name = string("doubled_81_interleave_0"), val = bool(false)]; tensor doubled_81_cast_fp16 = concat(axis = var_4114, interleave = doubled_81_interleave_0, values = (hidden_states_99_cast_fp16, var_4116_cast_fp16))[name = string("doubled_81_cast_fp16")]; tensor out_41_axes_0 = const()[name = string("out_41_axes_0"), val = tensor([1])]; tensor out_41_gamma_0_to_fp16 = const()[name = string("out_41_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405904320)))]; fp16 var_4126_to_fp16 = const()[name = string("op_4126_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_41_cast_fp16 = layer_norm(axes = out_41_axes_0, epsilon = var_4126_to_fp16, gamma = out_41_gamma_0_to_fp16, x = doubled_81_cast_fp16)[name = string("out_41_cast_fp16")]; tensor var_4137_split_sizes_0 = const()[name = string("op_4137_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4137_axis_0 = const()[name = string("op_4137_axis_0"), val = int32(1)]; tensor var_4137_cast_fp16_0, tensor var_4137_cast_fp16_1 = split(axis = var_4137_axis_0, split_sizes = var_4137_split_sizes_0, x = out_41_cast_fp16)[name = string("op_4137_cast_fp16")]; tensor layers_10_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405912576)))]; tensor query_states_61_strides_0 = const()[name = string("query_states_61_strides_0"), val = tensor([1, 1])]; string query_states_61_pad_type_0 = const()[name = string("query_states_61_pad_type_0"), val = string("valid")]; tensor query_states_61_pad_0 = const()[name = string("query_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_61_dilations_0 = const()[name = string("query_states_61_dilations_0"), val = tensor([1, 1])]; int32 query_states_61_groups_0 = const()[name = string("query_states_61_groups_0"), val = int32(1)]; tensor query_states_61_cast_fp16 = conv(dilations = query_states_61_dilations_0, groups = query_states_61_groups_0, pad = query_states_61_pad_0, pad_type = query_states_61_pad_type_0, strides = query_states_61_strides_0, weight = layers_10_self_attn_q_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("query_states_61_cast_fp16")]; tensor layers_10_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1414301248)))]; tensor key_states_101_strides_0 = const()[name = string("key_states_101_strides_0"), val = tensor([1, 1])]; string key_states_101_pad_type_0 = const()[name = string("key_states_101_pad_type_0"), val = string("valid")]; tensor key_states_101_pad_0 = const()[name = string("key_states_101_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_101_dilations_0 = const()[name = string("key_states_101_dilations_0"), val = tensor([1, 1])]; int32 key_states_101_groups_0 = const()[name = string("key_states_101_groups_0"), val = int32(1)]; tensor key_states_101_cast_fp16 = conv(dilations = key_states_101_dilations_0, groups = key_states_101_groups_0, pad = key_states_101_pad_0, pad_type = key_states_101_pad_type_0, strides = key_states_101_strides_0, weight = layers_10_self_attn_k_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("key_states_101_cast_fp16")]; tensor value_states_61_strides_0 = const()[name = string("value_states_61_strides_0"), val = tensor([1, 1])]; string value_states_61_pad_type_0 = const()[name = string("value_states_61_pad_type_0"), val = string("valid")]; tensor value_states_61_pad_0 = const()[name = string("value_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_61_dilations_0 = const()[name = string("value_states_61_dilations_0"), val = tensor([1, 1])]; int32 value_states_61_groups_0 = const()[name = string("value_states_61_groups_0"), val = int32(1)]; tensor value_states_61_cast_fp16 = conv(dilations = value_states_61_dilations_0, groups = value_states_61_groups_0, pad = value_states_61_pad_0, pad_type = value_states_61_pad_type_0, strides = value_states_61_strides_0, weight = layers_10_self_attn_v_proj_weight_cast_fp16, x = var_4137_cast_fp16_0)[name = string("value_states_61_cast_fp16")]; tensor concat_120x = const()[name = string("concat_120x"), val = tensor([1, 16, 128, -1])]; tensor x_101_cast_fp16 = reshape(shape = concat_120x, x = query_states_61_cast_fp16)[name = string("x_101_cast_fp16")]; tensor concat_121x = const()[name = string("concat_121x"), val = tensor([1, 2, 128, -1])]; tensor var_4194_cast_fp16 = reshape(shape = concat_121x, x = key_states_101_cast_fp16)[name = string("op_4194_cast_fp16")]; tensor concat_122x = const()[name = string("concat_122x"), val = tensor([1, 2, 128, -1])]; tensor var_4201_cast_fp16 = reshape(shape = concat_122x, x = value_states_61_cast_fp16)[name = string("op_4201_cast_fp16")]; tensor var_4205_cast_fp16 = mul(x = x_101_cast_fp16, y = var_869_cast_fp16)[name = string("op_4205_cast_fp16")]; tensor var_4206_split_sizes_0 = const()[name = string("op_4206_split_sizes_0"), val = tensor([64, 64])]; int32 var_4206_axis_0 = const()[name = string("op_4206_axis_0"), val = int32(-2)]; tensor var_4206_cast_fp16_0, tensor var_4206_cast_fp16_1 = split(axis = var_4206_axis_0, split_sizes = var_4206_split_sizes_0, x = x_101_cast_fp16)[name = string("op_4206_cast_fp16")]; fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4208_cast_fp16 = mul(x = var_4206_cast_fp16_1, y = const_102_promoted_to_fp16)[name = string("op_4208_cast_fp16")]; int32 var_4210 = const()[name = string("op_4210"), val = int32(-2)]; bool var_4211_interleave_0 = const()[name = string("op_4211_interleave_0"), val = bool(false)]; tensor var_4211_cast_fp16 = concat(axis = var_4210, interleave = var_4211_interleave_0, values = (var_4208_cast_fp16, var_4206_cast_fp16_0))[name = string("op_4211_cast_fp16")]; tensor var_4212_cast_fp16 = mul(x = var_4211_cast_fp16, y = var_878_cast_fp16)[name = string("op_4212_cast_fp16")]; tensor query_states_63_cast_fp16 = add(x = var_4205_cast_fp16, y = var_4212_cast_fp16)[name = string("query_states_63_cast_fp16")]; tensor var_4218_cast_fp16 = mul(x = var_4194_cast_fp16, y = var_869_cast_fp16)[name = string("op_4218_cast_fp16")]; tensor var_4219_split_sizes_0 = const()[name = string("op_4219_split_sizes_0"), val = tensor([64, 64])]; int32 var_4219_axis_0 = const()[name = string("op_4219_axis_0"), val = int32(-2)]; tensor var_4219_cast_fp16_0, tensor var_4219_cast_fp16_1 = split(axis = var_4219_axis_0, split_sizes = var_4219_split_sizes_0, x = var_4194_cast_fp16)[name = string("op_4219_cast_fp16")]; fp16 const_103_promoted_to_fp16 = const()[name = string("const_103_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4221_cast_fp16 = mul(x = var_4219_cast_fp16_1, y = const_103_promoted_to_fp16)[name = string("op_4221_cast_fp16")]; int32 var_4223 = const()[name = string("op_4223"), val = int32(-2)]; bool var_4224_interleave_0 = const()[name = string("op_4224_interleave_0"), val = bool(false)]; tensor var_4224_cast_fp16 = concat(axis = var_4223, interleave = var_4224_interleave_0, values = (var_4221_cast_fp16, var_4219_cast_fp16_0))[name = string("op_4224_cast_fp16")]; tensor var_4225_cast_fp16 = mul(x = var_4224_cast_fp16, y = var_878_cast_fp16)[name = string("op_4225_cast_fp16")]; tensor key_states_105_cast_fp16 = add(x = var_4218_cast_fp16, y = var_4225_cast_fp16)[name = string("key_states_105_cast_fp16")]; tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; int32 concat_125_axis_0 = const()[name = string("concat_125_axis_0"), val = int32(0)]; bool concat_125_interleave_0 = const()[name = string("concat_125_interleave_0"), val = bool(false)]; tensor concat_125 = concat(axis = concat_125_axis_0, interleave = concat_125_interleave_0, values = (expand_dims_120, expand_dims_121, position_id, expand_dims_123))[name = string("concat_125")]; tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; tensor concat_126_values1_0 = const()[name = string("concat_126_values1_0"), val = tensor([0])]; tensor concat_126_values3_0 = const()[name = string("concat_126_values3_0"), val = tensor([0])]; int32 concat_126_axis_0 = const()[name = string("concat_126_axis_0"), val = int32(0)]; bool concat_126_interleave_0 = const()[name = string("concat_126_interleave_0"), val = bool(false)]; tensor concat_126 = concat(axis = concat_126_axis_0, interleave = concat_126_interleave_0, values = (expand_dims_124, concat_126_values1_0, cache_position_end, concat_126_values3_0))[name = string("concat_126")]; tensor key_states_107_perm_0 = const()[name = string("key_states_107_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_11_stride_0 = const()[name = string("key_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_107_cast_fp16 = transpose(perm = key_states_107_perm_0, x = key_states_105_cast_fp16)[name = string("transpose_225")]; tensor key_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = key_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = key_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_11_squeeze_mask_0, stride = key_cache_internal_tensor_assign_11_stride_0, update = key_states_107_cast_fp16, x = coreml_update_state_130)[name = string("key_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_11_cast_fp16, input = key_cache)[name = string("coreml_update_state_132_write_state")]; tensor coreml_update_state_132 = read_state(input = key_cache)[name = string("coreml_update_state_132")]; tensor value_states_63_perm_0 = const()[name = string("value_states_63_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_11_stride_0 = const()[name = string("value_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_63_cast_fp16 = transpose(perm = value_states_63_perm_0, x = var_4201_cast_fp16)[name = string("transpose_224")]; tensor value_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = value_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = value_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_11_squeeze_mask_0, stride = value_cache_internal_tensor_assign_11_stride_0, update = value_states_63_cast_fp16, x = coreml_update_state_131)[name = string("value_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_11_cast_fp16, input = value_cache)[name = string("coreml_update_state_133_write_state")]; tensor coreml_update_state_133 = read_state(input = value_cache)[name = string("coreml_update_state_133")]; tensor var_4295_begin_0 = const()[name = string("op_4295_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4295_end_0 = const()[name = string("op_4295_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4295_end_mask_0 = const()[name = string("op_4295_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4295_cast_fp16 = slice_by_index(begin = var_4295_begin_0, end = var_4295_end_0, end_mask = var_4295_end_mask_0, x = coreml_update_state_132)[name = string("op_4295_cast_fp16")]; tensor tile_20 = const()[name = string("tile_20"), val = tensor([1, 1])]; int32 var_4298_axis_0 = const()[name = string("op_4298_axis_0"), val = int32(1)]; tensor var_4298_cast_fp16_0, tensor var_4298_cast_fp16_1 = split(axis = var_4298_axis_0, split_sizes = tile_20, x = var_4295_cast_fp16)[name = string("op_4298_cast_fp16")]; tensor var_4305_begin_0 = const()[name = string("op_4305_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4305_end_0 = const()[name = string("op_4305_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4305_end_mask_0 = const()[name = string("op_4305_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4305_cast_fp16 = slice_by_index(begin = var_4305_begin_0, end = var_4305_end_0, end_mask = var_4305_end_mask_0, x = coreml_update_state_133)[name = string("op_4305_cast_fp16")]; tensor tile_21 = const()[name = string("tile_21"), val = tensor([1, 1])]; int32 var_4308_axis_0 = const()[name = string("op_4308_axis_0"), val = int32(1)]; tensor var_4308_cast_fp16_0, tensor var_4308_cast_fp16_1 = split(axis = var_4308_axis_0, split_sizes = tile_21, x = var_4305_cast_fp16)[name = string("op_4308_cast_fp16")]; tensor var_4311_split_sizes_0 = const()[name = string("op_4311_split_sizes_0"), val = tensor([8, 8])]; int32 var_4311_axis_0 = const()[name = string("op_4311_axis_0"), val = int32(1)]; tensor var_4311_0, tensor var_4311_1 = split(axis = var_4311_axis_0, split_sizes = var_4311_split_sizes_0, x = query_states_63_cast_fp16)[name = string("op_4311")]; bool attn_weights_161_transpose_x_0 = const()[name = string("attn_weights_161_transpose_x_0"), val = bool(false)]; bool attn_weights_161_transpose_y_0 = const()[name = string("attn_weights_161_transpose_y_0"), val = bool(false)]; tensor attn_weights_161_cast_fp16 = matmul(transpose_x = attn_weights_161_transpose_x_0, transpose_y = attn_weights_161_transpose_y_0, x = var_4298_cast_fp16_0, y = var_4311_0)[name = string("attn_weights_161_cast_fp16")]; fp16 var_4314_to_fp16 = const()[name = string("op_4314_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_163_cast_fp16 = mul(x = attn_weights_161_cast_fp16, y = var_4314_to_fp16)[name = string("attn_weights_163_cast_fp16")]; tensor attn_weights_165_cast_fp16 = add(x = attn_weights_163_cast_fp16, y = attn_mask_1)[name = string("attn_weights_165_cast_fp16")]; int32 var_4318 = const()[name = string("op_4318"), val = int32(-2)]; tensor attn_weights_167_cast_fp16 = softmax(axis = var_4318, x = attn_weights_165_cast_fp16)[name = string("attn_weights_167_cast_fp16")]; bool var_4324_transpose_x_1 = const()[name = string("op_4324_transpose_x_1"), val = bool(true)]; bool var_4324_transpose_y_1 = const()[name = string("op_4324_transpose_y_1"), val = bool(false)]; tensor var_4324_cast_fp16 = matmul(transpose_x = var_4324_transpose_x_1, transpose_y = var_4324_transpose_y_1, x = attn_weights_167_cast_fp16, y = var_4308_cast_fp16_0)[name = string("op_4324_cast_fp16")]; bool attn_weights_169_transpose_x_0 = const()[name = string("attn_weights_169_transpose_x_0"), val = bool(false)]; bool attn_weights_169_transpose_y_0 = const()[name = string("attn_weights_169_transpose_y_0"), val = bool(false)]; tensor attn_weights_169_cast_fp16 = matmul(transpose_x = attn_weights_169_transpose_x_0, transpose_y = attn_weights_169_transpose_y_0, x = var_4298_cast_fp16_1, y = var_4311_1)[name = string("attn_weights_169_cast_fp16")]; fp16 var_4326_to_fp16 = const()[name = string("op_4326_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_171_cast_fp16 = mul(x = attn_weights_169_cast_fp16, y = var_4326_to_fp16)[name = string("attn_weights_171_cast_fp16")]; tensor attn_weights_173_cast_fp16 = add(x = attn_weights_171_cast_fp16, y = attn_mask_1)[name = string("attn_weights_173_cast_fp16")]; int32 var_4330 = const()[name = string("op_4330"), val = int32(-2)]; tensor attn_weights_175_cast_fp16 = softmax(axis = var_4330, x = attn_weights_173_cast_fp16)[name = string("attn_weights_175_cast_fp16")]; bool attn_output_81_transpose_x_1 = const()[name = string("attn_output_81_transpose_x_1"), val = bool(true)]; bool attn_output_81_transpose_y_1 = const()[name = string("attn_output_81_transpose_y_1"), val = bool(false)]; tensor attn_output_81_cast_fp16 = matmul(transpose_x = attn_output_81_transpose_x_1, transpose_y = attn_output_81_transpose_y_1, x = attn_weights_175_cast_fp16, y = var_4308_cast_fp16_1)[name = string("attn_output_81_cast_fp16")]; int32 var_4338 = const()[name = string("op_4338"), val = int32(1)]; bool attn_output_83_interleave_0 = const()[name = string("attn_output_83_interleave_0"), val = bool(false)]; tensor attn_output_83_cast_fp16 = concat(axis = var_4338, interleave = attn_output_83_interleave_0, values = (var_4324_cast_fp16, attn_output_81_cast_fp16))[name = string("attn_output_83_cast_fp16")]; tensor var_4342_perm_0 = const()[name = string("op_4342_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_131x = const()[name = string("concat_131x"), val = tensor([1, 2048, 1, -1])]; tensor var_4342_cast_fp16 = transpose(perm = var_4342_perm_0, x = attn_output_83_cast_fp16)[name = string("transpose_223")]; tensor attn_output_87_cast_fp16 = reshape(shape = concat_131x, x = var_4342_cast_fp16)[name = string("attn_output_87_cast_fp16")]; tensor hidden_states_103_strides_0 = const()[name = string("hidden_states_103_strides_0"), val = tensor([1, 1])]; string hidden_states_103_pad_type_0 = const()[name = string("hidden_states_103_pad_type_0"), val = string("valid")]; tensor hidden_states_103_pad_0 = const()[name = string("hidden_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_103_dilations_0 = const()[name = string("hidden_states_103_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_103_groups_0 = const()[name = string("hidden_states_103_groups_0"), val = int32(1)]; tensor hidden_states_103_cast_fp16 = conv(dilations = hidden_states_103_dilations_0, groups = hidden_states_103_groups_0, pad = hidden_states_103_pad_0, pad_type = hidden_states_103_pad_type_0, strides = hidden_states_103_strides_0, weight = layers_10_self_attn_o_proj_weight_cast_fp16, x = attn_output_87_cast_fp16)[name = string("hidden_states_103_cast_fp16")]; tensor hidden_states_105_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = hidden_states_103_cast_fp16)[name = string("hidden_states_105_cast_fp16")]; fp16 const_108_promoted_to_fp16 = const()[name = string("const_108_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4375_cast_fp16 = mul(x = hidden_states_105_cast_fp16, y = const_108_promoted_to_fp16)[name = string("op_4375_cast_fp16")]; int32 var_4373 = const()[name = string("op_4373"), val = int32(1)]; bool doubled_85_interleave_0 = const()[name = string("doubled_85_interleave_0"), val = bool(false)]; tensor doubled_85_cast_fp16 = concat(axis = var_4373, interleave = doubled_85_interleave_0, values = (hidden_states_105_cast_fp16, var_4375_cast_fp16))[name = string("doubled_85_cast_fp16")]; tensor out_43_axes_0 = const()[name = string("out_43_axes_0"), val = tensor([1])]; tensor out_43_gamma_0_to_fp16 = const()[name = string("out_43_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415349888)))]; fp16 var_4385_to_fp16 = const()[name = string("op_4385_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_43_cast_fp16 = layer_norm(axes = out_43_axes_0, epsilon = var_4385_to_fp16, gamma = out_43_gamma_0_to_fp16, x = doubled_85_cast_fp16)[name = string("out_43_cast_fp16")]; tensor var_4396_split_sizes_0 = const()[name = string("op_4396_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4396_axis_0 = const()[name = string("op_4396_axis_0"), val = int32(1)]; tensor var_4396_cast_fp16_0, tensor var_4396_cast_fp16_1 = split(axis = var_4396_axis_0, split_sizes = var_4396_split_sizes_0, x = out_43_cast_fp16)[name = string("op_4396_cast_fp16")]; tensor input_21_strides_0 = const()[name = string("input_21_strides_0"), val = tensor([1, 1])]; string input_21_pad_type_0 = const()[name = string("input_21_pad_type_0"), val = string("valid")]; tensor input_21_pad_0 = const()[name = string("input_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_21_dilations_0 = const()[name = string("input_21_dilations_0"), val = tensor([1, 1])]; int32 input_21_groups_0 = const()[name = string("input_21_groups_0"), val = int32(1)]; tensor input_21_cast_fp16 = conv(dilations = input_21_dilations_0, groups = input_21_groups_0, pad = input_21_pad_0, pad_type = input_21_pad_type_0, strides = input_21_strides_0, weight = layers_10_mlp_gate_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("input_21_cast_fp16")]; tensor var_4413_cast_fp16 = silu(x = input_21_cast_fp16)[name = string("op_4413_cast_fp16")]; tensor var_4419_strides_0 = const()[name = string("op_4419_strides_0"), val = tensor([1, 1])]; string var_4419_pad_type_0 = const()[name = string("op_4419_pad_type_0"), val = string("valid")]; tensor var_4419_pad_0 = const()[name = string("op_4419_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4419_dilations_0 = const()[name = string("op_4419_dilations_0"), val = tensor([1, 1])]; int32 var_4419_groups_0 = const()[name = string("op_4419_groups_0"), val = int32(1)]; tensor var_4419_cast_fp16 = conv(dilations = var_4419_dilations_0, groups = var_4419_groups_0, pad = var_4419_pad_0, pad_type = var_4419_pad_type_0, strides = var_4419_strides_0, weight = layers_10_mlp_up_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("op_4419_cast_fp16")]; tensor x_109_cast_fp16 = mul(x = var_4413_cast_fp16, y = var_4419_cast_fp16)[name = string("x_109_cast_fp16")]; tensor hidden_states_107_strides_0 = const()[name = string("hidden_states_107_strides_0"), val = tensor([1, 1])]; string hidden_states_107_pad_type_0 = const()[name = string("hidden_states_107_pad_type_0"), val = string("valid")]; tensor hidden_states_107_pad_0 = const()[name = string("hidden_states_107_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_107_dilations_0 = const()[name = string("hidden_states_107_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_107_groups_0 = const()[name = string("hidden_states_107_groups_0"), val = int32(1)]; tensor hidden_states_107_cast_fp16 = conv(dilations = hidden_states_107_dilations_0, groups = hidden_states_107_groups_0, pad = hidden_states_107_pad_0, pad_type = hidden_states_107_pad_type_0, strides = hidden_states_107_strides_0, weight = layers_10_mlp_down_proj_weight_cast_fp16, x = x_109_cast_fp16)[name = string("hidden_states_107_cast_fp16")]; tensor hidden_states_109_cast_fp16 = add(x = hidden_states_105_cast_fp16, y = hidden_states_107_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; fp16 const_110_promoted_to_fp16 = const()[name = string("const_110_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4437_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_110_promoted_to_fp16)[name = string("op_4437_cast_fp16")]; int32 var_4435 = const()[name = string("op_4435"), val = int32(1)]; bool doubled_89_interleave_0 = const()[name = string("doubled_89_interleave_0"), val = bool(false)]; tensor doubled_89_cast_fp16 = concat(axis = var_4435, interleave = doubled_89_interleave_0, values = (hidden_states_109_cast_fp16, var_4437_cast_fp16))[name = string("doubled_89_cast_fp16")]; tensor out_45_axes_0 = const()[name = string("out_45_axes_0"), val = tensor([1])]; tensor out_45_gamma_0_to_fp16 = const()[name = string("out_45_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415358144)))]; fp16 var_4447_to_fp16 = const()[name = string("op_4447_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_45_cast_fp16 = layer_norm(axes = out_45_axes_0, epsilon = var_4447_to_fp16, gamma = out_45_gamma_0_to_fp16, x = doubled_89_cast_fp16)[name = string("out_45_cast_fp16")]; tensor var_4458_split_sizes_0 = const()[name = string("op_4458_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4458_axis_0 = const()[name = string("op_4458_axis_0"), val = int32(1)]; tensor var_4458_cast_fp16_0, tensor var_4458_cast_fp16_1 = split(axis = var_4458_axis_0, split_sizes = var_4458_split_sizes_0, x = out_45_cast_fp16)[name = string("op_4458_cast_fp16")]; tensor query_states_67_strides_0 = const()[name = string("query_states_67_strides_0"), val = tensor([1, 1])]; string query_states_67_pad_type_0 = const()[name = string("query_states_67_pad_type_0"), val = string("valid")]; tensor query_states_67_pad_0 = const()[name = string("query_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_67_dilations_0 = const()[name = string("query_states_67_dilations_0"), val = tensor([1, 1])]; int32 query_states_67_groups_0 = const()[name = string("query_states_67_groups_0"), val = int32(1)]; tensor query_states_67_cast_fp16 = conv(dilations = query_states_67_dilations_0, groups = query_states_67_groups_0, pad = query_states_67_pad_0, pad_type = query_states_67_pad_type_0, strides = query_states_67_strides_0, weight = layers_11_self_attn_q_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("query_states_67_cast_fp16")]; tensor key_states_111_strides_0 = const()[name = string("key_states_111_strides_0"), val = tensor([1, 1])]; string key_states_111_pad_type_0 = const()[name = string("key_states_111_pad_type_0"), val = string("valid")]; tensor key_states_111_pad_0 = const()[name = string("key_states_111_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_111_dilations_0 = const()[name = string("key_states_111_dilations_0"), val = tensor([1, 1])]; int32 key_states_111_groups_0 = const()[name = string("key_states_111_groups_0"), val = int32(1)]; tensor key_states_111_cast_fp16 = conv(dilations = key_states_111_dilations_0, groups = key_states_111_groups_0, pad = key_states_111_pad_0, pad_type = key_states_111_pad_type_0, strides = key_states_111_strides_0, weight = layers_11_self_attn_k_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("key_states_111_cast_fp16")]; tensor value_states_67_strides_0 = const()[name = string("value_states_67_strides_0"), val = tensor([1, 1])]; string value_states_67_pad_type_0 = const()[name = string("value_states_67_pad_type_0"), val = string("valid")]; tensor value_states_67_pad_0 = const()[name = string("value_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_67_dilations_0 = const()[name = string("value_states_67_dilations_0"), val = tensor([1, 1])]; int32 value_states_67_groups_0 = const()[name = string("value_states_67_groups_0"), val = int32(1)]; tensor value_states_67_cast_fp16 = conv(dilations = value_states_67_dilations_0, groups = value_states_67_groups_0, pad = value_states_67_pad_0, pad_type = value_states_67_pad_type_0, strides = value_states_67_strides_0, weight = layers_11_self_attn_v_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("value_states_67_cast_fp16")]; tensor concat_132x = const()[name = string("concat_132x"), val = tensor([1, 16, 128, -1])]; tensor x_111_cast_fp16 = reshape(shape = concat_132x, x = query_states_67_cast_fp16)[name = string("x_111_cast_fp16")]; tensor concat_133x = const()[name = string("concat_133x"), val = tensor([1, 2, 128, -1])]; tensor var_4515_cast_fp16 = reshape(shape = concat_133x, x = key_states_111_cast_fp16)[name = string("op_4515_cast_fp16")]; tensor concat_134x = const()[name = string("concat_134x"), val = tensor([1, 2, 128, -1])]; tensor var_4522_cast_fp16 = reshape(shape = concat_134x, x = value_states_67_cast_fp16)[name = string("op_4522_cast_fp16")]; tensor var_4526_cast_fp16 = mul(x = x_111_cast_fp16, y = var_869_cast_fp16)[name = string("op_4526_cast_fp16")]; tensor var_4527_split_sizes_0 = const()[name = string("op_4527_split_sizes_0"), val = tensor([64, 64])]; int32 var_4527_axis_0 = const()[name = string("op_4527_axis_0"), val = int32(-2)]; tensor var_4527_cast_fp16_0, tensor var_4527_cast_fp16_1 = split(axis = var_4527_axis_0, split_sizes = var_4527_split_sizes_0, x = x_111_cast_fp16)[name = string("op_4527_cast_fp16")]; fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4529_cast_fp16 = mul(x = var_4527_cast_fp16_1, y = const_112_promoted_to_fp16)[name = string("op_4529_cast_fp16")]; int32 var_4531 = const()[name = string("op_4531"), val = int32(-2)]; bool var_4532_interleave_0 = const()[name = string("op_4532_interleave_0"), val = bool(false)]; tensor var_4532_cast_fp16 = concat(axis = var_4531, interleave = var_4532_interleave_0, values = (var_4529_cast_fp16, var_4527_cast_fp16_0))[name = string("op_4532_cast_fp16")]; tensor var_4533_cast_fp16 = mul(x = var_4532_cast_fp16, y = var_878_cast_fp16)[name = string("op_4533_cast_fp16")]; tensor query_states_69_cast_fp16 = add(x = var_4526_cast_fp16, y = var_4533_cast_fp16)[name = string("query_states_69_cast_fp16")]; tensor var_4539_cast_fp16 = mul(x = var_4515_cast_fp16, y = var_869_cast_fp16)[name = string("op_4539_cast_fp16")]; tensor var_4540_split_sizes_0 = const()[name = string("op_4540_split_sizes_0"), val = tensor([64, 64])]; int32 var_4540_axis_0 = const()[name = string("op_4540_axis_0"), val = int32(-2)]; tensor var_4540_cast_fp16_0, tensor var_4540_cast_fp16_1 = split(axis = var_4540_axis_0, split_sizes = var_4540_split_sizes_0, x = var_4515_cast_fp16)[name = string("op_4540_cast_fp16")]; fp16 const_113_promoted_to_fp16 = const()[name = string("const_113_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4542_cast_fp16 = mul(x = var_4540_cast_fp16_1, y = const_113_promoted_to_fp16)[name = string("op_4542_cast_fp16")]; int32 var_4544 = const()[name = string("op_4544"), val = int32(-2)]; bool var_4545_interleave_0 = const()[name = string("op_4545_interleave_0"), val = bool(false)]; tensor var_4545_cast_fp16 = concat(axis = var_4544, interleave = var_4545_interleave_0, values = (var_4542_cast_fp16, var_4540_cast_fp16_0))[name = string("op_4545_cast_fp16")]; tensor var_4546_cast_fp16 = mul(x = var_4545_cast_fp16, y = var_878_cast_fp16)[name = string("op_4546_cast_fp16")]; tensor key_states_115_cast_fp16 = add(x = var_4539_cast_fp16, y = var_4546_cast_fp16)[name = string("key_states_115_cast_fp16")]; tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; int32 concat_137_axis_0 = const()[name = string("concat_137_axis_0"), val = int32(0)]; bool concat_137_interleave_0 = const()[name = string("concat_137_interleave_0"), val = bool(false)]; tensor concat_137 = concat(axis = concat_137_axis_0, interleave = concat_137_interleave_0, values = (expand_dims_132, expand_dims_133, position_id, expand_dims_135))[name = string("concat_137")]; tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; tensor concat_138_values1_0 = const()[name = string("concat_138_values1_0"), val = tensor([0])]; tensor concat_138_values3_0 = const()[name = string("concat_138_values3_0"), val = tensor([0])]; int32 concat_138_axis_0 = const()[name = string("concat_138_axis_0"), val = int32(0)]; bool concat_138_interleave_0 = const()[name = string("concat_138_interleave_0"), val = bool(false)]; tensor concat_138 = concat(axis = concat_138_axis_0, interleave = concat_138_interleave_0, values = (expand_dims_136, concat_138_values1_0, cache_position_end, concat_138_values3_0))[name = string("concat_138")]; tensor key_states_117_perm_0 = const()[name = string("key_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_12_stride_0 = const()[name = string("key_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_117_cast_fp16 = transpose(perm = key_states_117_perm_0, x = key_states_115_cast_fp16)[name = string("transpose_222")]; tensor key_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = key_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = key_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_12_squeeze_mask_0, stride = key_cache_internal_tensor_assign_12_stride_0, update = key_states_117_cast_fp16, x = coreml_update_state_132)[name = string("key_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_12_cast_fp16, input = key_cache)[name = string("coreml_update_state_134_write_state")]; tensor coreml_update_state_134 = read_state(input = key_cache)[name = string("coreml_update_state_134")]; tensor value_states_69_perm_0 = const()[name = string("value_states_69_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_12_stride_0 = const()[name = string("value_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_69_cast_fp16 = transpose(perm = value_states_69_perm_0, x = var_4522_cast_fp16)[name = string("transpose_221")]; tensor value_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = value_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = value_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_12_squeeze_mask_0, stride = value_cache_internal_tensor_assign_12_stride_0, update = value_states_69_cast_fp16, x = coreml_update_state_133)[name = string("value_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_12_cast_fp16, input = value_cache)[name = string("coreml_update_state_135_write_state")]; tensor coreml_update_state_135 = read_state(input = value_cache)[name = string("coreml_update_state_135")]; tensor var_4616_begin_0 = const()[name = string("op_4616_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4616_end_0 = const()[name = string("op_4616_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4616_end_mask_0 = const()[name = string("op_4616_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4616_cast_fp16 = slice_by_index(begin = var_4616_begin_0, end = var_4616_end_0, end_mask = var_4616_end_mask_0, x = coreml_update_state_134)[name = string("op_4616_cast_fp16")]; tensor tile_22 = const()[name = string("tile_22"), val = tensor([1, 1])]; int32 var_4619_axis_0 = const()[name = string("op_4619_axis_0"), val = int32(1)]; tensor var_4619_cast_fp16_0, tensor var_4619_cast_fp16_1 = split(axis = var_4619_axis_0, split_sizes = tile_22, x = var_4616_cast_fp16)[name = string("op_4619_cast_fp16")]; tensor var_4626_begin_0 = const()[name = string("op_4626_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4626_end_0 = const()[name = string("op_4626_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4626_end_mask_0 = const()[name = string("op_4626_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4626_cast_fp16 = slice_by_index(begin = var_4626_begin_0, end = var_4626_end_0, end_mask = var_4626_end_mask_0, x = coreml_update_state_135)[name = string("op_4626_cast_fp16")]; tensor tile_23 = const()[name = string("tile_23"), val = tensor([1, 1])]; int32 var_4629_axis_0 = const()[name = string("op_4629_axis_0"), val = int32(1)]; tensor var_4629_cast_fp16_0, tensor var_4629_cast_fp16_1 = split(axis = var_4629_axis_0, split_sizes = tile_23, x = var_4626_cast_fp16)[name = string("op_4629_cast_fp16")]; tensor var_4632_split_sizes_0 = const()[name = string("op_4632_split_sizes_0"), val = tensor([8, 8])]; int32 var_4632_axis_0 = const()[name = string("op_4632_axis_0"), val = int32(1)]; tensor var_4632_0, tensor var_4632_1 = split(axis = var_4632_axis_0, split_sizes = var_4632_split_sizes_0, x = query_states_69_cast_fp16)[name = string("op_4632")]; bool attn_weights_177_transpose_x_0 = const()[name = string("attn_weights_177_transpose_x_0"), val = bool(false)]; bool attn_weights_177_transpose_y_0 = const()[name = string("attn_weights_177_transpose_y_0"), val = bool(false)]; tensor attn_weights_177_cast_fp16 = matmul(transpose_x = attn_weights_177_transpose_x_0, transpose_y = attn_weights_177_transpose_y_0, x = var_4619_cast_fp16_0, y = var_4632_0)[name = string("attn_weights_177_cast_fp16")]; fp16 var_4635_to_fp16 = const()[name = string("op_4635_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_179_cast_fp16 = mul(x = attn_weights_177_cast_fp16, y = var_4635_to_fp16)[name = string("attn_weights_179_cast_fp16")]; tensor attn_weights_181_cast_fp16 = add(x = attn_weights_179_cast_fp16, y = attn_mask_1)[name = string("attn_weights_181_cast_fp16")]; int32 var_4639 = const()[name = string("op_4639"), val = int32(-2)]; tensor attn_weights_183_cast_fp16 = softmax(axis = var_4639, x = attn_weights_181_cast_fp16)[name = string("attn_weights_183_cast_fp16")]; bool var_4645_transpose_x_1 = const()[name = string("op_4645_transpose_x_1"), val = bool(true)]; bool var_4645_transpose_y_1 = const()[name = string("op_4645_transpose_y_1"), val = bool(false)]; tensor var_4645_cast_fp16 = matmul(transpose_x = var_4645_transpose_x_1, transpose_y = var_4645_transpose_y_1, x = attn_weights_183_cast_fp16, y = var_4629_cast_fp16_0)[name = string("op_4645_cast_fp16")]; bool attn_weights_185_transpose_x_0 = const()[name = string("attn_weights_185_transpose_x_0"), val = bool(false)]; bool attn_weights_185_transpose_y_0 = const()[name = string("attn_weights_185_transpose_y_0"), val = bool(false)]; tensor attn_weights_185_cast_fp16 = matmul(transpose_x = attn_weights_185_transpose_x_0, transpose_y = attn_weights_185_transpose_y_0, x = var_4619_cast_fp16_1, y = var_4632_1)[name = string("attn_weights_185_cast_fp16")]; fp16 var_4647_to_fp16 = const()[name = string("op_4647_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_187_cast_fp16 = mul(x = attn_weights_185_cast_fp16, y = var_4647_to_fp16)[name = string("attn_weights_187_cast_fp16")]; tensor attn_weights_189_cast_fp16 = add(x = attn_weights_187_cast_fp16, y = attn_mask_1)[name = string("attn_weights_189_cast_fp16")]; int32 var_4651 = const()[name = string("op_4651"), val = int32(-2)]; tensor attn_weights_191_cast_fp16 = softmax(axis = var_4651, x = attn_weights_189_cast_fp16)[name = string("attn_weights_191_cast_fp16")]; bool attn_output_89_transpose_x_1 = const()[name = string("attn_output_89_transpose_x_1"), val = bool(true)]; bool attn_output_89_transpose_y_1 = const()[name = string("attn_output_89_transpose_y_1"), val = bool(false)]; tensor attn_output_89_cast_fp16 = matmul(transpose_x = attn_output_89_transpose_x_1, transpose_y = attn_output_89_transpose_y_1, x = attn_weights_191_cast_fp16, y = var_4629_cast_fp16_1)[name = string("attn_output_89_cast_fp16")]; int32 var_4659 = const()[name = string("op_4659"), val = int32(1)]; bool attn_output_91_interleave_0 = const()[name = string("attn_output_91_interleave_0"), val = bool(false)]; tensor attn_output_91_cast_fp16 = concat(axis = var_4659, interleave = attn_output_91_interleave_0, values = (var_4645_cast_fp16, attn_output_89_cast_fp16))[name = string("attn_output_91_cast_fp16")]; tensor var_4663_perm_0 = const()[name = string("op_4663_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_143x = const()[name = string("concat_143x"), val = tensor([1, 2048, 1, -1])]; tensor var_4663_cast_fp16 = transpose(perm = var_4663_perm_0, x = attn_output_91_cast_fp16)[name = string("transpose_220")]; tensor attn_output_95_cast_fp16 = reshape(shape = concat_143x, x = var_4663_cast_fp16)[name = string("attn_output_95_cast_fp16")]; tensor hidden_states_113_strides_0 = const()[name = string("hidden_states_113_strides_0"), val = tensor([1, 1])]; string hidden_states_113_pad_type_0 = const()[name = string("hidden_states_113_pad_type_0"), val = string("valid")]; tensor hidden_states_113_pad_0 = const()[name = string("hidden_states_113_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_113_dilations_0 = const()[name = string("hidden_states_113_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_113_groups_0 = const()[name = string("hidden_states_113_groups_0"), val = int32(1)]; tensor hidden_states_113_cast_fp16 = conv(dilations = hidden_states_113_dilations_0, groups = hidden_states_113_groups_0, pad = hidden_states_113_pad_0, pad_type = hidden_states_113_pad_type_0, strides = hidden_states_113_strides_0, weight = layers_11_self_attn_o_proj_weight_cast_fp16, x = attn_output_95_cast_fp16)[name = string("hidden_states_113_cast_fp16")]; tensor hidden_states_115_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = hidden_states_113_cast_fp16)[name = string("hidden_states_115_cast_fp16")]; fp16 const_118_promoted_to_fp16 = const()[name = string("const_118_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4696_cast_fp16 = mul(x = hidden_states_115_cast_fp16, y = const_118_promoted_to_fp16)[name = string("op_4696_cast_fp16")]; int32 var_4694 = const()[name = string("op_4694"), val = int32(1)]; bool doubled_93_interleave_0 = const()[name = string("doubled_93_interleave_0"), val = bool(false)]; tensor doubled_93_cast_fp16 = concat(axis = var_4694, interleave = doubled_93_interleave_0, values = (hidden_states_115_cast_fp16, var_4696_cast_fp16))[name = string("doubled_93_cast_fp16")]; tensor out_47_axes_0 = const()[name = string("out_47_axes_0"), val = tensor([1])]; tensor out_47_gamma_0_to_fp16 = const()[name = string("out_47_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415366400)))]; fp16 var_4706_to_fp16 = const()[name = string("op_4706_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_47_cast_fp16 = layer_norm(axes = out_47_axes_0, epsilon = var_4706_to_fp16, gamma = out_47_gamma_0_to_fp16, x = doubled_93_cast_fp16)[name = string("out_47_cast_fp16")]; tensor var_4717_split_sizes_0 = const()[name = string("op_4717_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4717_axis_0 = const()[name = string("op_4717_axis_0"), val = int32(1)]; tensor var_4717_cast_fp16_0, tensor var_4717_cast_fp16_1 = split(axis = var_4717_axis_0, split_sizes = var_4717_split_sizes_0, x = out_47_cast_fp16)[name = string("op_4717_cast_fp16")]; tensor input_23_strides_0 = const()[name = string("input_23_strides_0"), val = tensor([1, 1])]; string input_23_pad_type_0 = const()[name = string("input_23_pad_type_0"), val = string("valid")]; tensor input_23_pad_0 = const()[name = string("input_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_23_dilations_0 = const()[name = string("input_23_dilations_0"), val = tensor([1, 1])]; int32 input_23_groups_0 = const()[name = string("input_23_groups_0"), val = int32(1)]; tensor input_23_cast_fp16 = conv(dilations = input_23_dilations_0, groups = input_23_groups_0, pad = input_23_pad_0, pad_type = input_23_pad_type_0, strides = input_23_strides_0, weight = layers_11_mlp_gate_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("input_23_cast_fp16")]; tensor var_4734_cast_fp16 = silu(x = input_23_cast_fp16)[name = string("op_4734_cast_fp16")]; tensor var_4740_strides_0 = const()[name = string("op_4740_strides_0"), val = tensor([1, 1])]; string var_4740_pad_type_0 = const()[name = string("op_4740_pad_type_0"), val = string("valid")]; tensor var_4740_pad_0 = const()[name = string("op_4740_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4740_dilations_0 = const()[name = string("op_4740_dilations_0"), val = tensor([1, 1])]; int32 var_4740_groups_0 = const()[name = string("op_4740_groups_0"), val = int32(1)]; tensor var_4740_cast_fp16 = conv(dilations = var_4740_dilations_0, groups = var_4740_groups_0, pad = var_4740_pad_0, pad_type = var_4740_pad_type_0, strides = var_4740_strides_0, weight = layers_11_mlp_up_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("op_4740_cast_fp16")]; tensor x_119_cast_fp16 = mul(x = var_4734_cast_fp16, y = var_4740_cast_fp16)[name = string("x_119_cast_fp16")]; tensor hidden_states_117_strides_0 = const()[name = string("hidden_states_117_strides_0"), val = tensor([1, 1])]; string hidden_states_117_pad_type_0 = const()[name = string("hidden_states_117_pad_type_0"), val = string("valid")]; tensor hidden_states_117_pad_0 = const()[name = string("hidden_states_117_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_117_dilations_0 = const()[name = string("hidden_states_117_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_117_groups_0 = const()[name = string("hidden_states_117_groups_0"), val = int32(1)]; tensor hidden_states_117_cast_fp16 = conv(dilations = hidden_states_117_dilations_0, groups = hidden_states_117_groups_0, pad = hidden_states_117_pad_0, pad_type = hidden_states_117_pad_type_0, strides = hidden_states_117_strides_0, weight = layers_11_mlp_down_proj_weight_cast_fp16, x = x_119_cast_fp16)[name = string("hidden_states_117_cast_fp16")]; tensor hidden_states_119_cast_fp16 = add(x = hidden_states_115_cast_fp16, y = hidden_states_117_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; fp16 const_120_promoted_to_fp16 = const()[name = string("const_120_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4758_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_120_promoted_to_fp16)[name = string("op_4758_cast_fp16")]; int32 var_4756 = const()[name = string("op_4756"), val = int32(1)]; bool doubled_97_interleave_0 = const()[name = string("doubled_97_interleave_0"), val = bool(false)]; tensor doubled_97_cast_fp16 = concat(axis = var_4756, interleave = doubled_97_interleave_0, values = (hidden_states_119_cast_fp16, var_4758_cast_fp16))[name = string("doubled_97_cast_fp16")]; tensor out_49_axes_0 = const()[name = string("out_49_axes_0"), val = tensor([1])]; tensor out_49_gamma_0_to_fp16 = const()[name = string("out_49_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415374656)))]; fp16 var_4768_to_fp16 = const()[name = string("op_4768_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_49_cast_fp16 = layer_norm(axes = out_49_axes_0, epsilon = var_4768_to_fp16, gamma = out_49_gamma_0_to_fp16, x = doubled_97_cast_fp16)[name = string("out_49_cast_fp16")]; tensor var_4779_split_sizes_0 = const()[name = string("op_4779_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4779_axis_0 = const()[name = string("op_4779_axis_0"), val = int32(1)]; tensor var_4779_cast_fp16_0, tensor var_4779_cast_fp16_1 = split(axis = var_4779_axis_0, split_sizes = var_4779_split_sizes_0, x = out_49_cast_fp16)[name = string("op_4779_cast_fp16")]; tensor query_states_73_strides_0 = const()[name = string("query_states_73_strides_0"), val = tensor([1, 1])]; string query_states_73_pad_type_0 = const()[name = string("query_states_73_pad_type_0"), val = string("valid")]; tensor query_states_73_pad_0 = const()[name = string("query_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_73_dilations_0 = const()[name = string("query_states_73_dilations_0"), val = tensor([1, 1])]; int32 query_states_73_groups_0 = const()[name = string("query_states_73_groups_0"), val = int32(1)]; tensor query_states_73_cast_fp16 = conv(dilations = query_states_73_dilations_0, groups = query_states_73_groups_0, pad = query_states_73_pad_0, pad_type = query_states_73_pad_type_0, strides = query_states_73_strides_0, weight = layers_12_self_attn_q_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("query_states_73_cast_fp16")]; tensor key_states_121_strides_0 = const()[name = string("key_states_121_strides_0"), val = tensor([1, 1])]; string key_states_121_pad_type_0 = const()[name = string("key_states_121_pad_type_0"), val = string("valid")]; tensor key_states_121_pad_0 = const()[name = string("key_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_121_dilations_0 = const()[name = string("key_states_121_dilations_0"), val = tensor([1, 1])]; int32 key_states_121_groups_0 = const()[name = string("key_states_121_groups_0"), val = int32(1)]; tensor key_states_121_cast_fp16 = conv(dilations = key_states_121_dilations_0, groups = key_states_121_groups_0, pad = key_states_121_pad_0, pad_type = key_states_121_pad_type_0, strides = key_states_121_strides_0, weight = layers_12_self_attn_k_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("key_states_121_cast_fp16")]; tensor value_states_73_strides_0 = const()[name = string("value_states_73_strides_0"), val = tensor([1, 1])]; string value_states_73_pad_type_0 = const()[name = string("value_states_73_pad_type_0"), val = string("valid")]; tensor value_states_73_pad_0 = const()[name = string("value_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_73_dilations_0 = const()[name = string("value_states_73_dilations_0"), val = tensor([1, 1])]; int32 value_states_73_groups_0 = const()[name = string("value_states_73_groups_0"), val = int32(1)]; tensor value_states_73_cast_fp16 = conv(dilations = value_states_73_dilations_0, groups = value_states_73_groups_0, pad = value_states_73_pad_0, pad_type = value_states_73_pad_type_0, strides = value_states_73_strides_0, weight = layers_12_self_attn_v_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("value_states_73_cast_fp16")]; tensor concat_144x = const()[name = string("concat_144x"), val = tensor([1, 16, 128, -1])]; tensor x_121_cast_fp16 = reshape(shape = concat_144x, x = query_states_73_cast_fp16)[name = string("x_121_cast_fp16")]; tensor concat_145x = const()[name = string("concat_145x"), val = tensor([1, 2, 128, -1])]; tensor var_4836_cast_fp16 = reshape(shape = concat_145x, x = key_states_121_cast_fp16)[name = string("op_4836_cast_fp16")]; tensor concat_146x = const()[name = string("concat_146x"), val = tensor([1, 2, 128, -1])]; tensor var_4843_cast_fp16 = reshape(shape = concat_146x, x = value_states_73_cast_fp16)[name = string("op_4843_cast_fp16")]; tensor var_4847_cast_fp16 = mul(x = x_121_cast_fp16, y = var_869_cast_fp16)[name = string("op_4847_cast_fp16")]; tensor var_4848_split_sizes_0 = const()[name = string("op_4848_split_sizes_0"), val = tensor([64, 64])]; int32 var_4848_axis_0 = const()[name = string("op_4848_axis_0"), val = int32(-2)]; tensor var_4848_cast_fp16_0, tensor var_4848_cast_fp16_1 = split(axis = var_4848_axis_0, split_sizes = var_4848_split_sizes_0, x = x_121_cast_fp16)[name = string("op_4848_cast_fp16")]; fp16 const_122_promoted_to_fp16 = const()[name = string("const_122_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4850_cast_fp16 = mul(x = var_4848_cast_fp16_1, y = const_122_promoted_to_fp16)[name = string("op_4850_cast_fp16")]; int32 var_4852 = const()[name = string("op_4852"), val = int32(-2)]; bool var_4853_interleave_0 = const()[name = string("op_4853_interleave_0"), val = bool(false)]; tensor var_4853_cast_fp16 = concat(axis = var_4852, interleave = var_4853_interleave_0, values = (var_4850_cast_fp16, var_4848_cast_fp16_0))[name = string("op_4853_cast_fp16")]; tensor var_4854_cast_fp16 = mul(x = var_4853_cast_fp16, y = var_878_cast_fp16)[name = string("op_4854_cast_fp16")]; tensor query_states_75_cast_fp16 = add(x = var_4847_cast_fp16, y = var_4854_cast_fp16)[name = string("query_states_75_cast_fp16")]; tensor var_4860_cast_fp16 = mul(x = var_4836_cast_fp16, y = var_869_cast_fp16)[name = string("op_4860_cast_fp16")]; tensor var_4861_split_sizes_0 = const()[name = string("op_4861_split_sizes_0"), val = tensor([64, 64])]; int32 var_4861_axis_0 = const()[name = string("op_4861_axis_0"), val = int32(-2)]; tensor var_4861_cast_fp16_0, tensor var_4861_cast_fp16_1 = split(axis = var_4861_axis_0, split_sizes = var_4861_split_sizes_0, x = var_4836_cast_fp16)[name = string("op_4861_cast_fp16")]; fp16 const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4863_cast_fp16 = mul(x = var_4861_cast_fp16_1, y = const_123_promoted_to_fp16)[name = string("op_4863_cast_fp16")]; int32 var_4865 = const()[name = string("op_4865"), val = int32(-2)]; bool var_4866_interleave_0 = const()[name = string("op_4866_interleave_0"), val = bool(false)]; tensor var_4866_cast_fp16 = concat(axis = var_4865, interleave = var_4866_interleave_0, values = (var_4863_cast_fp16, var_4861_cast_fp16_0))[name = string("op_4866_cast_fp16")]; tensor var_4867_cast_fp16 = mul(x = var_4866_cast_fp16, y = var_878_cast_fp16)[name = string("op_4867_cast_fp16")]; tensor key_states_125_cast_fp16 = add(x = var_4860_cast_fp16, y = var_4867_cast_fp16)[name = string("key_states_125_cast_fp16")]; tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; int32 concat_149_axis_0 = const()[name = string("concat_149_axis_0"), val = int32(0)]; bool concat_149_interleave_0 = const()[name = string("concat_149_interleave_0"), val = bool(false)]; tensor concat_149 = concat(axis = concat_149_axis_0, interleave = concat_149_interleave_0, values = (expand_dims_144, expand_dims_145, position_id, expand_dims_147))[name = string("concat_149")]; tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; tensor concat_150_values1_0 = const()[name = string("concat_150_values1_0"), val = tensor([0])]; tensor concat_150_values3_0 = const()[name = string("concat_150_values3_0"), val = tensor([0])]; int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_148, concat_150_values1_0, cache_position_end, concat_150_values3_0))[name = string("concat_150")]; tensor key_states_127_perm_0 = const()[name = string("key_states_127_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_13_stride_0 = const()[name = string("key_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_127_cast_fp16 = transpose(perm = key_states_127_perm_0, x = key_states_125_cast_fp16)[name = string("transpose_219")]; tensor key_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = key_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = key_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_13_squeeze_mask_0, stride = key_cache_internal_tensor_assign_13_stride_0, update = key_states_127_cast_fp16, x = coreml_update_state_134)[name = string("key_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_13_cast_fp16, input = key_cache)[name = string("coreml_update_state_136_write_state")]; tensor coreml_update_state_136 = read_state(input = key_cache)[name = string("coreml_update_state_136")]; tensor value_states_75_perm_0 = const()[name = string("value_states_75_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_13_stride_0 = const()[name = string("value_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_75_cast_fp16 = transpose(perm = value_states_75_perm_0, x = var_4843_cast_fp16)[name = string("transpose_218")]; tensor value_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = value_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = value_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_13_squeeze_mask_0, stride = value_cache_internal_tensor_assign_13_stride_0, update = value_states_75_cast_fp16, x = coreml_update_state_135)[name = string("value_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_13_cast_fp16, input = value_cache)[name = string("coreml_update_state_137_write_state")]; tensor coreml_update_state_137 = read_state(input = value_cache)[name = string("coreml_update_state_137")]; tensor var_4937_begin_0 = const()[name = string("op_4937_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4937_end_0 = const()[name = string("op_4937_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4937_end_mask_0 = const()[name = string("op_4937_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4937_cast_fp16 = slice_by_index(begin = var_4937_begin_0, end = var_4937_end_0, end_mask = var_4937_end_mask_0, x = coreml_update_state_136)[name = string("op_4937_cast_fp16")]; tensor tile_24 = const()[name = string("tile_24"), val = tensor([1, 1])]; int32 var_4940_axis_0 = const()[name = string("op_4940_axis_0"), val = int32(1)]; tensor var_4940_cast_fp16_0, tensor var_4940_cast_fp16_1 = split(axis = var_4940_axis_0, split_sizes = tile_24, x = var_4937_cast_fp16)[name = string("op_4940_cast_fp16")]; tensor var_4947_begin_0 = const()[name = string("op_4947_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4947_end_0 = const()[name = string("op_4947_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4947_end_mask_0 = const()[name = string("op_4947_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4947_cast_fp16 = slice_by_index(begin = var_4947_begin_0, end = var_4947_end_0, end_mask = var_4947_end_mask_0, x = coreml_update_state_137)[name = string("op_4947_cast_fp16")]; tensor tile_25 = const()[name = string("tile_25"), val = tensor([1, 1])]; int32 var_4950_axis_0 = const()[name = string("op_4950_axis_0"), val = int32(1)]; tensor var_4950_cast_fp16_0, tensor var_4950_cast_fp16_1 = split(axis = var_4950_axis_0, split_sizes = tile_25, x = var_4947_cast_fp16)[name = string("op_4950_cast_fp16")]; tensor var_4953_split_sizes_0 = const()[name = string("op_4953_split_sizes_0"), val = tensor([8, 8])]; int32 var_4953_axis_0 = const()[name = string("op_4953_axis_0"), val = int32(1)]; tensor var_4953_0, tensor var_4953_1 = split(axis = var_4953_axis_0, split_sizes = var_4953_split_sizes_0, x = query_states_75_cast_fp16)[name = string("op_4953")]; bool attn_weights_193_transpose_x_0 = const()[name = string("attn_weights_193_transpose_x_0"), val = bool(false)]; bool attn_weights_193_transpose_y_0 = const()[name = string("attn_weights_193_transpose_y_0"), val = bool(false)]; tensor attn_weights_193_cast_fp16 = matmul(transpose_x = attn_weights_193_transpose_x_0, transpose_y = attn_weights_193_transpose_y_0, x = var_4940_cast_fp16_0, y = var_4953_0)[name = string("attn_weights_193_cast_fp16")]; fp16 var_4956_to_fp16 = const()[name = string("op_4956_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_195_cast_fp16 = mul(x = attn_weights_193_cast_fp16, y = var_4956_to_fp16)[name = string("attn_weights_195_cast_fp16")]; tensor attn_weights_197_cast_fp16 = add(x = attn_weights_195_cast_fp16, y = attn_mask_1)[name = string("attn_weights_197_cast_fp16")]; int32 var_4960 = const()[name = string("op_4960"), val = int32(-2)]; tensor attn_weights_199_cast_fp16 = softmax(axis = var_4960, x = attn_weights_197_cast_fp16)[name = string("attn_weights_199_cast_fp16")]; bool var_4966_transpose_x_1 = const()[name = string("op_4966_transpose_x_1"), val = bool(true)]; bool var_4966_transpose_y_1 = const()[name = string("op_4966_transpose_y_1"), val = bool(false)]; tensor var_4966_cast_fp16 = matmul(transpose_x = var_4966_transpose_x_1, transpose_y = var_4966_transpose_y_1, x = attn_weights_199_cast_fp16, y = var_4950_cast_fp16_0)[name = string("op_4966_cast_fp16")]; bool attn_weights_201_transpose_x_0 = const()[name = string("attn_weights_201_transpose_x_0"), val = bool(false)]; bool attn_weights_201_transpose_y_0 = const()[name = string("attn_weights_201_transpose_y_0"), val = bool(false)]; tensor attn_weights_201_cast_fp16 = matmul(transpose_x = attn_weights_201_transpose_x_0, transpose_y = attn_weights_201_transpose_y_0, x = var_4940_cast_fp16_1, y = var_4953_1)[name = string("attn_weights_201_cast_fp16")]; fp16 var_4968_to_fp16 = const()[name = string("op_4968_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_203_cast_fp16 = mul(x = attn_weights_201_cast_fp16, y = var_4968_to_fp16)[name = string("attn_weights_203_cast_fp16")]; tensor attn_weights_205_cast_fp16 = add(x = attn_weights_203_cast_fp16, y = attn_mask_1)[name = string("attn_weights_205_cast_fp16")]; int32 var_4972 = const()[name = string("op_4972"), val = int32(-2)]; tensor attn_weights_207_cast_fp16 = softmax(axis = var_4972, x = attn_weights_205_cast_fp16)[name = string("attn_weights_207_cast_fp16")]; bool attn_output_97_transpose_x_1 = const()[name = string("attn_output_97_transpose_x_1"), val = bool(true)]; bool attn_output_97_transpose_y_1 = const()[name = string("attn_output_97_transpose_y_1"), val = bool(false)]; tensor attn_output_97_cast_fp16 = matmul(transpose_x = attn_output_97_transpose_x_1, transpose_y = attn_output_97_transpose_y_1, x = attn_weights_207_cast_fp16, y = var_4950_cast_fp16_1)[name = string("attn_output_97_cast_fp16")]; int32 var_4980 = const()[name = string("op_4980"), val = int32(1)]; bool attn_output_99_interleave_0 = const()[name = string("attn_output_99_interleave_0"), val = bool(false)]; tensor attn_output_99_cast_fp16 = concat(axis = var_4980, interleave = attn_output_99_interleave_0, values = (var_4966_cast_fp16, attn_output_97_cast_fp16))[name = string("attn_output_99_cast_fp16")]; tensor var_4984_perm_0 = const()[name = string("op_4984_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_155x = const()[name = string("concat_155x"), val = tensor([1, 2048, 1, -1])]; tensor var_4984_cast_fp16 = transpose(perm = var_4984_perm_0, x = attn_output_99_cast_fp16)[name = string("transpose_217")]; tensor attn_output_103_cast_fp16 = reshape(shape = concat_155x, x = var_4984_cast_fp16)[name = string("attn_output_103_cast_fp16")]; tensor hidden_states_123_strides_0 = const()[name = string("hidden_states_123_strides_0"), val = tensor([1, 1])]; string hidden_states_123_pad_type_0 = const()[name = string("hidden_states_123_pad_type_0"), val = string("valid")]; tensor hidden_states_123_pad_0 = const()[name = string("hidden_states_123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_123_dilations_0 = const()[name = string("hidden_states_123_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_123_groups_0 = const()[name = string("hidden_states_123_groups_0"), val = int32(1)]; tensor hidden_states_123_cast_fp16 = conv(dilations = hidden_states_123_dilations_0, groups = hidden_states_123_groups_0, pad = hidden_states_123_pad_0, pad_type = hidden_states_123_pad_type_0, strides = hidden_states_123_strides_0, weight = layers_12_self_attn_o_proj_weight_cast_fp16, x = attn_output_103_cast_fp16)[name = string("hidden_states_123_cast_fp16")]; tensor hidden_states_125_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = hidden_states_123_cast_fp16)[name = string("hidden_states_125_cast_fp16")]; fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5017_cast_fp16 = mul(x = hidden_states_125_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_5017_cast_fp16")]; int32 var_5015 = const()[name = string("op_5015"), val = int32(1)]; bool doubled_101_interleave_0 = const()[name = string("doubled_101_interleave_0"), val = bool(false)]; tensor doubled_101_cast_fp16 = concat(axis = var_5015, interleave = doubled_101_interleave_0, values = (hidden_states_125_cast_fp16, var_5017_cast_fp16))[name = string("doubled_101_cast_fp16")]; tensor out_51_axes_0 = const()[name = string("out_51_axes_0"), val = tensor([1])]; tensor out_51_gamma_0_to_fp16 = const()[name = string("out_51_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415382912)))]; fp16 var_5027_to_fp16 = const()[name = string("op_5027_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_51_cast_fp16 = layer_norm(axes = out_51_axes_0, epsilon = var_5027_to_fp16, gamma = out_51_gamma_0_to_fp16, x = doubled_101_cast_fp16)[name = string("out_51_cast_fp16")]; tensor var_5038_split_sizes_0 = const()[name = string("op_5038_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5038_axis_0 = const()[name = string("op_5038_axis_0"), val = int32(1)]; tensor var_5038_cast_fp16_0, tensor var_5038_cast_fp16_1 = split(axis = var_5038_axis_0, split_sizes = var_5038_split_sizes_0, x = out_51_cast_fp16)[name = string("op_5038_cast_fp16")]; tensor input_25_strides_0 = const()[name = string("input_25_strides_0"), val = tensor([1, 1])]; string input_25_pad_type_0 = const()[name = string("input_25_pad_type_0"), val = string("valid")]; tensor input_25_pad_0 = const()[name = string("input_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_25_dilations_0 = const()[name = string("input_25_dilations_0"), val = tensor([1, 1])]; int32 input_25_groups_0 = const()[name = string("input_25_groups_0"), val = int32(1)]; tensor input_25_cast_fp16 = conv(dilations = input_25_dilations_0, groups = input_25_groups_0, pad = input_25_pad_0, pad_type = input_25_pad_type_0, strides = input_25_strides_0, weight = layers_12_mlp_gate_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("input_25_cast_fp16")]; tensor var_5055_cast_fp16 = silu(x = input_25_cast_fp16)[name = string("op_5055_cast_fp16")]; tensor var_5061_strides_0 = const()[name = string("op_5061_strides_0"), val = tensor([1, 1])]; string var_5061_pad_type_0 = const()[name = string("op_5061_pad_type_0"), val = string("valid")]; tensor var_5061_pad_0 = const()[name = string("op_5061_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5061_dilations_0 = const()[name = string("op_5061_dilations_0"), val = tensor([1, 1])]; int32 var_5061_groups_0 = const()[name = string("op_5061_groups_0"), val = int32(1)]; tensor var_5061_cast_fp16 = conv(dilations = var_5061_dilations_0, groups = var_5061_groups_0, pad = var_5061_pad_0, pad_type = var_5061_pad_type_0, strides = var_5061_strides_0, weight = layers_12_mlp_up_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("op_5061_cast_fp16")]; tensor x_129_cast_fp16 = mul(x = var_5055_cast_fp16, y = var_5061_cast_fp16)[name = string("x_129_cast_fp16")]; tensor hidden_states_127_strides_0 = const()[name = string("hidden_states_127_strides_0"), val = tensor([1, 1])]; string hidden_states_127_pad_type_0 = const()[name = string("hidden_states_127_pad_type_0"), val = string("valid")]; tensor hidden_states_127_pad_0 = const()[name = string("hidden_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_127_dilations_0 = const()[name = string("hidden_states_127_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_127_groups_0 = const()[name = string("hidden_states_127_groups_0"), val = int32(1)]; tensor hidden_states_127_cast_fp16 = conv(dilations = hidden_states_127_dilations_0, groups = hidden_states_127_groups_0, pad = hidden_states_127_pad_0, pad_type = hidden_states_127_pad_type_0, strides = hidden_states_127_strides_0, weight = layers_12_mlp_down_proj_weight_cast_fp16, x = x_129_cast_fp16)[name = string("hidden_states_127_cast_fp16")]; tensor hidden_states_129_cast_fp16 = add(x = hidden_states_125_cast_fp16, y = hidden_states_127_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5079_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_130_promoted_to_fp16)[name = string("op_5079_cast_fp16")]; int32 var_5077 = const()[name = string("op_5077"), val = int32(1)]; bool doubled_105_interleave_0 = const()[name = string("doubled_105_interleave_0"), val = bool(false)]; tensor doubled_105_cast_fp16 = concat(axis = var_5077, interleave = doubled_105_interleave_0, values = (hidden_states_129_cast_fp16, var_5079_cast_fp16))[name = string("doubled_105_cast_fp16")]; tensor out_53_axes_0 = const()[name = string("out_53_axes_0"), val = tensor([1])]; tensor out_53_gamma_0_to_fp16 = const()[name = string("out_53_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415391168)))]; fp16 var_5089_to_fp16 = const()[name = string("op_5089_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_53_cast_fp16 = layer_norm(axes = out_53_axes_0, epsilon = var_5089_to_fp16, gamma = out_53_gamma_0_to_fp16, x = doubled_105_cast_fp16)[name = string("out_53_cast_fp16")]; tensor var_5100_split_sizes_0 = const()[name = string("op_5100_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5100_axis_0 = const()[name = string("op_5100_axis_0"), val = int32(1)]; tensor var_5100_cast_fp16_0, tensor var_5100_cast_fp16_1 = split(axis = var_5100_axis_0, split_sizes = var_5100_split_sizes_0, x = out_53_cast_fp16)[name = string("op_5100_cast_fp16")]; tensor query_states_79_strides_0 = const()[name = string("query_states_79_strides_0"), val = tensor([1, 1])]; string query_states_79_pad_type_0 = const()[name = string("query_states_79_pad_type_0"), val = string("valid")]; tensor query_states_79_pad_0 = const()[name = string("query_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_79_dilations_0 = const()[name = string("query_states_79_dilations_0"), val = tensor([1, 1])]; int32 query_states_79_groups_0 = const()[name = string("query_states_79_groups_0"), val = int32(1)]; tensor query_states_79_cast_fp16 = conv(dilations = query_states_79_dilations_0, groups = query_states_79_groups_0, pad = query_states_79_pad_0, pad_type = query_states_79_pad_type_0, strides = query_states_79_strides_0, weight = layers_13_self_attn_q_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("query_states_79_cast_fp16")]; tensor key_states_131_strides_0 = const()[name = string("key_states_131_strides_0"), val = tensor([1, 1])]; string key_states_131_pad_type_0 = const()[name = string("key_states_131_pad_type_0"), val = string("valid")]; tensor key_states_131_pad_0 = const()[name = string("key_states_131_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_131_dilations_0 = const()[name = string("key_states_131_dilations_0"), val = tensor([1, 1])]; int32 key_states_131_groups_0 = const()[name = string("key_states_131_groups_0"), val = int32(1)]; tensor key_states_131_cast_fp16 = conv(dilations = key_states_131_dilations_0, groups = key_states_131_groups_0, pad = key_states_131_pad_0, pad_type = key_states_131_pad_type_0, strides = key_states_131_strides_0, weight = layers_13_self_attn_k_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("key_states_131_cast_fp16")]; tensor value_states_79_strides_0 = const()[name = string("value_states_79_strides_0"), val = tensor([1, 1])]; string value_states_79_pad_type_0 = const()[name = string("value_states_79_pad_type_0"), val = string("valid")]; tensor value_states_79_pad_0 = const()[name = string("value_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_79_dilations_0 = const()[name = string("value_states_79_dilations_0"), val = tensor([1, 1])]; int32 value_states_79_groups_0 = const()[name = string("value_states_79_groups_0"), val = int32(1)]; tensor value_states_79_cast_fp16 = conv(dilations = value_states_79_dilations_0, groups = value_states_79_groups_0, pad = value_states_79_pad_0, pad_type = value_states_79_pad_type_0, strides = value_states_79_strides_0, weight = layers_13_self_attn_v_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("value_states_79_cast_fp16")]; tensor concat_156x = const()[name = string("concat_156x"), val = tensor([1, 16, 128, -1])]; tensor x_131_cast_fp16 = reshape(shape = concat_156x, x = query_states_79_cast_fp16)[name = string("x_131_cast_fp16")]; tensor concat_157x = const()[name = string("concat_157x"), val = tensor([1, 2, 128, -1])]; tensor var_5157_cast_fp16 = reshape(shape = concat_157x, x = key_states_131_cast_fp16)[name = string("op_5157_cast_fp16")]; tensor concat_158x = const()[name = string("concat_158x"), val = tensor([1, 2, 128, -1])]; tensor var_5164_cast_fp16 = reshape(shape = concat_158x, x = value_states_79_cast_fp16)[name = string("op_5164_cast_fp16")]; tensor var_5168_cast_fp16 = mul(x = x_131_cast_fp16, y = var_869_cast_fp16)[name = string("op_5168_cast_fp16")]; tensor var_5169_split_sizes_0 = const()[name = string("op_5169_split_sizes_0"), val = tensor([64, 64])]; int32 var_5169_axis_0 = const()[name = string("op_5169_axis_0"), val = int32(-2)]; tensor var_5169_cast_fp16_0, tensor var_5169_cast_fp16_1 = split(axis = var_5169_axis_0, split_sizes = var_5169_split_sizes_0, x = x_131_cast_fp16)[name = string("op_5169_cast_fp16")]; fp16 const_132_promoted_to_fp16 = const()[name = string("const_132_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5171_cast_fp16 = mul(x = var_5169_cast_fp16_1, y = const_132_promoted_to_fp16)[name = string("op_5171_cast_fp16")]; int32 var_5173 = const()[name = string("op_5173"), val = int32(-2)]; bool var_5174_interleave_0 = const()[name = string("op_5174_interleave_0"), val = bool(false)]; tensor var_5174_cast_fp16 = concat(axis = var_5173, interleave = var_5174_interleave_0, values = (var_5171_cast_fp16, var_5169_cast_fp16_0))[name = string("op_5174_cast_fp16")]; tensor var_5175_cast_fp16 = mul(x = var_5174_cast_fp16, y = var_878_cast_fp16)[name = string("op_5175_cast_fp16")]; tensor query_states_81_cast_fp16 = add(x = var_5168_cast_fp16, y = var_5175_cast_fp16)[name = string("query_states_81_cast_fp16")]; tensor var_5181_cast_fp16 = mul(x = var_5157_cast_fp16, y = var_869_cast_fp16)[name = string("op_5181_cast_fp16")]; tensor var_5182_split_sizes_0 = const()[name = string("op_5182_split_sizes_0"), val = tensor([64, 64])]; int32 var_5182_axis_0 = const()[name = string("op_5182_axis_0"), val = int32(-2)]; tensor var_5182_cast_fp16_0, tensor var_5182_cast_fp16_1 = split(axis = var_5182_axis_0, split_sizes = var_5182_split_sizes_0, x = var_5157_cast_fp16)[name = string("op_5182_cast_fp16")]; fp16 const_133_promoted_to_fp16 = const()[name = string("const_133_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5184_cast_fp16 = mul(x = var_5182_cast_fp16_1, y = const_133_promoted_to_fp16)[name = string("op_5184_cast_fp16")]; int32 var_5186 = const()[name = string("op_5186"), val = int32(-2)]; bool var_5187_interleave_0 = const()[name = string("op_5187_interleave_0"), val = bool(false)]; tensor var_5187_cast_fp16 = concat(axis = var_5186, interleave = var_5187_interleave_0, values = (var_5184_cast_fp16, var_5182_cast_fp16_0))[name = string("op_5187_cast_fp16")]; tensor var_5188_cast_fp16 = mul(x = var_5187_cast_fp16, y = var_878_cast_fp16)[name = string("op_5188_cast_fp16")]; tensor key_states_135_cast_fp16 = add(x = var_5181_cast_fp16, y = var_5188_cast_fp16)[name = string("key_states_135_cast_fp16")]; tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; int32 concat_161_axis_0 = const()[name = string("concat_161_axis_0"), val = int32(0)]; bool concat_161_interleave_0 = const()[name = string("concat_161_interleave_0"), val = bool(false)]; tensor concat_161 = concat(axis = concat_161_axis_0, interleave = concat_161_interleave_0, values = (expand_dims_156, expand_dims_157, position_id, expand_dims_159))[name = string("concat_161")]; tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; tensor concat_162_values1_0 = const()[name = string("concat_162_values1_0"), val = tensor([0])]; tensor concat_162_values3_0 = const()[name = string("concat_162_values3_0"), val = tensor([0])]; int32 concat_162_axis_0 = const()[name = string("concat_162_axis_0"), val = int32(0)]; bool concat_162_interleave_0 = const()[name = string("concat_162_interleave_0"), val = bool(false)]; tensor concat_162 = concat(axis = concat_162_axis_0, interleave = concat_162_interleave_0, values = (expand_dims_160, concat_162_values1_0, cache_position_end, concat_162_values3_0))[name = string("concat_162")]; tensor key_states_137_perm_0 = const()[name = string("key_states_137_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_14_stride_0 = const()[name = string("key_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_137_cast_fp16 = transpose(perm = key_states_137_perm_0, x = key_states_135_cast_fp16)[name = string("transpose_216")]; tensor key_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = key_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = key_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_14_squeeze_mask_0, stride = key_cache_internal_tensor_assign_14_stride_0, update = key_states_137_cast_fp16, x = coreml_update_state_136)[name = string("key_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_14_cast_fp16, input = key_cache)[name = string("coreml_update_state_138_write_state")]; tensor coreml_update_state_138 = read_state(input = key_cache)[name = string("coreml_update_state_138")]; tensor value_states_81_perm_0 = const()[name = string("value_states_81_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_14_stride_0 = const()[name = string("value_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_81_cast_fp16 = transpose(perm = value_states_81_perm_0, x = var_5164_cast_fp16)[name = string("transpose_215")]; tensor value_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = value_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = value_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_14_squeeze_mask_0, stride = value_cache_internal_tensor_assign_14_stride_0, update = value_states_81_cast_fp16, x = coreml_update_state_137)[name = string("value_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_14_cast_fp16, input = value_cache)[name = string("coreml_update_state_139_write_state")]; tensor coreml_update_state_139 = read_state(input = value_cache)[name = string("coreml_update_state_139")]; tensor var_5258_begin_0 = const()[name = string("op_5258_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5258_end_0 = const()[name = string("op_5258_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5258_end_mask_0 = const()[name = string("op_5258_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5258_cast_fp16 = slice_by_index(begin = var_5258_begin_0, end = var_5258_end_0, end_mask = var_5258_end_mask_0, x = coreml_update_state_138)[name = string("op_5258_cast_fp16")]; tensor tile_26 = const()[name = string("tile_26"), val = tensor([1, 1])]; int32 var_5261_axis_0 = const()[name = string("op_5261_axis_0"), val = int32(1)]; tensor var_5261_cast_fp16_0, tensor var_5261_cast_fp16_1 = split(axis = var_5261_axis_0, split_sizes = tile_26, x = var_5258_cast_fp16)[name = string("op_5261_cast_fp16")]; tensor var_5268_begin_0 = const()[name = string("op_5268_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5268_end_0 = const()[name = string("op_5268_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5268_end_mask_0 = const()[name = string("op_5268_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5268_cast_fp16 = slice_by_index(begin = var_5268_begin_0, end = var_5268_end_0, end_mask = var_5268_end_mask_0, x = coreml_update_state_139)[name = string("op_5268_cast_fp16")]; tensor tile_27 = const()[name = string("tile_27"), val = tensor([1, 1])]; int32 var_5271_axis_0 = const()[name = string("op_5271_axis_0"), val = int32(1)]; tensor var_5271_cast_fp16_0, tensor var_5271_cast_fp16_1 = split(axis = var_5271_axis_0, split_sizes = tile_27, x = var_5268_cast_fp16)[name = string("op_5271_cast_fp16")]; tensor var_5274_split_sizes_0 = const()[name = string("op_5274_split_sizes_0"), val = tensor([8, 8])]; int32 var_5274_axis_0 = const()[name = string("op_5274_axis_0"), val = int32(1)]; tensor var_5274_0, tensor var_5274_1 = split(axis = var_5274_axis_0, split_sizes = var_5274_split_sizes_0, x = query_states_81_cast_fp16)[name = string("op_5274")]; bool attn_weights_209_transpose_x_0 = const()[name = string("attn_weights_209_transpose_x_0"), val = bool(false)]; bool attn_weights_209_transpose_y_0 = const()[name = string("attn_weights_209_transpose_y_0"), val = bool(false)]; tensor attn_weights_209_cast_fp16 = matmul(transpose_x = attn_weights_209_transpose_x_0, transpose_y = attn_weights_209_transpose_y_0, x = var_5261_cast_fp16_0, y = var_5274_0)[name = string("attn_weights_209_cast_fp16")]; fp16 var_5277_to_fp16 = const()[name = string("op_5277_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_211_cast_fp16 = mul(x = attn_weights_209_cast_fp16, y = var_5277_to_fp16)[name = string("attn_weights_211_cast_fp16")]; tensor attn_weights_213_cast_fp16 = add(x = attn_weights_211_cast_fp16, y = attn_mask_1)[name = string("attn_weights_213_cast_fp16")]; int32 var_5281 = const()[name = string("op_5281"), val = int32(-2)]; tensor attn_weights_215_cast_fp16 = softmax(axis = var_5281, x = attn_weights_213_cast_fp16)[name = string("attn_weights_215_cast_fp16")]; bool var_5287_transpose_x_1 = const()[name = string("op_5287_transpose_x_1"), val = bool(true)]; bool var_5287_transpose_y_1 = const()[name = string("op_5287_transpose_y_1"), val = bool(false)]; tensor var_5287_cast_fp16 = matmul(transpose_x = var_5287_transpose_x_1, transpose_y = var_5287_transpose_y_1, x = attn_weights_215_cast_fp16, y = var_5271_cast_fp16_0)[name = string("op_5287_cast_fp16")]; bool attn_weights_217_transpose_x_0 = const()[name = string("attn_weights_217_transpose_x_0"), val = bool(false)]; bool attn_weights_217_transpose_y_0 = const()[name = string("attn_weights_217_transpose_y_0"), val = bool(false)]; tensor attn_weights_217_cast_fp16 = matmul(transpose_x = attn_weights_217_transpose_x_0, transpose_y = attn_weights_217_transpose_y_0, x = var_5261_cast_fp16_1, y = var_5274_1)[name = string("attn_weights_217_cast_fp16")]; fp16 var_5289_to_fp16 = const()[name = string("op_5289_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_219_cast_fp16 = mul(x = attn_weights_217_cast_fp16, y = var_5289_to_fp16)[name = string("attn_weights_219_cast_fp16")]; tensor attn_weights_221_cast_fp16 = add(x = attn_weights_219_cast_fp16, y = attn_mask_1)[name = string("attn_weights_221_cast_fp16")]; int32 var_5293 = const()[name = string("op_5293"), val = int32(-2)]; tensor attn_weights_223_cast_fp16 = softmax(axis = var_5293, x = attn_weights_221_cast_fp16)[name = string("attn_weights_223_cast_fp16")]; bool attn_output_105_transpose_x_1 = const()[name = string("attn_output_105_transpose_x_1"), val = bool(true)]; bool attn_output_105_transpose_y_1 = const()[name = string("attn_output_105_transpose_y_1"), val = bool(false)]; tensor attn_output_105_cast_fp16 = matmul(transpose_x = attn_output_105_transpose_x_1, transpose_y = attn_output_105_transpose_y_1, x = attn_weights_223_cast_fp16, y = var_5271_cast_fp16_1)[name = string("attn_output_105_cast_fp16")]; int32 var_5301 = const()[name = string("op_5301"), val = int32(1)]; bool attn_output_107_interleave_0 = const()[name = string("attn_output_107_interleave_0"), val = bool(false)]; tensor attn_output_107_cast_fp16 = concat(axis = var_5301, interleave = attn_output_107_interleave_0, values = (var_5287_cast_fp16, attn_output_105_cast_fp16))[name = string("attn_output_107_cast_fp16")]; tensor var_5305_perm_0 = const()[name = string("op_5305_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_167x = const()[name = string("concat_167x"), val = tensor([1, 2048, 1, -1])]; tensor var_5305_cast_fp16 = transpose(perm = var_5305_perm_0, x = attn_output_107_cast_fp16)[name = string("transpose_214")]; tensor attn_output_111_cast_fp16 = reshape(shape = concat_167x, x = var_5305_cast_fp16)[name = string("attn_output_111_cast_fp16")]; tensor hidden_states_133_strides_0 = const()[name = string("hidden_states_133_strides_0"), val = tensor([1, 1])]; string hidden_states_133_pad_type_0 = const()[name = string("hidden_states_133_pad_type_0"), val = string("valid")]; tensor hidden_states_133_pad_0 = const()[name = string("hidden_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_133_dilations_0 = const()[name = string("hidden_states_133_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_133_groups_0 = const()[name = string("hidden_states_133_groups_0"), val = int32(1)]; tensor hidden_states_133_cast_fp16 = conv(dilations = hidden_states_133_dilations_0, groups = hidden_states_133_groups_0, pad = hidden_states_133_pad_0, pad_type = hidden_states_133_pad_type_0, strides = hidden_states_133_strides_0, weight = layers_13_self_attn_o_proj_weight_cast_fp16, x = attn_output_111_cast_fp16)[name = string("hidden_states_133_cast_fp16")]; tensor hidden_states_135_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = hidden_states_133_cast_fp16)[name = string("hidden_states_135_cast_fp16")]; fp16 const_138_promoted_to_fp16 = const()[name = string("const_138_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5338_cast_fp16 = mul(x = hidden_states_135_cast_fp16, y = const_138_promoted_to_fp16)[name = string("op_5338_cast_fp16")]; int32 var_5336 = const()[name = string("op_5336"), val = int32(1)]; bool doubled_109_interleave_0 = const()[name = string("doubled_109_interleave_0"), val = bool(false)]; tensor doubled_109_cast_fp16 = concat(axis = var_5336, interleave = doubled_109_interleave_0, values = (hidden_states_135_cast_fp16, var_5338_cast_fp16))[name = string("doubled_109_cast_fp16")]; tensor out_55_axes_0 = const()[name = string("out_55_axes_0"), val = tensor([1])]; tensor out_55_gamma_0_to_fp16 = const()[name = string("out_55_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415399424)))]; fp16 var_5348_to_fp16 = const()[name = string("op_5348_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_55_cast_fp16 = layer_norm(axes = out_55_axes_0, epsilon = var_5348_to_fp16, gamma = out_55_gamma_0_to_fp16, x = doubled_109_cast_fp16)[name = string("out_55_cast_fp16")]; tensor var_5359_split_sizes_0 = const()[name = string("op_5359_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5359_axis_0 = const()[name = string("op_5359_axis_0"), val = int32(1)]; tensor var_5359_cast_fp16_0, tensor var_5359_cast_fp16_1 = split(axis = var_5359_axis_0, split_sizes = var_5359_split_sizes_0, x = out_55_cast_fp16)[name = string("op_5359_cast_fp16")]; tensor input_27_strides_0 = const()[name = string("input_27_strides_0"), val = tensor([1, 1])]; string input_27_pad_type_0 = const()[name = string("input_27_pad_type_0"), val = string("valid")]; tensor input_27_pad_0 = const()[name = string("input_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_27_dilations_0 = const()[name = string("input_27_dilations_0"), val = tensor([1, 1])]; int32 input_27_groups_0 = const()[name = string("input_27_groups_0"), val = int32(1)]; tensor input_27_cast_fp16 = conv(dilations = input_27_dilations_0, groups = input_27_groups_0, pad = input_27_pad_0, pad_type = input_27_pad_type_0, strides = input_27_strides_0, weight = layers_13_mlp_gate_proj_weight_cast_fp16, x = var_5359_cast_fp16_0)[name = string("input_27_cast_fp16")]; tensor var_5376_cast_fp16 = silu(x = input_27_cast_fp16)[name = string("op_5376_cast_fp16")]; tensor layers_13_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_13_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415407680)))]; tensor var_5382_strides_0 = const()[name = string("op_5382_strides_0"), val = tensor([1, 1])]; string var_5382_pad_type_0 = const()[name = string("op_5382_pad_type_0"), val = string("valid")]; tensor var_5382_pad_0 = const()[name = string("op_5382_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5382_dilations_0 = const()[name = string("op_5382_dilations_0"), val = tensor([1, 1])]; int32 var_5382_groups_0 = const()[name = string("op_5382_groups_0"), val = int32(1)]; tensor var_5382_cast_fp16 = conv(dilations = var_5382_dilations_0, groups = var_5382_groups_0, pad = var_5382_pad_0, pad_type = var_5382_pad_type_0, strides = var_5382_strides_0, weight = layers_13_mlp_up_proj_weight_to_fp16, x = var_5359_cast_fp16_0)[name = string("op_5382_cast_fp16")]; tensor x_139_cast_fp16 = mul(x = var_5376_cast_fp16, y = var_5382_cast_fp16)[name = string("x_139_cast_fp16")]; tensor hidden_states_137_strides_0 = const()[name = string("hidden_states_137_strides_0"), val = tensor([1, 1])]; string hidden_states_137_pad_type_0 = const()[name = string("hidden_states_137_pad_type_0"), val = string("valid")]; tensor hidden_states_137_pad_0 = const()[name = string("hidden_states_137_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_137_dilations_0 = const()[name = string("hidden_states_137_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_137_groups_0 = const()[name = string("hidden_states_137_groups_0"), val = int32(1)]; tensor hidden_states_137_cast_fp16 = conv(dilations = hidden_states_137_dilations_0, groups = hidden_states_137_groups_0, pad = hidden_states_137_pad_0, pad_type = hidden_states_137_pad_type_0, strides = hidden_states_137_strides_0, weight = layers_13_mlp_down_proj_weight_cast_fp16, x = x_139_cast_fp16)[name = string("hidden_states_137_cast_fp16")]; tensor hidden_states_139_cast_fp16 = add(x = hidden_states_135_cast_fp16, y = hidden_states_137_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; fp16 const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5400_cast_fp16 = mul(x = hidden_states_139_cast_fp16, y = const_140_promoted_to_fp16)[name = string("op_5400_cast_fp16")]; int32 var_5398 = const()[name = string("op_5398"), val = int32(1)]; bool doubled_113_interleave_0 = const()[name = string("doubled_113_interleave_0"), val = bool(false)]; tensor doubled_113_cast_fp16 = concat(axis = var_5398, interleave = doubled_113_interleave_0, values = (hidden_states_139_cast_fp16, var_5400_cast_fp16))[name = string("doubled_113_cast_fp16")]; tensor out_57_axes_0 = const()[name = string("out_57_axes_0"), val = tensor([1])]; tensor out_57_gamma_0_to_fp16 = const()[name = string("out_57_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440573568)))]; fp16 var_5410_to_fp16 = const()[name = string("op_5410_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_57_cast_fp16 = layer_norm(axes = out_57_axes_0, epsilon = var_5410_to_fp16, gamma = out_57_gamma_0_to_fp16, x = doubled_113_cast_fp16)[name = string("out_57_cast_fp16")]; tensor var_5421_split_sizes_0 = const()[name = string("op_5421_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5421_axis_0 = const()[name = string("op_5421_axis_0"), val = int32(1)]; tensor var_5421_cast_fp16_0, tensor var_5421_cast_fp16_1 = split(axis = var_5421_axis_0, split_sizes = var_5421_split_sizes_0, x = out_57_cast_fp16)[name = string("op_5421_cast_fp16")]; tensor query_states_85_strides_0 = const()[name = string("query_states_85_strides_0"), val = tensor([1, 1])]; string query_states_85_pad_type_0 = const()[name = string("query_states_85_pad_type_0"), val = string("valid")]; tensor query_states_85_pad_0 = const()[name = string("query_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_85_dilations_0 = const()[name = string("query_states_85_dilations_0"), val = tensor([1, 1])]; int32 query_states_85_groups_0 = const()[name = string("query_states_85_groups_0"), val = int32(1)]; tensor query_states_85_cast_fp16 = conv(dilations = query_states_85_dilations_0, groups = query_states_85_groups_0, pad = query_states_85_pad_0, pad_type = query_states_85_pad_type_0, strides = query_states_85_strides_0, weight = layers_14_self_attn_q_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("query_states_85_cast_fp16")]; tensor layers_14_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_14_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440581824)))]; tensor key_states_141_strides_0 = const()[name = string("key_states_141_strides_0"), val = tensor([1, 1])]; string key_states_141_pad_type_0 = const()[name = string("key_states_141_pad_type_0"), val = string("valid")]; tensor key_states_141_pad_0 = const()[name = string("key_states_141_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_141_dilations_0 = const()[name = string("key_states_141_dilations_0"), val = tensor([1, 1])]; int32 key_states_141_groups_0 = const()[name = string("key_states_141_groups_0"), val = int32(1)]; tensor key_states_141_cast_fp16 = conv(dilations = key_states_141_dilations_0, groups = key_states_141_groups_0, pad = key_states_141_pad_0, pad_type = key_states_141_pad_type_0, strides = key_states_141_strides_0, weight = layers_14_self_attn_k_proj_weight_to_fp16, x = var_5421_cast_fp16_0)[name = string("key_states_141_cast_fp16")]; tensor value_states_85_strides_0 = const()[name = string("value_states_85_strides_0"), val = tensor([1, 1])]; string value_states_85_pad_type_0 = const()[name = string("value_states_85_pad_type_0"), val = string("valid")]; tensor value_states_85_pad_0 = const()[name = string("value_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_85_dilations_0 = const()[name = string("value_states_85_dilations_0"), val = tensor([1, 1])]; int32 value_states_85_groups_0 = const()[name = string("value_states_85_groups_0"), val = int32(1)]; tensor value_states_85_cast_fp16 = conv(dilations = value_states_85_dilations_0, groups = value_states_85_groups_0, pad = value_states_85_pad_0, pad_type = value_states_85_pad_type_0, strides = value_states_85_strides_0, weight = layers_14_self_attn_v_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("value_states_85_cast_fp16")]; tensor concat_168x = const()[name = string("concat_168x"), val = tensor([1, 16, 128, -1])]; tensor x_141_cast_fp16 = reshape(shape = concat_168x, x = query_states_85_cast_fp16)[name = string("x_141_cast_fp16")]; tensor concat_169x = const()[name = string("concat_169x"), val = tensor([1, 2, 128, -1])]; tensor var_5478_cast_fp16 = reshape(shape = concat_169x, x = key_states_141_cast_fp16)[name = string("op_5478_cast_fp16")]; tensor concat_170x = const()[name = string("concat_170x"), val = tensor([1, 2, 128, -1])]; tensor var_5485_cast_fp16 = reshape(shape = concat_170x, x = value_states_85_cast_fp16)[name = string("op_5485_cast_fp16")]; tensor var_5489_cast_fp16 = mul(x = x_141_cast_fp16, y = var_869_cast_fp16)[name = string("op_5489_cast_fp16")]; tensor var_5490_split_sizes_0 = const()[name = string("op_5490_split_sizes_0"), val = tensor([64, 64])]; int32 var_5490_axis_0 = const()[name = string("op_5490_axis_0"), val = int32(-2)]; tensor var_5490_cast_fp16_0, tensor var_5490_cast_fp16_1 = split(axis = var_5490_axis_0, split_sizes = var_5490_split_sizes_0, x = x_141_cast_fp16)[name = string("op_5490_cast_fp16")]; fp16 const_142_promoted_to_fp16 = const()[name = string("const_142_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5492_cast_fp16 = mul(x = var_5490_cast_fp16_1, y = const_142_promoted_to_fp16)[name = string("op_5492_cast_fp16")]; int32 var_5494 = const()[name = string("op_5494"), val = int32(-2)]; bool var_5495_interleave_0 = const()[name = string("op_5495_interleave_0"), val = bool(false)]; tensor var_5495_cast_fp16 = concat(axis = var_5494, interleave = var_5495_interleave_0, values = (var_5492_cast_fp16, var_5490_cast_fp16_0))[name = string("op_5495_cast_fp16")]; tensor var_5496_cast_fp16 = mul(x = var_5495_cast_fp16, y = var_878_cast_fp16)[name = string("op_5496_cast_fp16")]; tensor query_states_87_cast_fp16 = add(x = var_5489_cast_fp16, y = var_5496_cast_fp16)[name = string("query_states_87_cast_fp16")]; tensor var_5502_cast_fp16 = mul(x = var_5478_cast_fp16, y = var_869_cast_fp16)[name = string("op_5502_cast_fp16")]; tensor var_5503_split_sizes_0 = const()[name = string("op_5503_split_sizes_0"), val = tensor([64, 64])]; int32 var_5503_axis_0 = const()[name = string("op_5503_axis_0"), val = int32(-2)]; tensor var_5503_cast_fp16_0, tensor var_5503_cast_fp16_1 = split(axis = var_5503_axis_0, split_sizes = var_5503_split_sizes_0, x = var_5478_cast_fp16)[name = string("op_5503_cast_fp16")]; fp16 const_143_promoted_to_fp16 = const()[name = string("const_143_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5505_cast_fp16 = mul(x = var_5503_cast_fp16_1, y = const_143_promoted_to_fp16)[name = string("op_5505_cast_fp16")]; int32 var_5507 = const()[name = string("op_5507"), val = int32(-2)]; bool var_5508_interleave_0 = const()[name = string("op_5508_interleave_0"), val = bool(false)]; tensor var_5508_cast_fp16 = concat(axis = var_5507, interleave = var_5508_interleave_0, values = (var_5505_cast_fp16, var_5503_cast_fp16_0))[name = string("op_5508_cast_fp16")]; tensor var_5509_cast_fp16 = mul(x = var_5508_cast_fp16, y = var_878_cast_fp16)[name = string("op_5509_cast_fp16")]; tensor key_states_145_cast_fp16 = add(x = var_5502_cast_fp16, y = var_5509_cast_fp16)[name = string("key_states_145_cast_fp16")]; tensor expand_dims_168 = const()[name = string("expand_dims_168"), val = tensor([14])]; tensor expand_dims_169 = const()[name = string("expand_dims_169"), val = tensor([0])]; tensor expand_dims_171 = const()[name = string("expand_dims_171"), val = tensor([0])]; int32 concat_173_axis_0 = const()[name = string("concat_173_axis_0"), val = int32(0)]; bool concat_173_interleave_0 = const()[name = string("concat_173_interleave_0"), val = bool(false)]; tensor concat_173 = concat(axis = concat_173_axis_0, interleave = concat_173_interleave_0, values = (expand_dims_168, expand_dims_169, position_id, expand_dims_171))[name = string("concat_173")]; tensor expand_dims_172 = const()[name = string("expand_dims_172"), val = tensor([15])]; tensor concat_174_values1_0 = const()[name = string("concat_174_values1_0"), val = tensor([0])]; tensor concat_174_values3_0 = const()[name = string("concat_174_values3_0"), val = tensor([0])]; int32 concat_174_axis_0 = const()[name = string("concat_174_axis_0"), val = int32(0)]; bool concat_174_interleave_0 = const()[name = string("concat_174_interleave_0"), val = bool(false)]; tensor concat_174 = concat(axis = concat_174_axis_0, interleave = concat_174_interleave_0, values = (expand_dims_172, concat_174_values1_0, cache_position_end, concat_174_values3_0))[name = string("concat_174")]; tensor key_states_147_perm_0 = const()[name = string("key_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_15_stride_0 = const()[name = string("key_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_147_cast_fp16 = transpose(perm = key_states_147_perm_0, x = key_states_145_cast_fp16)[name = string("transpose_213")]; tensor key_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = key_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = key_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_15_squeeze_mask_0, stride = key_cache_internal_tensor_assign_15_stride_0, update = key_states_147_cast_fp16, x = coreml_update_state_138)[name = string("key_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_15_cast_fp16, input = key_cache)[name = string("coreml_update_state_140_write_state")]; tensor coreml_update_state_140 = read_state(input = key_cache)[name = string("coreml_update_state_140")]; tensor value_states_87_perm_0 = const()[name = string("value_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_15_stride_0 = const()[name = string("value_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_87_cast_fp16 = transpose(perm = value_states_87_perm_0, x = var_5485_cast_fp16)[name = string("transpose_212")]; tensor value_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = value_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = value_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_15_squeeze_mask_0, stride = value_cache_internal_tensor_assign_15_stride_0, update = value_states_87_cast_fp16, x = coreml_update_state_139)[name = string("value_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_15_cast_fp16, input = value_cache)[name = string("coreml_update_state_141_write_state")]; tensor coreml_update_state_141 = read_state(input = value_cache)[name = string("coreml_update_state_141")]; tensor var_5579_begin_0 = const()[name = string("op_5579_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5579_end_0 = const()[name = string("op_5579_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5579_end_mask_0 = const()[name = string("op_5579_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5579_cast_fp16 = slice_by_index(begin = var_5579_begin_0, end = var_5579_end_0, end_mask = var_5579_end_mask_0, x = coreml_update_state_140)[name = string("op_5579_cast_fp16")]; tensor tile_28 = const()[name = string("tile_28"), val = tensor([1, 1])]; int32 var_5582_axis_0 = const()[name = string("op_5582_axis_0"), val = int32(1)]; tensor var_5582_cast_fp16_0, tensor var_5582_cast_fp16_1 = split(axis = var_5582_axis_0, split_sizes = tile_28, x = var_5579_cast_fp16)[name = string("op_5582_cast_fp16")]; tensor var_5589_begin_0 = const()[name = string("op_5589_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5589_end_0 = const()[name = string("op_5589_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5589_end_mask_0 = const()[name = string("op_5589_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5589_cast_fp16 = slice_by_index(begin = var_5589_begin_0, end = var_5589_end_0, end_mask = var_5589_end_mask_0, x = coreml_update_state_141)[name = string("op_5589_cast_fp16")]; tensor tile_29 = const()[name = string("tile_29"), val = tensor([1, 1])]; int32 var_5592_axis_0 = const()[name = string("op_5592_axis_0"), val = int32(1)]; tensor var_5592_cast_fp16_0, tensor var_5592_cast_fp16_1 = split(axis = var_5592_axis_0, split_sizes = tile_29, x = var_5589_cast_fp16)[name = string("op_5592_cast_fp16")]; tensor var_5595_split_sizes_0 = const()[name = string("op_5595_split_sizes_0"), val = tensor([8, 8])]; int32 var_5595_axis_0 = const()[name = string("op_5595_axis_0"), val = int32(1)]; tensor var_5595_0, tensor var_5595_1 = split(axis = var_5595_axis_0, split_sizes = var_5595_split_sizes_0, x = query_states_87_cast_fp16)[name = string("op_5595")]; bool attn_weights_225_transpose_x_0 = const()[name = string("attn_weights_225_transpose_x_0"), val = bool(false)]; bool attn_weights_225_transpose_y_0 = const()[name = string("attn_weights_225_transpose_y_0"), val = bool(false)]; tensor attn_weights_225_cast_fp16 = matmul(transpose_x = attn_weights_225_transpose_x_0, transpose_y = attn_weights_225_transpose_y_0, x = var_5582_cast_fp16_0, y = var_5595_0)[name = string("attn_weights_225_cast_fp16")]; fp16 var_5598_to_fp16 = const()[name = string("op_5598_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_227_cast_fp16 = mul(x = attn_weights_225_cast_fp16, y = var_5598_to_fp16)[name = string("attn_weights_227_cast_fp16")]; tensor attn_weights_229_cast_fp16 = add(x = attn_weights_227_cast_fp16, y = attn_mask_1)[name = string("attn_weights_229_cast_fp16")]; int32 var_5602 = const()[name = string("op_5602"), val = int32(-2)]; tensor attn_weights_231_cast_fp16 = softmax(axis = var_5602, x = attn_weights_229_cast_fp16)[name = string("attn_weights_231_cast_fp16")]; bool var_5608_transpose_x_1 = const()[name = string("op_5608_transpose_x_1"), val = bool(true)]; bool var_5608_transpose_y_1 = const()[name = string("op_5608_transpose_y_1"), val = bool(false)]; tensor var_5608_cast_fp16 = matmul(transpose_x = var_5608_transpose_x_1, transpose_y = var_5608_transpose_y_1, x = attn_weights_231_cast_fp16, y = var_5592_cast_fp16_0)[name = string("op_5608_cast_fp16")]; bool attn_weights_233_transpose_x_0 = const()[name = string("attn_weights_233_transpose_x_0"), val = bool(false)]; bool attn_weights_233_transpose_y_0 = const()[name = string("attn_weights_233_transpose_y_0"), val = bool(false)]; tensor attn_weights_233_cast_fp16 = matmul(transpose_x = attn_weights_233_transpose_x_0, transpose_y = attn_weights_233_transpose_y_0, x = var_5582_cast_fp16_1, y = var_5595_1)[name = string("attn_weights_233_cast_fp16")]; fp16 var_5610_to_fp16 = const()[name = string("op_5610_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_235_cast_fp16 = mul(x = attn_weights_233_cast_fp16, y = var_5610_to_fp16)[name = string("attn_weights_235_cast_fp16")]; tensor attn_weights_237_cast_fp16 = add(x = attn_weights_235_cast_fp16, y = attn_mask_1)[name = string("attn_weights_237_cast_fp16")]; int32 var_5614 = const()[name = string("op_5614"), val = int32(-2)]; tensor attn_weights_239_cast_fp16 = softmax(axis = var_5614, x = attn_weights_237_cast_fp16)[name = string("attn_weights_239_cast_fp16")]; bool attn_output_113_transpose_x_1 = const()[name = string("attn_output_113_transpose_x_1"), val = bool(true)]; bool attn_output_113_transpose_y_1 = const()[name = string("attn_output_113_transpose_y_1"), val = bool(false)]; tensor attn_output_113_cast_fp16 = matmul(transpose_x = attn_output_113_transpose_x_1, transpose_y = attn_output_113_transpose_y_1, x = attn_weights_239_cast_fp16, y = var_5592_cast_fp16_1)[name = string("attn_output_113_cast_fp16")]; int32 var_5622 = const()[name = string("op_5622"), val = int32(1)]; bool attn_output_115_interleave_0 = const()[name = string("attn_output_115_interleave_0"), val = bool(false)]; tensor attn_output_115_cast_fp16 = concat(axis = var_5622, interleave = attn_output_115_interleave_0, values = (var_5608_cast_fp16, attn_output_113_cast_fp16))[name = string("attn_output_115_cast_fp16")]; tensor var_5626_perm_0 = const()[name = string("op_5626_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_179x = const()[name = string("concat_179x"), val = tensor([1, 2048, 1, -1])]; tensor var_5626_cast_fp16 = transpose(perm = var_5626_perm_0, x = attn_output_115_cast_fp16)[name = string("transpose_211")]; tensor attn_output_119_cast_fp16 = reshape(shape = concat_179x, x = var_5626_cast_fp16)[name = string("attn_output_119_cast_fp16")]; tensor hidden_states_143_strides_0 = const()[name = string("hidden_states_143_strides_0"), val = tensor([1, 1])]; string hidden_states_143_pad_type_0 = const()[name = string("hidden_states_143_pad_type_0"), val = string("valid")]; tensor hidden_states_143_pad_0 = const()[name = string("hidden_states_143_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_143_dilations_0 = const()[name = string("hidden_states_143_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_143_groups_0 = const()[name = string("hidden_states_143_groups_0"), val = int32(1)]; tensor hidden_states_143_cast_fp16 = conv(dilations = hidden_states_143_dilations_0, groups = hidden_states_143_groups_0, pad = hidden_states_143_pad_0, pad_type = hidden_states_143_pad_type_0, strides = hidden_states_143_strides_0, weight = layers_14_self_attn_o_proj_weight_cast_fp16, x = attn_output_119_cast_fp16)[name = string("hidden_states_143_cast_fp16")]; tensor hidden_states_145_cast_fp16 = add(x = hidden_states_139_cast_fp16, y = hidden_states_143_cast_fp16)[name = string("hidden_states_145_cast_fp16")]; fp16 const_148_promoted_to_fp16 = const()[name = string("const_148_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5659_cast_fp16 = mul(x = hidden_states_145_cast_fp16, y = const_148_promoted_to_fp16)[name = string("op_5659_cast_fp16")]; int32 var_5657 = const()[name = string("op_5657"), val = int32(1)]; bool doubled_117_interleave_0 = const()[name = string("doubled_117_interleave_0"), val = bool(false)]; tensor doubled_117_cast_fp16 = concat(axis = var_5657, interleave = doubled_117_interleave_0, values = (hidden_states_145_cast_fp16, var_5659_cast_fp16))[name = string("doubled_117_cast_fp16")]; tensor out_59_axes_0 = const()[name = string("out_59_axes_0"), val = tensor([1])]; tensor out_59_gamma_0_to_fp16 = const()[name = string("out_59_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441630464)))]; fp16 var_5669_to_fp16 = const()[name = string("op_5669_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_59_cast_fp16 = layer_norm(axes = out_59_axes_0, epsilon = var_5669_to_fp16, gamma = out_59_gamma_0_to_fp16, x = doubled_117_cast_fp16)[name = string("out_59_cast_fp16")]; tensor var_5680_split_sizes_0 = const()[name = string("op_5680_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5680_axis_0 = const()[name = string("op_5680_axis_0"), val = int32(1)]; tensor var_5680_cast_fp16_0, tensor var_5680_cast_fp16_1 = split(axis = var_5680_axis_0, split_sizes = var_5680_split_sizes_0, x = out_59_cast_fp16)[name = string("op_5680_cast_fp16")]; tensor input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor([1, 1])]; string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; tensor input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor([1, 1])]; int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; tensor input_29_cast_fp16 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_14_mlp_gate_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("input_29_cast_fp16")]; tensor var_5697_cast_fp16 = silu(x = input_29_cast_fp16)[name = string("op_5697_cast_fp16")]; tensor var_5703_strides_0 = const()[name = string("op_5703_strides_0"), val = tensor([1, 1])]; string var_5703_pad_type_0 = const()[name = string("op_5703_pad_type_0"), val = string("valid")]; tensor var_5703_pad_0 = const()[name = string("op_5703_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5703_dilations_0 = const()[name = string("op_5703_dilations_0"), val = tensor([1, 1])]; int32 var_5703_groups_0 = const()[name = string("op_5703_groups_0"), val = int32(1)]; tensor var_5703_cast_fp16 = conv(dilations = var_5703_dilations_0, groups = var_5703_groups_0, pad = var_5703_pad_0, pad_type = var_5703_pad_type_0, strides = var_5703_strides_0, weight = layers_14_mlp_up_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("op_5703_cast_fp16")]; tensor x_149_cast_fp16 = mul(x = var_5697_cast_fp16, y = var_5703_cast_fp16)[name = string("x_149_cast_fp16")]; tensor hidden_states_147_strides_0 = const()[name = string("hidden_states_147_strides_0"), val = tensor([1, 1])]; string hidden_states_147_pad_type_0 = const()[name = string("hidden_states_147_pad_type_0"), val = string("valid")]; tensor hidden_states_147_pad_0 = const()[name = string("hidden_states_147_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_147_dilations_0 = const()[name = string("hidden_states_147_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_147_groups_0 = const()[name = string("hidden_states_147_groups_0"), val = int32(1)]; tensor hidden_states_147_cast_fp16 = conv(dilations = hidden_states_147_dilations_0, groups = hidden_states_147_groups_0, pad = hidden_states_147_pad_0, pad_type = hidden_states_147_pad_type_0, strides = hidden_states_147_strides_0, weight = layers_14_mlp_down_proj_weight_cast_fp16, x = x_149_cast_fp16)[name = string("hidden_states_147_cast_fp16")]; tensor hidden_states_149_cast_fp16 = add(x = hidden_states_145_cast_fp16, y = hidden_states_147_cast_fp16)[name = string("hidden_states_149_cast_fp16")]; fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5721_cast_fp16 = mul(x = hidden_states_149_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_5721_cast_fp16")]; int32 var_5719 = const()[name = string("op_5719"), val = int32(1)]; bool doubled_121_interleave_0 = const()[name = string("doubled_121_interleave_0"), val = bool(false)]; tensor doubled_121_cast_fp16 = concat(axis = var_5719, interleave = doubled_121_interleave_0, values = (hidden_states_149_cast_fp16, var_5721_cast_fp16))[name = string("doubled_121_cast_fp16")]; tensor out_61_axes_0 = const()[name = string("out_61_axes_0"), val = tensor([1])]; tensor out_61_gamma_0_to_fp16 = const()[name = string("out_61_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441638720)))]; fp16 var_5731_to_fp16 = const()[name = string("op_5731_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_61_cast_fp16 = layer_norm(axes = out_61_axes_0, epsilon = var_5731_to_fp16, gamma = out_61_gamma_0_to_fp16, x = doubled_121_cast_fp16)[name = string("out_61_cast_fp16")]; tensor var_5742_split_sizes_0 = const()[name = string("op_5742_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5742_axis_0 = const()[name = string("op_5742_axis_0"), val = int32(1)]; tensor var_5742_cast_fp16_0, tensor var_5742_cast_fp16_1 = split(axis = var_5742_axis_0, split_sizes = var_5742_split_sizes_0, x = out_61_cast_fp16)[name = string("op_5742_cast_fp16")]; tensor query_states_91_strides_0 = const()[name = string("query_states_91_strides_0"), val = tensor([1, 1])]; string query_states_91_pad_type_0 = const()[name = string("query_states_91_pad_type_0"), val = string("valid")]; tensor query_states_91_pad_0 = const()[name = string("query_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_91_dilations_0 = const()[name = string("query_states_91_dilations_0"), val = tensor([1, 1])]; int32 query_states_91_groups_0 = const()[name = string("query_states_91_groups_0"), val = int32(1)]; tensor query_states_91_cast_fp16 = conv(dilations = query_states_91_dilations_0, groups = query_states_91_groups_0, pad = query_states_91_pad_0, pad_type = query_states_91_pad_type_0, strides = query_states_91_strides_0, weight = layers_15_self_attn_q_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("query_states_91_cast_fp16")]; tensor key_states_151_strides_0 = const()[name = string("key_states_151_strides_0"), val = tensor([1, 1])]; string key_states_151_pad_type_0 = const()[name = string("key_states_151_pad_type_0"), val = string("valid")]; tensor key_states_151_pad_0 = const()[name = string("key_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_151_dilations_0 = const()[name = string("key_states_151_dilations_0"), val = tensor([1, 1])]; int32 key_states_151_groups_0 = const()[name = string("key_states_151_groups_0"), val = int32(1)]; tensor key_states_151_cast_fp16 = conv(dilations = key_states_151_dilations_0, groups = key_states_151_groups_0, pad = key_states_151_pad_0, pad_type = key_states_151_pad_type_0, strides = key_states_151_strides_0, weight = layers_15_self_attn_k_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("key_states_151_cast_fp16")]; tensor value_states_91_strides_0 = const()[name = string("value_states_91_strides_0"), val = tensor([1, 1])]; string value_states_91_pad_type_0 = const()[name = string("value_states_91_pad_type_0"), val = string("valid")]; tensor value_states_91_pad_0 = const()[name = string("value_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_91_dilations_0 = const()[name = string("value_states_91_dilations_0"), val = tensor([1, 1])]; int32 value_states_91_groups_0 = const()[name = string("value_states_91_groups_0"), val = int32(1)]; tensor value_states_91_cast_fp16 = conv(dilations = value_states_91_dilations_0, groups = value_states_91_groups_0, pad = value_states_91_pad_0, pad_type = value_states_91_pad_type_0, strides = value_states_91_strides_0, weight = layers_15_self_attn_v_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("value_states_91_cast_fp16")]; tensor concat_180x = const()[name = string("concat_180x"), val = tensor([1, 16, 128, -1])]; tensor x_151_cast_fp16 = reshape(shape = concat_180x, x = query_states_91_cast_fp16)[name = string("x_151_cast_fp16")]; tensor concat_181x = const()[name = string("concat_181x"), val = tensor([1, 2, 128, -1])]; tensor var_5799_cast_fp16 = reshape(shape = concat_181x, x = key_states_151_cast_fp16)[name = string("op_5799_cast_fp16")]; tensor concat_182x = const()[name = string("concat_182x"), val = tensor([1, 2, 128, -1])]; tensor var_5806_cast_fp16 = reshape(shape = concat_182x, x = value_states_91_cast_fp16)[name = string("op_5806_cast_fp16")]; tensor var_5810_cast_fp16 = mul(x = x_151_cast_fp16, y = var_869_cast_fp16)[name = string("op_5810_cast_fp16")]; tensor var_5811_split_sizes_0 = const()[name = string("op_5811_split_sizes_0"), val = tensor([64, 64])]; int32 var_5811_axis_0 = const()[name = string("op_5811_axis_0"), val = int32(-2)]; tensor var_5811_cast_fp16_0, tensor var_5811_cast_fp16_1 = split(axis = var_5811_axis_0, split_sizes = var_5811_split_sizes_0, x = x_151_cast_fp16)[name = string("op_5811_cast_fp16")]; fp16 const_152_promoted_to_fp16 = const()[name = string("const_152_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5813_cast_fp16 = mul(x = var_5811_cast_fp16_1, y = const_152_promoted_to_fp16)[name = string("op_5813_cast_fp16")]; int32 var_5815 = const()[name = string("op_5815"), val = int32(-2)]; bool var_5816_interleave_0 = const()[name = string("op_5816_interleave_0"), val = bool(false)]; tensor var_5816_cast_fp16 = concat(axis = var_5815, interleave = var_5816_interleave_0, values = (var_5813_cast_fp16, var_5811_cast_fp16_0))[name = string("op_5816_cast_fp16")]; tensor var_5817_cast_fp16 = mul(x = var_5816_cast_fp16, y = var_878_cast_fp16)[name = string("op_5817_cast_fp16")]; tensor query_states_93_cast_fp16 = add(x = var_5810_cast_fp16, y = var_5817_cast_fp16)[name = string("query_states_93_cast_fp16")]; tensor var_5823_cast_fp16 = mul(x = var_5799_cast_fp16, y = var_869_cast_fp16)[name = string("op_5823_cast_fp16")]; tensor var_5824_split_sizes_0 = const()[name = string("op_5824_split_sizes_0"), val = tensor([64, 64])]; int32 var_5824_axis_0 = const()[name = string("op_5824_axis_0"), val = int32(-2)]; tensor var_5824_cast_fp16_0, tensor var_5824_cast_fp16_1 = split(axis = var_5824_axis_0, split_sizes = var_5824_split_sizes_0, x = var_5799_cast_fp16)[name = string("op_5824_cast_fp16")]; fp16 const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5826_cast_fp16 = mul(x = var_5824_cast_fp16_1, y = const_153_promoted_to_fp16)[name = string("op_5826_cast_fp16")]; int32 var_5828 = const()[name = string("op_5828"), val = int32(-2)]; bool var_5829_interleave_0 = const()[name = string("op_5829_interleave_0"), val = bool(false)]; tensor var_5829_cast_fp16 = concat(axis = var_5828, interleave = var_5829_interleave_0, values = (var_5826_cast_fp16, var_5824_cast_fp16_0))[name = string("op_5829_cast_fp16")]; tensor var_5830_cast_fp16 = mul(x = var_5829_cast_fp16, y = var_878_cast_fp16)[name = string("op_5830_cast_fp16")]; tensor key_states_155_cast_fp16 = add(x = var_5823_cast_fp16, y = var_5830_cast_fp16)[name = string("key_states_155_cast_fp16")]; tensor expand_dims_180 = const()[name = string("expand_dims_180"), val = tensor([15])]; tensor expand_dims_181 = const()[name = string("expand_dims_181"), val = tensor([0])]; tensor expand_dims_183 = const()[name = string("expand_dims_183"), val = tensor([0])]; int32 concat_185_axis_0 = const()[name = string("concat_185_axis_0"), val = int32(0)]; bool concat_185_interleave_0 = const()[name = string("concat_185_interleave_0"), val = bool(false)]; tensor concat_185 = concat(axis = concat_185_axis_0, interleave = concat_185_interleave_0, values = (expand_dims_180, expand_dims_181, position_id, expand_dims_183))[name = string("concat_185")]; tensor expand_dims_184 = const()[name = string("expand_dims_184"), val = tensor([16])]; tensor concat_186_values1_0 = const()[name = string("concat_186_values1_0"), val = tensor([0])]; tensor concat_186_values3_0 = const()[name = string("concat_186_values3_0"), val = tensor([0])]; int32 concat_186_axis_0 = const()[name = string("concat_186_axis_0"), val = int32(0)]; bool concat_186_interleave_0 = const()[name = string("concat_186_interleave_0"), val = bool(false)]; tensor concat_186 = concat(axis = concat_186_axis_0, interleave = concat_186_interleave_0, values = (expand_dims_184, concat_186_values1_0, cache_position_end, concat_186_values3_0))[name = string("concat_186")]; tensor key_states_157_perm_0 = const()[name = string("key_states_157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_16_stride_0 = const()[name = string("key_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_157_cast_fp16 = transpose(perm = key_states_157_perm_0, x = key_states_155_cast_fp16)[name = string("transpose_210")]; tensor key_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = key_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = key_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_16_squeeze_mask_0, stride = key_cache_internal_tensor_assign_16_stride_0, update = key_states_157_cast_fp16, x = coreml_update_state_140)[name = string("key_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_16_cast_fp16, input = key_cache)[name = string("coreml_update_state_142_write_state")]; tensor coreml_update_state_142 = read_state(input = key_cache)[name = string("coreml_update_state_142")]; tensor value_states_93_perm_0 = const()[name = string("value_states_93_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_16_stride_0 = const()[name = string("value_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_93_cast_fp16 = transpose(perm = value_states_93_perm_0, x = var_5806_cast_fp16)[name = string("transpose_209")]; tensor value_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = value_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = value_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_16_squeeze_mask_0, stride = value_cache_internal_tensor_assign_16_stride_0, update = value_states_93_cast_fp16, x = coreml_update_state_141)[name = string("value_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_16_cast_fp16, input = value_cache)[name = string("coreml_update_state_143_write_state")]; tensor coreml_update_state_143 = read_state(input = value_cache)[name = string("coreml_update_state_143")]; tensor var_5900_begin_0 = const()[name = string("op_5900_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5900_end_0 = const()[name = string("op_5900_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5900_end_mask_0 = const()[name = string("op_5900_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5900_cast_fp16 = slice_by_index(begin = var_5900_begin_0, end = var_5900_end_0, end_mask = var_5900_end_mask_0, x = coreml_update_state_142)[name = string("op_5900_cast_fp16")]; tensor tile_30 = const()[name = string("tile_30"), val = tensor([1, 1])]; int32 var_5903_axis_0 = const()[name = string("op_5903_axis_0"), val = int32(1)]; tensor var_5903_cast_fp16_0, tensor var_5903_cast_fp16_1 = split(axis = var_5903_axis_0, split_sizes = tile_30, x = var_5900_cast_fp16)[name = string("op_5903_cast_fp16")]; tensor var_5910_begin_0 = const()[name = string("op_5910_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5910_end_0 = const()[name = string("op_5910_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5910_end_mask_0 = const()[name = string("op_5910_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5910_cast_fp16 = slice_by_index(begin = var_5910_begin_0, end = var_5910_end_0, end_mask = var_5910_end_mask_0, x = coreml_update_state_143)[name = string("op_5910_cast_fp16")]; tensor tile_31 = const()[name = string("tile_31"), val = tensor([1, 1])]; int32 var_5913_axis_0 = const()[name = string("op_5913_axis_0"), val = int32(1)]; tensor var_5913_cast_fp16_0, tensor var_5913_cast_fp16_1 = split(axis = var_5913_axis_0, split_sizes = tile_31, x = var_5910_cast_fp16)[name = string("op_5913_cast_fp16")]; tensor var_5916_split_sizes_0 = const()[name = string("op_5916_split_sizes_0"), val = tensor([8, 8])]; int32 var_5916_axis_0 = const()[name = string("op_5916_axis_0"), val = int32(1)]; tensor var_5916_0, tensor var_5916_1 = split(axis = var_5916_axis_0, split_sizes = var_5916_split_sizes_0, x = query_states_93_cast_fp16)[name = string("op_5916")]; bool attn_weights_241_transpose_x_0 = const()[name = string("attn_weights_241_transpose_x_0"), val = bool(false)]; bool attn_weights_241_transpose_y_0 = const()[name = string("attn_weights_241_transpose_y_0"), val = bool(false)]; tensor attn_weights_241_cast_fp16 = matmul(transpose_x = attn_weights_241_transpose_x_0, transpose_y = attn_weights_241_transpose_y_0, x = var_5903_cast_fp16_0, y = var_5916_0)[name = string("attn_weights_241_cast_fp16")]; fp16 var_5919_to_fp16 = const()[name = string("op_5919_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_243_cast_fp16 = mul(x = attn_weights_241_cast_fp16, y = var_5919_to_fp16)[name = string("attn_weights_243_cast_fp16")]; tensor attn_weights_245_cast_fp16 = add(x = attn_weights_243_cast_fp16, y = attn_mask_1)[name = string("attn_weights_245_cast_fp16")]; int32 var_5923 = const()[name = string("op_5923"), val = int32(-2)]; tensor attn_weights_247_cast_fp16 = softmax(axis = var_5923, x = attn_weights_245_cast_fp16)[name = string("attn_weights_247_cast_fp16")]; bool var_5929_transpose_x_1 = const()[name = string("op_5929_transpose_x_1"), val = bool(true)]; bool var_5929_transpose_y_1 = const()[name = string("op_5929_transpose_y_1"), val = bool(false)]; tensor var_5929_cast_fp16 = matmul(transpose_x = var_5929_transpose_x_1, transpose_y = var_5929_transpose_y_1, x = attn_weights_247_cast_fp16, y = var_5913_cast_fp16_0)[name = string("op_5929_cast_fp16")]; bool attn_weights_249_transpose_x_0 = const()[name = string("attn_weights_249_transpose_x_0"), val = bool(false)]; bool attn_weights_249_transpose_y_0 = const()[name = string("attn_weights_249_transpose_y_0"), val = bool(false)]; tensor attn_weights_249_cast_fp16 = matmul(transpose_x = attn_weights_249_transpose_x_0, transpose_y = attn_weights_249_transpose_y_0, x = var_5903_cast_fp16_1, y = var_5916_1)[name = string("attn_weights_249_cast_fp16")]; fp16 var_5931_to_fp16 = const()[name = string("op_5931_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_251_cast_fp16 = mul(x = attn_weights_249_cast_fp16, y = var_5931_to_fp16)[name = string("attn_weights_251_cast_fp16")]; tensor attn_weights_253_cast_fp16 = add(x = attn_weights_251_cast_fp16, y = attn_mask_1)[name = string("attn_weights_253_cast_fp16")]; int32 var_5935 = const()[name = string("op_5935"), val = int32(-2)]; tensor attn_weights_255_cast_fp16 = softmax(axis = var_5935, x = attn_weights_253_cast_fp16)[name = string("attn_weights_255_cast_fp16")]; bool attn_output_121_transpose_x_1 = const()[name = string("attn_output_121_transpose_x_1"), val = bool(true)]; bool attn_output_121_transpose_y_1 = const()[name = string("attn_output_121_transpose_y_1"), val = bool(false)]; tensor attn_output_121_cast_fp16 = matmul(transpose_x = attn_output_121_transpose_x_1, transpose_y = attn_output_121_transpose_y_1, x = attn_weights_255_cast_fp16, y = var_5913_cast_fp16_1)[name = string("attn_output_121_cast_fp16")]; int32 var_5943 = const()[name = string("op_5943"), val = int32(1)]; bool attn_output_123_interleave_0 = const()[name = string("attn_output_123_interleave_0"), val = bool(false)]; tensor attn_output_123_cast_fp16 = concat(axis = var_5943, interleave = attn_output_123_interleave_0, values = (var_5929_cast_fp16, attn_output_121_cast_fp16))[name = string("attn_output_123_cast_fp16")]; tensor var_5947_perm_0 = const()[name = string("op_5947_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_191x = const()[name = string("concat_191x"), val = tensor([1, 2048, 1, -1])]; tensor var_5947_cast_fp16 = transpose(perm = var_5947_perm_0, x = attn_output_123_cast_fp16)[name = string("transpose_208")]; tensor attn_output_127_cast_fp16 = reshape(shape = concat_191x, x = var_5947_cast_fp16)[name = string("attn_output_127_cast_fp16")]; tensor hidden_states_153_strides_0 = const()[name = string("hidden_states_153_strides_0"), val = tensor([1, 1])]; string hidden_states_153_pad_type_0 = const()[name = string("hidden_states_153_pad_type_0"), val = string("valid")]; tensor hidden_states_153_pad_0 = const()[name = string("hidden_states_153_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_153_dilations_0 = const()[name = string("hidden_states_153_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_153_groups_0 = const()[name = string("hidden_states_153_groups_0"), val = int32(1)]; tensor hidden_states_153_cast_fp16 = conv(dilations = hidden_states_153_dilations_0, groups = hidden_states_153_groups_0, pad = hidden_states_153_pad_0, pad_type = hidden_states_153_pad_type_0, strides = hidden_states_153_strides_0, weight = layers_15_self_attn_o_proj_weight_cast_fp16, x = attn_output_127_cast_fp16)[name = string("hidden_states_153_cast_fp16")]; tensor hidden_states_155_cast_fp16 = add(x = hidden_states_149_cast_fp16, y = hidden_states_153_cast_fp16)[name = string("hidden_states_155_cast_fp16")]; fp16 const_158_promoted_to_fp16 = const()[name = string("const_158_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5980_cast_fp16 = mul(x = hidden_states_155_cast_fp16, y = const_158_promoted_to_fp16)[name = string("op_5980_cast_fp16")]; int32 var_5978 = const()[name = string("op_5978"), val = int32(1)]; bool doubled_125_interleave_0 = const()[name = string("doubled_125_interleave_0"), val = bool(false)]; tensor doubled_125_cast_fp16 = concat(axis = var_5978, interleave = doubled_125_interleave_0, values = (hidden_states_155_cast_fp16, var_5980_cast_fp16))[name = string("doubled_125_cast_fp16")]; tensor out_63_axes_0 = const()[name = string("out_63_axes_0"), val = tensor([1])]; tensor out_63_gamma_0_to_fp16 = const()[name = string("out_63_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441646976)))]; fp16 var_5990_to_fp16 = const()[name = string("op_5990_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_63_cast_fp16 = layer_norm(axes = out_63_axes_0, epsilon = var_5990_to_fp16, gamma = out_63_gamma_0_to_fp16, x = doubled_125_cast_fp16)[name = string("out_63_cast_fp16")]; tensor var_6001_split_sizes_0 = const()[name = string("op_6001_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6001_axis_0 = const()[name = string("op_6001_axis_0"), val = int32(1)]; tensor var_6001_cast_fp16_0, tensor var_6001_cast_fp16_1 = split(axis = var_6001_axis_0, split_sizes = var_6001_split_sizes_0, x = out_63_cast_fp16)[name = string("op_6001_cast_fp16")]; tensor input_31_strides_0 = const()[name = string("input_31_strides_0"), val = tensor([1, 1])]; string input_31_pad_type_0 = const()[name = string("input_31_pad_type_0"), val = string("valid")]; tensor input_31_pad_0 = const()[name = string("input_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_31_dilations_0 = const()[name = string("input_31_dilations_0"), val = tensor([1, 1])]; int32 input_31_groups_0 = const()[name = string("input_31_groups_0"), val = int32(1)]; tensor input_31_cast_fp16 = conv(dilations = input_31_dilations_0, groups = input_31_groups_0, pad = input_31_pad_0, pad_type = input_31_pad_type_0, strides = input_31_strides_0, weight = layers_15_mlp_gate_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("input_31_cast_fp16")]; tensor var_6018_cast_fp16 = silu(x = input_31_cast_fp16)[name = string("op_6018_cast_fp16")]; tensor var_6024_strides_0 = const()[name = string("op_6024_strides_0"), val = tensor([1, 1])]; string var_6024_pad_type_0 = const()[name = string("op_6024_pad_type_0"), val = string("valid")]; tensor var_6024_pad_0 = const()[name = string("op_6024_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6024_dilations_0 = const()[name = string("op_6024_dilations_0"), val = tensor([1, 1])]; int32 var_6024_groups_0 = const()[name = string("op_6024_groups_0"), val = int32(1)]; tensor var_6024_cast_fp16 = conv(dilations = var_6024_dilations_0, groups = var_6024_groups_0, pad = var_6024_pad_0, pad_type = var_6024_pad_type_0, strides = var_6024_strides_0, weight = layers_15_mlp_up_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("op_6024_cast_fp16")]; tensor x_159_cast_fp16 = mul(x = var_6018_cast_fp16, y = var_6024_cast_fp16)[name = string("x_159_cast_fp16")]; tensor hidden_states_157_strides_0 = const()[name = string("hidden_states_157_strides_0"), val = tensor([1, 1])]; string hidden_states_157_pad_type_0 = const()[name = string("hidden_states_157_pad_type_0"), val = string("valid")]; tensor hidden_states_157_pad_0 = const()[name = string("hidden_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_157_dilations_0 = const()[name = string("hidden_states_157_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_157_groups_0 = const()[name = string("hidden_states_157_groups_0"), val = int32(1)]; tensor hidden_states_157_cast_fp16 = conv(dilations = hidden_states_157_dilations_0, groups = hidden_states_157_groups_0, pad = hidden_states_157_pad_0, pad_type = hidden_states_157_pad_type_0, strides = hidden_states_157_strides_0, weight = layers_15_mlp_down_proj_weight_cast_fp16, x = x_159_cast_fp16)[name = string("hidden_states_157_cast_fp16")]; tensor hidden_states_159_cast_fp16 = add(x = hidden_states_155_cast_fp16, y = hidden_states_157_cast_fp16)[name = string("hidden_states_159_cast_fp16")]; fp16 const_160_promoted_to_fp16 = const()[name = string("const_160_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6042_cast_fp16 = mul(x = hidden_states_159_cast_fp16, y = const_160_promoted_to_fp16)[name = string("op_6042_cast_fp16")]; int32 var_6040 = const()[name = string("op_6040"), val = int32(1)]; bool doubled_129_interleave_0 = const()[name = string("doubled_129_interleave_0"), val = bool(false)]; tensor doubled_129_cast_fp16 = concat(axis = var_6040, interleave = doubled_129_interleave_0, values = (hidden_states_159_cast_fp16, var_6042_cast_fp16))[name = string("doubled_129_cast_fp16")]; tensor out_65_axes_0 = const()[name = string("out_65_axes_0"), val = tensor([1])]; tensor out_65_gamma_0_to_fp16 = const()[name = string("out_65_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441655232)))]; fp16 var_6052_to_fp16 = const()[name = string("op_6052_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_65_cast_fp16 = layer_norm(axes = out_65_axes_0, epsilon = var_6052_to_fp16, gamma = out_65_gamma_0_to_fp16, x = doubled_129_cast_fp16)[name = string("out_65_cast_fp16")]; tensor var_6063_split_sizes_0 = const()[name = string("op_6063_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6063_axis_0 = const()[name = string("op_6063_axis_0"), val = int32(1)]; tensor var_6063_cast_fp16_0, tensor var_6063_cast_fp16_1 = split(axis = var_6063_axis_0, split_sizes = var_6063_split_sizes_0, x = out_65_cast_fp16)[name = string("op_6063_cast_fp16")]; tensor query_states_97_strides_0 = const()[name = string("query_states_97_strides_0"), val = tensor([1, 1])]; string query_states_97_pad_type_0 = const()[name = string("query_states_97_pad_type_0"), val = string("valid")]; tensor query_states_97_pad_0 = const()[name = string("query_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_97_dilations_0 = const()[name = string("query_states_97_dilations_0"), val = tensor([1, 1])]; int32 query_states_97_groups_0 = const()[name = string("query_states_97_groups_0"), val = int32(1)]; tensor query_states_97_cast_fp16 = conv(dilations = query_states_97_dilations_0, groups = query_states_97_groups_0, pad = query_states_97_pad_0, pad_type = query_states_97_pad_type_0, strides = query_states_97_strides_0, weight = layers_16_self_attn_q_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("query_states_97_cast_fp16")]; tensor key_states_161_strides_0 = const()[name = string("key_states_161_strides_0"), val = tensor([1, 1])]; string key_states_161_pad_type_0 = const()[name = string("key_states_161_pad_type_0"), val = string("valid")]; tensor key_states_161_pad_0 = const()[name = string("key_states_161_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_161_dilations_0 = const()[name = string("key_states_161_dilations_0"), val = tensor([1, 1])]; int32 key_states_161_groups_0 = const()[name = string("key_states_161_groups_0"), val = int32(1)]; tensor key_states_161_cast_fp16 = conv(dilations = key_states_161_dilations_0, groups = key_states_161_groups_0, pad = key_states_161_pad_0, pad_type = key_states_161_pad_type_0, strides = key_states_161_strides_0, weight = layers_16_self_attn_k_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("key_states_161_cast_fp16")]; tensor value_states_97_strides_0 = const()[name = string("value_states_97_strides_0"), val = tensor([1, 1])]; string value_states_97_pad_type_0 = const()[name = string("value_states_97_pad_type_0"), val = string("valid")]; tensor value_states_97_pad_0 = const()[name = string("value_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_97_dilations_0 = const()[name = string("value_states_97_dilations_0"), val = tensor([1, 1])]; int32 value_states_97_groups_0 = const()[name = string("value_states_97_groups_0"), val = int32(1)]; tensor value_states_97_cast_fp16 = conv(dilations = value_states_97_dilations_0, groups = value_states_97_groups_0, pad = value_states_97_pad_0, pad_type = value_states_97_pad_type_0, strides = value_states_97_strides_0, weight = layers_16_self_attn_v_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("value_states_97_cast_fp16")]; tensor concat_192x = const()[name = string("concat_192x"), val = tensor([1, 16, 128, -1])]; tensor x_161_cast_fp16 = reshape(shape = concat_192x, x = query_states_97_cast_fp16)[name = string("x_161_cast_fp16")]; tensor concat_193x = const()[name = string("concat_193x"), val = tensor([1, 2, 128, -1])]; tensor var_6120_cast_fp16 = reshape(shape = concat_193x, x = key_states_161_cast_fp16)[name = string("op_6120_cast_fp16")]; tensor concat_194x = const()[name = string("concat_194x"), val = tensor([1, 2, 128, -1])]; tensor var_6127_cast_fp16 = reshape(shape = concat_194x, x = value_states_97_cast_fp16)[name = string("op_6127_cast_fp16")]; tensor var_6131_cast_fp16 = mul(x = x_161_cast_fp16, y = var_869_cast_fp16)[name = string("op_6131_cast_fp16")]; tensor var_6132_split_sizes_0 = const()[name = string("op_6132_split_sizes_0"), val = tensor([64, 64])]; int32 var_6132_axis_0 = const()[name = string("op_6132_axis_0"), val = int32(-2)]; tensor var_6132_cast_fp16_0, tensor var_6132_cast_fp16_1 = split(axis = var_6132_axis_0, split_sizes = var_6132_split_sizes_0, x = x_161_cast_fp16)[name = string("op_6132_cast_fp16")]; fp16 const_162_promoted_to_fp16 = const()[name = string("const_162_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6134_cast_fp16 = mul(x = var_6132_cast_fp16_1, y = const_162_promoted_to_fp16)[name = string("op_6134_cast_fp16")]; int32 var_6136 = const()[name = string("op_6136"), val = int32(-2)]; bool var_6137_interleave_0 = const()[name = string("op_6137_interleave_0"), val = bool(false)]; tensor var_6137_cast_fp16 = concat(axis = var_6136, interleave = var_6137_interleave_0, values = (var_6134_cast_fp16, var_6132_cast_fp16_0))[name = string("op_6137_cast_fp16")]; tensor var_6138_cast_fp16 = mul(x = var_6137_cast_fp16, y = var_878_cast_fp16)[name = string("op_6138_cast_fp16")]; tensor query_states_99_cast_fp16 = add(x = var_6131_cast_fp16, y = var_6138_cast_fp16)[name = string("query_states_99_cast_fp16")]; tensor var_6144_cast_fp16 = mul(x = var_6120_cast_fp16, y = var_869_cast_fp16)[name = string("op_6144_cast_fp16")]; tensor var_6145_split_sizes_0 = const()[name = string("op_6145_split_sizes_0"), val = tensor([64, 64])]; int32 var_6145_axis_0 = const()[name = string("op_6145_axis_0"), val = int32(-2)]; tensor var_6145_cast_fp16_0, tensor var_6145_cast_fp16_1 = split(axis = var_6145_axis_0, split_sizes = var_6145_split_sizes_0, x = var_6120_cast_fp16)[name = string("op_6145_cast_fp16")]; fp16 const_163_promoted_to_fp16 = const()[name = string("const_163_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6147_cast_fp16 = mul(x = var_6145_cast_fp16_1, y = const_163_promoted_to_fp16)[name = string("op_6147_cast_fp16")]; int32 var_6149 = const()[name = string("op_6149"), val = int32(-2)]; bool var_6150_interleave_0 = const()[name = string("op_6150_interleave_0"), val = bool(false)]; tensor var_6150_cast_fp16 = concat(axis = var_6149, interleave = var_6150_interleave_0, values = (var_6147_cast_fp16, var_6145_cast_fp16_0))[name = string("op_6150_cast_fp16")]; tensor var_6151_cast_fp16 = mul(x = var_6150_cast_fp16, y = var_878_cast_fp16)[name = string("op_6151_cast_fp16")]; tensor key_states_165_cast_fp16 = add(x = var_6144_cast_fp16, y = var_6151_cast_fp16)[name = string("key_states_165_cast_fp16")]; tensor expand_dims_192 = const()[name = string("expand_dims_192"), val = tensor([16])]; tensor expand_dims_193 = const()[name = string("expand_dims_193"), val = tensor([0])]; tensor expand_dims_195 = const()[name = string("expand_dims_195"), val = tensor([0])]; int32 concat_197_axis_0 = const()[name = string("concat_197_axis_0"), val = int32(0)]; bool concat_197_interleave_0 = const()[name = string("concat_197_interleave_0"), val = bool(false)]; tensor concat_197 = concat(axis = concat_197_axis_0, interleave = concat_197_interleave_0, values = (expand_dims_192, expand_dims_193, position_id, expand_dims_195))[name = string("concat_197")]; tensor expand_dims_196 = const()[name = string("expand_dims_196"), val = tensor([17])]; tensor concat_198_values1_0 = const()[name = string("concat_198_values1_0"), val = tensor([0])]; tensor concat_198_values3_0 = const()[name = string("concat_198_values3_0"), val = tensor([0])]; int32 concat_198_axis_0 = const()[name = string("concat_198_axis_0"), val = int32(0)]; bool concat_198_interleave_0 = const()[name = string("concat_198_interleave_0"), val = bool(false)]; tensor concat_198 = concat(axis = concat_198_axis_0, interleave = concat_198_interleave_0, values = (expand_dims_196, concat_198_values1_0, cache_position_end, concat_198_values3_0))[name = string("concat_198")]; tensor key_states_167_perm_0 = const()[name = string("key_states_167_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_17_stride_0 = const()[name = string("key_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_167_cast_fp16 = transpose(perm = key_states_167_perm_0, x = key_states_165_cast_fp16)[name = string("transpose_207")]; tensor key_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = key_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = key_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_17_squeeze_mask_0, stride = key_cache_internal_tensor_assign_17_stride_0, update = key_states_167_cast_fp16, x = coreml_update_state_142)[name = string("key_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_17_cast_fp16, input = key_cache)[name = string("coreml_update_state_144_write_state")]; tensor coreml_update_state_144 = read_state(input = key_cache)[name = string("coreml_update_state_144")]; tensor value_states_99_perm_0 = const()[name = string("value_states_99_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_17_stride_0 = const()[name = string("value_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_99_cast_fp16 = transpose(perm = value_states_99_perm_0, x = var_6127_cast_fp16)[name = string("transpose_206")]; tensor value_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = value_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = value_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_17_squeeze_mask_0, stride = value_cache_internal_tensor_assign_17_stride_0, update = value_states_99_cast_fp16, x = coreml_update_state_143)[name = string("value_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_17_cast_fp16, input = value_cache)[name = string("coreml_update_state_145_write_state")]; tensor coreml_update_state_145 = read_state(input = value_cache)[name = string("coreml_update_state_145")]; tensor var_6221_begin_0 = const()[name = string("op_6221_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6221_end_0 = const()[name = string("op_6221_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6221_end_mask_0 = const()[name = string("op_6221_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6221_cast_fp16 = slice_by_index(begin = var_6221_begin_0, end = var_6221_end_0, end_mask = var_6221_end_mask_0, x = coreml_update_state_144)[name = string("op_6221_cast_fp16")]; tensor tile_32 = const()[name = string("tile_32"), val = tensor([1, 1])]; int32 var_6224_axis_0 = const()[name = string("op_6224_axis_0"), val = int32(1)]; tensor var_6224_cast_fp16_0, tensor var_6224_cast_fp16_1 = split(axis = var_6224_axis_0, split_sizes = tile_32, x = var_6221_cast_fp16)[name = string("op_6224_cast_fp16")]; tensor var_6231_begin_0 = const()[name = string("op_6231_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6231_end_0 = const()[name = string("op_6231_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6231_end_mask_0 = const()[name = string("op_6231_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6231_cast_fp16 = slice_by_index(begin = var_6231_begin_0, end = var_6231_end_0, end_mask = var_6231_end_mask_0, x = coreml_update_state_145)[name = string("op_6231_cast_fp16")]; tensor tile_33 = const()[name = string("tile_33"), val = tensor([1, 1])]; int32 var_6234_axis_0 = const()[name = string("op_6234_axis_0"), val = int32(1)]; tensor var_6234_cast_fp16_0, tensor var_6234_cast_fp16_1 = split(axis = var_6234_axis_0, split_sizes = tile_33, x = var_6231_cast_fp16)[name = string("op_6234_cast_fp16")]; tensor var_6237_split_sizes_0 = const()[name = string("op_6237_split_sizes_0"), val = tensor([8, 8])]; int32 var_6237_axis_0 = const()[name = string("op_6237_axis_0"), val = int32(1)]; tensor var_6237_0, tensor var_6237_1 = split(axis = var_6237_axis_0, split_sizes = var_6237_split_sizes_0, x = query_states_99_cast_fp16)[name = string("op_6237")]; bool attn_weights_257_transpose_x_0 = const()[name = string("attn_weights_257_transpose_x_0"), val = bool(false)]; bool attn_weights_257_transpose_y_0 = const()[name = string("attn_weights_257_transpose_y_0"), val = bool(false)]; tensor attn_weights_257_cast_fp16 = matmul(transpose_x = attn_weights_257_transpose_x_0, transpose_y = attn_weights_257_transpose_y_0, x = var_6224_cast_fp16_0, y = var_6237_0)[name = string("attn_weights_257_cast_fp16")]; fp16 var_6240_to_fp16 = const()[name = string("op_6240_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_259_cast_fp16 = mul(x = attn_weights_257_cast_fp16, y = var_6240_to_fp16)[name = string("attn_weights_259_cast_fp16")]; tensor attn_weights_261_cast_fp16 = add(x = attn_weights_259_cast_fp16, y = attn_mask_1)[name = string("attn_weights_261_cast_fp16")]; int32 var_6244 = const()[name = string("op_6244"), val = int32(-2)]; tensor attn_weights_263_cast_fp16 = softmax(axis = var_6244, x = attn_weights_261_cast_fp16)[name = string("attn_weights_263_cast_fp16")]; bool var_6250_transpose_x_1 = const()[name = string("op_6250_transpose_x_1"), val = bool(true)]; bool var_6250_transpose_y_1 = const()[name = string("op_6250_transpose_y_1"), val = bool(false)]; tensor var_6250_cast_fp16 = matmul(transpose_x = var_6250_transpose_x_1, transpose_y = var_6250_transpose_y_1, x = attn_weights_263_cast_fp16, y = var_6234_cast_fp16_0)[name = string("op_6250_cast_fp16")]; bool attn_weights_265_transpose_x_0 = const()[name = string("attn_weights_265_transpose_x_0"), val = bool(false)]; bool attn_weights_265_transpose_y_0 = const()[name = string("attn_weights_265_transpose_y_0"), val = bool(false)]; tensor attn_weights_265_cast_fp16 = matmul(transpose_x = attn_weights_265_transpose_x_0, transpose_y = attn_weights_265_transpose_y_0, x = var_6224_cast_fp16_1, y = var_6237_1)[name = string("attn_weights_265_cast_fp16")]; fp16 var_6252_to_fp16 = const()[name = string("op_6252_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_267_cast_fp16 = mul(x = attn_weights_265_cast_fp16, y = var_6252_to_fp16)[name = string("attn_weights_267_cast_fp16")]; tensor attn_weights_269_cast_fp16 = add(x = attn_weights_267_cast_fp16, y = attn_mask_1)[name = string("attn_weights_269_cast_fp16")]; int32 var_6256 = const()[name = string("op_6256"), val = int32(-2)]; tensor attn_weights_271_cast_fp16 = softmax(axis = var_6256, x = attn_weights_269_cast_fp16)[name = string("attn_weights_271_cast_fp16")]; bool attn_output_129_transpose_x_1 = const()[name = string("attn_output_129_transpose_x_1"), val = bool(true)]; bool attn_output_129_transpose_y_1 = const()[name = string("attn_output_129_transpose_y_1"), val = bool(false)]; tensor attn_output_129_cast_fp16 = matmul(transpose_x = attn_output_129_transpose_x_1, transpose_y = attn_output_129_transpose_y_1, x = attn_weights_271_cast_fp16, y = var_6234_cast_fp16_1)[name = string("attn_output_129_cast_fp16")]; int32 var_6264 = const()[name = string("op_6264"), val = int32(1)]; bool attn_output_131_interleave_0 = const()[name = string("attn_output_131_interleave_0"), val = bool(false)]; tensor attn_output_131_cast_fp16 = concat(axis = var_6264, interleave = attn_output_131_interleave_0, values = (var_6250_cast_fp16, attn_output_129_cast_fp16))[name = string("attn_output_131_cast_fp16")]; tensor var_6268_perm_0 = const()[name = string("op_6268_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_203x = const()[name = string("concat_203x"), val = tensor([1, 2048, 1, -1])]; tensor var_6268_cast_fp16 = transpose(perm = var_6268_perm_0, x = attn_output_131_cast_fp16)[name = string("transpose_205")]; tensor attn_output_135_cast_fp16 = reshape(shape = concat_203x, x = var_6268_cast_fp16)[name = string("attn_output_135_cast_fp16")]; tensor hidden_states_163_strides_0 = const()[name = string("hidden_states_163_strides_0"), val = tensor([1, 1])]; string hidden_states_163_pad_type_0 = const()[name = string("hidden_states_163_pad_type_0"), val = string("valid")]; tensor hidden_states_163_pad_0 = const()[name = string("hidden_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_163_dilations_0 = const()[name = string("hidden_states_163_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_163_groups_0 = const()[name = string("hidden_states_163_groups_0"), val = int32(1)]; tensor hidden_states_163_cast_fp16 = conv(dilations = hidden_states_163_dilations_0, groups = hidden_states_163_groups_0, pad = hidden_states_163_pad_0, pad_type = hidden_states_163_pad_type_0, strides = hidden_states_163_strides_0, weight = layers_16_self_attn_o_proj_weight_cast_fp16, x = attn_output_135_cast_fp16)[name = string("hidden_states_163_cast_fp16")]; tensor hidden_states_165_cast_fp16 = add(x = hidden_states_159_cast_fp16, y = hidden_states_163_cast_fp16)[name = string("hidden_states_165_cast_fp16")]; fp16 const_168_promoted_to_fp16 = const()[name = string("const_168_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6301_cast_fp16 = mul(x = hidden_states_165_cast_fp16, y = const_168_promoted_to_fp16)[name = string("op_6301_cast_fp16")]; int32 var_6299 = const()[name = string("op_6299"), val = int32(1)]; bool doubled_133_interleave_0 = const()[name = string("doubled_133_interleave_0"), val = bool(false)]; tensor doubled_133_cast_fp16 = concat(axis = var_6299, interleave = doubled_133_interleave_0, values = (hidden_states_165_cast_fp16, var_6301_cast_fp16))[name = string("doubled_133_cast_fp16")]; tensor out_67_axes_0 = const()[name = string("out_67_axes_0"), val = tensor([1])]; tensor out_67_gamma_0_to_fp16 = const()[name = string("out_67_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441663488)))]; fp16 var_6311_to_fp16 = const()[name = string("op_6311_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_67_cast_fp16 = layer_norm(axes = out_67_axes_0, epsilon = var_6311_to_fp16, gamma = out_67_gamma_0_to_fp16, x = doubled_133_cast_fp16)[name = string("out_67_cast_fp16")]; tensor var_6322_split_sizes_0 = const()[name = string("op_6322_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6322_axis_0 = const()[name = string("op_6322_axis_0"), val = int32(1)]; tensor var_6322_cast_fp16_0, tensor var_6322_cast_fp16_1 = split(axis = var_6322_axis_0, split_sizes = var_6322_split_sizes_0, x = out_67_cast_fp16)[name = string("op_6322_cast_fp16")]; tensor layers_16_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441671744)))]; tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; tensor input_33_cast_fp16 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = layers_16_mlp_gate_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("input_33_cast_fp16")]; tensor var_6339_cast_fp16 = silu(x = input_33_cast_fp16)[name = string("op_6339_cast_fp16")]; tensor layers_16_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1466837632)))]; tensor var_6345_strides_0 = const()[name = string("op_6345_strides_0"), val = tensor([1, 1])]; string var_6345_pad_type_0 = const()[name = string("op_6345_pad_type_0"), val = string("valid")]; tensor var_6345_pad_0 = const()[name = string("op_6345_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6345_dilations_0 = const()[name = string("op_6345_dilations_0"), val = tensor([1, 1])]; int32 var_6345_groups_0 = const()[name = string("op_6345_groups_0"), val = int32(1)]; tensor var_6345_cast_fp16 = conv(dilations = var_6345_dilations_0, groups = var_6345_groups_0, pad = var_6345_pad_0, pad_type = var_6345_pad_type_0, strides = var_6345_strides_0, weight = layers_16_mlp_up_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("op_6345_cast_fp16")]; tensor x_169_cast_fp16 = mul(x = var_6339_cast_fp16, y = var_6345_cast_fp16)[name = string("x_169_cast_fp16")]; tensor hidden_states_167_strides_0 = const()[name = string("hidden_states_167_strides_0"), val = tensor([1, 1])]; string hidden_states_167_pad_type_0 = const()[name = string("hidden_states_167_pad_type_0"), val = string("valid")]; tensor hidden_states_167_pad_0 = const()[name = string("hidden_states_167_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_167_dilations_0 = const()[name = string("hidden_states_167_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_167_groups_0 = const()[name = string("hidden_states_167_groups_0"), val = int32(1)]; tensor hidden_states_167_cast_fp16 = conv(dilations = hidden_states_167_dilations_0, groups = hidden_states_167_groups_0, pad = hidden_states_167_pad_0, pad_type = hidden_states_167_pad_type_0, strides = hidden_states_167_strides_0, weight = layers_16_mlp_down_proj_weight_cast_fp16, x = x_169_cast_fp16)[name = string("hidden_states_167_cast_fp16")]; tensor hidden_states_169_cast_fp16 = add(x = hidden_states_165_cast_fp16, y = hidden_states_167_cast_fp16)[name = string("hidden_states_169_cast_fp16")]; fp16 const_170_promoted_to_fp16 = const()[name = string("const_170_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6363_cast_fp16 = mul(x = hidden_states_169_cast_fp16, y = const_170_promoted_to_fp16)[name = string("op_6363_cast_fp16")]; int32 var_6361 = const()[name = string("op_6361"), val = int32(1)]; bool doubled_137_interleave_0 = const()[name = string("doubled_137_interleave_0"), val = bool(false)]; tensor doubled_137_cast_fp16 = concat(axis = var_6361, interleave = doubled_137_interleave_0, values = (hidden_states_169_cast_fp16, var_6363_cast_fp16))[name = string("doubled_137_cast_fp16")]; tensor out_69_axes_0 = const()[name = string("out_69_axes_0"), val = tensor([1])]; tensor out_69_gamma_0_to_fp16 = const()[name = string("out_69_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492003520)))]; fp16 var_6373_to_fp16 = const()[name = string("op_6373_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_69_cast_fp16 = layer_norm(axes = out_69_axes_0, epsilon = var_6373_to_fp16, gamma = out_69_gamma_0_to_fp16, x = doubled_137_cast_fp16)[name = string("out_69_cast_fp16")]; tensor var_6384_split_sizes_0 = const()[name = string("op_6384_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6384_axis_0 = const()[name = string("op_6384_axis_0"), val = int32(1)]; tensor var_6384_cast_fp16_0, tensor var_6384_cast_fp16_1 = split(axis = var_6384_axis_0, split_sizes = var_6384_split_sizes_0, x = out_69_cast_fp16)[name = string("op_6384_cast_fp16")]; tensor query_states_103_strides_0 = const()[name = string("query_states_103_strides_0"), val = tensor([1, 1])]; string query_states_103_pad_type_0 = const()[name = string("query_states_103_pad_type_0"), val = string("valid")]; tensor query_states_103_pad_0 = const()[name = string("query_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_103_dilations_0 = const()[name = string("query_states_103_dilations_0"), val = tensor([1, 1])]; int32 query_states_103_groups_0 = const()[name = string("query_states_103_groups_0"), val = int32(1)]; tensor query_states_103_cast_fp16 = conv(dilations = query_states_103_dilations_0, groups = query_states_103_groups_0, pad = query_states_103_pad_0, pad_type = query_states_103_pad_type_0, strides = query_states_103_strides_0, weight = layers_17_self_attn_q_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("query_states_103_cast_fp16")]; tensor key_states_171_strides_0 = const()[name = string("key_states_171_strides_0"), val = tensor([1, 1])]; string key_states_171_pad_type_0 = const()[name = string("key_states_171_pad_type_0"), val = string("valid")]; tensor key_states_171_pad_0 = const()[name = string("key_states_171_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_171_dilations_0 = const()[name = string("key_states_171_dilations_0"), val = tensor([1, 1])]; int32 key_states_171_groups_0 = const()[name = string("key_states_171_groups_0"), val = int32(1)]; tensor key_states_171_cast_fp16 = conv(dilations = key_states_171_dilations_0, groups = key_states_171_groups_0, pad = key_states_171_pad_0, pad_type = key_states_171_pad_type_0, strides = key_states_171_strides_0, weight = layers_17_self_attn_k_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("key_states_171_cast_fp16")]; tensor value_states_103_strides_0 = const()[name = string("value_states_103_strides_0"), val = tensor([1, 1])]; string value_states_103_pad_type_0 = const()[name = string("value_states_103_pad_type_0"), val = string("valid")]; tensor value_states_103_pad_0 = const()[name = string("value_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_103_dilations_0 = const()[name = string("value_states_103_dilations_0"), val = tensor([1, 1])]; int32 value_states_103_groups_0 = const()[name = string("value_states_103_groups_0"), val = int32(1)]; tensor value_states_103_cast_fp16 = conv(dilations = value_states_103_dilations_0, groups = value_states_103_groups_0, pad = value_states_103_pad_0, pad_type = value_states_103_pad_type_0, strides = value_states_103_strides_0, weight = layers_17_self_attn_v_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("value_states_103_cast_fp16")]; tensor concat_204x = const()[name = string("concat_204x"), val = tensor([1, 16, 128, -1])]; tensor x_171_cast_fp16 = reshape(shape = concat_204x, x = query_states_103_cast_fp16)[name = string("x_171_cast_fp16")]; tensor concat_205x = const()[name = string("concat_205x"), val = tensor([1, 2, 128, -1])]; tensor var_6441_cast_fp16 = reshape(shape = concat_205x, x = key_states_171_cast_fp16)[name = string("op_6441_cast_fp16")]; tensor concat_206x = const()[name = string("concat_206x"), val = tensor([1, 2, 128, -1])]; tensor var_6448_cast_fp16 = reshape(shape = concat_206x, x = value_states_103_cast_fp16)[name = string("op_6448_cast_fp16")]; tensor var_6452_cast_fp16 = mul(x = x_171_cast_fp16, y = var_869_cast_fp16)[name = string("op_6452_cast_fp16")]; tensor var_6453_split_sizes_0 = const()[name = string("op_6453_split_sizes_0"), val = tensor([64, 64])]; int32 var_6453_axis_0 = const()[name = string("op_6453_axis_0"), val = int32(-2)]; tensor var_6453_cast_fp16_0, tensor var_6453_cast_fp16_1 = split(axis = var_6453_axis_0, split_sizes = var_6453_split_sizes_0, x = x_171_cast_fp16)[name = string("op_6453_cast_fp16")]; fp16 const_172_promoted_to_fp16 = const()[name = string("const_172_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6455_cast_fp16 = mul(x = var_6453_cast_fp16_1, y = const_172_promoted_to_fp16)[name = string("op_6455_cast_fp16")]; int32 var_6457 = const()[name = string("op_6457"), val = int32(-2)]; bool var_6458_interleave_0 = const()[name = string("op_6458_interleave_0"), val = bool(false)]; tensor var_6458_cast_fp16 = concat(axis = var_6457, interleave = var_6458_interleave_0, values = (var_6455_cast_fp16, var_6453_cast_fp16_0))[name = string("op_6458_cast_fp16")]; tensor var_6459_cast_fp16 = mul(x = var_6458_cast_fp16, y = var_878_cast_fp16)[name = string("op_6459_cast_fp16")]; tensor query_states_105_cast_fp16 = add(x = var_6452_cast_fp16, y = var_6459_cast_fp16)[name = string("query_states_105_cast_fp16")]; tensor var_6465_cast_fp16 = mul(x = var_6441_cast_fp16, y = var_869_cast_fp16)[name = string("op_6465_cast_fp16")]; tensor var_6466_split_sizes_0 = const()[name = string("op_6466_split_sizes_0"), val = tensor([64, 64])]; int32 var_6466_axis_0 = const()[name = string("op_6466_axis_0"), val = int32(-2)]; tensor var_6466_cast_fp16_0, tensor var_6466_cast_fp16_1 = split(axis = var_6466_axis_0, split_sizes = var_6466_split_sizes_0, x = var_6441_cast_fp16)[name = string("op_6466_cast_fp16")]; fp16 const_173_promoted_to_fp16 = const()[name = string("const_173_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6468_cast_fp16 = mul(x = var_6466_cast_fp16_1, y = const_173_promoted_to_fp16)[name = string("op_6468_cast_fp16")]; int32 var_6470 = const()[name = string("op_6470"), val = int32(-2)]; bool var_6471_interleave_0 = const()[name = string("op_6471_interleave_0"), val = bool(false)]; tensor var_6471_cast_fp16 = concat(axis = var_6470, interleave = var_6471_interleave_0, values = (var_6468_cast_fp16, var_6466_cast_fp16_0))[name = string("op_6471_cast_fp16")]; tensor var_6472_cast_fp16 = mul(x = var_6471_cast_fp16, y = var_878_cast_fp16)[name = string("op_6472_cast_fp16")]; tensor key_states_175_cast_fp16 = add(x = var_6465_cast_fp16, y = var_6472_cast_fp16)[name = string("key_states_175_cast_fp16")]; tensor expand_dims_204 = const()[name = string("expand_dims_204"), val = tensor([17])]; tensor expand_dims_205 = const()[name = string("expand_dims_205"), val = tensor([0])]; tensor expand_dims_207 = const()[name = string("expand_dims_207"), val = tensor([0])]; int32 concat_209_axis_0 = const()[name = string("concat_209_axis_0"), val = int32(0)]; bool concat_209_interleave_0 = const()[name = string("concat_209_interleave_0"), val = bool(false)]; tensor concat_209 = concat(axis = concat_209_axis_0, interleave = concat_209_interleave_0, values = (expand_dims_204, expand_dims_205, position_id, expand_dims_207))[name = string("concat_209")]; tensor expand_dims_208 = const()[name = string("expand_dims_208"), val = tensor([18])]; tensor concat_210_values1_0 = const()[name = string("concat_210_values1_0"), val = tensor([0])]; tensor concat_210_values3_0 = const()[name = string("concat_210_values3_0"), val = tensor([0])]; int32 concat_210_axis_0 = const()[name = string("concat_210_axis_0"), val = int32(0)]; bool concat_210_interleave_0 = const()[name = string("concat_210_interleave_0"), val = bool(false)]; tensor concat_210 = concat(axis = concat_210_axis_0, interleave = concat_210_interleave_0, values = (expand_dims_208, concat_210_values1_0, cache_position_end, concat_210_values3_0))[name = string("concat_210")]; tensor key_states_177_perm_0 = const()[name = string("key_states_177_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_18_stride_0 = const()[name = string("key_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_177_cast_fp16 = transpose(perm = key_states_177_perm_0, x = key_states_175_cast_fp16)[name = string("transpose_204")]; tensor key_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = key_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = key_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_18_squeeze_mask_0, stride = key_cache_internal_tensor_assign_18_stride_0, update = key_states_177_cast_fp16, x = coreml_update_state_144)[name = string("key_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_18_cast_fp16, input = key_cache)[name = string("coreml_update_state_146_write_state")]; tensor coreml_update_state_146 = read_state(input = key_cache)[name = string("coreml_update_state_146")]; tensor value_states_105_perm_0 = const()[name = string("value_states_105_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_18_stride_0 = const()[name = string("value_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_105_cast_fp16 = transpose(perm = value_states_105_perm_0, x = var_6448_cast_fp16)[name = string("transpose_203")]; tensor value_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = value_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = value_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_18_squeeze_mask_0, stride = value_cache_internal_tensor_assign_18_stride_0, update = value_states_105_cast_fp16, x = coreml_update_state_145)[name = string("value_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_18_cast_fp16, input = value_cache)[name = string("coreml_update_state_147_write_state")]; tensor coreml_update_state_147 = read_state(input = value_cache)[name = string("coreml_update_state_147")]; tensor var_6542_begin_0 = const()[name = string("op_6542_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6542_end_0 = const()[name = string("op_6542_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6542_end_mask_0 = const()[name = string("op_6542_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6542_cast_fp16 = slice_by_index(begin = var_6542_begin_0, end = var_6542_end_0, end_mask = var_6542_end_mask_0, x = coreml_update_state_146)[name = string("op_6542_cast_fp16")]; tensor tile_34 = const()[name = string("tile_34"), val = tensor([1, 1])]; int32 var_6545_axis_0 = const()[name = string("op_6545_axis_0"), val = int32(1)]; tensor var_6545_cast_fp16_0, tensor var_6545_cast_fp16_1 = split(axis = var_6545_axis_0, split_sizes = tile_34, x = var_6542_cast_fp16)[name = string("op_6545_cast_fp16")]; tensor var_6552_begin_0 = const()[name = string("op_6552_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6552_end_0 = const()[name = string("op_6552_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6552_end_mask_0 = const()[name = string("op_6552_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6552_cast_fp16 = slice_by_index(begin = var_6552_begin_0, end = var_6552_end_0, end_mask = var_6552_end_mask_0, x = coreml_update_state_147)[name = string("op_6552_cast_fp16")]; tensor tile_35 = const()[name = string("tile_35"), val = tensor([1, 1])]; int32 var_6555_axis_0 = const()[name = string("op_6555_axis_0"), val = int32(1)]; tensor var_6555_cast_fp16_0, tensor var_6555_cast_fp16_1 = split(axis = var_6555_axis_0, split_sizes = tile_35, x = var_6552_cast_fp16)[name = string("op_6555_cast_fp16")]; tensor var_6558_split_sizes_0 = const()[name = string("op_6558_split_sizes_0"), val = tensor([8, 8])]; int32 var_6558_axis_0 = const()[name = string("op_6558_axis_0"), val = int32(1)]; tensor var_6558_0, tensor var_6558_1 = split(axis = var_6558_axis_0, split_sizes = var_6558_split_sizes_0, x = query_states_105_cast_fp16)[name = string("op_6558")]; bool attn_weights_273_transpose_x_0 = const()[name = string("attn_weights_273_transpose_x_0"), val = bool(false)]; bool attn_weights_273_transpose_y_0 = const()[name = string("attn_weights_273_transpose_y_0"), val = bool(false)]; tensor attn_weights_273_cast_fp16 = matmul(transpose_x = attn_weights_273_transpose_x_0, transpose_y = attn_weights_273_transpose_y_0, x = var_6545_cast_fp16_0, y = var_6558_0)[name = string("attn_weights_273_cast_fp16")]; fp16 var_6561_to_fp16 = const()[name = string("op_6561_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_275_cast_fp16 = mul(x = attn_weights_273_cast_fp16, y = var_6561_to_fp16)[name = string("attn_weights_275_cast_fp16")]; tensor attn_weights_277_cast_fp16 = add(x = attn_weights_275_cast_fp16, y = attn_mask_1)[name = string("attn_weights_277_cast_fp16")]; int32 var_6565 = const()[name = string("op_6565"), val = int32(-2)]; tensor attn_weights_279_cast_fp16 = softmax(axis = var_6565, x = attn_weights_277_cast_fp16)[name = string("attn_weights_279_cast_fp16")]; bool var_6571_transpose_x_1 = const()[name = string("op_6571_transpose_x_1"), val = bool(true)]; bool var_6571_transpose_y_1 = const()[name = string("op_6571_transpose_y_1"), val = bool(false)]; tensor var_6571_cast_fp16 = matmul(transpose_x = var_6571_transpose_x_1, transpose_y = var_6571_transpose_y_1, x = attn_weights_279_cast_fp16, y = var_6555_cast_fp16_0)[name = string("op_6571_cast_fp16")]; bool attn_weights_281_transpose_x_0 = const()[name = string("attn_weights_281_transpose_x_0"), val = bool(false)]; bool attn_weights_281_transpose_y_0 = const()[name = string("attn_weights_281_transpose_y_0"), val = bool(false)]; tensor attn_weights_281_cast_fp16 = matmul(transpose_x = attn_weights_281_transpose_x_0, transpose_y = attn_weights_281_transpose_y_0, x = var_6545_cast_fp16_1, y = var_6558_1)[name = string("attn_weights_281_cast_fp16")]; fp16 var_6573_to_fp16 = const()[name = string("op_6573_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_283_cast_fp16 = mul(x = attn_weights_281_cast_fp16, y = var_6573_to_fp16)[name = string("attn_weights_283_cast_fp16")]; tensor attn_weights_285_cast_fp16 = add(x = attn_weights_283_cast_fp16, y = attn_mask_1)[name = string("attn_weights_285_cast_fp16")]; int32 var_6577 = const()[name = string("op_6577"), val = int32(-2)]; tensor attn_weights_287_cast_fp16 = softmax(axis = var_6577, x = attn_weights_285_cast_fp16)[name = string("attn_weights_287_cast_fp16")]; bool attn_output_137_transpose_x_1 = const()[name = string("attn_output_137_transpose_x_1"), val = bool(true)]; bool attn_output_137_transpose_y_1 = const()[name = string("attn_output_137_transpose_y_1"), val = bool(false)]; tensor attn_output_137_cast_fp16 = matmul(transpose_x = attn_output_137_transpose_x_1, transpose_y = attn_output_137_transpose_y_1, x = attn_weights_287_cast_fp16, y = var_6555_cast_fp16_1)[name = string("attn_output_137_cast_fp16")]; int32 var_6585 = const()[name = string("op_6585"), val = int32(1)]; bool attn_output_139_interleave_0 = const()[name = string("attn_output_139_interleave_0"), val = bool(false)]; tensor attn_output_139_cast_fp16 = concat(axis = var_6585, interleave = attn_output_139_interleave_0, values = (var_6571_cast_fp16, attn_output_137_cast_fp16))[name = string("attn_output_139_cast_fp16")]; tensor var_6589_perm_0 = const()[name = string("op_6589_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_215x = const()[name = string("concat_215x"), val = tensor([1, 2048, 1, -1])]; tensor var_6589_cast_fp16 = transpose(perm = var_6589_perm_0, x = attn_output_139_cast_fp16)[name = string("transpose_202")]; tensor attn_output_143_cast_fp16 = reshape(shape = concat_215x, x = var_6589_cast_fp16)[name = string("attn_output_143_cast_fp16")]; tensor hidden_states_173_strides_0 = const()[name = string("hidden_states_173_strides_0"), val = tensor([1, 1])]; string hidden_states_173_pad_type_0 = const()[name = string("hidden_states_173_pad_type_0"), val = string("valid")]; tensor hidden_states_173_pad_0 = const()[name = string("hidden_states_173_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_173_dilations_0 = const()[name = string("hidden_states_173_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_173_groups_0 = const()[name = string("hidden_states_173_groups_0"), val = int32(1)]; tensor hidden_states_173_cast_fp16 = conv(dilations = hidden_states_173_dilations_0, groups = hidden_states_173_groups_0, pad = hidden_states_173_pad_0, pad_type = hidden_states_173_pad_type_0, strides = hidden_states_173_strides_0, weight = layers_17_self_attn_o_proj_weight_cast_fp16, x = attn_output_143_cast_fp16)[name = string("hidden_states_173_cast_fp16")]; tensor hidden_states_175_cast_fp16 = add(x = hidden_states_169_cast_fp16, y = hidden_states_173_cast_fp16)[name = string("hidden_states_175_cast_fp16")]; fp16 const_178_promoted_to_fp16 = const()[name = string("const_178_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6622_cast_fp16 = mul(x = hidden_states_175_cast_fp16, y = const_178_promoted_to_fp16)[name = string("op_6622_cast_fp16")]; int32 var_6620 = const()[name = string("op_6620"), val = int32(1)]; bool doubled_141_interleave_0 = const()[name = string("doubled_141_interleave_0"), val = bool(false)]; tensor doubled_141_cast_fp16 = concat(axis = var_6620, interleave = doubled_141_interleave_0, values = (hidden_states_175_cast_fp16, var_6622_cast_fp16))[name = string("doubled_141_cast_fp16")]; tensor out_71_axes_0 = const()[name = string("out_71_axes_0"), val = tensor([1])]; tensor out_71_gamma_0_to_fp16 = const()[name = string("out_71_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492011776)))]; fp16 var_6632_to_fp16 = const()[name = string("op_6632_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_71_cast_fp16 = layer_norm(axes = out_71_axes_0, epsilon = var_6632_to_fp16, gamma = out_71_gamma_0_to_fp16, x = doubled_141_cast_fp16)[name = string("out_71_cast_fp16")]; tensor var_6643_split_sizes_0 = const()[name = string("op_6643_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6643_axis_0 = const()[name = string("op_6643_axis_0"), val = int32(1)]; tensor var_6643_cast_fp16_0, tensor var_6643_cast_fp16_1 = split(axis = var_6643_axis_0, split_sizes = var_6643_split_sizes_0, x = out_71_cast_fp16)[name = string("op_6643_cast_fp16")]; tensor input_35_strides_0 = const()[name = string("input_35_strides_0"), val = tensor([1, 1])]; string input_35_pad_type_0 = const()[name = string("input_35_pad_type_0"), val = string("valid")]; tensor input_35_pad_0 = const()[name = string("input_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_35_dilations_0 = const()[name = string("input_35_dilations_0"), val = tensor([1, 1])]; int32 input_35_groups_0 = const()[name = string("input_35_groups_0"), val = int32(1)]; tensor input_35_cast_fp16 = conv(dilations = input_35_dilations_0, groups = input_35_groups_0, pad = input_35_pad_0, pad_type = input_35_pad_type_0, strides = input_35_strides_0, weight = layers_17_mlp_gate_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("input_35_cast_fp16")]; tensor var_6660_cast_fp16 = silu(x = input_35_cast_fp16)[name = string("op_6660_cast_fp16")]; tensor var_6666_strides_0 = const()[name = string("op_6666_strides_0"), val = tensor([1, 1])]; string var_6666_pad_type_0 = const()[name = string("op_6666_pad_type_0"), val = string("valid")]; tensor var_6666_pad_0 = const()[name = string("op_6666_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6666_dilations_0 = const()[name = string("op_6666_dilations_0"), val = tensor([1, 1])]; int32 var_6666_groups_0 = const()[name = string("op_6666_groups_0"), val = int32(1)]; tensor var_6666_cast_fp16 = conv(dilations = var_6666_dilations_0, groups = var_6666_groups_0, pad = var_6666_pad_0, pad_type = var_6666_pad_type_0, strides = var_6666_strides_0, weight = layers_17_mlp_up_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("op_6666_cast_fp16")]; tensor x_179_cast_fp16 = mul(x = var_6660_cast_fp16, y = var_6666_cast_fp16)[name = string("x_179_cast_fp16")]; tensor hidden_states_177_strides_0 = const()[name = string("hidden_states_177_strides_0"), val = tensor([1, 1])]; string hidden_states_177_pad_type_0 = const()[name = string("hidden_states_177_pad_type_0"), val = string("valid")]; tensor hidden_states_177_pad_0 = const()[name = string("hidden_states_177_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_177_dilations_0 = const()[name = string("hidden_states_177_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_177_groups_0 = const()[name = string("hidden_states_177_groups_0"), val = int32(1)]; tensor hidden_states_177_cast_fp16 = conv(dilations = hidden_states_177_dilations_0, groups = hidden_states_177_groups_0, pad = hidden_states_177_pad_0, pad_type = hidden_states_177_pad_type_0, strides = hidden_states_177_strides_0, weight = layers_17_mlp_down_proj_weight_cast_fp16, x = x_179_cast_fp16)[name = string("hidden_states_177_cast_fp16")]; tensor hidden_states_179_cast_fp16 = add(x = hidden_states_175_cast_fp16, y = hidden_states_177_cast_fp16)[name = string("hidden_states_179_cast_fp16")]; fp16 const_180_promoted_to_fp16 = const()[name = string("const_180_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6684_cast_fp16 = mul(x = hidden_states_179_cast_fp16, y = const_180_promoted_to_fp16)[name = string("op_6684_cast_fp16")]; int32 var_6682 = const()[name = string("op_6682"), val = int32(1)]; bool doubled_145_interleave_0 = const()[name = string("doubled_145_interleave_0"), val = bool(false)]; tensor doubled_145_cast_fp16 = concat(axis = var_6682, interleave = doubled_145_interleave_0, values = (hidden_states_179_cast_fp16, var_6684_cast_fp16))[name = string("doubled_145_cast_fp16")]; tensor out_73_axes_0 = const()[name = string("out_73_axes_0"), val = tensor([1])]; tensor out_73_gamma_0_to_fp16 = const()[name = string("out_73_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492020032)))]; fp16 var_6694_to_fp16 = const()[name = string("op_6694_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_73_cast_fp16 = layer_norm(axes = out_73_axes_0, epsilon = var_6694_to_fp16, gamma = out_73_gamma_0_to_fp16, x = doubled_145_cast_fp16)[name = string("out_73_cast_fp16")]; tensor var_6705_split_sizes_0 = const()[name = string("op_6705_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6705_axis_0 = const()[name = string("op_6705_axis_0"), val = int32(1)]; tensor var_6705_cast_fp16_0, tensor var_6705_cast_fp16_1 = split(axis = var_6705_axis_0, split_sizes = var_6705_split_sizes_0, x = out_73_cast_fp16)[name = string("op_6705_cast_fp16")]; tensor query_states_109_strides_0 = const()[name = string("query_states_109_strides_0"), val = tensor([1, 1])]; string query_states_109_pad_type_0 = const()[name = string("query_states_109_pad_type_0"), val = string("valid")]; tensor query_states_109_pad_0 = const()[name = string("query_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_109_dilations_0 = const()[name = string("query_states_109_dilations_0"), val = tensor([1, 1])]; int32 query_states_109_groups_0 = const()[name = string("query_states_109_groups_0"), val = int32(1)]; tensor query_states_109_cast_fp16 = conv(dilations = query_states_109_dilations_0, groups = query_states_109_groups_0, pad = query_states_109_pad_0, pad_type = query_states_109_pad_type_0, strides = query_states_109_strides_0, weight = layers_18_self_attn_q_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("query_states_109_cast_fp16")]; tensor key_states_181_strides_0 = const()[name = string("key_states_181_strides_0"), val = tensor([1, 1])]; string key_states_181_pad_type_0 = const()[name = string("key_states_181_pad_type_0"), val = string("valid")]; tensor key_states_181_pad_0 = const()[name = string("key_states_181_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_181_dilations_0 = const()[name = string("key_states_181_dilations_0"), val = tensor([1, 1])]; int32 key_states_181_groups_0 = const()[name = string("key_states_181_groups_0"), val = int32(1)]; tensor key_states_181_cast_fp16 = conv(dilations = key_states_181_dilations_0, groups = key_states_181_groups_0, pad = key_states_181_pad_0, pad_type = key_states_181_pad_type_0, strides = key_states_181_strides_0, weight = layers_18_self_attn_k_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("key_states_181_cast_fp16")]; tensor value_states_109_strides_0 = const()[name = string("value_states_109_strides_0"), val = tensor([1, 1])]; string value_states_109_pad_type_0 = const()[name = string("value_states_109_pad_type_0"), val = string("valid")]; tensor value_states_109_pad_0 = const()[name = string("value_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_109_dilations_0 = const()[name = string("value_states_109_dilations_0"), val = tensor([1, 1])]; int32 value_states_109_groups_0 = const()[name = string("value_states_109_groups_0"), val = int32(1)]; tensor value_states_109_cast_fp16 = conv(dilations = value_states_109_dilations_0, groups = value_states_109_groups_0, pad = value_states_109_pad_0, pad_type = value_states_109_pad_type_0, strides = value_states_109_strides_0, weight = layers_18_self_attn_v_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("value_states_109_cast_fp16")]; tensor concat_216x = const()[name = string("concat_216x"), val = tensor([1, 16, 128, -1])]; tensor x_181_cast_fp16 = reshape(shape = concat_216x, x = query_states_109_cast_fp16)[name = string("x_181_cast_fp16")]; tensor concat_217x = const()[name = string("concat_217x"), val = tensor([1, 2, 128, -1])]; tensor var_6762_cast_fp16 = reshape(shape = concat_217x, x = key_states_181_cast_fp16)[name = string("op_6762_cast_fp16")]; tensor concat_218x = const()[name = string("concat_218x"), val = tensor([1, 2, 128, -1])]; tensor var_6769_cast_fp16 = reshape(shape = concat_218x, x = value_states_109_cast_fp16)[name = string("op_6769_cast_fp16")]; tensor var_6773_cast_fp16 = mul(x = x_181_cast_fp16, y = var_869_cast_fp16)[name = string("op_6773_cast_fp16")]; tensor var_6774_split_sizes_0 = const()[name = string("op_6774_split_sizes_0"), val = tensor([64, 64])]; int32 var_6774_axis_0 = const()[name = string("op_6774_axis_0"), val = int32(-2)]; tensor var_6774_cast_fp16_0, tensor var_6774_cast_fp16_1 = split(axis = var_6774_axis_0, split_sizes = var_6774_split_sizes_0, x = x_181_cast_fp16)[name = string("op_6774_cast_fp16")]; fp16 const_182_promoted_to_fp16 = const()[name = string("const_182_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6776_cast_fp16 = mul(x = var_6774_cast_fp16_1, y = const_182_promoted_to_fp16)[name = string("op_6776_cast_fp16")]; int32 var_6778 = const()[name = string("op_6778"), val = int32(-2)]; bool var_6779_interleave_0 = const()[name = string("op_6779_interleave_0"), val = bool(false)]; tensor var_6779_cast_fp16 = concat(axis = var_6778, interleave = var_6779_interleave_0, values = (var_6776_cast_fp16, var_6774_cast_fp16_0))[name = string("op_6779_cast_fp16")]; tensor var_6780_cast_fp16 = mul(x = var_6779_cast_fp16, y = var_878_cast_fp16)[name = string("op_6780_cast_fp16")]; tensor query_states_111_cast_fp16 = add(x = var_6773_cast_fp16, y = var_6780_cast_fp16)[name = string("query_states_111_cast_fp16")]; tensor var_6786_cast_fp16 = mul(x = var_6762_cast_fp16, y = var_869_cast_fp16)[name = string("op_6786_cast_fp16")]; tensor var_6787_split_sizes_0 = const()[name = string("op_6787_split_sizes_0"), val = tensor([64, 64])]; int32 var_6787_axis_0 = const()[name = string("op_6787_axis_0"), val = int32(-2)]; tensor var_6787_cast_fp16_0, tensor var_6787_cast_fp16_1 = split(axis = var_6787_axis_0, split_sizes = var_6787_split_sizes_0, x = var_6762_cast_fp16)[name = string("op_6787_cast_fp16")]; fp16 const_183_promoted_to_fp16 = const()[name = string("const_183_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6789_cast_fp16 = mul(x = var_6787_cast_fp16_1, y = const_183_promoted_to_fp16)[name = string("op_6789_cast_fp16")]; int32 var_6791 = const()[name = string("op_6791"), val = int32(-2)]; bool var_6792_interleave_0 = const()[name = string("op_6792_interleave_0"), val = bool(false)]; tensor var_6792_cast_fp16 = concat(axis = var_6791, interleave = var_6792_interleave_0, values = (var_6789_cast_fp16, var_6787_cast_fp16_0))[name = string("op_6792_cast_fp16")]; tensor var_6793_cast_fp16 = mul(x = var_6792_cast_fp16, y = var_878_cast_fp16)[name = string("op_6793_cast_fp16")]; tensor key_states_185_cast_fp16 = add(x = var_6786_cast_fp16, y = var_6793_cast_fp16)[name = string("key_states_185_cast_fp16")]; tensor expand_dims_216 = const()[name = string("expand_dims_216"), val = tensor([18])]; tensor expand_dims_217 = const()[name = string("expand_dims_217"), val = tensor([0])]; tensor expand_dims_219 = const()[name = string("expand_dims_219"), val = tensor([0])]; int32 concat_221_axis_0 = const()[name = string("concat_221_axis_0"), val = int32(0)]; bool concat_221_interleave_0 = const()[name = string("concat_221_interleave_0"), val = bool(false)]; tensor concat_221 = concat(axis = concat_221_axis_0, interleave = concat_221_interleave_0, values = (expand_dims_216, expand_dims_217, position_id, expand_dims_219))[name = string("concat_221")]; tensor expand_dims_220 = const()[name = string("expand_dims_220"), val = tensor([19])]; tensor concat_222_values1_0 = const()[name = string("concat_222_values1_0"), val = tensor([0])]; tensor concat_222_values3_0 = const()[name = string("concat_222_values3_0"), val = tensor([0])]; int32 concat_222_axis_0 = const()[name = string("concat_222_axis_0"), val = int32(0)]; bool concat_222_interleave_0 = const()[name = string("concat_222_interleave_0"), val = bool(false)]; tensor concat_222 = concat(axis = concat_222_axis_0, interleave = concat_222_interleave_0, values = (expand_dims_220, concat_222_values1_0, cache_position_end, concat_222_values3_0))[name = string("concat_222")]; tensor key_states_187_perm_0 = const()[name = string("key_states_187_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_19_stride_0 = const()[name = string("key_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_187_cast_fp16 = transpose(perm = key_states_187_perm_0, x = key_states_185_cast_fp16)[name = string("transpose_201")]; tensor key_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = key_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = key_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_19_squeeze_mask_0, stride = key_cache_internal_tensor_assign_19_stride_0, update = key_states_187_cast_fp16, x = coreml_update_state_146)[name = string("key_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_19_cast_fp16, input = key_cache)[name = string("coreml_update_state_148_write_state")]; tensor coreml_update_state_148 = read_state(input = key_cache)[name = string("coreml_update_state_148")]; tensor value_states_111_perm_0 = const()[name = string("value_states_111_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_19_stride_0 = const()[name = string("value_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_111_cast_fp16 = transpose(perm = value_states_111_perm_0, x = var_6769_cast_fp16)[name = string("transpose_200")]; tensor value_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = value_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = value_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_19_squeeze_mask_0, stride = value_cache_internal_tensor_assign_19_stride_0, update = value_states_111_cast_fp16, x = coreml_update_state_147)[name = string("value_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_19_cast_fp16, input = value_cache)[name = string("coreml_update_state_149_write_state")]; tensor coreml_update_state_149 = read_state(input = value_cache)[name = string("coreml_update_state_149")]; tensor var_6863_begin_0 = const()[name = string("op_6863_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6863_end_0 = const()[name = string("op_6863_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6863_end_mask_0 = const()[name = string("op_6863_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6863_cast_fp16 = slice_by_index(begin = var_6863_begin_0, end = var_6863_end_0, end_mask = var_6863_end_mask_0, x = coreml_update_state_148)[name = string("op_6863_cast_fp16")]; tensor tile_36 = const()[name = string("tile_36"), val = tensor([1, 1])]; int32 var_6866_axis_0 = const()[name = string("op_6866_axis_0"), val = int32(1)]; tensor var_6866_cast_fp16_0, tensor var_6866_cast_fp16_1 = split(axis = var_6866_axis_0, split_sizes = tile_36, x = var_6863_cast_fp16)[name = string("op_6866_cast_fp16")]; tensor var_6873_begin_0 = const()[name = string("op_6873_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6873_end_0 = const()[name = string("op_6873_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6873_end_mask_0 = const()[name = string("op_6873_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6873_cast_fp16 = slice_by_index(begin = var_6873_begin_0, end = var_6873_end_0, end_mask = var_6873_end_mask_0, x = coreml_update_state_149)[name = string("op_6873_cast_fp16")]; tensor tile_37 = const()[name = string("tile_37"), val = tensor([1, 1])]; int32 var_6876_axis_0 = const()[name = string("op_6876_axis_0"), val = int32(1)]; tensor var_6876_cast_fp16_0, tensor var_6876_cast_fp16_1 = split(axis = var_6876_axis_0, split_sizes = tile_37, x = var_6873_cast_fp16)[name = string("op_6876_cast_fp16")]; tensor var_6879_split_sizes_0 = const()[name = string("op_6879_split_sizes_0"), val = tensor([8, 8])]; int32 var_6879_axis_0 = const()[name = string("op_6879_axis_0"), val = int32(1)]; tensor var_6879_0, tensor var_6879_1 = split(axis = var_6879_axis_0, split_sizes = var_6879_split_sizes_0, x = query_states_111_cast_fp16)[name = string("op_6879")]; bool attn_weights_289_transpose_x_0 = const()[name = string("attn_weights_289_transpose_x_0"), val = bool(false)]; bool attn_weights_289_transpose_y_0 = const()[name = string("attn_weights_289_transpose_y_0"), val = bool(false)]; tensor attn_weights_289_cast_fp16 = matmul(transpose_x = attn_weights_289_transpose_x_0, transpose_y = attn_weights_289_transpose_y_0, x = var_6866_cast_fp16_0, y = var_6879_0)[name = string("attn_weights_289_cast_fp16")]; fp16 var_6882_to_fp16 = const()[name = string("op_6882_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_291_cast_fp16 = mul(x = attn_weights_289_cast_fp16, y = var_6882_to_fp16)[name = string("attn_weights_291_cast_fp16")]; tensor attn_weights_293_cast_fp16 = add(x = attn_weights_291_cast_fp16, y = attn_mask_1)[name = string("attn_weights_293_cast_fp16")]; int32 var_6886 = const()[name = string("op_6886"), val = int32(-2)]; tensor attn_weights_295_cast_fp16 = softmax(axis = var_6886, x = attn_weights_293_cast_fp16)[name = string("attn_weights_295_cast_fp16")]; bool var_6892_transpose_x_1 = const()[name = string("op_6892_transpose_x_1"), val = bool(true)]; bool var_6892_transpose_y_1 = const()[name = string("op_6892_transpose_y_1"), val = bool(false)]; tensor var_6892_cast_fp16 = matmul(transpose_x = var_6892_transpose_x_1, transpose_y = var_6892_transpose_y_1, x = attn_weights_295_cast_fp16, y = var_6876_cast_fp16_0)[name = string("op_6892_cast_fp16")]; bool attn_weights_297_transpose_x_0 = const()[name = string("attn_weights_297_transpose_x_0"), val = bool(false)]; bool attn_weights_297_transpose_y_0 = const()[name = string("attn_weights_297_transpose_y_0"), val = bool(false)]; tensor attn_weights_297_cast_fp16 = matmul(transpose_x = attn_weights_297_transpose_x_0, transpose_y = attn_weights_297_transpose_y_0, x = var_6866_cast_fp16_1, y = var_6879_1)[name = string("attn_weights_297_cast_fp16")]; fp16 var_6894_to_fp16 = const()[name = string("op_6894_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_299_cast_fp16 = mul(x = attn_weights_297_cast_fp16, y = var_6894_to_fp16)[name = string("attn_weights_299_cast_fp16")]; tensor attn_weights_301_cast_fp16 = add(x = attn_weights_299_cast_fp16, y = attn_mask_1)[name = string("attn_weights_301_cast_fp16")]; int32 var_6898 = const()[name = string("op_6898"), val = int32(-2)]; tensor attn_weights_303_cast_fp16 = softmax(axis = var_6898, x = attn_weights_301_cast_fp16)[name = string("attn_weights_303_cast_fp16")]; bool attn_output_145_transpose_x_1 = const()[name = string("attn_output_145_transpose_x_1"), val = bool(true)]; bool attn_output_145_transpose_y_1 = const()[name = string("attn_output_145_transpose_y_1"), val = bool(false)]; tensor attn_output_145_cast_fp16 = matmul(transpose_x = attn_output_145_transpose_x_1, transpose_y = attn_output_145_transpose_y_1, x = attn_weights_303_cast_fp16, y = var_6876_cast_fp16_1)[name = string("attn_output_145_cast_fp16")]; int32 var_6906 = const()[name = string("op_6906"), val = int32(1)]; bool attn_output_147_interleave_0 = const()[name = string("attn_output_147_interleave_0"), val = bool(false)]; tensor attn_output_147_cast_fp16 = concat(axis = var_6906, interleave = attn_output_147_interleave_0, values = (var_6892_cast_fp16, attn_output_145_cast_fp16))[name = string("attn_output_147_cast_fp16")]; tensor var_6910_perm_0 = const()[name = string("op_6910_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_227x = const()[name = string("concat_227x"), val = tensor([1, 2048, 1, -1])]; tensor var_6910_cast_fp16 = transpose(perm = var_6910_perm_0, x = attn_output_147_cast_fp16)[name = string("transpose_199")]; tensor attn_output_151_cast_fp16 = reshape(shape = concat_227x, x = var_6910_cast_fp16)[name = string("attn_output_151_cast_fp16")]; tensor hidden_states_183_strides_0 = const()[name = string("hidden_states_183_strides_0"), val = tensor([1, 1])]; string hidden_states_183_pad_type_0 = const()[name = string("hidden_states_183_pad_type_0"), val = string("valid")]; tensor hidden_states_183_pad_0 = const()[name = string("hidden_states_183_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_183_dilations_0 = const()[name = string("hidden_states_183_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_183_groups_0 = const()[name = string("hidden_states_183_groups_0"), val = int32(1)]; tensor hidden_states_183_cast_fp16 = conv(dilations = hidden_states_183_dilations_0, groups = hidden_states_183_groups_0, pad = hidden_states_183_pad_0, pad_type = hidden_states_183_pad_type_0, strides = hidden_states_183_strides_0, weight = layers_18_self_attn_o_proj_weight_cast_fp16, x = attn_output_151_cast_fp16)[name = string("hidden_states_183_cast_fp16")]; tensor hidden_states_185_cast_fp16 = add(x = hidden_states_179_cast_fp16, y = hidden_states_183_cast_fp16)[name = string("hidden_states_185_cast_fp16")]; fp16 const_188_promoted_to_fp16 = const()[name = string("const_188_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6943_cast_fp16 = mul(x = hidden_states_185_cast_fp16, y = const_188_promoted_to_fp16)[name = string("op_6943_cast_fp16")]; int32 var_6941 = const()[name = string("op_6941"), val = int32(1)]; bool doubled_149_interleave_0 = const()[name = string("doubled_149_interleave_0"), val = bool(false)]; tensor doubled_149_cast_fp16 = concat(axis = var_6941, interleave = doubled_149_interleave_0, values = (hidden_states_185_cast_fp16, var_6943_cast_fp16))[name = string("doubled_149_cast_fp16")]; tensor out_75_axes_0 = const()[name = string("out_75_axes_0"), val = tensor([1])]; tensor out_75_gamma_0_to_fp16 = const()[name = string("out_75_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492028288)))]; fp16 var_6953_to_fp16 = const()[name = string("op_6953_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_75_cast_fp16 = layer_norm(axes = out_75_axes_0, epsilon = var_6953_to_fp16, gamma = out_75_gamma_0_to_fp16, x = doubled_149_cast_fp16)[name = string("out_75_cast_fp16")]; tensor var_6964_split_sizes_0 = const()[name = string("op_6964_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6964_axis_0 = const()[name = string("op_6964_axis_0"), val = int32(1)]; tensor var_6964_cast_fp16_0, tensor var_6964_cast_fp16_1 = split(axis = var_6964_axis_0, split_sizes = var_6964_split_sizes_0, x = out_75_cast_fp16)[name = string("op_6964_cast_fp16")]; tensor input_37_strides_0 = const()[name = string("input_37_strides_0"), val = tensor([1, 1])]; string input_37_pad_type_0 = const()[name = string("input_37_pad_type_0"), val = string("valid")]; tensor input_37_pad_0 = const()[name = string("input_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_37_dilations_0 = const()[name = string("input_37_dilations_0"), val = tensor([1, 1])]; int32 input_37_groups_0 = const()[name = string("input_37_groups_0"), val = int32(1)]; tensor input_37_cast_fp16 = conv(dilations = input_37_dilations_0, groups = input_37_groups_0, pad = input_37_pad_0, pad_type = input_37_pad_type_0, strides = input_37_strides_0, weight = layers_18_mlp_gate_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("input_37_cast_fp16")]; tensor var_6981_cast_fp16 = silu(x = input_37_cast_fp16)[name = string("op_6981_cast_fp16")]; tensor var_6987_strides_0 = const()[name = string("op_6987_strides_0"), val = tensor([1, 1])]; string var_6987_pad_type_0 = const()[name = string("op_6987_pad_type_0"), val = string("valid")]; tensor var_6987_pad_0 = const()[name = string("op_6987_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6987_dilations_0 = const()[name = string("op_6987_dilations_0"), val = tensor([1, 1])]; int32 var_6987_groups_0 = const()[name = string("op_6987_groups_0"), val = int32(1)]; tensor var_6987_cast_fp16 = conv(dilations = var_6987_dilations_0, groups = var_6987_groups_0, pad = var_6987_pad_0, pad_type = var_6987_pad_type_0, strides = var_6987_strides_0, weight = layers_18_mlp_up_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("op_6987_cast_fp16")]; tensor x_189_cast_fp16 = mul(x = var_6981_cast_fp16, y = var_6987_cast_fp16)[name = string("x_189_cast_fp16")]; tensor hidden_states_187_strides_0 = const()[name = string("hidden_states_187_strides_0"), val = tensor([1, 1])]; string hidden_states_187_pad_type_0 = const()[name = string("hidden_states_187_pad_type_0"), val = string("valid")]; tensor hidden_states_187_pad_0 = const()[name = string("hidden_states_187_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_187_dilations_0 = const()[name = string("hidden_states_187_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_187_groups_0 = const()[name = string("hidden_states_187_groups_0"), val = int32(1)]; tensor hidden_states_187_cast_fp16 = conv(dilations = hidden_states_187_dilations_0, groups = hidden_states_187_groups_0, pad = hidden_states_187_pad_0, pad_type = hidden_states_187_pad_type_0, strides = hidden_states_187_strides_0, weight = layers_18_mlp_down_proj_weight_cast_fp16, x = x_189_cast_fp16)[name = string("hidden_states_187_cast_fp16")]; tensor hidden_states_189_cast_fp16 = add(x = hidden_states_185_cast_fp16, y = hidden_states_187_cast_fp16)[name = string("hidden_states_189_cast_fp16")]; fp16 const_190_promoted_to_fp16 = const()[name = string("const_190_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7005_cast_fp16 = mul(x = hidden_states_189_cast_fp16, y = const_190_promoted_to_fp16)[name = string("op_7005_cast_fp16")]; int32 var_7003 = const()[name = string("op_7003"), val = int32(1)]; bool doubled_153_interleave_0 = const()[name = string("doubled_153_interleave_0"), val = bool(false)]; tensor doubled_153_cast_fp16 = concat(axis = var_7003, interleave = doubled_153_interleave_0, values = (hidden_states_189_cast_fp16, var_7005_cast_fp16))[name = string("doubled_153_cast_fp16")]; tensor out_77_axes_0 = const()[name = string("out_77_axes_0"), val = tensor([1])]; tensor out_77_gamma_0_to_fp16 = const()[name = string("out_77_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492036544)))]; fp16 var_7015_to_fp16 = const()[name = string("op_7015_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_77_cast_fp16 = layer_norm(axes = out_77_axes_0, epsilon = var_7015_to_fp16, gamma = out_77_gamma_0_to_fp16, x = doubled_153_cast_fp16)[name = string("out_77_cast_fp16")]; tensor var_7026_split_sizes_0 = const()[name = string("op_7026_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7026_axis_0 = const()[name = string("op_7026_axis_0"), val = int32(1)]; tensor var_7026_cast_fp16_0, tensor var_7026_cast_fp16_1 = split(axis = var_7026_axis_0, split_sizes = var_7026_split_sizes_0, x = out_77_cast_fp16)[name = string("op_7026_cast_fp16")]; tensor query_states_115_strides_0 = const()[name = string("query_states_115_strides_0"), val = tensor([1, 1])]; string query_states_115_pad_type_0 = const()[name = string("query_states_115_pad_type_0"), val = string("valid")]; tensor query_states_115_pad_0 = const()[name = string("query_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_115_dilations_0 = const()[name = string("query_states_115_dilations_0"), val = tensor([1, 1])]; int32 query_states_115_groups_0 = const()[name = string("query_states_115_groups_0"), val = int32(1)]; tensor query_states_115_cast_fp16 = conv(dilations = query_states_115_dilations_0, groups = query_states_115_groups_0, pad = query_states_115_pad_0, pad_type = query_states_115_pad_type_0, strides = query_states_115_strides_0, weight = layers_19_self_attn_q_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("query_states_115_cast_fp16")]; tensor key_states_191_strides_0 = const()[name = string("key_states_191_strides_0"), val = tensor([1, 1])]; string key_states_191_pad_type_0 = const()[name = string("key_states_191_pad_type_0"), val = string("valid")]; tensor key_states_191_pad_0 = const()[name = string("key_states_191_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_191_dilations_0 = const()[name = string("key_states_191_dilations_0"), val = tensor([1, 1])]; int32 key_states_191_groups_0 = const()[name = string("key_states_191_groups_0"), val = int32(1)]; tensor key_states_191_cast_fp16 = conv(dilations = key_states_191_dilations_0, groups = key_states_191_groups_0, pad = key_states_191_pad_0, pad_type = key_states_191_pad_type_0, strides = key_states_191_strides_0, weight = layers_19_self_attn_k_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("key_states_191_cast_fp16")]; tensor layers_19_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492044800)))]; tensor value_states_115_strides_0 = const()[name = string("value_states_115_strides_0"), val = tensor([1, 1])]; string value_states_115_pad_type_0 = const()[name = string("value_states_115_pad_type_0"), val = string("valid")]; tensor value_states_115_pad_0 = const()[name = string("value_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_115_dilations_0 = const()[name = string("value_states_115_dilations_0"), val = tensor([1, 1])]; int32 value_states_115_groups_0 = const()[name = string("value_states_115_groups_0"), val = int32(1)]; tensor value_states_115_cast_fp16 = conv(dilations = value_states_115_dilations_0, groups = value_states_115_groups_0, pad = value_states_115_pad_0, pad_type = value_states_115_pad_type_0, strides = value_states_115_strides_0, weight = layers_19_self_attn_v_proj_weight_to_fp16, x = var_7026_cast_fp16_0)[name = string("value_states_115_cast_fp16")]; tensor concat_228x = const()[name = string("concat_228x"), val = tensor([1, 16, 128, -1])]; tensor x_191_cast_fp16 = reshape(shape = concat_228x, x = query_states_115_cast_fp16)[name = string("x_191_cast_fp16")]; tensor concat_229x = const()[name = string("concat_229x"), val = tensor([1, 2, 128, -1])]; tensor var_7083_cast_fp16 = reshape(shape = concat_229x, x = key_states_191_cast_fp16)[name = string("op_7083_cast_fp16")]; tensor concat_230x = const()[name = string("concat_230x"), val = tensor([1, 2, 128, -1])]; tensor var_7090_cast_fp16 = reshape(shape = concat_230x, x = value_states_115_cast_fp16)[name = string("op_7090_cast_fp16")]; tensor var_7094_cast_fp16 = mul(x = x_191_cast_fp16, y = var_869_cast_fp16)[name = string("op_7094_cast_fp16")]; tensor var_7095_split_sizes_0 = const()[name = string("op_7095_split_sizes_0"), val = tensor([64, 64])]; int32 var_7095_axis_0 = const()[name = string("op_7095_axis_0"), val = int32(-2)]; tensor var_7095_cast_fp16_0, tensor var_7095_cast_fp16_1 = split(axis = var_7095_axis_0, split_sizes = var_7095_split_sizes_0, x = x_191_cast_fp16)[name = string("op_7095_cast_fp16")]; fp16 const_192_promoted_to_fp16 = const()[name = string("const_192_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7097_cast_fp16 = mul(x = var_7095_cast_fp16_1, y = const_192_promoted_to_fp16)[name = string("op_7097_cast_fp16")]; int32 var_7099 = const()[name = string("op_7099"), val = int32(-2)]; bool var_7100_interleave_0 = const()[name = string("op_7100_interleave_0"), val = bool(false)]; tensor var_7100_cast_fp16 = concat(axis = var_7099, interleave = var_7100_interleave_0, values = (var_7097_cast_fp16, var_7095_cast_fp16_0))[name = string("op_7100_cast_fp16")]; tensor var_7101_cast_fp16 = mul(x = var_7100_cast_fp16, y = var_878_cast_fp16)[name = string("op_7101_cast_fp16")]; tensor query_states_117_cast_fp16 = add(x = var_7094_cast_fp16, y = var_7101_cast_fp16)[name = string("query_states_117_cast_fp16")]; tensor var_7107_cast_fp16 = mul(x = var_7083_cast_fp16, y = var_869_cast_fp16)[name = string("op_7107_cast_fp16")]; tensor var_7108_split_sizes_0 = const()[name = string("op_7108_split_sizes_0"), val = tensor([64, 64])]; int32 var_7108_axis_0 = const()[name = string("op_7108_axis_0"), val = int32(-2)]; tensor var_7108_cast_fp16_0, tensor var_7108_cast_fp16_1 = split(axis = var_7108_axis_0, split_sizes = var_7108_split_sizes_0, x = var_7083_cast_fp16)[name = string("op_7108_cast_fp16")]; fp16 const_193_promoted_to_fp16 = const()[name = string("const_193_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7110_cast_fp16 = mul(x = var_7108_cast_fp16_1, y = const_193_promoted_to_fp16)[name = string("op_7110_cast_fp16")]; int32 var_7112 = const()[name = string("op_7112"), val = int32(-2)]; bool var_7113_interleave_0 = const()[name = string("op_7113_interleave_0"), val = bool(false)]; tensor var_7113_cast_fp16 = concat(axis = var_7112, interleave = var_7113_interleave_0, values = (var_7110_cast_fp16, var_7108_cast_fp16_0))[name = string("op_7113_cast_fp16")]; tensor var_7114_cast_fp16 = mul(x = var_7113_cast_fp16, y = var_878_cast_fp16)[name = string("op_7114_cast_fp16")]; tensor key_states_195_cast_fp16 = add(x = var_7107_cast_fp16, y = var_7114_cast_fp16)[name = string("key_states_195_cast_fp16")]; tensor expand_dims_228 = const()[name = string("expand_dims_228"), val = tensor([19])]; tensor expand_dims_229 = const()[name = string("expand_dims_229"), val = tensor([0])]; tensor expand_dims_231 = const()[name = string("expand_dims_231"), val = tensor([0])]; int32 concat_233_axis_0 = const()[name = string("concat_233_axis_0"), val = int32(0)]; bool concat_233_interleave_0 = const()[name = string("concat_233_interleave_0"), val = bool(false)]; tensor concat_233 = concat(axis = concat_233_axis_0, interleave = concat_233_interleave_0, values = (expand_dims_228, expand_dims_229, position_id, expand_dims_231))[name = string("concat_233")]; tensor expand_dims_232 = const()[name = string("expand_dims_232"), val = tensor([20])]; tensor concat_234_values1_0 = const()[name = string("concat_234_values1_0"), val = tensor([0])]; tensor concat_234_values3_0 = const()[name = string("concat_234_values3_0"), val = tensor([0])]; int32 concat_234_axis_0 = const()[name = string("concat_234_axis_0"), val = int32(0)]; bool concat_234_interleave_0 = const()[name = string("concat_234_interleave_0"), val = bool(false)]; tensor concat_234 = concat(axis = concat_234_axis_0, interleave = concat_234_interleave_0, values = (expand_dims_232, concat_234_values1_0, cache_position_end, concat_234_values3_0))[name = string("concat_234")]; tensor key_states_197_perm_0 = const()[name = string("key_states_197_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_20_stride_0 = const()[name = string("key_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_197_cast_fp16 = transpose(perm = key_states_197_perm_0, x = key_states_195_cast_fp16)[name = string("transpose_198")]; tensor key_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = key_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = key_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_20_squeeze_mask_0, stride = key_cache_internal_tensor_assign_20_stride_0, update = key_states_197_cast_fp16, x = coreml_update_state_148)[name = string("key_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_20_cast_fp16, input = key_cache)[name = string("coreml_update_state_150_write_state")]; tensor coreml_update_state_150 = read_state(input = key_cache)[name = string("coreml_update_state_150")]; tensor value_states_117_perm_0 = const()[name = string("value_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_20_stride_0 = const()[name = string("value_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_117_cast_fp16 = transpose(perm = value_states_117_perm_0, x = var_7090_cast_fp16)[name = string("transpose_197")]; tensor value_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = value_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = value_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_20_squeeze_mask_0, stride = value_cache_internal_tensor_assign_20_stride_0, update = value_states_117_cast_fp16, x = coreml_update_state_149)[name = string("value_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_20_cast_fp16, input = value_cache)[name = string("coreml_update_state_151_write_state")]; tensor coreml_update_state_151 = read_state(input = value_cache)[name = string("coreml_update_state_151")]; tensor var_7184_begin_0 = const()[name = string("op_7184_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7184_end_0 = const()[name = string("op_7184_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7184_end_mask_0 = const()[name = string("op_7184_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7184_cast_fp16 = slice_by_index(begin = var_7184_begin_0, end = var_7184_end_0, end_mask = var_7184_end_mask_0, x = coreml_update_state_150)[name = string("op_7184_cast_fp16")]; tensor tile_38 = const()[name = string("tile_38"), val = tensor([1, 1])]; int32 var_7187_axis_0 = const()[name = string("op_7187_axis_0"), val = int32(1)]; tensor var_7187_cast_fp16_0, tensor var_7187_cast_fp16_1 = split(axis = var_7187_axis_0, split_sizes = tile_38, x = var_7184_cast_fp16)[name = string("op_7187_cast_fp16")]; tensor var_7194_begin_0 = const()[name = string("op_7194_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7194_end_0 = const()[name = string("op_7194_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7194_end_mask_0 = const()[name = string("op_7194_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7194_cast_fp16 = slice_by_index(begin = var_7194_begin_0, end = var_7194_end_0, end_mask = var_7194_end_mask_0, x = coreml_update_state_151)[name = string("op_7194_cast_fp16")]; tensor tile_39 = const()[name = string("tile_39"), val = tensor([1, 1])]; int32 var_7197_axis_0 = const()[name = string("op_7197_axis_0"), val = int32(1)]; tensor var_7197_cast_fp16_0, tensor var_7197_cast_fp16_1 = split(axis = var_7197_axis_0, split_sizes = tile_39, x = var_7194_cast_fp16)[name = string("op_7197_cast_fp16")]; tensor var_7200_split_sizes_0 = const()[name = string("op_7200_split_sizes_0"), val = tensor([8, 8])]; int32 var_7200_axis_0 = const()[name = string("op_7200_axis_0"), val = int32(1)]; tensor var_7200_0, tensor var_7200_1 = split(axis = var_7200_axis_0, split_sizes = var_7200_split_sizes_0, x = query_states_117_cast_fp16)[name = string("op_7200")]; bool attn_weights_305_transpose_x_0 = const()[name = string("attn_weights_305_transpose_x_0"), val = bool(false)]; bool attn_weights_305_transpose_y_0 = const()[name = string("attn_weights_305_transpose_y_0"), val = bool(false)]; tensor attn_weights_305_cast_fp16 = matmul(transpose_x = attn_weights_305_transpose_x_0, transpose_y = attn_weights_305_transpose_y_0, x = var_7187_cast_fp16_0, y = var_7200_0)[name = string("attn_weights_305_cast_fp16")]; fp16 var_7203_to_fp16 = const()[name = string("op_7203_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_307_cast_fp16 = mul(x = attn_weights_305_cast_fp16, y = var_7203_to_fp16)[name = string("attn_weights_307_cast_fp16")]; tensor attn_weights_309_cast_fp16 = add(x = attn_weights_307_cast_fp16, y = attn_mask_1)[name = string("attn_weights_309_cast_fp16")]; int32 var_7207 = const()[name = string("op_7207"), val = int32(-2)]; tensor attn_weights_311_cast_fp16 = softmax(axis = var_7207, x = attn_weights_309_cast_fp16)[name = string("attn_weights_311_cast_fp16")]; bool var_7213_transpose_x_1 = const()[name = string("op_7213_transpose_x_1"), val = bool(true)]; bool var_7213_transpose_y_1 = const()[name = string("op_7213_transpose_y_1"), val = bool(false)]; tensor var_7213_cast_fp16 = matmul(transpose_x = var_7213_transpose_x_1, transpose_y = var_7213_transpose_y_1, x = attn_weights_311_cast_fp16, y = var_7197_cast_fp16_0)[name = string("op_7213_cast_fp16")]; bool attn_weights_313_transpose_x_0 = const()[name = string("attn_weights_313_transpose_x_0"), val = bool(false)]; bool attn_weights_313_transpose_y_0 = const()[name = string("attn_weights_313_transpose_y_0"), val = bool(false)]; tensor attn_weights_313_cast_fp16 = matmul(transpose_x = attn_weights_313_transpose_x_0, transpose_y = attn_weights_313_transpose_y_0, x = var_7187_cast_fp16_1, y = var_7200_1)[name = string("attn_weights_313_cast_fp16")]; fp16 var_7215_to_fp16 = const()[name = string("op_7215_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_315_cast_fp16 = mul(x = attn_weights_313_cast_fp16, y = var_7215_to_fp16)[name = string("attn_weights_315_cast_fp16")]; tensor attn_weights_317_cast_fp16 = add(x = attn_weights_315_cast_fp16, y = attn_mask_1)[name = string("attn_weights_317_cast_fp16")]; int32 var_7219 = const()[name = string("op_7219"), val = int32(-2)]; tensor attn_weights_319_cast_fp16 = softmax(axis = var_7219, x = attn_weights_317_cast_fp16)[name = string("attn_weights_319_cast_fp16")]; bool attn_output_153_transpose_x_1 = const()[name = string("attn_output_153_transpose_x_1"), val = bool(true)]; bool attn_output_153_transpose_y_1 = const()[name = string("attn_output_153_transpose_y_1"), val = bool(false)]; tensor attn_output_153_cast_fp16 = matmul(transpose_x = attn_output_153_transpose_x_1, transpose_y = attn_output_153_transpose_y_1, x = attn_weights_319_cast_fp16, y = var_7197_cast_fp16_1)[name = string("attn_output_153_cast_fp16")]; int32 var_7227 = const()[name = string("op_7227"), val = int32(1)]; bool attn_output_155_interleave_0 = const()[name = string("attn_output_155_interleave_0"), val = bool(false)]; tensor attn_output_155_cast_fp16 = concat(axis = var_7227, interleave = attn_output_155_interleave_0, values = (var_7213_cast_fp16, attn_output_153_cast_fp16))[name = string("attn_output_155_cast_fp16")]; tensor var_7231_perm_0 = const()[name = string("op_7231_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_239x = const()[name = string("concat_239x"), val = tensor([1, 2048, 1, -1])]; tensor var_7231_cast_fp16 = transpose(perm = var_7231_perm_0, x = attn_output_155_cast_fp16)[name = string("transpose_196")]; tensor attn_output_159_cast_fp16 = reshape(shape = concat_239x, x = var_7231_cast_fp16)[name = string("attn_output_159_cast_fp16")]; tensor layers_19_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1493093440)))]; tensor hidden_states_193_strides_0 = const()[name = string("hidden_states_193_strides_0"), val = tensor([1, 1])]; string hidden_states_193_pad_type_0 = const()[name = string("hidden_states_193_pad_type_0"), val = string("valid")]; tensor hidden_states_193_pad_0 = const()[name = string("hidden_states_193_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_193_dilations_0 = const()[name = string("hidden_states_193_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_193_groups_0 = const()[name = string("hidden_states_193_groups_0"), val = int32(1)]; tensor hidden_states_193_cast_fp16 = conv(dilations = hidden_states_193_dilations_0, groups = hidden_states_193_groups_0, pad = hidden_states_193_pad_0, pad_type = hidden_states_193_pad_type_0, strides = hidden_states_193_strides_0, weight = layers_19_self_attn_o_proj_weight_to_fp16, x = attn_output_159_cast_fp16)[name = string("hidden_states_193_cast_fp16")]; tensor hidden_states_195_cast_fp16 = add(x = hidden_states_189_cast_fp16, y = hidden_states_193_cast_fp16)[name = string("hidden_states_195_cast_fp16")]; fp16 const_198_promoted_to_fp16 = const()[name = string("const_198_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7264_cast_fp16 = mul(x = hidden_states_195_cast_fp16, y = const_198_promoted_to_fp16)[name = string("op_7264_cast_fp16")]; int32 var_7262 = const()[name = string("op_7262"), val = int32(1)]; bool doubled_157_interleave_0 = const()[name = string("doubled_157_interleave_0"), val = bool(false)]; tensor doubled_157_cast_fp16 = concat(axis = var_7262, interleave = doubled_157_interleave_0, values = (hidden_states_195_cast_fp16, var_7264_cast_fp16))[name = string("doubled_157_cast_fp16")]; tensor out_79_axes_0 = const()[name = string("out_79_axes_0"), val = tensor([1])]; tensor out_79_gamma_0_to_fp16 = const()[name = string("out_79_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501482112)))]; fp16 var_7274_to_fp16 = const()[name = string("op_7274_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_79_cast_fp16 = layer_norm(axes = out_79_axes_0, epsilon = var_7274_to_fp16, gamma = out_79_gamma_0_to_fp16, x = doubled_157_cast_fp16)[name = string("out_79_cast_fp16")]; tensor var_7285_split_sizes_0 = const()[name = string("op_7285_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7285_axis_0 = const()[name = string("op_7285_axis_0"), val = int32(1)]; tensor var_7285_cast_fp16_0, tensor var_7285_cast_fp16_1 = split(axis = var_7285_axis_0, split_sizes = var_7285_split_sizes_0, x = out_79_cast_fp16)[name = string("op_7285_cast_fp16")]; tensor input_39_strides_0 = const()[name = string("input_39_strides_0"), val = tensor([1, 1])]; string input_39_pad_type_0 = const()[name = string("input_39_pad_type_0"), val = string("valid")]; tensor input_39_pad_0 = const()[name = string("input_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_39_dilations_0 = const()[name = string("input_39_dilations_0"), val = tensor([1, 1])]; int32 input_39_groups_0 = const()[name = string("input_39_groups_0"), val = int32(1)]; tensor input_39_cast_fp16 = conv(dilations = input_39_dilations_0, groups = input_39_groups_0, pad = input_39_pad_0, pad_type = input_39_pad_type_0, strides = input_39_strides_0, weight = layers_19_mlp_gate_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("input_39_cast_fp16")]; tensor var_7302_cast_fp16 = silu(x = input_39_cast_fp16)[name = string("op_7302_cast_fp16")]; tensor var_7308_strides_0 = const()[name = string("op_7308_strides_0"), val = tensor([1, 1])]; string var_7308_pad_type_0 = const()[name = string("op_7308_pad_type_0"), val = string("valid")]; tensor var_7308_pad_0 = const()[name = string("op_7308_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7308_dilations_0 = const()[name = string("op_7308_dilations_0"), val = tensor([1, 1])]; int32 var_7308_groups_0 = const()[name = string("op_7308_groups_0"), val = int32(1)]; tensor var_7308_cast_fp16 = conv(dilations = var_7308_dilations_0, groups = var_7308_groups_0, pad = var_7308_pad_0, pad_type = var_7308_pad_type_0, strides = var_7308_strides_0, weight = layers_19_mlp_up_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("op_7308_cast_fp16")]; tensor x_199_cast_fp16 = mul(x = var_7302_cast_fp16, y = var_7308_cast_fp16)[name = string("x_199_cast_fp16")]; tensor hidden_states_197_strides_0 = const()[name = string("hidden_states_197_strides_0"), val = tensor([1, 1])]; string hidden_states_197_pad_type_0 = const()[name = string("hidden_states_197_pad_type_0"), val = string("valid")]; tensor hidden_states_197_pad_0 = const()[name = string("hidden_states_197_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_197_dilations_0 = const()[name = string("hidden_states_197_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_197_groups_0 = const()[name = string("hidden_states_197_groups_0"), val = int32(1)]; tensor hidden_states_197_cast_fp16 = conv(dilations = hidden_states_197_dilations_0, groups = hidden_states_197_groups_0, pad = hidden_states_197_pad_0, pad_type = hidden_states_197_pad_type_0, strides = hidden_states_197_strides_0, weight = layers_19_mlp_down_proj_weight_cast_fp16, x = x_199_cast_fp16)[name = string("hidden_states_197_cast_fp16")]; tensor hidden_states_199_cast_fp16 = add(x = hidden_states_195_cast_fp16, y = hidden_states_197_cast_fp16)[name = string("hidden_states_199_cast_fp16")]; fp16 const_200_promoted_to_fp16 = const()[name = string("const_200_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7326_cast_fp16 = mul(x = hidden_states_199_cast_fp16, y = const_200_promoted_to_fp16)[name = string("op_7326_cast_fp16")]; int32 var_7324 = const()[name = string("op_7324"), val = int32(1)]; bool doubled_161_interleave_0 = const()[name = string("doubled_161_interleave_0"), val = bool(false)]; tensor doubled_161_cast_fp16 = concat(axis = var_7324, interleave = doubled_161_interleave_0, values = (hidden_states_199_cast_fp16, var_7326_cast_fp16))[name = string("doubled_161_cast_fp16")]; tensor out_81_axes_0 = const()[name = string("out_81_axes_0"), val = tensor([1])]; tensor out_81_gamma_0_to_fp16 = const()[name = string("out_81_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501490368)))]; fp16 var_7336_to_fp16 = const()[name = string("op_7336_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_81_cast_fp16 = layer_norm(axes = out_81_axes_0, epsilon = var_7336_to_fp16, gamma = out_81_gamma_0_to_fp16, x = doubled_161_cast_fp16)[name = string("out_81_cast_fp16")]; tensor var_7347_split_sizes_0 = const()[name = string("op_7347_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7347_axis_0 = const()[name = string("op_7347_axis_0"), val = int32(1)]; tensor var_7347_cast_fp16_0, tensor var_7347_cast_fp16_1 = split(axis = var_7347_axis_0, split_sizes = var_7347_split_sizes_0, x = out_81_cast_fp16)[name = string("op_7347_cast_fp16")]; tensor query_states_121_strides_0 = const()[name = string("query_states_121_strides_0"), val = tensor([1, 1])]; string query_states_121_pad_type_0 = const()[name = string("query_states_121_pad_type_0"), val = string("valid")]; tensor query_states_121_pad_0 = const()[name = string("query_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_121_dilations_0 = const()[name = string("query_states_121_dilations_0"), val = tensor([1, 1])]; int32 query_states_121_groups_0 = const()[name = string("query_states_121_groups_0"), val = int32(1)]; tensor query_states_121_cast_fp16 = conv(dilations = query_states_121_dilations_0, groups = query_states_121_groups_0, pad = query_states_121_pad_0, pad_type = query_states_121_pad_type_0, strides = query_states_121_strides_0, weight = layers_20_self_attn_q_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("query_states_121_cast_fp16")]; tensor key_states_201_strides_0 = const()[name = string("key_states_201_strides_0"), val = tensor([1, 1])]; string key_states_201_pad_type_0 = const()[name = string("key_states_201_pad_type_0"), val = string("valid")]; tensor key_states_201_pad_0 = const()[name = string("key_states_201_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_201_dilations_0 = const()[name = string("key_states_201_dilations_0"), val = tensor([1, 1])]; int32 key_states_201_groups_0 = const()[name = string("key_states_201_groups_0"), val = int32(1)]; tensor key_states_201_cast_fp16 = conv(dilations = key_states_201_dilations_0, groups = key_states_201_groups_0, pad = key_states_201_pad_0, pad_type = key_states_201_pad_type_0, strides = key_states_201_strides_0, weight = layers_20_self_attn_k_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("key_states_201_cast_fp16")]; tensor layers_20_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_20_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501498624)))]; tensor value_states_121_strides_0 = const()[name = string("value_states_121_strides_0"), val = tensor([1, 1])]; string value_states_121_pad_type_0 = const()[name = string("value_states_121_pad_type_0"), val = string("valid")]; tensor value_states_121_pad_0 = const()[name = string("value_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_121_dilations_0 = const()[name = string("value_states_121_dilations_0"), val = tensor([1, 1])]; int32 value_states_121_groups_0 = const()[name = string("value_states_121_groups_0"), val = int32(1)]; tensor value_states_121_cast_fp16 = conv(dilations = value_states_121_dilations_0, groups = value_states_121_groups_0, pad = value_states_121_pad_0, pad_type = value_states_121_pad_type_0, strides = value_states_121_strides_0, weight = layers_20_self_attn_v_proj_weight_to_fp16, x = var_7347_cast_fp16_0)[name = string("value_states_121_cast_fp16")]; tensor concat_240x = const()[name = string("concat_240x"), val = tensor([1, 16, 128, -1])]; tensor x_201_cast_fp16 = reshape(shape = concat_240x, x = query_states_121_cast_fp16)[name = string("x_201_cast_fp16")]; tensor concat_241x = const()[name = string("concat_241x"), val = tensor([1, 2, 128, -1])]; tensor var_7404_cast_fp16 = reshape(shape = concat_241x, x = key_states_201_cast_fp16)[name = string("op_7404_cast_fp16")]; tensor concat_242x = const()[name = string("concat_242x"), val = tensor([1, 2, 128, -1])]; tensor var_7411_cast_fp16 = reshape(shape = concat_242x, x = value_states_121_cast_fp16)[name = string("op_7411_cast_fp16")]; tensor var_7415_cast_fp16 = mul(x = x_201_cast_fp16, y = var_869_cast_fp16)[name = string("op_7415_cast_fp16")]; tensor var_7416_split_sizes_0 = const()[name = string("op_7416_split_sizes_0"), val = tensor([64, 64])]; int32 var_7416_axis_0 = const()[name = string("op_7416_axis_0"), val = int32(-2)]; tensor var_7416_cast_fp16_0, tensor var_7416_cast_fp16_1 = split(axis = var_7416_axis_0, split_sizes = var_7416_split_sizes_0, x = x_201_cast_fp16)[name = string("op_7416_cast_fp16")]; fp16 const_202_promoted_to_fp16 = const()[name = string("const_202_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7418_cast_fp16 = mul(x = var_7416_cast_fp16_1, y = const_202_promoted_to_fp16)[name = string("op_7418_cast_fp16")]; int32 var_7420 = const()[name = string("op_7420"), val = int32(-2)]; bool var_7421_interleave_0 = const()[name = string("op_7421_interleave_0"), val = bool(false)]; tensor var_7421_cast_fp16 = concat(axis = var_7420, interleave = var_7421_interleave_0, values = (var_7418_cast_fp16, var_7416_cast_fp16_0))[name = string("op_7421_cast_fp16")]; tensor var_7422_cast_fp16 = mul(x = var_7421_cast_fp16, y = var_878_cast_fp16)[name = string("op_7422_cast_fp16")]; tensor query_states_123_cast_fp16 = add(x = var_7415_cast_fp16, y = var_7422_cast_fp16)[name = string("query_states_123_cast_fp16")]; tensor var_7428_cast_fp16 = mul(x = var_7404_cast_fp16, y = var_869_cast_fp16)[name = string("op_7428_cast_fp16")]; tensor var_7429_split_sizes_0 = const()[name = string("op_7429_split_sizes_0"), val = tensor([64, 64])]; int32 var_7429_axis_0 = const()[name = string("op_7429_axis_0"), val = int32(-2)]; tensor var_7429_cast_fp16_0, tensor var_7429_cast_fp16_1 = split(axis = var_7429_axis_0, split_sizes = var_7429_split_sizes_0, x = var_7404_cast_fp16)[name = string("op_7429_cast_fp16")]; fp16 const_203_promoted_to_fp16 = const()[name = string("const_203_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7431_cast_fp16 = mul(x = var_7429_cast_fp16_1, y = const_203_promoted_to_fp16)[name = string("op_7431_cast_fp16")]; int32 var_7433 = const()[name = string("op_7433"), val = int32(-2)]; bool var_7434_interleave_0 = const()[name = string("op_7434_interleave_0"), val = bool(false)]; tensor var_7434_cast_fp16 = concat(axis = var_7433, interleave = var_7434_interleave_0, values = (var_7431_cast_fp16, var_7429_cast_fp16_0))[name = string("op_7434_cast_fp16")]; tensor var_7435_cast_fp16 = mul(x = var_7434_cast_fp16, y = var_878_cast_fp16)[name = string("op_7435_cast_fp16")]; tensor key_states_205_cast_fp16 = add(x = var_7428_cast_fp16, y = var_7435_cast_fp16)[name = string("key_states_205_cast_fp16")]; tensor expand_dims_240 = const()[name = string("expand_dims_240"), val = tensor([20])]; tensor expand_dims_241 = const()[name = string("expand_dims_241"), val = tensor([0])]; tensor expand_dims_243 = const()[name = string("expand_dims_243"), val = tensor([0])]; int32 concat_245_axis_0 = const()[name = string("concat_245_axis_0"), val = int32(0)]; bool concat_245_interleave_0 = const()[name = string("concat_245_interleave_0"), val = bool(false)]; tensor concat_245 = concat(axis = concat_245_axis_0, interleave = concat_245_interleave_0, values = (expand_dims_240, expand_dims_241, position_id, expand_dims_243))[name = string("concat_245")]; tensor expand_dims_244 = const()[name = string("expand_dims_244"), val = tensor([21])]; tensor concat_246_values1_0 = const()[name = string("concat_246_values1_0"), val = tensor([0])]; tensor concat_246_values3_0 = const()[name = string("concat_246_values3_0"), val = tensor([0])]; int32 concat_246_axis_0 = const()[name = string("concat_246_axis_0"), val = int32(0)]; bool concat_246_interleave_0 = const()[name = string("concat_246_interleave_0"), val = bool(false)]; tensor concat_246 = concat(axis = concat_246_axis_0, interleave = concat_246_interleave_0, values = (expand_dims_244, concat_246_values1_0, cache_position_end, concat_246_values3_0))[name = string("concat_246")]; tensor key_states_207_perm_0 = const()[name = string("key_states_207_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_21_stride_0 = const()[name = string("key_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_207_cast_fp16 = transpose(perm = key_states_207_perm_0, x = key_states_205_cast_fp16)[name = string("transpose_195")]; tensor key_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = key_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = key_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_21_squeeze_mask_0, stride = key_cache_internal_tensor_assign_21_stride_0, update = key_states_207_cast_fp16, x = coreml_update_state_150)[name = string("key_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_21_cast_fp16, input = key_cache)[name = string("coreml_update_state_152_write_state")]; tensor coreml_update_state_152 = read_state(input = key_cache)[name = string("coreml_update_state_152")]; tensor value_states_123_perm_0 = const()[name = string("value_states_123_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_21_stride_0 = const()[name = string("value_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_123_cast_fp16 = transpose(perm = value_states_123_perm_0, x = var_7411_cast_fp16)[name = string("transpose_194")]; tensor value_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = value_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = value_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_21_squeeze_mask_0, stride = value_cache_internal_tensor_assign_21_stride_0, update = value_states_123_cast_fp16, x = coreml_update_state_151)[name = string("value_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_21_cast_fp16, input = value_cache)[name = string("coreml_update_state_153_write_state")]; tensor coreml_update_state_153 = read_state(input = value_cache)[name = string("coreml_update_state_153")]; tensor var_7505_begin_0 = const()[name = string("op_7505_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7505_end_0 = const()[name = string("op_7505_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7505_end_mask_0 = const()[name = string("op_7505_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7505_cast_fp16 = slice_by_index(begin = var_7505_begin_0, end = var_7505_end_0, end_mask = var_7505_end_mask_0, x = coreml_update_state_152)[name = string("op_7505_cast_fp16")]; tensor tile_40 = const()[name = string("tile_40"), val = tensor([1, 1])]; int32 var_7508_axis_0 = const()[name = string("op_7508_axis_0"), val = int32(1)]; tensor var_7508_cast_fp16_0, tensor var_7508_cast_fp16_1 = split(axis = var_7508_axis_0, split_sizes = tile_40, x = var_7505_cast_fp16)[name = string("op_7508_cast_fp16")]; tensor var_7515_begin_0 = const()[name = string("op_7515_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7515_end_0 = const()[name = string("op_7515_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7515_end_mask_0 = const()[name = string("op_7515_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7515_cast_fp16 = slice_by_index(begin = var_7515_begin_0, end = var_7515_end_0, end_mask = var_7515_end_mask_0, x = coreml_update_state_153)[name = string("op_7515_cast_fp16")]; tensor tile_41 = const()[name = string("tile_41"), val = tensor([1, 1])]; int32 var_7518_axis_0 = const()[name = string("op_7518_axis_0"), val = int32(1)]; tensor var_7518_cast_fp16_0, tensor var_7518_cast_fp16_1 = split(axis = var_7518_axis_0, split_sizes = tile_41, x = var_7515_cast_fp16)[name = string("op_7518_cast_fp16")]; tensor var_7521_split_sizes_0 = const()[name = string("op_7521_split_sizes_0"), val = tensor([8, 8])]; int32 var_7521_axis_0 = const()[name = string("op_7521_axis_0"), val = int32(1)]; tensor var_7521_0, tensor var_7521_1 = split(axis = var_7521_axis_0, split_sizes = var_7521_split_sizes_0, x = query_states_123_cast_fp16)[name = string("op_7521")]; bool attn_weights_321_transpose_x_0 = const()[name = string("attn_weights_321_transpose_x_0"), val = bool(false)]; bool attn_weights_321_transpose_y_0 = const()[name = string("attn_weights_321_transpose_y_0"), val = bool(false)]; tensor attn_weights_321_cast_fp16 = matmul(transpose_x = attn_weights_321_transpose_x_0, transpose_y = attn_weights_321_transpose_y_0, x = var_7508_cast_fp16_0, y = var_7521_0)[name = string("attn_weights_321_cast_fp16")]; fp16 var_7524_to_fp16 = const()[name = string("op_7524_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_323_cast_fp16 = mul(x = attn_weights_321_cast_fp16, y = var_7524_to_fp16)[name = string("attn_weights_323_cast_fp16")]; tensor attn_weights_325_cast_fp16 = add(x = attn_weights_323_cast_fp16, y = attn_mask_1)[name = string("attn_weights_325_cast_fp16")]; int32 var_7528 = const()[name = string("op_7528"), val = int32(-2)]; tensor attn_weights_327_cast_fp16 = softmax(axis = var_7528, x = attn_weights_325_cast_fp16)[name = string("attn_weights_327_cast_fp16")]; bool var_7534_transpose_x_1 = const()[name = string("op_7534_transpose_x_1"), val = bool(true)]; bool var_7534_transpose_y_1 = const()[name = string("op_7534_transpose_y_1"), val = bool(false)]; tensor var_7534_cast_fp16 = matmul(transpose_x = var_7534_transpose_x_1, transpose_y = var_7534_transpose_y_1, x = attn_weights_327_cast_fp16, y = var_7518_cast_fp16_0)[name = string("op_7534_cast_fp16")]; bool attn_weights_329_transpose_x_0 = const()[name = string("attn_weights_329_transpose_x_0"), val = bool(false)]; bool attn_weights_329_transpose_y_0 = const()[name = string("attn_weights_329_transpose_y_0"), val = bool(false)]; tensor attn_weights_329_cast_fp16 = matmul(transpose_x = attn_weights_329_transpose_x_0, transpose_y = attn_weights_329_transpose_y_0, x = var_7508_cast_fp16_1, y = var_7521_1)[name = string("attn_weights_329_cast_fp16")]; fp16 var_7536_to_fp16 = const()[name = string("op_7536_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_331_cast_fp16 = mul(x = attn_weights_329_cast_fp16, y = var_7536_to_fp16)[name = string("attn_weights_331_cast_fp16")]; tensor attn_weights_333_cast_fp16 = add(x = attn_weights_331_cast_fp16, y = attn_mask_1)[name = string("attn_weights_333_cast_fp16")]; int32 var_7540 = const()[name = string("op_7540"), val = int32(-2)]; tensor attn_weights_335_cast_fp16 = softmax(axis = var_7540, x = attn_weights_333_cast_fp16)[name = string("attn_weights_335_cast_fp16")]; bool attn_output_161_transpose_x_1 = const()[name = string("attn_output_161_transpose_x_1"), val = bool(true)]; bool attn_output_161_transpose_y_1 = const()[name = string("attn_output_161_transpose_y_1"), val = bool(false)]; tensor attn_output_161_cast_fp16 = matmul(transpose_x = attn_output_161_transpose_x_1, transpose_y = attn_output_161_transpose_y_1, x = attn_weights_335_cast_fp16, y = var_7518_cast_fp16_1)[name = string("attn_output_161_cast_fp16")]; int32 var_7548 = const()[name = string("op_7548"), val = int32(1)]; bool attn_output_163_interleave_0 = const()[name = string("attn_output_163_interleave_0"), val = bool(false)]; tensor attn_output_163_cast_fp16 = concat(axis = var_7548, interleave = attn_output_163_interleave_0, values = (var_7534_cast_fp16, attn_output_161_cast_fp16))[name = string("attn_output_163_cast_fp16")]; tensor var_7552_perm_0 = const()[name = string("op_7552_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_251x = const()[name = string("concat_251x"), val = tensor([1, 2048, 1, -1])]; tensor var_7552_cast_fp16 = transpose(perm = var_7552_perm_0, x = attn_output_163_cast_fp16)[name = string("transpose_193")]; tensor attn_output_167_cast_fp16 = reshape(shape = concat_251x, x = var_7552_cast_fp16)[name = string("attn_output_167_cast_fp16")]; tensor hidden_states_203_strides_0 = const()[name = string("hidden_states_203_strides_0"), val = tensor([1, 1])]; string hidden_states_203_pad_type_0 = const()[name = string("hidden_states_203_pad_type_0"), val = string("valid")]; tensor hidden_states_203_pad_0 = const()[name = string("hidden_states_203_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_203_dilations_0 = const()[name = string("hidden_states_203_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_203_groups_0 = const()[name = string("hidden_states_203_groups_0"), val = int32(1)]; tensor hidden_states_203_cast_fp16 = conv(dilations = hidden_states_203_dilations_0, groups = hidden_states_203_groups_0, pad = hidden_states_203_pad_0, pad_type = hidden_states_203_pad_type_0, strides = hidden_states_203_strides_0, weight = layers_20_self_attn_o_proj_weight_cast_fp16, x = attn_output_167_cast_fp16)[name = string("hidden_states_203_cast_fp16")]; tensor hidden_states_205_cast_fp16 = add(x = hidden_states_199_cast_fp16, y = hidden_states_203_cast_fp16)[name = string("hidden_states_205_cast_fp16")]; fp16 const_208_promoted_to_fp16 = const()[name = string("const_208_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7585_cast_fp16 = mul(x = hidden_states_205_cast_fp16, y = const_208_promoted_to_fp16)[name = string("op_7585_cast_fp16")]; int32 var_7583 = const()[name = string("op_7583"), val = int32(1)]; bool doubled_165_interleave_0 = const()[name = string("doubled_165_interleave_0"), val = bool(false)]; tensor doubled_165_cast_fp16 = concat(axis = var_7583, interleave = doubled_165_interleave_0, values = (hidden_states_205_cast_fp16, var_7585_cast_fp16))[name = string("doubled_165_cast_fp16")]; tensor out_83_axes_0 = const()[name = string("out_83_axes_0"), val = tensor([1])]; tensor out_83_gamma_0_to_fp16 = const()[name = string("out_83_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502547264)))]; fp16 var_7595_to_fp16 = const()[name = string("op_7595_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_83_cast_fp16 = layer_norm(axes = out_83_axes_0, epsilon = var_7595_to_fp16, gamma = out_83_gamma_0_to_fp16, x = doubled_165_cast_fp16)[name = string("out_83_cast_fp16")]; tensor var_7606_split_sizes_0 = const()[name = string("op_7606_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7606_axis_0 = const()[name = string("op_7606_axis_0"), val = int32(1)]; tensor var_7606_cast_fp16_0, tensor var_7606_cast_fp16_1 = split(axis = var_7606_axis_0, split_sizes = var_7606_split_sizes_0, x = out_83_cast_fp16)[name = string("op_7606_cast_fp16")]; tensor input_41_strides_0 = const()[name = string("input_41_strides_0"), val = tensor([1, 1])]; string input_41_pad_type_0 = const()[name = string("input_41_pad_type_0"), val = string("valid")]; tensor input_41_pad_0 = const()[name = string("input_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_41_dilations_0 = const()[name = string("input_41_dilations_0"), val = tensor([1, 1])]; int32 input_41_groups_0 = const()[name = string("input_41_groups_0"), val = int32(1)]; tensor input_41_cast_fp16 = conv(dilations = input_41_dilations_0, groups = input_41_groups_0, pad = input_41_pad_0, pad_type = input_41_pad_type_0, strides = input_41_strides_0, weight = layers_20_mlp_gate_proj_weight_cast_fp16, x = var_7606_cast_fp16_0)[name = string("input_41_cast_fp16")]; tensor var_7623_cast_fp16 = silu(x = input_41_cast_fp16)[name = string("op_7623_cast_fp16")]; tensor layers_20_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_20_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502555520)))]; tensor var_7629_strides_0 = const()[name = string("op_7629_strides_0"), val = tensor([1, 1])]; string var_7629_pad_type_0 = const()[name = string("op_7629_pad_type_0"), val = string("valid")]; tensor var_7629_pad_0 = const()[name = string("op_7629_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7629_dilations_0 = const()[name = string("op_7629_dilations_0"), val = tensor([1, 1])]; int32 var_7629_groups_0 = const()[name = string("op_7629_groups_0"), val = int32(1)]; tensor var_7629_cast_fp16 = conv(dilations = var_7629_dilations_0, groups = var_7629_groups_0, pad = var_7629_pad_0, pad_type = var_7629_pad_type_0, strides = var_7629_strides_0, weight = layers_20_mlp_up_proj_weight_to_fp16, x = var_7606_cast_fp16_0)[name = string("op_7629_cast_fp16")]; tensor x_209_cast_fp16 = mul(x = var_7623_cast_fp16, y = var_7629_cast_fp16)[name = string("x_209_cast_fp16")]; tensor hidden_states_207_strides_0 = const()[name = string("hidden_states_207_strides_0"), val = tensor([1, 1])]; string hidden_states_207_pad_type_0 = const()[name = string("hidden_states_207_pad_type_0"), val = string("valid")]; tensor hidden_states_207_pad_0 = const()[name = string("hidden_states_207_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_207_dilations_0 = const()[name = string("hidden_states_207_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_207_groups_0 = const()[name = string("hidden_states_207_groups_0"), val = int32(1)]; tensor hidden_states_207_cast_fp16 = conv(dilations = hidden_states_207_dilations_0, groups = hidden_states_207_groups_0, pad = hidden_states_207_pad_0, pad_type = hidden_states_207_pad_type_0, strides = hidden_states_207_strides_0, weight = layers_20_mlp_down_proj_weight_cast_fp16, x = x_209_cast_fp16)[name = string("hidden_states_207_cast_fp16")]; tensor hidden_states_209_cast_fp16 = add(x = hidden_states_205_cast_fp16, y = hidden_states_207_cast_fp16)[name = string("hidden_states_209_cast_fp16")]; fp16 const_210_promoted_to_fp16 = const()[name = string("const_210_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7647_cast_fp16 = mul(x = hidden_states_209_cast_fp16, y = const_210_promoted_to_fp16)[name = string("op_7647_cast_fp16")]; int32 var_7645 = const()[name = string("op_7645"), val = int32(1)]; bool doubled_169_interleave_0 = const()[name = string("doubled_169_interleave_0"), val = bool(false)]; tensor doubled_169_cast_fp16 = concat(axis = var_7645, interleave = doubled_169_interleave_0, values = (hidden_states_209_cast_fp16, var_7647_cast_fp16))[name = string("doubled_169_cast_fp16")]; tensor out_85_axes_0 = const()[name = string("out_85_axes_0"), val = tensor([1])]; tensor out_85_gamma_0_to_fp16 = const()[name = string("out_85_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527721408)))]; fp16 var_7657_to_fp16 = const()[name = string("op_7657_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_85_cast_fp16 = layer_norm(axes = out_85_axes_0, epsilon = var_7657_to_fp16, gamma = out_85_gamma_0_to_fp16, x = doubled_169_cast_fp16)[name = string("out_85_cast_fp16")]; tensor var_7668_split_sizes_0 = const()[name = string("op_7668_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7668_axis_0 = const()[name = string("op_7668_axis_0"), val = int32(1)]; tensor var_7668_cast_fp16_0, tensor var_7668_cast_fp16_1 = split(axis = var_7668_axis_0, split_sizes = var_7668_split_sizes_0, x = out_85_cast_fp16)[name = string("op_7668_cast_fp16")]; tensor query_states_127_strides_0 = const()[name = string("query_states_127_strides_0"), val = tensor([1, 1])]; string query_states_127_pad_type_0 = const()[name = string("query_states_127_pad_type_0"), val = string("valid")]; tensor query_states_127_pad_0 = const()[name = string("query_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_127_dilations_0 = const()[name = string("query_states_127_dilations_0"), val = tensor([1, 1])]; int32 query_states_127_groups_0 = const()[name = string("query_states_127_groups_0"), val = int32(1)]; tensor query_states_127_cast_fp16 = conv(dilations = query_states_127_dilations_0, groups = query_states_127_groups_0, pad = query_states_127_pad_0, pad_type = query_states_127_pad_type_0, strides = query_states_127_strides_0, weight = layers_21_self_attn_q_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("query_states_127_cast_fp16")]; tensor key_states_211_strides_0 = const()[name = string("key_states_211_strides_0"), val = tensor([1, 1])]; string key_states_211_pad_type_0 = const()[name = string("key_states_211_pad_type_0"), val = string("valid")]; tensor key_states_211_pad_0 = const()[name = string("key_states_211_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_211_dilations_0 = const()[name = string("key_states_211_dilations_0"), val = tensor([1, 1])]; int32 key_states_211_groups_0 = const()[name = string("key_states_211_groups_0"), val = int32(1)]; tensor key_states_211_cast_fp16 = conv(dilations = key_states_211_dilations_0, groups = key_states_211_groups_0, pad = key_states_211_pad_0, pad_type = key_states_211_pad_type_0, strides = key_states_211_strides_0, weight = layers_21_self_attn_k_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("key_states_211_cast_fp16")]; tensor layers_21_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_21_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527729664)))]; tensor value_states_127_strides_0 = const()[name = string("value_states_127_strides_0"), val = tensor([1, 1])]; string value_states_127_pad_type_0 = const()[name = string("value_states_127_pad_type_0"), val = string("valid")]; tensor value_states_127_pad_0 = const()[name = string("value_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_127_dilations_0 = const()[name = string("value_states_127_dilations_0"), val = tensor([1, 1])]; int32 value_states_127_groups_0 = const()[name = string("value_states_127_groups_0"), val = int32(1)]; tensor value_states_127_cast_fp16 = conv(dilations = value_states_127_dilations_0, groups = value_states_127_groups_0, pad = value_states_127_pad_0, pad_type = value_states_127_pad_type_0, strides = value_states_127_strides_0, weight = layers_21_self_attn_v_proj_weight_to_fp16, x = var_7668_cast_fp16_0)[name = string("value_states_127_cast_fp16")]; tensor concat_252x = const()[name = string("concat_252x"), val = tensor([1, 16, 128, -1])]; tensor x_211_cast_fp16 = reshape(shape = concat_252x, x = query_states_127_cast_fp16)[name = string("x_211_cast_fp16")]; tensor concat_253x = const()[name = string("concat_253x"), val = tensor([1, 2, 128, -1])]; tensor var_7725_cast_fp16 = reshape(shape = concat_253x, x = key_states_211_cast_fp16)[name = string("op_7725_cast_fp16")]; tensor concat_254x = const()[name = string("concat_254x"), val = tensor([1, 2, 128, -1])]; tensor var_7732_cast_fp16 = reshape(shape = concat_254x, x = value_states_127_cast_fp16)[name = string("op_7732_cast_fp16")]; tensor var_7736_cast_fp16 = mul(x = x_211_cast_fp16, y = var_869_cast_fp16)[name = string("op_7736_cast_fp16")]; tensor var_7737_split_sizes_0 = const()[name = string("op_7737_split_sizes_0"), val = tensor([64, 64])]; int32 var_7737_axis_0 = const()[name = string("op_7737_axis_0"), val = int32(-2)]; tensor var_7737_cast_fp16_0, tensor var_7737_cast_fp16_1 = split(axis = var_7737_axis_0, split_sizes = var_7737_split_sizes_0, x = x_211_cast_fp16)[name = string("op_7737_cast_fp16")]; fp16 const_212_promoted_to_fp16 = const()[name = string("const_212_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7739_cast_fp16 = mul(x = var_7737_cast_fp16_1, y = const_212_promoted_to_fp16)[name = string("op_7739_cast_fp16")]; int32 var_7741 = const()[name = string("op_7741"), val = int32(-2)]; bool var_7742_interleave_0 = const()[name = string("op_7742_interleave_0"), val = bool(false)]; tensor var_7742_cast_fp16 = concat(axis = var_7741, interleave = var_7742_interleave_0, values = (var_7739_cast_fp16, var_7737_cast_fp16_0))[name = string("op_7742_cast_fp16")]; tensor var_7743_cast_fp16 = mul(x = var_7742_cast_fp16, y = var_878_cast_fp16)[name = string("op_7743_cast_fp16")]; tensor query_states_129_cast_fp16 = add(x = var_7736_cast_fp16, y = var_7743_cast_fp16)[name = string("query_states_129_cast_fp16")]; tensor var_7749_cast_fp16 = mul(x = var_7725_cast_fp16, y = var_869_cast_fp16)[name = string("op_7749_cast_fp16")]; tensor var_7750_split_sizes_0 = const()[name = string("op_7750_split_sizes_0"), val = tensor([64, 64])]; int32 var_7750_axis_0 = const()[name = string("op_7750_axis_0"), val = int32(-2)]; tensor var_7750_cast_fp16_0, tensor var_7750_cast_fp16_1 = split(axis = var_7750_axis_0, split_sizes = var_7750_split_sizes_0, x = var_7725_cast_fp16)[name = string("op_7750_cast_fp16")]; fp16 const_213_promoted_to_fp16 = const()[name = string("const_213_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7752_cast_fp16 = mul(x = var_7750_cast_fp16_1, y = const_213_promoted_to_fp16)[name = string("op_7752_cast_fp16")]; int32 var_7754 = const()[name = string("op_7754"), val = int32(-2)]; bool var_7755_interleave_0 = const()[name = string("op_7755_interleave_0"), val = bool(false)]; tensor var_7755_cast_fp16 = concat(axis = var_7754, interleave = var_7755_interleave_0, values = (var_7752_cast_fp16, var_7750_cast_fp16_0))[name = string("op_7755_cast_fp16")]; tensor var_7756_cast_fp16 = mul(x = var_7755_cast_fp16, y = var_878_cast_fp16)[name = string("op_7756_cast_fp16")]; tensor key_states_215_cast_fp16 = add(x = var_7749_cast_fp16, y = var_7756_cast_fp16)[name = string("key_states_215_cast_fp16")]; tensor expand_dims_252 = const()[name = string("expand_dims_252"), val = tensor([21])]; tensor expand_dims_253 = const()[name = string("expand_dims_253"), val = tensor([0])]; tensor expand_dims_255 = const()[name = string("expand_dims_255"), val = tensor([0])]; int32 concat_257_axis_0 = const()[name = string("concat_257_axis_0"), val = int32(0)]; bool concat_257_interleave_0 = const()[name = string("concat_257_interleave_0"), val = bool(false)]; tensor concat_257 = concat(axis = concat_257_axis_0, interleave = concat_257_interleave_0, values = (expand_dims_252, expand_dims_253, position_id, expand_dims_255))[name = string("concat_257")]; tensor expand_dims_256 = const()[name = string("expand_dims_256"), val = tensor([22])]; tensor concat_258_values1_0 = const()[name = string("concat_258_values1_0"), val = tensor([0])]; tensor concat_258_values3_0 = const()[name = string("concat_258_values3_0"), val = tensor([0])]; int32 concat_258_axis_0 = const()[name = string("concat_258_axis_0"), val = int32(0)]; bool concat_258_interleave_0 = const()[name = string("concat_258_interleave_0"), val = bool(false)]; tensor concat_258 = concat(axis = concat_258_axis_0, interleave = concat_258_interleave_0, values = (expand_dims_256, concat_258_values1_0, cache_position_end, concat_258_values3_0))[name = string("concat_258")]; tensor key_states_217_perm_0 = const()[name = string("key_states_217_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_22_stride_0 = const()[name = string("key_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_217_cast_fp16 = transpose(perm = key_states_217_perm_0, x = key_states_215_cast_fp16)[name = string("transpose_192")]; tensor key_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = key_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = key_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_22_squeeze_mask_0, stride = key_cache_internal_tensor_assign_22_stride_0, update = key_states_217_cast_fp16, x = coreml_update_state_152)[name = string("key_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_22_cast_fp16, input = key_cache)[name = string("coreml_update_state_154_write_state")]; tensor coreml_update_state_154 = read_state(input = key_cache)[name = string("coreml_update_state_154")]; tensor value_states_129_perm_0 = const()[name = string("value_states_129_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_22_stride_0 = const()[name = string("value_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_129_cast_fp16 = transpose(perm = value_states_129_perm_0, x = var_7732_cast_fp16)[name = string("transpose_191")]; tensor value_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = value_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = value_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_22_squeeze_mask_0, stride = value_cache_internal_tensor_assign_22_stride_0, update = value_states_129_cast_fp16, x = coreml_update_state_153)[name = string("value_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_22_cast_fp16, input = value_cache)[name = string("coreml_update_state_155_write_state")]; tensor coreml_update_state_155 = read_state(input = value_cache)[name = string("coreml_update_state_155")]; tensor var_7826_begin_0 = const()[name = string("op_7826_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7826_end_0 = const()[name = string("op_7826_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7826_end_mask_0 = const()[name = string("op_7826_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7826_cast_fp16 = slice_by_index(begin = var_7826_begin_0, end = var_7826_end_0, end_mask = var_7826_end_mask_0, x = coreml_update_state_154)[name = string("op_7826_cast_fp16")]; tensor tile_42 = const()[name = string("tile_42"), val = tensor([1, 1])]; int32 var_7829_axis_0 = const()[name = string("op_7829_axis_0"), val = int32(1)]; tensor var_7829_cast_fp16_0, tensor var_7829_cast_fp16_1 = split(axis = var_7829_axis_0, split_sizes = tile_42, x = var_7826_cast_fp16)[name = string("op_7829_cast_fp16")]; tensor var_7836_begin_0 = const()[name = string("op_7836_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7836_end_0 = const()[name = string("op_7836_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7836_end_mask_0 = const()[name = string("op_7836_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7836_cast_fp16 = slice_by_index(begin = var_7836_begin_0, end = var_7836_end_0, end_mask = var_7836_end_mask_0, x = coreml_update_state_155)[name = string("op_7836_cast_fp16")]; tensor tile_43 = const()[name = string("tile_43"), val = tensor([1, 1])]; int32 var_7839_axis_0 = const()[name = string("op_7839_axis_0"), val = int32(1)]; tensor var_7839_cast_fp16_0, tensor var_7839_cast_fp16_1 = split(axis = var_7839_axis_0, split_sizes = tile_43, x = var_7836_cast_fp16)[name = string("op_7839_cast_fp16")]; tensor var_7842_split_sizes_0 = const()[name = string("op_7842_split_sizes_0"), val = tensor([8, 8])]; int32 var_7842_axis_0 = const()[name = string("op_7842_axis_0"), val = int32(1)]; tensor var_7842_0, tensor var_7842_1 = split(axis = var_7842_axis_0, split_sizes = var_7842_split_sizes_0, x = query_states_129_cast_fp16)[name = string("op_7842")]; bool attn_weights_337_transpose_x_0 = const()[name = string("attn_weights_337_transpose_x_0"), val = bool(false)]; bool attn_weights_337_transpose_y_0 = const()[name = string("attn_weights_337_transpose_y_0"), val = bool(false)]; tensor attn_weights_337_cast_fp16 = matmul(transpose_x = attn_weights_337_transpose_x_0, transpose_y = attn_weights_337_transpose_y_0, x = var_7829_cast_fp16_0, y = var_7842_0)[name = string("attn_weights_337_cast_fp16")]; fp16 var_7845_to_fp16 = const()[name = string("op_7845_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_339_cast_fp16 = mul(x = attn_weights_337_cast_fp16, y = var_7845_to_fp16)[name = string("attn_weights_339_cast_fp16")]; tensor attn_weights_341_cast_fp16 = add(x = attn_weights_339_cast_fp16, y = attn_mask_1)[name = string("attn_weights_341_cast_fp16")]; int32 var_7849 = const()[name = string("op_7849"), val = int32(-2)]; tensor attn_weights_343_cast_fp16 = softmax(axis = var_7849, x = attn_weights_341_cast_fp16)[name = string("attn_weights_343_cast_fp16")]; bool var_7855_transpose_x_1 = const()[name = string("op_7855_transpose_x_1"), val = bool(true)]; bool var_7855_transpose_y_1 = const()[name = string("op_7855_transpose_y_1"), val = bool(false)]; tensor var_7855_cast_fp16 = matmul(transpose_x = var_7855_transpose_x_1, transpose_y = var_7855_transpose_y_1, x = attn_weights_343_cast_fp16, y = var_7839_cast_fp16_0)[name = string("op_7855_cast_fp16")]; bool attn_weights_345_transpose_x_0 = const()[name = string("attn_weights_345_transpose_x_0"), val = bool(false)]; bool attn_weights_345_transpose_y_0 = const()[name = string("attn_weights_345_transpose_y_0"), val = bool(false)]; tensor attn_weights_345_cast_fp16 = matmul(transpose_x = attn_weights_345_transpose_x_0, transpose_y = attn_weights_345_transpose_y_0, x = var_7829_cast_fp16_1, y = var_7842_1)[name = string("attn_weights_345_cast_fp16")]; fp16 var_7857_to_fp16 = const()[name = string("op_7857_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_347_cast_fp16 = mul(x = attn_weights_345_cast_fp16, y = var_7857_to_fp16)[name = string("attn_weights_347_cast_fp16")]; tensor attn_weights_349_cast_fp16 = add(x = attn_weights_347_cast_fp16, y = attn_mask_1)[name = string("attn_weights_349_cast_fp16")]; int32 var_7861 = const()[name = string("op_7861"), val = int32(-2)]; tensor attn_weights_351_cast_fp16 = softmax(axis = var_7861, x = attn_weights_349_cast_fp16)[name = string("attn_weights_351_cast_fp16")]; bool attn_output_169_transpose_x_1 = const()[name = string("attn_output_169_transpose_x_1"), val = bool(true)]; bool attn_output_169_transpose_y_1 = const()[name = string("attn_output_169_transpose_y_1"), val = bool(false)]; tensor attn_output_169_cast_fp16 = matmul(transpose_x = attn_output_169_transpose_x_1, transpose_y = attn_output_169_transpose_y_1, x = attn_weights_351_cast_fp16, y = var_7839_cast_fp16_1)[name = string("attn_output_169_cast_fp16")]; int32 var_7869 = const()[name = string("op_7869"), val = int32(1)]; bool attn_output_171_interleave_0 = const()[name = string("attn_output_171_interleave_0"), val = bool(false)]; tensor attn_output_171_cast_fp16 = concat(axis = var_7869, interleave = attn_output_171_interleave_0, values = (var_7855_cast_fp16, attn_output_169_cast_fp16))[name = string("attn_output_171_cast_fp16")]; tensor var_7873_perm_0 = const()[name = string("op_7873_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_263x = const()[name = string("concat_263x"), val = tensor([1, 2048, 1, -1])]; tensor var_7873_cast_fp16 = transpose(perm = var_7873_perm_0, x = attn_output_171_cast_fp16)[name = string("transpose_190")]; tensor attn_output_175_cast_fp16 = reshape(shape = concat_263x, x = var_7873_cast_fp16)[name = string("attn_output_175_cast_fp16")]; tensor hidden_states_213_strides_0 = const()[name = string("hidden_states_213_strides_0"), val = tensor([1, 1])]; string hidden_states_213_pad_type_0 = const()[name = string("hidden_states_213_pad_type_0"), val = string("valid")]; tensor hidden_states_213_pad_0 = const()[name = string("hidden_states_213_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_213_dilations_0 = const()[name = string("hidden_states_213_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_213_groups_0 = const()[name = string("hidden_states_213_groups_0"), val = int32(1)]; tensor hidden_states_213_cast_fp16 = conv(dilations = hidden_states_213_dilations_0, groups = hidden_states_213_groups_0, pad = hidden_states_213_pad_0, pad_type = hidden_states_213_pad_type_0, strides = hidden_states_213_strides_0, weight = layers_21_self_attn_o_proj_weight_cast_fp16, x = attn_output_175_cast_fp16)[name = string("hidden_states_213_cast_fp16")]; tensor hidden_states_215_cast_fp16 = add(x = hidden_states_209_cast_fp16, y = hidden_states_213_cast_fp16)[name = string("hidden_states_215_cast_fp16")]; fp16 const_218_promoted_to_fp16 = const()[name = string("const_218_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7906_cast_fp16 = mul(x = hidden_states_215_cast_fp16, y = const_218_promoted_to_fp16)[name = string("op_7906_cast_fp16")]; int32 var_7904 = const()[name = string("op_7904"), val = int32(1)]; bool doubled_173_interleave_0 = const()[name = string("doubled_173_interleave_0"), val = bool(false)]; tensor doubled_173_cast_fp16 = concat(axis = var_7904, interleave = doubled_173_interleave_0, values = (hidden_states_215_cast_fp16, var_7906_cast_fp16))[name = string("doubled_173_cast_fp16")]; tensor out_87_axes_0 = const()[name = string("out_87_axes_0"), val = tensor([1])]; tensor out_87_gamma_0_to_fp16 = const()[name = string("out_87_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528778304)))]; fp16 var_7916_to_fp16 = const()[name = string("op_7916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_87_cast_fp16 = layer_norm(axes = out_87_axes_0, epsilon = var_7916_to_fp16, gamma = out_87_gamma_0_to_fp16, x = doubled_173_cast_fp16)[name = string("out_87_cast_fp16")]; tensor var_7927_split_sizes_0 = const()[name = string("op_7927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7927_axis_0 = const()[name = string("op_7927_axis_0"), val = int32(1)]; tensor var_7927_cast_fp16_0, tensor var_7927_cast_fp16_1 = split(axis = var_7927_axis_0, split_sizes = var_7927_split_sizes_0, x = out_87_cast_fp16)[name = string("op_7927_cast_fp16")]; tensor input_43_strides_0 = const()[name = string("input_43_strides_0"), val = tensor([1, 1])]; string input_43_pad_type_0 = const()[name = string("input_43_pad_type_0"), val = string("valid")]; tensor input_43_pad_0 = const()[name = string("input_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_43_dilations_0 = const()[name = string("input_43_dilations_0"), val = tensor([1, 1])]; int32 input_43_groups_0 = const()[name = string("input_43_groups_0"), val = int32(1)]; tensor input_43_cast_fp16 = conv(dilations = input_43_dilations_0, groups = input_43_groups_0, pad = input_43_pad_0, pad_type = input_43_pad_type_0, strides = input_43_strides_0, weight = layers_21_mlp_gate_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("input_43_cast_fp16")]; tensor var_7944_cast_fp16 = silu(x = input_43_cast_fp16)[name = string("op_7944_cast_fp16")]; tensor var_7950_strides_0 = const()[name = string("op_7950_strides_0"), val = tensor([1, 1])]; string var_7950_pad_type_0 = const()[name = string("op_7950_pad_type_0"), val = string("valid")]; tensor var_7950_pad_0 = const()[name = string("op_7950_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7950_dilations_0 = const()[name = string("op_7950_dilations_0"), val = tensor([1, 1])]; int32 var_7950_groups_0 = const()[name = string("op_7950_groups_0"), val = int32(1)]; tensor var_7950_cast_fp16 = conv(dilations = var_7950_dilations_0, groups = var_7950_groups_0, pad = var_7950_pad_0, pad_type = var_7950_pad_type_0, strides = var_7950_strides_0, weight = layers_21_mlp_up_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("op_7950_cast_fp16")]; tensor x_219_cast_fp16 = mul(x = var_7944_cast_fp16, y = var_7950_cast_fp16)[name = string("x_219_cast_fp16")]; tensor hidden_states_217_strides_0 = const()[name = string("hidden_states_217_strides_0"), val = tensor([1, 1])]; string hidden_states_217_pad_type_0 = const()[name = string("hidden_states_217_pad_type_0"), val = string("valid")]; tensor hidden_states_217_pad_0 = const()[name = string("hidden_states_217_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_217_dilations_0 = const()[name = string("hidden_states_217_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_217_groups_0 = const()[name = string("hidden_states_217_groups_0"), val = int32(1)]; tensor hidden_states_217_cast_fp16 = conv(dilations = hidden_states_217_dilations_0, groups = hidden_states_217_groups_0, pad = hidden_states_217_pad_0, pad_type = hidden_states_217_pad_type_0, strides = hidden_states_217_strides_0, weight = layers_21_mlp_down_proj_weight_cast_fp16, x = x_219_cast_fp16)[name = string("hidden_states_217_cast_fp16")]; tensor hidden_states_219_cast_fp16 = add(x = hidden_states_215_cast_fp16, y = hidden_states_217_cast_fp16)[name = string("hidden_states_219_cast_fp16")]; fp16 const_220_promoted_to_fp16 = const()[name = string("const_220_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7968_cast_fp16 = mul(x = hidden_states_219_cast_fp16, y = const_220_promoted_to_fp16)[name = string("op_7968_cast_fp16")]; int32 var_7966 = const()[name = string("op_7966"), val = int32(1)]; bool doubled_177_interleave_0 = const()[name = string("doubled_177_interleave_0"), val = bool(false)]; tensor doubled_177_cast_fp16 = concat(axis = var_7966, interleave = doubled_177_interleave_0, values = (hidden_states_219_cast_fp16, var_7968_cast_fp16))[name = string("doubled_177_cast_fp16")]; tensor out_89_axes_0 = const()[name = string("out_89_axes_0"), val = tensor([1])]; tensor out_89_gamma_0_to_fp16 = const()[name = string("out_89_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528786560)))]; fp16 var_7978_to_fp16 = const()[name = string("op_7978_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_89_cast_fp16 = layer_norm(axes = out_89_axes_0, epsilon = var_7978_to_fp16, gamma = out_89_gamma_0_to_fp16, x = doubled_177_cast_fp16)[name = string("out_89_cast_fp16")]; tensor var_7989_split_sizes_0 = const()[name = string("op_7989_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7989_axis_0 = const()[name = string("op_7989_axis_0"), val = int32(1)]; tensor var_7989_cast_fp16_0, tensor var_7989_cast_fp16_1 = split(axis = var_7989_axis_0, split_sizes = var_7989_split_sizes_0, x = out_89_cast_fp16)[name = string("op_7989_cast_fp16")]; tensor query_states_133_strides_0 = const()[name = string("query_states_133_strides_0"), val = tensor([1, 1])]; string query_states_133_pad_type_0 = const()[name = string("query_states_133_pad_type_0"), val = string("valid")]; tensor query_states_133_pad_0 = const()[name = string("query_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_133_dilations_0 = const()[name = string("query_states_133_dilations_0"), val = tensor([1, 1])]; int32 query_states_133_groups_0 = const()[name = string("query_states_133_groups_0"), val = int32(1)]; tensor query_states_133_cast_fp16 = conv(dilations = query_states_133_dilations_0, groups = query_states_133_groups_0, pad = query_states_133_pad_0, pad_type = query_states_133_pad_type_0, strides = query_states_133_strides_0, weight = layers_22_self_attn_q_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("query_states_133_cast_fp16")]; tensor key_states_221_strides_0 = const()[name = string("key_states_221_strides_0"), val = tensor([1, 1])]; string key_states_221_pad_type_0 = const()[name = string("key_states_221_pad_type_0"), val = string("valid")]; tensor key_states_221_pad_0 = const()[name = string("key_states_221_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_221_dilations_0 = const()[name = string("key_states_221_dilations_0"), val = tensor([1, 1])]; int32 key_states_221_groups_0 = const()[name = string("key_states_221_groups_0"), val = int32(1)]; tensor key_states_221_cast_fp16 = conv(dilations = key_states_221_dilations_0, groups = key_states_221_groups_0, pad = key_states_221_pad_0, pad_type = key_states_221_pad_type_0, strides = key_states_221_strides_0, weight = layers_22_self_attn_k_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("key_states_221_cast_fp16")]; tensor layers_22_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528794816)))]; tensor value_states_133_strides_0 = const()[name = string("value_states_133_strides_0"), val = tensor([1, 1])]; string value_states_133_pad_type_0 = const()[name = string("value_states_133_pad_type_0"), val = string("valid")]; tensor value_states_133_pad_0 = const()[name = string("value_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_133_dilations_0 = const()[name = string("value_states_133_dilations_0"), val = tensor([1, 1])]; int32 value_states_133_groups_0 = const()[name = string("value_states_133_groups_0"), val = int32(1)]; tensor value_states_133_cast_fp16 = conv(dilations = value_states_133_dilations_0, groups = value_states_133_groups_0, pad = value_states_133_pad_0, pad_type = value_states_133_pad_type_0, strides = value_states_133_strides_0, weight = layers_22_self_attn_v_proj_weight_to_fp16, x = var_7989_cast_fp16_0)[name = string("value_states_133_cast_fp16")]; tensor concat_264x = const()[name = string("concat_264x"), val = tensor([1, 16, 128, -1])]; tensor x_221_cast_fp16 = reshape(shape = concat_264x, x = query_states_133_cast_fp16)[name = string("x_221_cast_fp16")]; tensor concat_265x = const()[name = string("concat_265x"), val = tensor([1, 2, 128, -1])]; tensor var_8046_cast_fp16 = reshape(shape = concat_265x, x = key_states_221_cast_fp16)[name = string("op_8046_cast_fp16")]; tensor concat_266x = const()[name = string("concat_266x"), val = tensor([1, 2, 128, -1])]; tensor var_8053_cast_fp16 = reshape(shape = concat_266x, x = value_states_133_cast_fp16)[name = string("op_8053_cast_fp16")]; tensor var_8057_cast_fp16 = mul(x = x_221_cast_fp16, y = var_869_cast_fp16)[name = string("op_8057_cast_fp16")]; tensor var_8058_split_sizes_0 = const()[name = string("op_8058_split_sizes_0"), val = tensor([64, 64])]; int32 var_8058_axis_0 = const()[name = string("op_8058_axis_0"), val = int32(-2)]; tensor var_8058_cast_fp16_0, tensor var_8058_cast_fp16_1 = split(axis = var_8058_axis_0, split_sizes = var_8058_split_sizes_0, x = x_221_cast_fp16)[name = string("op_8058_cast_fp16")]; fp16 const_222_promoted_to_fp16 = const()[name = string("const_222_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8060_cast_fp16 = mul(x = var_8058_cast_fp16_1, y = const_222_promoted_to_fp16)[name = string("op_8060_cast_fp16")]; int32 var_8062 = const()[name = string("op_8062"), val = int32(-2)]; bool var_8063_interleave_0 = const()[name = string("op_8063_interleave_0"), val = bool(false)]; tensor var_8063_cast_fp16 = concat(axis = var_8062, interleave = var_8063_interleave_0, values = (var_8060_cast_fp16, var_8058_cast_fp16_0))[name = string("op_8063_cast_fp16")]; tensor var_8064_cast_fp16 = mul(x = var_8063_cast_fp16, y = var_878_cast_fp16)[name = string("op_8064_cast_fp16")]; tensor query_states_135_cast_fp16 = add(x = var_8057_cast_fp16, y = var_8064_cast_fp16)[name = string("query_states_135_cast_fp16")]; tensor var_8070_cast_fp16 = mul(x = var_8046_cast_fp16, y = var_869_cast_fp16)[name = string("op_8070_cast_fp16")]; tensor var_8071_split_sizes_0 = const()[name = string("op_8071_split_sizes_0"), val = tensor([64, 64])]; int32 var_8071_axis_0 = const()[name = string("op_8071_axis_0"), val = int32(-2)]; tensor var_8071_cast_fp16_0, tensor var_8071_cast_fp16_1 = split(axis = var_8071_axis_0, split_sizes = var_8071_split_sizes_0, x = var_8046_cast_fp16)[name = string("op_8071_cast_fp16")]; fp16 const_223_promoted_to_fp16 = const()[name = string("const_223_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8073_cast_fp16 = mul(x = var_8071_cast_fp16_1, y = const_223_promoted_to_fp16)[name = string("op_8073_cast_fp16")]; int32 var_8075 = const()[name = string("op_8075"), val = int32(-2)]; bool var_8076_interleave_0 = const()[name = string("op_8076_interleave_0"), val = bool(false)]; tensor var_8076_cast_fp16 = concat(axis = var_8075, interleave = var_8076_interleave_0, values = (var_8073_cast_fp16, var_8071_cast_fp16_0))[name = string("op_8076_cast_fp16")]; tensor var_8077_cast_fp16 = mul(x = var_8076_cast_fp16, y = var_878_cast_fp16)[name = string("op_8077_cast_fp16")]; tensor key_states_225_cast_fp16 = add(x = var_8070_cast_fp16, y = var_8077_cast_fp16)[name = string("key_states_225_cast_fp16")]; tensor expand_dims_264 = const()[name = string("expand_dims_264"), val = tensor([22])]; tensor expand_dims_265 = const()[name = string("expand_dims_265"), val = tensor([0])]; tensor expand_dims_267 = const()[name = string("expand_dims_267"), val = tensor([0])]; int32 concat_269_axis_0 = const()[name = string("concat_269_axis_0"), val = int32(0)]; bool concat_269_interleave_0 = const()[name = string("concat_269_interleave_0"), val = bool(false)]; tensor concat_269 = concat(axis = concat_269_axis_0, interleave = concat_269_interleave_0, values = (expand_dims_264, expand_dims_265, position_id, expand_dims_267))[name = string("concat_269")]; tensor expand_dims_268 = const()[name = string("expand_dims_268"), val = tensor([23])]; tensor concat_270_values1_0 = const()[name = string("concat_270_values1_0"), val = tensor([0])]; tensor concat_270_values3_0 = const()[name = string("concat_270_values3_0"), val = tensor([0])]; int32 concat_270_axis_0 = const()[name = string("concat_270_axis_0"), val = int32(0)]; bool concat_270_interleave_0 = const()[name = string("concat_270_interleave_0"), val = bool(false)]; tensor concat_270 = concat(axis = concat_270_axis_0, interleave = concat_270_interleave_0, values = (expand_dims_268, concat_270_values1_0, cache_position_end, concat_270_values3_0))[name = string("concat_270")]; tensor key_states_227_perm_0 = const()[name = string("key_states_227_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_23_stride_0 = const()[name = string("key_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_227_cast_fp16 = transpose(perm = key_states_227_perm_0, x = key_states_225_cast_fp16)[name = string("transpose_189")]; tensor key_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = key_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = key_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_23_squeeze_mask_0, stride = key_cache_internal_tensor_assign_23_stride_0, update = key_states_227_cast_fp16, x = coreml_update_state_154)[name = string("key_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_23_cast_fp16, input = key_cache)[name = string("coreml_update_state_156_write_state")]; tensor coreml_update_state_156 = read_state(input = key_cache)[name = string("coreml_update_state_156")]; tensor value_states_135_perm_0 = const()[name = string("value_states_135_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_23_stride_0 = const()[name = string("value_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_135_cast_fp16 = transpose(perm = value_states_135_perm_0, x = var_8053_cast_fp16)[name = string("transpose_188")]; tensor value_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = value_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = value_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_23_squeeze_mask_0, stride = value_cache_internal_tensor_assign_23_stride_0, update = value_states_135_cast_fp16, x = coreml_update_state_155)[name = string("value_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_23_cast_fp16, input = value_cache)[name = string("coreml_update_state_157_write_state")]; tensor coreml_update_state_157 = read_state(input = value_cache)[name = string("coreml_update_state_157")]; tensor var_8147_begin_0 = const()[name = string("op_8147_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8147_end_0 = const()[name = string("op_8147_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8147_end_mask_0 = const()[name = string("op_8147_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8147_cast_fp16 = slice_by_index(begin = var_8147_begin_0, end = var_8147_end_0, end_mask = var_8147_end_mask_0, x = coreml_update_state_156)[name = string("op_8147_cast_fp16")]; tensor tile_44 = const()[name = string("tile_44"), val = tensor([1, 1])]; int32 var_8150_axis_0 = const()[name = string("op_8150_axis_0"), val = int32(1)]; tensor var_8150_cast_fp16_0, tensor var_8150_cast_fp16_1 = split(axis = var_8150_axis_0, split_sizes = tile_44, x = var_8147_cast_fp16)[name = string("op_8150_cast_fp16")]; tensor var_8157_begin_0 = const()[name = string("op_8157_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8157_end_0 = const()[name = string("op_8157_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8157_end_mask_0 = const()[name = string("op_8157_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8157_cast_fp16 = slice_by_index(begin = var_8157_begin_0, end = var_8157_end_0, end_mask = var_8157_end_mask_0, x = coreml_update_state_157)[name = string("op_8157_cast_fp16")]; tensor tile_45 = const()[name = string("tile_45"), val = tensor([1, 1])]; int32 var_8160_axis_0 = const()[name = string("op_8160_axis_0"), val = int32(1)]; tensor var_8160_cast_fp16_0, tensor var_8160_cast_fp16_1 = split(axis = var_8160_axis_0, split_sizes = tile_45, x = var_8157_cast_fp16)[name = string("op_8160_cast_fp16")]; tensor var_8163_split_sizes_0 = const()[name = string("op_8163_split_sizes_0"), val = tensor([8, 8])]; int32 var_8163_axis_0 = const()[name = string("op_8163_axis_0"), val = int32(1)]; tensor var_8163_0, tensor var_8163_1 = split(axis = var_8163_axis_0, split_sizes = var_8163_split_sizes_0, x = query_states_135_cast_fp16)[name = string("op_8163")]; bool attn_weights_353_transpose_x_0 = const()[name = string("attn_weights_353_transpose_x_0"), val = bool(false)]; bool attn_weights_353_transpose_y_0 = const()[name = string("attn_weights_353_transpose_y_0"), val = bool(false)]; tensor attn_weights_353_cast_fp16 = matmul(transpose_x = attn_weights_353_transpose_x_0, transpose_y = attn_weights_353_transpose_y_0, x = var_8150_cast_fp16_0, y = var_8163_0)[name = string("attn_weights_353_cast_fp16")]; fp16 var_8166_to_fp16 = const()[name = string("op_8166_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_355_cast_fp16 = mul(x = attn_weights_353_cast_fp16, y = var_8166_to_fp16)[name = string("attn_weights_355_cast_fp16")]; tensor attn_weights_357_cast_fp16 = add(x = attn_weights_355_cast_fp16, y = attn_mask_1)[name = string("attn_weights_357_cast_fp16")]; int32 var_8170 = const()[name = string("op_8170"), val = int32(-2)]; tensor attn_weights_359_cast_fp16 = softmax(axis = var_8170, x = attn_weights_357_cast_fp16)[name = string("attn_weights_359_cast_fp16")]; bool var_8176_transpose_x_1 = const()[name = string("op_8176_transpose_x_1"), val = bool(true)]; bool var_8176_transpose_y_1 = const()[name = string("op_8176_transpose_y_1"), val = bool(false)]; tensor var_8176_cast_fp16 = matmul(transpose_x = var_8176_transpose_x_1, transpose_y = var_8176_transpose_y_1, x = attn_weights_359_cast_fp16, y = var_8160_cast_fp16_0)[name = string("op_8176_cast_fp16")]; bool attn_weights_361_transpose_x_0 = const()[name = string("attn_weights_361_transpose_x_0"), val = bool(false)]; bool attn_weights_361_transpose_y_0 = const()[name = string("attn_weights_361_transpose_y_0"), val = bool(false)]; tensor attn_weights_361_cast_fp16 = matmul(transpose_x = attn_weights_361_transpose_x_0, transpose_y = attn_weights_361_transpose_y_0, x = var_8150_cast_fp16_1, y = var_8163_1)[name = string("attn_weights_361_cast_fp16")]; fp16 var_8178_to_fp16 = const()[name = string("op_8178_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_363_cast_fp16 = mul(x = attn_weights_361_cast_fp16, y = var_8178_to_fp16)[name = string("attn_weights_363_cast_fp16")]; tensor attn_weights_365_cast_fp16 = add(x = attn_weights_363_cast_fp16, y = attn_mask_1)[name = string("attn_weights_365_cast_fp16")]; int32 var_8182 = const()[name = string("op_8182"), val = int32(-2)]; tensor attn_weights_367_cast_fp16 = softmax(axis = var_8182, x = attn_weights_365_cast_fp16)[name = string("attn_weights_367_cast_fp16")]; bool attn_output_177_transpose_x_1 = const()[name = string("attn_output_177_transpose_x_1"), val = bool(true)]; bool attn_output_177_transpose_y_1 = const()[name = string("attn_output_177_transpose_y_1"), val = bool(false)]; tensor attn_output_177_cast_fp16 = matmul(transpose_x = attn_output_177_transpose_x_1, transpose_y = attn_output_177_transpose_y_1, x = attn_weights_367_cast_fp16, y = var_8160_cast_fp16_1)[name = string("attn_output_177_cast_fp16")]; int32 var_8190 = const()[name = string("op_8190"), val = int32(1)]; bool attn_output_179_interleave_0 = const()[name = string("attn_output_179_interleave_0"), val = bool(false)]; tensor attn_output_179_cast_fp16 = concat(axis = var_8190, interleave = attn_output_179_interleave_0, values = (var_8176_cast_fp16, attn_output_177_cast_fp16))[name = string("attn_output_179_cast_fp16")]; tensor var_8194_perm_0 = const()[name = string("op_8194_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_275x = const()[name = string("concat_275x"), val = tensor([1, 2048, 1, -1])]; tensor var_8194_cast_fp16 = transpose(perm = var_8194_perm_0, x = attn_output_179_cast_fp16)[name = string("transpose_187")]; tensor attn_output_183_cast_fp16 = reshape(shape = concat_275x, x = var_8194_cast_fp16)[name = string("attn_output_183_cast_fp16")]; tensor layers_22_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1529843456)))]; tensor hidden_states_223_strides_0 = const()[name = string("hidden_states_223_strides_0"), val = tensor([1, 1])]; string hidden_states_223_pad_type_0 = const()[name = string("hidden_states_223_pad_type_0"), val = string("valid")]; tensor hidden_states_223_pad_0 = const()[name = string("hidden_states_223_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_223_dilations_0 = const()[name = string("hidden_states_223_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_223_groups_0 = const()[name = string("hidden_states_223_groups_0"), val = int32(1)]; tensor hidden_states_223_cast_fp16 = conv(dilations = hidden_states_223_dilations_0, groups = hidden_states_223_groups_0, pad = hidden_states_223_pad_0, pad_type = hidden_states_223_pad_type_0, strides = hidden_states_223_strides_0, weight = layers_22_self_attn_o_proj_weight_to_fp16, x = attn_output_183_cast_fp16)[name = string("hidden_states_223_cast_fp16")]; tensor hidden_states_225_cast_fp16 = add(x = hidden_states_219_cast_fp16, y = hidden_states_223_cast_fp16)[name = string("hidden_states_225_cast_fp16")]; fp16 const_228_promoted_to_fp16 = const()[name = string("const_228_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8227_cast_fp16 = mul(x = hidden_states_225_cast_fp16, y = const_228_promoted_to_fp16)[name = string("op_8227_cast_fp16")]; int32 var_8225 = const()[name = string("op_8225"), val = int32(1)]; bool doubled_181_interleave_0 = const()[name = string("doubled_181_interleave_0"), val = bool(false)]; tensor doubled_181_cast_fp16 = concat(axis = var_8225, interleave = doubled_181_interleave_0, values = (hidden_states_225_cast_fp16, var_8227_cast_fp16))[name = string("doubled_181_cast_fp16")]; tensor out_91_axes_0 = const()[name = string("out_91_axes_0"), val = tensor([1])]; tensor out_91_gamma_0_to_fp16 = const()[name = string("out_91_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538232128)))]; fp16 var_8237_to_fp16 = const()[name = string("op_8237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_91_cast_fp16 = layer_norm(axes = out_91_axes_0, epsilon = var_8237_to_fp16, gamma = out_91_gamma_0_to_fp16, x = doubled_181_cast_fp16)[name = string("out_91_cast_fp16")]; tensor var_8248_split_sizes_0 = const()[name = string("op_8248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8248_axis_0 = const()[name = string("op_8248_axis_0"), val = int32(1)]; tensor var_8248_cast_fp16_0, tensor var_8248_cast_fp16_1 = split(axis = var_8248_axis_0, split_sizes = var_8248_split_sizes_0, x = out_91_cast_fp16)[name = string("op_8248_cast_fp16")]; tensor input_45_strides_0 = const()[name = string("input_45_strides_0"), val = tensor([1, 1])]; string input_45_pad_type_0 = const()[name = string("input_45_pad_type_0"), val = string("valid")]; tensor input_45_pad_0 = const()[name = string("input_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_45_dilations_0 = const()[name = string("input_45_dilations_0"), val = tensor([1, 1])]; int32 input_45_groups_0 = const()[name = string("input_45_groups_0"), val = int32(1)]; tensor input_45_cast_fp16 = conv(dilations = input_45_dilations_0, groups = input_45_groups_0, pad = input_45_pad_0, pad_type = input_45_pad_type_0, strides = input_45_strides_0, weight = layers_22_mlp_gate_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("input_45_cast_fp16")]; tensor var_8265_cast_fp16 = silu(x = input_45_cast_fp16)[name = string("op_8265_cast_fp16")]; tensor var_8271_strides_0 = const()[name = string("op_8271_strides_0"), val = tensor([1, 1])]; string var_8271_pad_type_0 = const()[name = string("op_8271_pad_type_0"), val = string("valid")]; tensor var_8271_pad_0 = const()[name = string("op_8271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8271_dilations_0 = const()[name = string("op_8271_dilations_0"), val = tensor([1, 1])]; int32 var_8271_groups_0 = const()[name = string("op_8271_groups_0"), val = int32(1)]; tensor var_8271_cast_fp16 = conv(dilations = var_8271_dilations_0, groups = var_8271_groups_0, pad = var_8271_pad_0, pad_type = var_8271_pad_type_0, strides = var_8271_strides_0, weight = layers_22_mlp_up_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("op_8271_cast_fp16")]; tensor x_229_cast_fp16 = mul(x = var_8265_cast_fp16, y = var_8271_cast_fp16)[name = string("x_229_cast_fp16")]; tensor hidden_states_227_strides_0 = const()[name = string("hidden_states_227_strides_0"), val = tensor([1, 1])]; string hidden_states_227_pad_type_0 = const()[name = string("hidden_states_227_pad_type_0"), val = string("valid")]; tensor hidden_states_227_pad_0 = const()[name = string("hidden_states_227_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_227_dilations_0 = const()[name = string("hidden_states_227_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_227_groups_0 = const()[name = string("hidden_states_227_groups_0"), val = int32(1)]; tensor hidden_states_227_cast_fp16 = conv(dilations = hidden_states_227_dilations_0, groups = hidden_states_227_groups_0, pad = hidden_states_227_pad_0, pad_type = hidden_states_227_pad_type_0, strides = hidden_states_227_strides_0, weight = layers_22_mlp_down_proj_weight_cast_fp16, x = x_229_cast_fp16)[name = string("hidden_states_227_cast_fp16")]; tensor hidden_states_229_cast_fp16 = add(x = hidden_states_225_cast_fp16, y = hidden_states_227_cast_fp16)[name = string("hidden_states_229_cast_fp16")]; fp16 const_230_promoted_to_fp16 = const()[name = string("const_230_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8289_cast_fp16 = mul(x = hidden_states_229_cast_fp16, y = const_230_promoted_to_fp16)[name = string("op_8289_cast_fp16")]; int32 var_8287 = const()[name = string("op_8287"), val = int32(1)]; bool doubled_185_interleave_0 = const()[name = string("doubled_185_interleave_0"), val = bool(false)]; tensor doubled_185_cast_fp16 = concat(axis = var_8287, interleave = doubled_185_interleave_0, values = (hidden_states_229_cast_fp16, var_8289_cast_fp16))[name = string("doubled_185_cast_fp16")]; tensor out_93_axes_0 = const()[name = string("out_93_axes_0"), val = tensor([1])]; tensor out_93_gamma_0_to_fp16 = const()[name = string("out_93_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538240384)))]; fp16 var_8299_to_fp16 = const()[name = string("op_8299_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_93_cast_fp16 = layer_norm(axes = out_93_axes_0, epsilon = var_8299_to_fp16, gamma = out_93_gamma_0_to_fp16, x = doubled_185_cast_fp16)[name = string("out_93_cast_fp16")]; tensor var_8310_split_sizes_0 = const()[name = string("op_8310_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8310_axis_0 = const()[name = string("op_8310_axis_0"), val = int32(1)]; tensor var_8310_cast_fp16_0, tensor var_8310_cast_fp16_1 = split(axis = var_8310_axis_0, split_sizes = var_8310_split_sizes_0, x = out_93_cast_fp16)[name = string("op_8310_cast_fp16")]; tensor query_states_139_strides_0 = const()[name = string("query_states_139_strides_0"), val = tensor([1, 1])]; string query_states_139_pad_type_0 = const()[name = string("query_states_139_pad_type_0"), val = string("valid")]; tensor query_states_139_pad_0 = const()[name = string("query_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_139_dilations_0 = const()[name = string("query_states_139_dilations_0"), val = tensor([1, 1])]; int32 query_states_139_groups_0 = const()[name = string("query_states_139_groups_0"), val = int32(1)]; tensor query_states_139_cast_fp16 = conv(dilations = query_states_139_dilations_0, groups = query_states_139_groups_0, pad = query_states_139_pad_0, pad_type = query_states_139_pad_type_0, strides = query_states_139_strides_0, weight = layers_23_self_attn_q_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("query_states_139_cast_fp16")]; tensor key_states_231_strides_0 = const()[name = string("key_states_231_strides_0"), val = tensor([1, 1])]; string key_states_231_pad_type_0 = const()[name = string("key_states_231_pad_type_0"), val = string("valid")]; tensor key_states_231_pad_0 = const()[name = string("key_states_231_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_231_dilations_0 = const()[name = string("key_states_231_dilations_0"), val = tensor([1, 1])]; int32 key_states_231_groups_0 = const()[name = string("key_states_231_groups_0"), val = int32(1)]; tensor key_states_231_cast_fp16 = conv(dilations = key_states_231_dilations_0, groups = key_states_231_groups_0, pad = key_states_231_pad_0, pad_type = key_states_231_pad_type_0, strides = key_states_231_strides_0, weight = layers_23_self_attn_k_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("key_states_231_cast_fp16")]; tensor layers_23_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_23_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538248640)))]; tensor value_states_139_strides_0 = const()[name = string("value_states_139_strides_0"), val = tensor([1, 1])]; string value_states_139_pad_type_0 = const()[name = string("value_states_139_pad_type_0"), val = string("valid")]; tensor value_states_139_pad_0 = const()[name = string("value_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_139_dilations_0 = const()[name = string("value_states_139_dilations_0"), val = tensor([1, 1])]; int32 value_states_139_groups_0 = const()[name = string("value_states_139_groups_0"), val = int32(1)]; tensor value_states_139_cast_fp16 = conv(dilations = value_states_139_dilations_0, groups = value_states_139_groups_0, pad = value_states_139_pad_0, pad_type = value_states_139_pad_type_0, strides = value_states_139_strides_0, weight = layers_23_self_attn_v_proj_weight_to_fp16, x = var_8310_cast_fp16_0)[name = string("value_states_139_cast_fp16")]; tensor concat_276x = const()[name = string("concat_276x"), val = tensor([1, 16, 128, -1])]; tensor x_231_cast_fp16 = reshape(shape = concat_276x, x = query_states_139_cast_fp16)[name = string("x_231_cast_fp16")]; tensor concat_277x = const()[name = string("concat_277x"), val = tensor([1, 2, 128, -1])]; tensor var_8367_cast_fp16 = reshape(shape = concat_277x, x = key_states_231_cast_fp16)[name = string("op_8367_cast_fp16")]; tensor concat_278x = const()[name = string("concat_278x"), val = tensor([1, 2, 128, -1])]; tensor var_8374_cast_fp16 = reshape(shape = concat_278x, x = value_states_139_cast_fp16)[name = string("op_8374_cast_fp16")]; tensor var_8378_cast_fp16 = mul(x = x_231_cast_fp16, y = var_869_cast_fp16)[name = string("op_8378_cast_fp16")]; tensor var_8379_split_sizes_0 = const()[name = string("op_8379_split_sizes_0"), val = tensor([64, 64])]; int32 var_8379_axis_0 = const()[name = string("op_8379_axis_0"), val = int32(-2)]; tensor var_8379_cast_fp16_0, tensor var_8379_cast_fp16_1 = split(axis = var_8379_axis_0, split_sizes = var_8379_split_sizes_0, x = x_231_cast_fp16)[name = string("op_8379_cast_fp16")]; fp16 const_232_promoted_to_fp16 = const()[name = string("const_232_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8381_cast_fp16 = mul(x = var_8379_cast_fp16_1, y = const_232_promoted_to_fp16)[name = string("op_8381_cast_fp16")]; int32 var_8383 = const()[name = string("op_8383"), val = int32(-2)]; bool var_8384_interleave_0 = const()[name = string("op_8384_interleave_0"), val = bool(false)]; tensor var_8384_cast_fp16 = concat(axis = var_8383, interleave = var_8384_interleave_0, values = (var_8381_cast_fp16, var_8379_cast_fp16_0))[name = string("op_8384_cast_fp16")]; tensor var_8385_cast_fp16 = mul(x = var_8384_cast_fp16, y = var_878_cast_fp16)[name = string("op_8385_cast_fp16")]; tensor query_states_141_cast_fp16 = add(x = var_8378_cast_fp16, y = var_8385_cast_fp16)[name = string("query_states_141_cast_fp16")]; tensor var_8391_cast_fp16 = mul(x = var_8367_cast_fp16, y = var_869_cast_fp16)[name = string("op_8391_cast_fp16")]; tensor var_8392_split_sizes_0 = const()[name = string("op_8392_split_sizes_0"), val = tensor([64, 64])]; int32 var_8392_axis_0 = const()[name = string("op_8392_axis_0"), val = int32(-2)]; tensor var_8392_cast_fp16_0, tensor var_8392_cast_fp16_1 = split(axis = var_8392_axis_0, split_sizes = var_8392_split_sizes_0, x = var_8367_cast_fp16)[name = string("op_8392_cast_fp16")]; fp16 const_233_promoted_to_fp16 = const()[name = string("const_233_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8394_cast_fp16 = mul(x = var_8392_cast_fp16_1, y = const_233_promoted_to_fp16)[name = string("op_8394_cast_fp16")]; int32 var_8396 = const()[name = string("op_8396"), val = int32(-2)]; bool var_8397_interleave_0 = const()[name = string("op_8397_interleave_0"), val = bool(false)]; tensor var_8397_cast_fp16 = concat(axis = var_8396, interleave = var_8397_interleave_0, values = (var_8394_cast_fp16, var_8392_cast_fp16_0))[name = string("op_8397_cast_fp16")]; tensor var_8398_cast_fp16 = mul(x = var_8397_cast_fp16, y = var_878_cast_fp16)[name = string("op_8398_cast_fp16")]; tensor key_states_235_cast_fp16 = add(x = var_8391_cast_fp16, y = var_8398_cast_fp16)[name = string("key_states_235_cast_fp16")]; tensor expand_dims_276 = const()[name = string("expand_dims_276"), val = tensor([23])]; tensor expand_dims_277 = const()[name = string("expand_dims_277"), val = tensor([0])]; tensor expand_dims_279 = const()[name = string("expand_dims_279"), val = tensor([0])]; int32 concat_281_axis_0 = const()[name = string("concat_281_axis_0"), val = int32(0)]; bool concat_281_interleave_0 = const()[name = string("concat_281_interleave_0"), val = bool(false)]; tensor concat_281 = concat(axis = concat_281_axis_0, interleave = concat_281_interleave_0, values = (expand_dims_276, expand_dims_277, position_id, expand_dims_279))[name = string("concat_281")]; tensor expand_dims_280 = const()[name = string("expand_dims_280"), val = tensor([24])]; tensor concat_282_values1_0 = const()[name = string("concat_282_values1_0"), val = tensor([0])]; tensor concat_282_values3_0 = const()[name = string("concat_282_values3_0"), val = tensor([0])]; int32 concat_282_axis_0 = const()[name = string("concat_282_axis_0"), val = int32(0)]; bool concat_282_interleave_0 = const()[name = string("concat_282_interleave_0"), val = bool(false)]; tensor concat_282 = concat(axis = concat_282_axis_0, interleave = concat_282_interleave_0, values = (expand_dims_280, concat_282_values1_0, cache_position_end, concat_282_values3_0))[name = string("concat_282")]; tensor key_states_237_perm_0 = const()[name = string("key_states_237_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_24_stride_0 = const()[name = string("key_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_237_cast_fp16 = transpose(perm = key_states_237_perm_0, x = key_states_235_cast_fp16)[name = string("transpose_186")]; tensor key_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = key_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = key_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_24_squeeze_mask_0, stride = key_cache_internal_tensor_assign_24_stride_0, update = key_states_237_cast_fp16, x = coreml_update_state_156)[name = string("key_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_24_cast_fp16, input = key_cache)[name = string("coreml_update_state_158_write_state")]; tensor coreml_update_state_158 = read_state(input = key_cache)[name = string("coreml_update_state_158")]; tensor value_states_141_perm_0 = const()[name = string("value_states_141_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_24_stride_0 = const()[name = string("value_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_141_cast_fp16 = transpose(perm = value_states_141_perm_0, x = var_8374_cast_fp16)[name = string("transpose_185")]; tensor value_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = value_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = value_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_24_squeeze_mask_0, stride = value_cache_internal_tensor_assign_24_stride_0, update = value_states_141_cast_fp16, x = coreml_update_state_157)[name = string("value_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_24_cast_fp16, input = value_cache)[name = string("coreml_update_state_159_write_state")]; tensor coreml_update_state_159 = read_state(input = value_cache)[name = string("coreml_update_state_159")]; tensor var_8468_begin_0 = const()[name = string("op_8468_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8468_end_0 = const()[name = string("op_8468_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8468_end_mask_0 = const()[name = string("op_8468_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8468_cast_fp16 = slice_by_index(begin = var_8468_begin_0, end = var_8468_end_0, end_mask = var_8468_end_mask_0, x = coreml_update_state_158)[name = string("op_8468_cast_fp16")]; tensor tile_46 = const()[name = string("tile_46"), val = tensor([1, 1])]; int32 var_8471_axis_0 = const()[name = string("op_8471_axis_0"), val = int32(1)]; tensor var_8471_cast_fp16_0, tensor var_8471_cast_fp16_1 = split(axis = var_8471_axis_0, split_sizes = tile_46, x = var_8468_cast_fp16)[name = string("op_8471_cast_fp16")]; tensor var_8478_begin_0 = const()[name = string("op_8478_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8478_end_0 = const()[name = string("op_8478_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8478_end_mask_0 = const()[name = string("op_8478_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8478_cast_fp16 = slice_by_index(begin = var_8478_begin_0, end = var_8478_end_0, end_mask = var_8478_end_mask_0, x = coreml_update_state_159)[name = string("op_8478_cast_fp16")]; tensor tile_47 = const()[name = string("tile_47"), val = tensor([1, 1])]; int32 var_8481_axis_0 = const()[name = string("op_8481_axis_0"), val = int32(1)]; tensor var_8481_cast_fp16_0, tensor var_8481_cast_fp16_1 = split(axis = var_8481_axis_0, split_sizes = tile_47, x = var_8478_cast_fp16)[name = string("op_8481_cast_fp16")]; tensor var_8484_split_sizes_0 = const()[name = string("op_8484_split_sizes_0"), val = tensor([8, 8])]; int32 var_8484_axis_0 = const()[name = string("op_8484_axis_0"), val = int32(1)]; tensor var_8484_0, tensor var_8484_1 = split(axis = var_8484_axis_0, split_sizes = var_8484_split_sizes_0, x = query_states_141_cast_fp16)[name = string("op_8484")]; bool attn_weights_369_transpose_x_0 = const()[name = string("attn_weights_369_transpose_x_0"), val = bool(false)]; bool attn_weights_369_transpose_y_0 = const()[name = string("attn_weights_369_transpose_y_0"), val = bool(false)]; tensor attn_weights_369_cast_fp16 = matmul(transpose_x = attn_weights_369_transpose_x_0, transpose_y = attn_weights_369_transpose_y_0, x = var_8471_cast_fp16_0, y = var_8484_0)[name = string("attn_weights_369_cast_fp16")]; fp16 var_8487_to_fp16 = const()[name = string("op_8487_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_371_cast_fp16 = mul(x = attn_weights_369_cast_fp16, y = var_8487_to_fp16)[name = string("attn_weights_371_cast_fp16")]; tensor attn_weights_373_cast_fp16 = add(x = attn_weights_371_cast_fp16, y = attn_mask_1)[name = string("attn_weights_373_cast_fp16")]; int32 var_8491 = const()[name = string("op_8491"), val = int32(-2)]; tensor attn_weights_375_cast_fp16 = softmax(axis = var_8491, x = attn_weights_373_cast_fp16)[name = string("attn_weights_375_cast_fp16")]; bool var_8497_transpose_x_1 = const()[name = string("op_8497_transpose_x_1"), val = bool(true)]; bool var_8497_transpose_y_1 = const()[name = string("op_8497_transpose_y_1"), val = bool(false)]; tensor var_8497_cast_fp16 = matmul(transpose_x = var_8497_transpose_x_1, transpose_y = var_8497_transpose_y_1, x = attn_weights_375_cast_fp16, y = var_8481_cast_fp16_0)[name = string("op_8497_cast_fp16")]; bool attn_weights_377_transpose_x_0 = const()[name = string("attn_weights_377_transpose_x_0"), val = bool(false)]; bool attn_weights_377_transpose_y_0 = const()[name = string("attn_weights_377_transpose_y_0"), val = bool(false)]; tensor attn_weights_377_cast_fp16 = matmul(transpose_x = attn_weights_377_transpose_x_0, transpose_y = attn_weights_377_transpose_y_0, x = var_8471_cast_fp16_1, y = var_8484_1)[name = string("attn_weights_377_cast_fp16")]; fp16 var_8499_to_fp16 = const()[name = string("op_8499_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_379_cast_fp16 = mul(x = attn_weights_377_cast_fp16, y = var_8499_to_fp16)[name = string("attn_weights_379_cast_fp16")]; tensor attn_weights_381_cast_fp16 = add(x = attn_weights_379_cast_fp16, y = attn_mask_1)[name = string("attn_weights_381_cast_fp16")]; int32 var_8503 = const()[name = string("op_8503"), val = int32(-2)]; tensor attn_weights_383_cast_fp16 = softmax(axis = var_8503, x = attn_weights_381_cast_fp16)[name = string("attn_weights_383_cast_fp16")]; bool attn_output_185_transpose_x_1 = const()[name = string("attn_output_185_transpose_x_1"), val = bool(true)]; bool attn_output_185_transpose_y_1 = const()[name = string("attn_output_185_transpose_y_1"), val = bool(false)]; tensor attn_output_185_cast_fp16 = matmul(transpose_x = attn_output_185_transpose_x_1, transpose_y = attn_output_185_transpose_y_1, x = attn_weights_383_cast_fp16, y = var_8481_cast_fp16_1)[name = string("attn_output_185_cast_fp16")]; int32 var_8511 = const()[name = string("op_8511"), val = int32(1)]; bool attn_output_187_interleave_0 = const()[name = string("attn_output_187_interleave_0"), val = bool(false)]; tensor attn_output_187_cast_fp16 = concat(axis = var_8511, interleave = attn_output_187_interleave_0, values = (var_8497_cast_fp16, attn_output_185_cast_fp16))[name = string("attn_output_187_cast_fp16")]; tensor var_8515_perm_0 = const()[name = string("op_8515_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_287x = const()[name = string("concat_287x"), val = tensor([1, 2048, 1, -1])]; tensor var_8515_cast_fp16 = transpose(perm = var_8515_perm_0, x = attn_output_187_cast_fp16)[name = string("transpose_184")]; tensor attn_output_191_cast_fp16 = reshape(shape = concat_287x, x = var_8515_cast_fp16)[name = string("attn_output_191_cast_fp16")]; tensor hidden_states_233_strides_0 = const()[name = string("hidden_states_233_strides_0"), val = tensor([1, 1])]; string hidden_states_233_pad_type_0 = const()[name = string("hidden_states_233_pad_type_0"), val = string("valid")]; tensor hidden_states_233_pad_0 = const()[name = string("hidden_states_233_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_233_dilations_0 = const()[name = string("hidden_states_233_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_233_groups_0 = const()[name = string("hidden_states_233_groups_0"), val = int32(1)]; tensor hidden_states_233_cast_fp16 = conv(dilations = hidden_states_233_dilations_0, groups = hidden_states_233_groups_0, pad = hidden_states_233_pad_0, pad_type = hidden_states_233_pad_type_0, strides = hidden_states_233_strides_0, weight = layers_23_self_attn_o_proj_weight_cast_fp16, x = attn_output_191_cast_fp16)[name = string("hidden_states_233_cast_fp16")]; tensor hidden_states_235_cast_fp16 = add(x = hidden_states_229_cast_fp16, y = hidden_states_233_cast_fp16)[name = string("hidden_states_235_cast_fp16")]; fp16 const_238_promoted_to_fp16 = const()[name = string("const_238_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8548_cast_fp16 = mul(x = hidden_states_235_cast_fp16, y = const_238_promoted_to_fp16)[name = string("op_8548_cast_fp16")]; int32 var_8546 = const()[name = string("op_8546"), val = int32(1)]; bool doubled_189_interleave_0 = const()[name = string("doubled_189_interleave_0"), val = bool(false)]; tensor doubled_189_cast_fp16 = concat(axis = var_8546, interleave = doubled_189_interleave_0, values = (hidden_states_235_cast_fp16, var_8548_cast_fp16))[name = string("doubled_189_cast_fp16")]; tensor out_95_axes_0 = const()[name = string("out_95_axes_0"), val = tensor([1])]; tensor out_95_gamma_0_to_fp16 = const()[name = string("out_95_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539297280)))]; fp16 var_8558_to_fp16 = const()[name = string("op_8558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_95_cast_fp16 = layer_norm(axes = out_95_axes_0, epsilon = var_8558_to_fp16, gamma = out_95_gamma_0_to_fp16, x = doubled_189_cast_fp16)[name = string("out_95_cast_fp16")]; tensor var_8569_split_sizes_0 = const()[name = string("op_8569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8569_axis_0 = const()[name = string("op_8569_axis_0"), val = int32(1)]; tensor var_8569_cast_fp16_0, tensor var_8569_cast_fp16_1 = split(axis = var_8569_axis_0, split_sizes = var_8569_split_sizes_0, x = out_95_cast_fp16)[name = string("op_8569_cast_fp16")]; tensor input_47_strides_0 = const()[name = string("input_47_strides_0"), val = tensor([1, 1])]; string input_47_pad_type_0 = const()[name = string("input_47_pad_type_0"), val = string("valid")]; tensor input_47_pad_0 = const()[name = string("input_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_47_dilations_0 = const()[name = string("input_47_dilations_0"), val = tensor([1, 1])]; int32 input_47_groups_0 = const()[name = string("input_47_groups_0"), val = int32(1)]; tensor input_47_cast_fp16 = conv(dilations = input_47_dilations_0, groups = input_47_groups_0, pad = input_47_pad_0, pad_type = input_47_pad_type_0, strides = input_47_strides_0, weight = layers_23_mlp_gate_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("input_47_cast_fp16")]; tensor var_8586_cast_fp16 = silu(x = input_47_cast_fp16)[name = string("op_8586_cast_fp16")]; tensor var_8592_strides_0 = const()[name = string("op_8592_strides_0"), val = tensor([1, 1])]; string var_8592_pad_type_0 = const()[name = string("op_8592_pad_type_0"), val = string("valid")]; tensor var_8592_pad_0 = const()[name = string("op_8592_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8592_dilations_0 = const()[name = string("op_8592_dilations_0"), val = tensor([1, 1])]; int32 var_8592_groups_0 = const()[name = string("op_8592_groups_0"), val = int32(1)]; tensor var_8592_cast_fp16 = conv(dilations = var_8592_dilations_0, groups = var_8592_groups_0, pad = var_8592_pad_0, pad_type = var_8592_pad_type_0, strides = var_8592_strides_0, weight = layers_23_mlp_up_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("op_8592_cast_fp16")]; tensor x_239_cast_fp16 = mul(x = var_8586_cast_fp16, y = var_8592_cast_fp16)[name = string("x_239_cast_fp16")]; tensor hidden_states_237_strides_0 = const()[name = string("hidden_states_237_strides_0"), val = tensor([1, 1])]; string hidden_states_237_pad_type_0 = const()[name = string("hidden_states_237_pad_type_0"), val = string("valid")]; tensor hidden_states_237_pad_0 = const()[name = string("hidden_states_237_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_237_dilations_0 = const()[name = string("hidden_states_237_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_237_groups_0 = const()[name = string("hidden_states_237_groups_0"), val = int32(1)]; tensor hidden_states_237_cast_fp16 = conv(dilations = hidden_states_237_dilations_0, groups = hidden_states_237_groups_0, pad = hidden_states_237_pad_0, pad_type = hidden_states_237_pad_type_0, strides = hidden_states_237_strides_0, weight = layers_23_mlp_down_proj_weight_cast_fp16, x = x_239_cast_fp16)[name = string("hidden_states_237_cast_fp16")]; tensor hidden_states_239_cast_fp16 = add(x = hidden_states_235_cast_fp16, y = hidden_states_237_cast_fp16)[name = string("hidden_states_239_cast_fp16")]; fp16 const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8610_cast_fp16 = mul(x = hidden_states_239_cast_fp16, y = const_240_promoted_to_fp16)[name = string("op_8610_cast_fp16")]; int32 var_8608 = const()[name = string("op_8608"), val = int32(1)]; bool doubled_193_interleave_0 = const()[name = string("doubled_193_interleave_0"), val = bool(false)]; tensor doubled_193_cast_fp16 = concat(axis = var_8608, interleave = doubled_193_interleave_0, values = (hidden_states_239_cast_fp16, var_8610_cast_fp16))[name = string("doubled_193_cast_fp16")]; tensor out_97_axes_0 = const()[name = string("out_97_axes_0"), val = tensor([1])]; tensor out_97_gamma_0_to_fp16 = const()[name = string("out_97_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539305536)))]; fp16 var_8620_to_fp16 = const()[name = string("op_8620_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_97_cast_fp16 = layer_norm(axes = out_97_axes_0, epsilon = var_8620_to_fp16, gamma = out_97_gamma_0_to_fp16, x = doubled_193_cast_fp16)[name = string("out_97_cast_fp16")]; tensor var_8631_split_sizes_0 = const()[name = string("op_8631_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8631_axis_0 = const()[name = string("op_8631_axis_0"), val = int32(1)]; tensor var_8631_cast_fp16_0, tensor var_8631_cast_fp16_1 = split(axis = var_8631_axis_0, split_sizes = var_8631_split_sizes_0, x = out_97_cast_fp16)[name = string("op_8631_cast_fp16")]; tensor query_states_145_strides_0 = const()[name = string("query_states_145_strides_0"), val = tensor([1, 1])]; string query_states_145_pad_type_0 = const()[name = string("query_states_145_pad_type_0"), val = string("valid")]; tensor query_states_145_pad_0 = const()[name = string("query_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_145_dilations_0 = const()[name = string("query_states_145_dilations_0"), val = tensor([1, 1])]; int32 query_states_145_groups_0 = const()[name = string("query_states_145_groups_0"), val = int32(1)]; tensor query_states_145_cast_fp16 = conv(dilations = query_states_145_dilations_0, groups = query_states_145_groups_0, pad = query_states_145_pad_0, pad_type = query_states_145_pad_type_0, strides = query_states_145_strides_0, weight = layers_24_self_attn_q_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("query_states_145_cast_fp16")]; tensor key_states_241_strides_0 = const()[name = string("key_states_241_strides_0"), val = tensor([1, 1])]; string key_states_241_pad_type_0 = const()[name = string("key_states_241_pad_type_0"), val = string("valid")]; tensor key_states_241_pad_0 = const()[name = string("key_states_241_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_241_dilations_0 = const()[name = string("key_states_241_dilations_0"), val = tensor([1, 1])]; int32 key_states_241_groups_0 = const()[name = string("key_states_241_groups_0"), val = int32(1)]; tensor key_states_241_cast_fp16 = conv(dilations = key_states_241_dilations_0, groups = key_states_241_groups_0, pad = key_states_241_pad_0, pad_type = key_states_241_pad_type_0, strides = key_states_241_strides_0, weight = layers_24_self_attn_k_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("key_states_241_cast_fp16")]; tensor layers_24_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_24_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539313792)))]; tensor value_states_145_strides_0 = const()[name = string("value_states_145_strides_0"), val = tensor([1, 1])]; string value_states_145_pad_type_0 = const()[name = string("value_states_145_pad_type_0"), val = string("valid")]; tensor value_states_145_pad_0 = const()[name = string("value_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_145_dilations_0 = const()[name = string("value_states_145_dilations_0"), val = tensor([1, 1])]; int32 value_states_145_groups_0 = const()[name = string("value_states_145_groups_0"), val = int32(1)]; tensor value_states_145_cast_fp16 = conv(dilations = value_states_145_dilations_0, groups = value_states_145_groups_0, pad = value_states_145_pad_0, pad_type = value_states_145_pad_type_0, strides = value_states_145_strides_0, weight = layers_24_self_attn_v_proj_weight_to_fp16, x = var_8631_cast_fp16_0)[name = string("value_states_145_cast_fp16")]; tensor concat_288x = const()[name = string("concat_288x"), val = tensor([1, 16, 128, -1])]; tensor x_241_cast_fp16 = reshape(shape = concat_288x, x = query_states_145_cast_fp16)[name = string("x_241_cast_fp16")]; tensor concat_289x = const()[name = string("concat_289x"), val = tensor([1, 2, 128, -1])]; tensor var_8688_cast_fp16 = reshape(shape = concat_289x, x = key_states_241_cast_fp16)[name = string("op_8688_cast_fp16")]; tensor concat_290x = const()[name = string("concat_290x"), val = tensor([1, 2, 128, -1])]; tensor var_8695_cast_fp16 = reshape(shape = concat_290x, x = value_states_145_cast_fp16)[name = string("op_8695_cast_fp16")]; tensor var_8699_cast_fp16 = mul(x = x_241_cast_fp16, y = var_869_cast_fp16)[name = string("op_8699_cast_fp16")]; tensor var_8700_split_sizes_0 = const()[name = string("op_8700_split_sizes_0"), val = tensor([64, 64])]; int32 var_8700_axis_0 = const()[name = string("op_8700_axis_0"), val = int32(-2)]; tensor var_8700_cast_fp16_0, tensor var_8700_cast_fp16_1 = split(axis = var_8700_axis_0, split_sizes = var_8700_split_sizes_0, x = x_241_cast_fp16)[name = string("op_8700_cast_fp16")]; fp16 const_242_promoted_to_fp16 = const()[name = string("const_242_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8702_cast_fp16 = mul(x = var_8700_cast_fp16_1, y = const_242_promoted_to_fp16)[name = string("op_8702_cast_fp16")]; int32 var_8704 = const()[name = string("op_8704"), val = int32(-2)]; bool var_8705_interleave_0 = const()[name = string("op_8705_interleave_0"), val = bool(false)]; tensor var_8705_cast_fp16 = concat(axis = var_8704, interleave = var_8705_interleave_0, values = (var_8702_cast_fp16, var_8700_cast_fp16_0))[name = string("op_8705_cast_fp16")]; tensor var_8706_cast_fp16 = mul(x = var_8705_cast_fp16, y = var_878_cast_fp16)[name = string("op_8706_cast_fp16")]; tensor query_states_147_cast_fp16 = add(x = var_8699_cast_fp16, y = var_8706_cast_fp16)[name = string("query_states_147_cast_fp16")]; tensor var_8712_cast_fp16 = mul(x = var_8688_cast_fp16, y = var_869_cast_fp16)[name = string("op_8712_cast_fp16")]; tensor var_8713_split_sizes_0 = const()[name = string("op_8713_split_sizes_0"), val = tensor([64, 64])]; int32 var_8713_axis_0 = const()[name = string("op_8713_axis_0"), val = int32(-2)]; tensor var_8713_cast_fp16_0, tensor var_8713_cast_fp16_1 = split(axis = var_8713_axis_0, split_sizes = var_8713_split_sizes_0, x = var_8688_cast_fp16)[name = string("op_8713_cast_fp16")]; fp16 const_243_promoted_to_fp16 = const()[name = string("const_243_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8715_cast_fp16 = mul(x = var_8713_cast_fp16_1, y = const_243_promoted_to_fp16)[name = string("op_8715_cast_fp16")]; int32 var_8717 = const()[name = string("op_8717"), val = int32(-2)]; bool var_8718_interleave_0 = const()[name = string("op_8718_interleave_0"), val = bool(false)]; tensor var_8718_cast_fp16 = concat(axis = var_8717, interleave = var_8718_interleave_0, values = (var_8715_cast_fp16, var_8713_cast_fp16_0))[name = string("op_8718_cast_fp16")]; tensor var_8719_cast_fp16 = mul(x = var_8718_cast_fp16, y = var_878_cast_fp16)[name = string("op_8719_cast_fp16")]; tensor key_states_245_cast_fp16 = add(x = var_8712_cast_fp16, y = var_8719_cast_fp16)[name = string("key_states_245_cast_fp16")]; tensor expand_dims_288 = const()[name = string("expand_dims_288"), val = tensor([24])]; tensor expand_dims_289 = const()[name = string("expand_dims_289"), val = tensor([0])]; tensor expand_dims_291 = const()[name = string("expand_dims_291"), val = tensor([0])]; int32 concat_293_axis_0 = const()[name = string("concat_293_axis_0"), val = int32(0)]; bool concat_293_interleave_0 = const()[name = string("concat_293_interleave_0"), val = bool(false)]; tensor concat_293 = concat(axis = concat_293_axis_0, interleave = concat_293_interleave_0, values = (expand_dims_288, expand_dims_289, position_id, expand_dims_291))[name = string("concat_293")]; tensor expand_dims_292 = const()[name = string("expand_dims_292"), val = tensor([25])]; tensor concat_294_values1_0 = const()[name = string("concat_294_values1_0"), val = tensor([0])]; tensor concat_294_values3_0 = const()[name = string("concat_294_values3_0"), val = tensor([0])]; int32 concat_294_axis_0 = const()[name = string("concat_294_axis_0"), val = int32(0)]; bool concat_294_interleave_0 = const()[name = string("concat_294_interleave_0"), val = bool(false)]; tensor concat_294 = concat(axis = concat_294_axis_0, interleave = concat_294_interleave_0, values = (expand_dims_292, concat_294_values1_0, cache_position_end, concat_294_values3_0))[name = string("concat_294")]; tensor key_states_247_perm_0 = const()[name = string("key_states_247_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_25_stride_0 = const()[name = string("key_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_247_cast_fp16 = transpose(perm = key_states_247_perm_0, x = key_states_245_cast_fp16)[name = string("transpose_183")]; tensor key_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = key_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = key_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_25_squeeze_mask_0, stride = key_cache_internal_tensor_assign_25_stride_0, update = key_states_247_cast_fp16, x = coreml_update_state_158)[name = string("key_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_25_cast_fp16, input = key_cache)[name = string("coreml_update_state_160_write_state")]; tensor coreml_update_state_160 = read_state(input = key_cache)[name = string("coreml_update_state_160")]; tensor value_states_147_perm_0 = const()[name = string("value_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_25_stride_0 = const()[name = string("value_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_147_cast_fp16 = transpose(perm = value_states_147_perm_0, x = var_8695_cast_fp16)[name = string("transpose_182")]; tensor value_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = value_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = value_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_25_squeeze_mask_0, stride = value_cache_internal_tensor_assign_25_stride_0, update = value_states_147_cast_fp16, x = coreml_update_state_159)[name = string("value_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_25_cast_fp16, input = value_cache)[name = string("coreml_update_state_161_write_state")]; tensor coreml_update_state_161 = read_state(input = value_cache)[name = string("coreml_update_state_161")]; tensor var_8789_begin_0 = const()[name = string("op_8789_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8789_end_0 = const()[name = string("op_8789_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8789_end_mask_0 = const()[name = string("op_8789_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8789_cast_fp16 = slice_by_index(begin = var_8789_begin_0, end = var_8789_end_0, end_mask = var_8789_end_mask_0, x = coreml_update_state_160)[name = string("op_8789_cast_fp16")]; tensor tile_48 = const()[name = string("tile_48"), val = tensor([1, 1])]; int32 var_8792_axis_0 = const()[name = string("op_8792_axis_0"), val = int32(1)]; tensor var_8792_cast_fp16_0, tensor var_8792_cast_fp16_1 = split(axis = var_8792_axis_0, split_sizes = tile_48, x = var_8789_cast_fp16)[name = string("op_8792_cast_fp16")]; tensor var_8799_begin_0 = const()[name = string("op_8799_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8799_end_0 = const()[name = string("op_8799_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8799_end_mask_0 = const()[name = string("op_8799_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8799_cast_fp16 = slice_by_index(begin = var_8799_begin_0, end = var_8799_end_0, end_mask = var_8799_end_mask_0, x = coreml_update_state_161)[name = string("op_8799_cast_fp16")]; tensor tile_49 = const()[name = string("tile_49"), val = tensor([1, 1])]; int32 var_8802_axis_0 = const()[name = string("op_8802_axis_0"), val = int32(1)]; tensor var_8802_cast_fp16_0, tensor var_8802_cast_fp16_1 = split(axis = var_8802_axis_0, split_sizes = tile_49, x = var_8799_cast_fp16)[name = string("op_8802_cast_fp16")]; tensor var_8805_split_sizes_0 = const()[name = string("op_8805_split_sizes_0"), val = tensor([8, 8])]; int32 var_8805_axis_0 = const()[name = string("op_8805_axis_0"), val = int32(1)]; tensor var_8805_0, tensor var_8805_1 = split(axis = var_8805_axis_0, split_sizes = var_8805_split_sizes_0, x = query_states_147_cast_fp16)[name = string("op_8805")]; bool attn_weights_385_transpose_x_0 = const()[name = string("attn_weights_385_transpose_x_0"), val = bool(false)]; bool attn_weights_385_transpose_y_0 = const()[name = string("attn_weights_385_transpose_y_0"), val = bool(false)]; tensor attn_weights_385_cast_fp16 = matmul(transpose_x = attn_weights_385_transpose_x_0, transpose_y = attn_weights_385_transpose_y_0, x = var_8792_cast_fp16_0, y = var_8805_0)[name = string("attn_weights_385_cast_fp16")]; fp16 var_8808_to_fp16 = const()[name = string("op_8808_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_387_cast_fp16 = mul(x = attn_weights_385_cast_fp16, y = var_8808_to_fp16)[name = string("attn_weights_387_cast_fp16")]; tensor attn_weights_389_cast_fp16 = add(x = attn_weights_387_cast_fp16, y = attn_mask_1)[name = string("attn_weights_389_cast_fp16")]; int32 var_8812 = const()[name = string("op_8812"), val = int32(-2)]; tensor attn_weights_391_cast_fp16 = softmax(axis = var_8812, x = attn_weights_389_cast_fp16)[name = string("attn_weights_391_cast_fp16")]; bool var_8818_transpose_x_1 = const()[name = string("op_8818_transpose_x_1"), val = bool(true)]; bool var_8818_transpose_y_1 = const()[name = string("op_8818_transpose_y_1"), val = bool(false)]; tensor var_8818_cast_fp16 = matmul(transpose_x = var_8818_transpose_x_1, transpose_y = var_8818_transpose_y_1, x = attn_weights_391_cast_fp16, y = var_8802_cast_fp16_0)[name = string("op_8818_cast_fp16")]; bool attn_weights_393_transpose_x_0 = const()[name = string("attn_weights_393_transpose_x_0"), val = bool(false)]; bool attn_weights_393_transpose_y_0 = const()[name = string("attn_weights_393_transpose_y_0"), val = bool(false)]; tensor attn_weights_393_cast_fp16 = matmul(transpose_x = attn_weights_393_transpose_x_0, transpose_y = attn_weights_393_transpose_y_0, x = var_8792_cast_fp16_1, y = var_8805_1)[name = string("attn_weights_393_cast_fp16")]; fp16 var_8820_to_fp16 = const()[name = string("op_8820_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_395_cast_fp16 = mul(x = attn_weights_393_cast_fp16, y = var_8820_to_fp16)[name = string("attn_weights_395_cast_fp16")]; tensor attn_weights_397_cast_fp16 = add(x = attn_weights_395_cast_fp16, y = attn_mask_1)[name = string("attn_weights_397_cast_fp16")]; int32 var_8824 = const()[name = string("op_8824"), val = int32(-2)]; tensor attn_weights_399_cast_fp16 = softmax(axis = var_8824, x = attn_weights_397_cast_fp16)[name = string("attn_weights_399_cast_fp16")]; bool attn_output_193_transpose_x_1 = const()[name = string("attn_output_193_transpose_x_1"), val = bool(true)]; bool attn_output_193_transpose_y_1 = const()[name = string("attn_output_193_transpose_y_1"), val = bool(false)]; tensor attn_output_193_cast_fp16 = matmul(transpose_x = attn_output_193_transpose_x_1, transpose_y = attn_output_193_transpose_y_1, x = attn_weights_399_cast_fp16, y = var_8802_cast_fp16_1)[name = string("attn_output_193_cast_fp16")]; int32 var_8832 = const()[name = string("op_8832"), val = int32(1)]; bool attn_output_195_interleave_0 = const()[name = string("attn_output_195_interleave_0"), val = bool(false)]; tensor attn_output_195_cast_fp16 = concat(axis = var_8832, interleave = attn_output_195_interleave_0, values = (var_8818_cast_fp16, attn_output_193_cast_fp16))[name = string("attn_output_195_cast_fp16")]; tensor var_8836_perm_0 = const()[name = string("op_8836_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_299x = const()[name = string("concat_299x"), val = tensor([1, 2048, 1, -1])]; tensor var_8836_cast_fp16 = transpose(perm = var_8836_perm_0, x = attn_output_195_cast_fp16)[name = string("transpose_181")]; tensor attn_output_199_cast_fp16 = reshape(shape = concat_299x, x = var_8836_cast_fp16)[name = string("attn_output_199_cast_fp16")]; tensor hidden_states_243_strides_0 = const()[name = string("hidden_states_243_strides_0"), val = tensor([1, 1])]; string hidden_states_243_pad_type_0 = const()[name = string("hidden_states_243_pad_type_0"), val = string("valid")]; tensor hidden_states_243_pad_0 = const()[name = string("hidden_states_243_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_243_dilations_0 = const()[name = string("hidden_states_243_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_243_groups_0 = const()[name = string("hidden_states_243_groups_0"), val = int32(1)]; tensor hidden_states_243_cast_fp16 = conv(dilations = hidden_states_243_dilations_0, groups = hidden_states_243_groups_0, pad = hidden_states_243_pad_0, pad_type = hidden_states_243_pad_type_0, strides = hidden_states_243_strides_0, weight = layers_24_self_attn_o_proj_weight_cast_fp16, x = attn_output_199_cast_fp16)[name = string("hidden_states_243_cast_fp16")]; tensor hidden_states_245_cast_fp16 = add(x = hidden_states_239_cast_fp16, y = hidden_states_243_cast_fp16)[name = string("hidden_states_245_cast_fp16")]; fp16 const_248_promoted_to_fp16 = const()[name = string("const_248_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8869_cast_fp16 = mul(x = hidden_states_245_cast_fp16, y = const_248_promoted_to_fp16)[name = string("op_8869_cast_fp16")]; int32 var_8867 = const()[name = string("op_8867"), val = int32(1)]; bool doubled_197_interleave_0 = const()[name = string("doubled_197_interleave_0"), val = bool(false)]; tensor doubled_197_cast_fp16 = concat(axis = var_8867, interleave = doubled_197_interleave_0, values = (hidden_states_245_cast_fp16, var_8869_cast_fp16))[name = string("doubled_197_cast_fp16")]; tensor out_99_axes_0 = const()[name = string("out_99_axes_0"), val = tensor([1])]; tensor out_99_gamma_0_to_fp16 = const()[name = string("out_99_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540362432)))]; fp16 var_8879_to_fp16 = const()[name = string("op_8879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_99_cast_fp16 = layer_norm(axes = out_99_axes_0, epsilon = var_8879_to_fp16, gamma = out_99_gamma_0_to_fp16, x = doubled_197_cast_fp16)[name = string("out_99_cast_fp16")]; tensor var_8890_split_sizes_0 = const()[name = string("op_8890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8890_axis_0 = const()[name = string("op_8890_axis_0"), val = int32(1)]; tensor var_8890_cast_fp16_0, tensor var_8890_cast_fp16_1 = split(axis = var_8890_axis_0, split_sizes = var_8890_split_sizes_0, x = out_99_cast_fp16)[name = string("op_8890_cast_fp16")]; tensor input_49_strides_0 = const()[name = string("input_49_strides_0"), val = tensor([1, 1])]; string input_49_pad_type_0 = const()[name = string("input_49_pad_type_0"), val = string("valid")]; tensor input_49_pad_0 = const()[name = string("input_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_49_dilations_0 = const()[name = string("input_49_dilations_0"), val = tensor([1, 1])]; int32 input_49_groups_0 = const()[name = string("input_49_groups_0"), val = int32(1)]; tensor input_49_cast_fp16 = conv(dilations = input_49_dilations_0, groups = input_49_groups_0, pad = input_49_pad_0, pad_type = input_49_pad_type_0, strides = input_49_strides_0, weight = layers_24_mlp_gate_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("input_49_cast_fp16")]; tensor var_8907_cast_fp16 = silu(x = input_49_cast_fp16)[name = string("op_8907_cast_fp16")]; tensor var_8913_strides_0 = const()[name = string("op_8913_strides_0"), val = tensor([1, 1])]; string var_8913_pad_type_0 = const()[name = string("op_8913_pad_type_0"), val = string("valid")]; tensor var_8913_pad_0 = const()[name = string("op_8913_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8913_dilations_0 = const()[name = string("op_8913_dilations_0"), val = tensor([1, 1])]; int32 var_8913_groups_0 = const()[name = string("op_8913_groups_0"), val = int32(1)]; tensor var_8913_cast_fp16 = conv(dilations = var_8913_dilations_0, groups = var_8913_groups_0, pad = var_8913_pad_0, pad_type = var_8913_pad_type_0, strides = var_8913_strides_0, weight = layers_24_mlp_up_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("op_8913_cast_fp16")]; tensor x_249_cast_fp16 = mul(x = var_8907_cast_fp16, y = var_8913_cast_fp16)[name = string("x_249_cast_fp16")]; tensor hidden_states_247_strides_0 = const()[name = string("hidden_states_247_strides_0"), val = tensor([1, 1])]; string hidden_states_247_pad_type_0 = const()[name = string("hidden_states_247_pad_type_0"), val = string("valid")]; tensor hidden_states_247_pad_0 = const()[name = string("hidden_states_247_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_247_dilations_0 = const()[name = string("hidden_states_247_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_247_groups_0 = const()[name = string("hidden_states_247_groups_0"), val = int32(1)]; tensor hidden_states_247_cast_fp16 = conv(dilations = hidden_states_247_dilations_0, groups = hidden_states_247_groups_0, pad = hidden_states_247_pad_0, pad_type = hidden_states_247_pad_type_0, strides = hidden_states_247_strides_0, weight = layers_24_mlp_down_proj_weight_cast_fp16, x = x_249_cast_fp16)[name = string("hidden_states_247_cast_fp16")]; tensor hidden_states_249_cast_fp16 = add(x = hidden_states_245_cast_fp16, y = hidden_states_247_cast_fp16)[name = string("hidden_states_249_cast_fp16")]; fp16 const_250_promoted_to_fp16 = const()[name = string("const_250_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8931_cast_fp16 = mul(x = hidden_states_249_cast_fp16, y = const_250_promoted_to_fp16)[name = string("op_8931_cast_fp16")]; int32 var_8929 = const()[name = string("op_8929"), val = int32(1)]; bool doubled_201_interleave_0 = const()[name = string("doubled_201_interleave_0"), val = bool(false)]; tensor doubled_201_cast_fp16 = concat(axis = var_8929, interleave = doubled_201_interleave_0, values = (hidden_states_249_cast_fp16, var_8931_cast_fp16))[name = string("doubled_201_cast_fp16")]; tensor out_101_axes_0 = const()[name = string("out_101_axes_0"), val = tensor([1])]; tensor out_101_gamma_0_to_fp16 = const()[name = string("out_101_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540370688)))]; fp16 var_8941_to_fp16 = const()[name = string("op_8941_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_101_cast_fp16 = layer_norm(axes = out_101_axes_0, epsilon = var_8941_to_fp16, gamma = out_101_gamma_0_to_fp16, x = doubled_201_cast_fp16)[name = string("out_101_cast_fp16")]; tensor var_8952_split_sizes_0 = const()[name = string("op_8952_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8952_axis_0 = const()[name = string("op_8952_axis_0"), val = int32(1)]; tensor var_8952_cast_fp16_0, tensor var_8952_cast_fp16_1 = split(axis = var_8952_axis_0, split_sizes = var_8952_split_sizes_0, x = out_101_cast_fp16)[name = string("op_8952_cast_fp16")]; tensor query_states_151_strides_0 = const()[name = string("query_states_151_strides_0"), val = tensor([1, 1])]; string query_states_151_pad_type_0 = const()[name = string("query_states_151_pad_type_0"), val = string("valid")]; tensor query_states_151_pad_0 = const()[name = string("query_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_151_dilations_0 = const()[name = string("query_states_151_dilations_0"), val = tensor([1, 1])]; int32 query_states_151_groups_0 = const()[name = string("query_states_151_groups_0"), val = int32(1)]; tensor query_states_151_cast_fp16 = conv(dilations = query_states_151_dilations_0, groups = query_states_151_groups_0, pad = query_states_151_pad_0, pad_type = query_states_151_pad_type_0, strides = query_states_151_strides_0, weight = layers_25_self_attn_q_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("query_states_151_cast_fp16")]; tensor key_states_251_strides_0 = const()[name = string("key_states_251_strides_0"), val = tensor([1, 1])]; string key_states_251_pad_type_0 = const()[name = string("key_states_251_pad_type_0"), val = string("valid")]; tensor key_states_251_pad_0 = const()[name = string("key_states_251_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_251_dilations_0 = const()[name = string("key_states_251_dilations_0"), val = tensor([1, 1])]; int32 key_states_251_groups_0 = const()[name = string("key_states_251_groups_0"), val = int32(1)]; tensor key_states_251_cast_fp16 = conv(dilations = key_states_251_dilations_0, groups = key_states_251_groups_0, pad = key_states_251_pad_0, pad_type = key_states_251_pad_type_0, strides = key_states_251_strides_0, weight = layers_25_self_attn_k_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("key_states_251_cast_fp16")]; tensor layers_25_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_25_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540378944)))]; tensor value_states_151_strides_0 = const()[name = string("value_states_151_strides_0"), val = tensor([1, 1])]; string value_states_151_pad_type_0 = const()[name = string("value_states_151_pad_type_0"), val = string("valid")]; tensor value_states_151_pad_0 = const()[name = string("value_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_151_dilations_0 = const()[name = string("value_states_151_dilations_0"), val = tensor([1, 1])]; int32 value_states_151_groups_0 = const()[name = string("value_states_151_groups_0"), val = int32(1)]; tensor value_states_151_cast_fp16 = conv(dilations = value_states_151_dilations_0, groups = value_states_151_groups_0, pad = value_states_151_pad_0, pad_type = value_states_151_pad_type_0, strides = value_states_151_strides_0, weight = layers_25_self_attn_v_proj_weight_to_fp16, x = var_8952_cast_fp16_0)[name = string("value_states_151_cast_fp16")]; tensor concat_300x = const()[name = string("concat_300x"), val = tensor([1, 16, 128, -1])]; tensor x_251_cast_fp16 = reshape(shape = concat_300x, x = query_states_151_cast_fp16)[name = string("x_251_cast_fp16")]; tensor concat_301x = const()[name = string("concat_301x"), val = tensor([1, 2, 128, -1])]; tensor var_9009_cast_fp16 = reshape(shape = concat_301x, x = key_states_251_cast_fp16)[name = string("op_9009_cast_fp16")]; tensor concat_302x = const()[name = string("concat_302x"), val = tensor([1, 2, 128, -1])]; tensor var_9016_cast_fp16 = reshape(shape = concat_302x, x = value_states_151_cast_fp16)[name = string("op_9016_cast_fp16")]; tensor var_9020_cast_fp16 = mul(x = x_251_cast_fp16, y = var_869_cast_fp16)[name = string("op_9020_cast_fp16")]; tensor var_9021_split_sizes_0 = const()[name = string("op_9021_split_sizes_0"), val = tensor([64, 64])]; int32 var_9021_axis_0 = const()[name = string("op_9021_axis_0"), val = int32(-2)]; tensor var_9021_cast_fp16_0, tensor var_9021_cast_fp16_1 = split(axis = var_9021_axis_0, split_sizes = var_9021_split_sizes_0, x = x_251_cast_fp16)[name = string("op_9021_cast_fp16")]; fp16 const_252_promoted_to_fp16 = const()[name = string("const_252_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9023_cast_fp16 = mul(x = var_9021_cast_fp16_1, y = const_252_promoted_to_fp16)[name = string("op_9023_cast_fp16")]; int32 var_9025 = const()[name = string("op_9025"), val = int32(-2)]; bool var_9026_interleave_0 = const()[name = string("op_9026_interleave_0"), val = bool(false)]; tensor var_9026_cast_fp16 = concat(axis = var_9025, interleave = var_9026_interleave_0, values = (var_9023_cast_fp16, var_9021_cast_fp16_0))[name = string("op_9026_cast_fp16")]; tensor var_9027_cast_fp16 = mul(x = var_9026_cast_fp16, y = var_878_cast_fp16)[name = string("op_9027_cast_fp16")]; tensor query_states_153_cast_fp16 = add(x = var_9020_cast_fp16, y = var_9027_cast_fp16)[name = string("query_states_153_cast_fp16")]; tensor var_9033_cast_fp16 = mul(x = var_9009_cast_fp16, y = var_869_cast_fp16)[name = string("op_9033_cast_fp16")]; tensor var_9034_split_sizes_0 = const()[name = string("op_9034_split_sizes_0"), val = tensor([64, 64])]; int32 var_9034_axis_0 = const()[name = string("op_9034_axis_0"), val = int32(-2)]; tensor var_9034_cast_fp16_0, tensor var_9034_cast_fp16_1 = split(axis = var_9034_axis_0, split_sizes = var_9034_split_sizes_0, x = var_9009_cast_fp16)[name = string("op_9034_cast_fp16")]; fp16 const_253_promoted_to_fp16 = const()[name = string("const_253_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9036_cast_fp16 = mul(x = var_9034_cast_fp16_1, y = const_253_promoted_to_fp16)[name = string("op_9036_cast_fp16")]; int32 var_9038 = const()[name = string("op_9038"), val = int32(-2)]; bool var_9039_interleave_0 = const()[name = string("op_9039_interleave_0"), val = bool(false)]; tensor var_9039_cast_fp16 = concat(axis = var_9038, interleave = var_9039_interleave_0, values = (var_9036_cast_fp16, var_9034_cast_fp16_0))[name = string("op_9039_cast_fp16")]; tensor var_9040_cast_fp16 = mul(x = var_9039_cast_fp16, y = var_878_cast_fp16)[name = string("op_9040_cast_fp16")]; tensor key_states_255_cast_fp16 = add(x = var_9033_cast_fp16, y = var_9040_cast_fp16)[name = string("key_states_255_cast_fp16")]; tensor expand_dims_300 = const()[name = string("expand_dims_300"), val = tensor([25])]; tensor expand_dims_301 = const()[name = string("expand_dims_301"), val = tensor([0])]; tensor expand_dims_303 = const()[name = string("expand_dims_303"), val = tensor([0])]; int32 concat_305_axis_0 = const()[name = string("concat_305_axis_0"), val = int32(0)]; bool concat_305_interleave_0 = const()[name = string("concat_305_interleave_0"), val = bool(false)]; tensor concat_305 = concat(axis = concat_305_axis_0, interleave = concat_305_interleave_0, values = (expand_dims_300, expand_dims_301, position_id, expand_dims_303))[name = string("concat_305")]; tensor expand_dims_304 = const()[name = string("expand_dims_304"), val = tensor([26])]; tensor concat_306_values1_0 = const()[name = string("concat_306_values1_0"), val = tensor([0])]; tensor concat_306_values3_0 = const()[name = string("concat_306_values3_0"), val = tensor([0])]; int32 concat_306_axis_0 = const()[name = string("concat_306_axis_0"), val = int32(0)]; bool concat_306_interleave_0 = const()[name = string("concat_306_interleave_0"), val = bool(false)]; tensor concat_306 = concat(axis = concat_306_axis_0, interleave = concat_306_interleave_0, values = (expand_dims_304, concat_306_values1_0, cache_position_end, concat_306_values3_0))[name = string("concat_306")]; tensor key_states_257_perm_0 = const()[name = string("key_states_257_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_26_stride_0 = const()[name = string("key_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_257_cast_fp16 = transpose(perm = key_states_257_perm_0, x = key_states_255_cast_fp16)[name = string("transpose_180")]; tensor key_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = key_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = key_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_26_squeeze_mask_0, stride = key_cache_internal_tensor_assign_26_stride_0, update = key_states_257_cast_fp16, x = coreml_update_state_160)[name = string("key_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_26_cast_fp16, input = key_cache)[name = string("coreml_update_state_162_write_state")]; tensor coreml_update_state_162 = read_state(input = key_cache)[name = string("coreml_update_state_162")]; tensor value_states_153_perm_0 = const()[name = string("value_states_153_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_26_stride_0 = const()[name = string("value_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_153_cast_fp16 = transpose(perm = value_states_153_perm_0, x = var_9016_cast_fp16)[name = string("transpose_179")]; tensor value_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = value_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = value_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_26_squeeze_mask_0, stride = value_cache_internal_tensor_assign_26_stride_0, update = value_states_153_cast_fp16, x = coreml_update_state_161)[name = string("value_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_26_cast_fp16, input = value_cache)[name = string("coreml_update_state_163_write_state")]; tensor coreml_update_state_163 = read_state(input = value_cache)[name = string("coreml_update_state_163")]; tensor var_9110_begin_0 = const()[name = string("op_9110_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9110_end_0 = const()[name = string("op_9110_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9110_end_mask_0 = const()[name = string("op_9110_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9110_cast_fp16 = slice_by_index(begin = var_9110_begin_0, end = var_9110_end_0, end_mask = var_9110_end_mask_0, x = coreml_update_state_162)[name = string("op_9110_cast_fp16")]; tensor tile_50 = const()[name = string("tile_50"), val = tensor([1, 1])]; int32 var_9113_axis_0 = const()[name = string("op_9113_axis_0"), val = int32(1)]; tensor var_9113_cast_fp16_0, tensor var_9113_cast_fp16_1 = split(axis = var_9113_axis_0, split_sizes = tile_50, x = var_9110_cast_fp16)[name = string("op_9113_cast_fp16")]; tensor var_9120_begin_0 = const()[name = string("op_9120_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9120_end_0 = const()[name = string("op_9120_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9120_end_mask_0 = const()[name = string("op_9120_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9120_cast_fp16 = slice_by_index(begin = var_9120_begin_0, end = var_9120_end_0, end_mask = var_9120_end_mask_0, x = coreml_update_state_163)[name = string("op_9120_cast_fp16")]; tensor tile_51 = const()[name = string("tile_51"), val = tensor([1, 1])]; int32 var_9123_axis_0 = const()[name = string("op_9123_axis_0"), val = int32(1)]; tensor var_9123_cast_fp16_0, tensor var_9123_cast_fp16_1 = split(axis = var_9123_axis_0, split_sizes = tile_51, x = var_9120_cast_fp16)[name = string("op_9123_cast_fp16")]; tensor var_9126_split_sizes_0 = const()[name = string("op_9126_split_sizes_0"), val = tensor([8, 8])]; int32 var_9126_axis_0 = const()[name = string("op_9126_axis_0"), val = int32(1)]; tensor var_9126_0, tensor var_9126_1 = split(axis = var_9126_axis_0, split_sizes = var_9126_split_sizes_0, x = query_states_153_cast_fp16)[name = string("op_9126")]; bool attn_weights_401_transpose_x_0 = const()[name = string("attn_weights_401_transpose_x_0"), val = bool(false)]; bool attn_weights_401_transpose_y_0 = const()[name = string("attn_weights_401_transpose_y_0"), val = bool(false)]; tensor attn_weights_401_cast_fp16 = matmul(transpose_x = attn_weights_401_transpose_x_0, transpose_y = attn_weights_401_transpose_y_0, x = var_9113_cast_fp16_0, y = var_9126_0)[name = string("attn_weights_401_cast_fp16")]; fp16 var_9129_to_fp16 = const()[name = string("op_9129_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_403_cast_fp16 = mul(x = attn_weights_401_cast_fp16, y = var_9129_to_fp16)[name = string("attn_weights_403_cast_fp16")]; tensor attn_weights_405_cast_fp16 = add(x = attn_weights_403_cast_fp16, y = attn_mask_1)[name = string("attn_weights_405_cast_fp16")]; int32 var_9133 = const()[name = string("op_9133"), val = int32(-2)]; tensor attn_weights_407_cast_fp16 = softmax(axis = var_9133, x = attn_weights_405_cast_fp16)[name = string("attn_weights_407_cast_fp16")]; bool var_9139_transpose_x_1 = const()[name = string("op_9139_transpose_x_1"), val = bool(true)]; bool var_9139_transpose_y_1 = const()[name = string("op_9139_transpose_y_1"), val = bool(false)]; tensor var_9139_cast_fp16 = matmul(transpose_x = var_9139_transpose_x_1, transpose_y = var_9139_transpose_y_1, x = attn_weights_407_cast_fp16, y = var_9123_cast_fp16_0)[name = string("op_9139_cast_fp16")]; bool attn_weights_409_transpose_x_0 = const()[name = string("attn_weights_409_transpose_x_0"), val = bool(false)]; bool attn_weights_409_transpose_y_0 = const()[name = string("attn_weights_409_transpose_y_0"), val = bool(false)]; tensor attn_weights_409_cast_fp16 = matmul(transpose_x = attn_weights_409_transpose_x_0, transpose_y = attn_weights_409_transpose_y_0, x = var_9113_cast_fp16_1, y = var_9126_1)[name = string("attn_weights_409_cast_fp16")]; fp16 var_9141_to_fp16 = const()[name = string("op_9141_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_411_cast_fp16 = mul(x = attn_weights_409_cast_fp16, y = var_9141_to_fp16)[name = string("attn_weights_411_cast_fp16")]; tensor attn_weights_413_cast_fp16 = add(x = attn_weights_411_cast_fp16, y = attn_mask_1)[name = string("attn_weights_413_cast_fp16")]; int32 var_9145 = const()[name = string("op_9145"), val = int32(-2)]; tensor attn_weights_415_cast_fp16 = softmax(axis = var_9145, x = attn_weights_413_cast_fp16)[name = string("attn_weights_415_cast_fp16")]; bool attn_output_201_transpose_x_1 = const()[name = string("attn_output_201_transpose_x_1"), val = bool(true)]; bool attn_output_201_transpose_y_1 = const()[name = string("attn_output_201_transpose_y_1"), val = bool(false)]; tensor attn_output_201_cast_fp16 = matmul(transpose_x = attn_output_201_transpose_x_1, transpose_y = attn_output_201_transpose_y_1, x = attn_weights_415_cast_fp16, y = var_9123_cast_fp16_1)[name = string("attn_output_201_cast_fp16")]; int32 var_9153 = const()[name = string("op_9153"), val = int32(1)]; bool attn_output_203_interleave_0 = const()[name = string("attn_output_203_interleave_0"), val = bool(false)]; tensor attn_output_203_cast_fp16 = concat(axis = var_9153, interleave = attn_output_203_interleave_0, values = (var_9139_cast_fp16, attn_output_201_cast_fp16))[name = string("attn_output_203_cast_fp16")]; tensor var_9157_perm_0 = const()[name = string("op_9157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_311x = const()[name = string("concat_311x"), val = tensor([1, 2048, 1, -1])]; tensor var_9157_cast_fp16 = transpose(perm = var_9157_perm_0, x = attn_output_203_cast_fp16)[name = string("transpose_178")]; tensor attn_output_207_cast_fp16 = reshape(shape = concat_311x, x = var_9157_cast_fp16)[name = string("attn_output_207_cast_fp16")]; tensor hidden_states_253_strides_0 = const()[name = string("hidden_states_253_strides_0"), val = tensor([1, 1])]; string hidden_states_253_pad_type_0 = const()[name = string("hidden_states_253_pad_type_0"), val = string("valid")]; tensor hidden_states_253_pad_0 = const()[name = string("hidden_states_253_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_253_dilations_0 = const()[name = string("hidden_states_253_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_253_groups_0 = const()[name = string("hidden_states_253_groups_0"), val = int32(1)]; tensor hidden_states_253_cast_fp16 = conv(dilations = hidden_states_253_dilations_0, groups = hidden_states_253_groups_0, pad = hidden_states_253_pad_0, pad_type = hidden_states_253_pad_type_0, strides = hidden_states_253_strides_0, weight = layers_25_self_attn_o_proj_weight_cast_fp16, x = attn_output_207_cast_fp16)[name = string("hidden_states_253_cast_fp16")]; tensor hidden_states_255_cast_fp16 = add(x = hidden_states_249_cast_fp16, y = hidden_states_253_cast_fp16)[name = string("hidden_states_255_cast_fp16")]; fp16 const_258_promoted_to_fp16 = const()[name = string("const_258_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9190_cast_fp16 = mul(x = hidden_states_255_cast_fp16, y = const_258_promoted_to_fp16)[name = string("op_9190_cast_fp16")]; int32 var_9188 = const()[name = string("op_9188"), val = int32(1)]; bool doubled_205_interleave_0 = const()[name = string("doubled_205_interleave_0"), val = bool(false)]; tensor doubled_205_cast_fp16 = concat(axis = var_9188, interleave = doubled_205_interleave_0, values = (hidden_states_255_cast_fp16, var_9190_cast_fp16))[name = string("doubled_205_cast_fp16")]; tensor out_103_axes_0 = const()[name = string("out_103_axes_0"), val = tensor([1])]; tensor out_103_gamma_0_to_fp16 = const()[name = string("out_103_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541427584)))]; fp16 var_9200_to_fp16 = const()[name = string("op_9200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_103_cast_fp16 = layer_norm(axes = out_103_axes_0, epsilon = var_9200_to_fp16, gamma = out_103_gamma_0_to_fp16, x = doubled_205_cast_fp16)[name = string("out_103_cast_fp16")]; tensor var_9211_split_sizes_0 = const()[name = string("op_9211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9211_axis_0 = const()[name = string("op_9211_axis_0"), val = int32(1)]; tensor var_9211_cast_fp16_0, tensor var_9211_cast_fp16_1 = split(axis = var_9211_axis_0, split_sizes = var_9211_split_sizes_0, x = out_103_cast_fp16)[name = string("op_9211_cast_fp16")]; tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; tensor input_51_cast_fp16 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = layers_25_mlp_gate_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("input_51_cast_fp16")]; tensor var_9228_cast_fp16 = silu(x = input_51_cast_fp16)[name = string("op_9228_cast_fp16")]; tensor var_9234_strides_0 = const()[name = string("op_9234_strides_0"), val = tensor([1, 1])]; string var_9234_pad_type_0 = const()[name = string("op_9234_pad_type_0"), val = string("valid")]; tensor var_9234_pad_0 = const()[name = string("op_9234_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9234_dilations_0 = const()[name = string("op_9234_dilations_0"), val = tensor([1, 1])]; int32 var_9234_groups_0 = const()[name = string("op_9234_groups_0"), val = int32(1)]; tensor var_9234_cast_fp16 = conv(dilations = var_9234_dilations_0, groups = var_9234_groups_0, pad = var_9234_pad_0, pad_type = var_9234_pad_type_0, strides = var_9234_strides_0, weight = layers_25_mlp_up_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("op_9234_cast_fp16")]; tensor x_259_cast_fp16 = mul(x = var_9228_cast_fp16, y = var_9234_cast_fp16)[name = string("x_259_cast_fp16")]; tensor layers_25_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_25_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541435840)))]; tensor hidden_states_257_strides_0 = const()[name = string("hidden_states_257_strides_0"), val = tensor([1, 1])]; string hidden_states_257_pad_type_0 = const()[name = string("hidden_states_257_pad_type_0"), val = string("valid")]; tensor hidden_states_257_pad_0 = const()[name = string("hidden_states_257_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_257_dilations_0 = const()[name = string("hidden_states_257_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_257_groups_0 = const()[name = string("hidden_states_257_groups_0"), val = int32(1)]; tensor hidden_states_257_cast_fp16 = conv(dilations = hidden_states_257_dilations_0, groups = hidden_states_257_groups_0, pad = hidden_states_257_pad_0, pad_type = hidden_states_257_pad_type_0, strides = hidden_states_257_strides_0, weight = layers_25_mlp_down_proj_weight_to_fp16, x = x_259_cast_fp16)[name = string("hidden_states_257_cast_fp16")]; tensor hidden_states_259_cast_fp16 = add(x = hidden_states_255_cast_fp16, y = hidden_states_257_cast_fp16)[name = string("hidden_states_259_cast_fp16")]; fp16 const_260_promoted_to_fp16 = const()[name = string("const_260_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9252_cast_fp16 = mul(x = hidden_states_259_cast_fp16, y = const_260_promoted_to_fp16)[name = string("op_9252_cast_fp16")]; int32 var_9250 = const()[name = string("op_9250"), val = int32(1)]; bool doubled_209_interleave_0 = const()[name = string("doubled_209_interleave_0"), val = bool(false)]; tensor doubled_209_cast_fp16 = concat(axis = var_9250, interleave = doubled_209_interleave_0, values = (hidden_states_259_cast_fp16, var_9252_cast_fp16))[name = string("doubled_209_cast_fp16")]; tensor out_105_axes_0 = const()[name = string("out_105_axes_0"), val = tensor([1])]; tensor out_105_gamma_0_to_fp16 = const()[name = string("out_105_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566601728)))]; fp16 var_9262_to_fp16 = const()[name = string("op_9262_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_105_cast_fp16 = layer_norm(axes = out_105_axes_0, epsilon = var_9262_to_fp16, gamma = out_105_gamma_0_to_fp16, x = doubled_209_cast_fp16)[name = string("out_105_cast_fp16")]; tensor var_9273_split_sizes_0 = const()[name = string("op_9273_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9273_axis_0 = const()[name = string("op_9273_axis_0"), val = int32(1)]; tensor var_9273_cast_fp16_0, tensor var_9273_cast_fp16_1 = split(axis = var_9273_axis_0, split_sizes = var_9273_split_sizes_0, x = out_105_cast_fp16)[name = string("op_9273_cast_fp16")]; tensor query_states_157_strides_0 = const()[name = string("query_states_157_strides_0"), val = tensor([1, 1])]; string query_states_157_pad_type_0 = const()[name = string("query_states_157_pad_type_0"), val = string("valid")]; tensor query_states_157_pad_0 = const()[name = string("query_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_157_dilations_0 = const()[name = string("query_states_157_dilations_0"), val = tensor([1, 1])]; int32 query_states_157_groups_0 = const()[name = string("query_states_157_groups_0"), val = int32(1)]; tensor query_states_157_cast_fp16 = conv(dilations = query_states_157_dilations_0, groups = query_states_157_groups_0, pad = query_states_157_pad_0, pad_type = query_states_157_pad_type_0, strides = query_states_157_strides_0, weight = layers_26_self_attn_q_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("query_states_157_cast_fp16")]; tensor key_states_261_strides_0 = const()[name = string("key_states_261_strides_0"), val = tensor([1, 1])]; string key_states_261_pad_type_0 = const()[name = string("key_states_261_pad_type_0"), val = string("valid")]; tensor key_states_261_pad_0 = const()[name = string("key_states_261_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_261_dilations_0 = const()[name = string("key_states_261_dilations_0"), val = tensor([1, 1])]; int32 key_states_261_groups_0 = const()[name = string("key_states_261_groups_0"), val = int32(1)]; tensor key_states_261_cast_fp16 = conv(dilations = key_states_261_dilations_0, groups = key_states_261_groups_0, pad = key_states_261_pad_0, pad_type = key_states_261_pad_type_0, strides = key_states_261_strides_0, weight = layers_26_self_attn_k_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("key_states_261_cast_fp16")]; tensor layers_26_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_26_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566609984)))]; tensor value_states_157_strides_0 = const()[name = string("value_states_157_strides_0"), val = tensor([1, 1])]; string value_states_157_pad_type_0 = const()[name = string("value_states_157_pad_type_0"), val = string("valid")]; tensor value_states_157_pad_0 = const()[name = string("value_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_157_dilations_0 = const()[name = string("value_states_157_dilations_0"), val = tensor([1, 1])]; int32 value_states_157_groups_0 = const()[name = string("value_states_157_groups_0"), val = int32(1)]; tensor value_states_157_cast_fp16 = conv(dilations = value_states_157_dilations_0, groups = value_states_157_groups_0, pad = value_states_157_pad_0, pad_type = value_states_157_pad_type_0, strides = value_states_157_strides_0, weight = layers_26_self_attn_v_proj_weight_to_fp16, x = var_9273_cast_fp16_0)[name = string("value_states_157_cast_fp16")]; tensor concat_312x = const()[name = string("concat_312x"), val = tensor([1, 16, 128, -1])]; tensor x_261_cast_fp16 = reshape(shape = concat_312x, x = query_states_157_cast_fp16)[name = string("x_261_cast_fp16")]; tensor concat_313x = const()[name = string("concat_313x"), val = tensor([1, 2, 128, -1])]; tensor var_9330_cast_fp16 = reshape(shape = concat_313x, x = key_states_261_cast_fp16)[name = string("op_9330_cast_fp16")]; tensor concat_314x = const()[name = string("concat_314x"), val = tensor([1, 2, 128, -1])]; tensor var_9337_cast_fp16 = reshape(shape = concat_314x, x = value_states_157_cast_fp16)[name = string("op_9337_cast_fp16")]; tensor var_9341_cast_fp16 = mul(x = x_261_cast_fp16, y = var_869_cast_fp16)[name = string("op_9341_cast_fp16")]; tensor var_9342_split_sizes_0 = const()[name = string("op_9342_split_sizes_0"), val = tensor([64, 64])]; int32 var_9342_axis_0 = const()[name = string("op_9342_axis_0"), val = int32(-2)]; tensor var_9342_cast_fp16_0, tensor var_9342_cast_fp16_1 = split(axis = var_9342_axis_0, split_sizes = var_9342_split_sizes_0, x = x_261_cast_fp16)[name = string("op_9342_cast_fp16")]; fp16 const_262_promoted_to_fp16 = const()[name = string("const_262_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9344_cast_fp16 = mul(x = var_9342_cast_fp16_1, y = const_262_promoted_to_fp16)[name = string("op_9344_cast_fp16")]; int32 var_9346 = const()[name = string("op_9346"), val = int32(-2)]; bool var_9347_interleave_0 = const()[name = string("op_9347_interleave_0"), val = bool(false)]; tensor var_9347_cast_fp16 = concat(axis = var_9346, interleave = var_9347_interleave_0, values = (var_9344_cast_fp16, var_9342_cast_fp16_0))[name = string("op_9347_cast_fp16")]; tensor var_9348_cast_fp16 = mul(x = var_9347_cast_fp16, y = var_878_cast_fp16)[name = string("op_9348_cast_fp16")]; tensor query_states_159_cast_fp16 = add(x = var_9341_cast_fp16, y = var_9348_cast_fp16)[name = string("query_states_159_cast_fp16")]; tensor var_9354_cast_fp16 = mul(x = var_9330_cast_fp16, y = var_869_cast_fp16)[name = string("op_9354_cast_fp16")]; tensor var_9355_split_sizes_0 = const()[name = string("op_9355_split_sizes_0"), val = tensor([64, 64])]; int32 var_9355_axis_0 = const()[name = string("op_9355_axis_0"), val = int32(-2)]; tensor var_9355_cast_fp16_0, tensor var_9355_cast_fp16_1 = split(axis = var_9355_axis_0, split_sizes = var_9355_split_sizes_0, x = var_9330_cast_fp16)[name = string("op_9355_cast_fp16")]; fp16 const_263_promoted_to_fp16 = const()[name = string("const_263_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9357_cast_fp16 = mul(x = var_9355_cast_fp16_1, y = const_263_promoted_to_fp16)[name = string("op_9357_cast_fp16")]; int32 var_9359 = const()[name = string("op_9359"), val = int32(-2)]; bool var_9360_interleave_0 = const()[name = string("op_9360_interleave_0"), val = bool(false)]; tensor var_9360_cast_fp16 = concat(axis = var_9359, interleave = var_9360_interleave_0, values = (var_9357_cast_fp16, var_9355_cast_fp16_0))[name = string("op_9360_cast_fp16")]; tensor var_9361_cast_fp16 = mul(x = var_9360_cast_fp16, y = var_878_cast_fp16)[name = string("op_9361_cast_fp16")]; tensor key_states_265_cast_fp16 = add(x = var_9354_cast_fp16, y = var_9361_cast_fp16)[name = string("key_states_265_cast_fp16")]; tensor expand_dims_312 = const()[name = string("expand_dims_312"), val = tensor([26])]; tensor expand_dims_313 = const()[name = string("expand_dims_313"), val = tensor([0])]; tensor expand_dims_315 = const()[name = string("expand_dims_315"), val = tensor([0])]; int32 concat_317_axis_0 = const()[name = string("concat_317_axis_0"), val = int32(0)]; bool concat_317_interleave_0 = const()[name = string("concat_317_interleave_0"), val = bool(false)]; tensor concat_317 = concat(axis = concat_317_axis_0, interleave = concat_317_interleave_0, values = (expand_dims_312, expand_dims_313, position_id, expand_dims_315))[name = string("concat_317")]; tensor expand_dims_316 = const()[name = string("expand_dims_316"), val = tensor([27])]; tensor concat_318_values1_0 = const()[name = string("concat_318_values1_0"), val = tensor([0])]; tensor concat_318_values3_0 = const()[name = string("concat_318_values3_0"), val = tensor([0])]; int32 concat_318_axis_0 = const()[name = string("concat_318_axis_0"), val = int32(0)]; bool concat_318_interleave_0 = const()[name = string("concat_318_interleave_0"), val = bool(false)]; tensor concat_318 = concat(axis = concat_318_axis_0, interleave = concat_318_interleave_0, values = (expand_dims_316, concat_318_values1_0, cache_position_end, concat_318_values3_0))[name = string("concat_318")]; tensor key_states_267_perm_0 = const()[name = string("key_states_267_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_27_stride_0 = const()[name = string("key_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_267_cast_fp16 = transpose(perm = key_states_267_perm_0, x = key_states_265_cast_fp16)[name = string("transpose_177")]; tensor key_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = key_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = key_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_27_squeeze_mask_0, stride = key_cache_internal_tensor_assign_27_stride_0, update = key_states_267_cast_fp16, x = coreml_update_state_162)[name = string("key_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_27_cast_fp16, input = key_cache)[name = string("coreml_update_state_164_write_state")]; tensor coreml_update_state_164 = read_state(input = key_cache)[name = string("coreml_update_state_164")]; tensor value_states_159_perm_0 = const()[name = string("value_states_159_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_27_stride_0 = const()[name = string("value_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_159_cast_fp16 = transpose(perm = value_states_159_perm_0, x = var_9337_cast_fp16)[name = string("transpose_176")]; tensor value_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = value_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = value_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_27_squeeze_mask_0, stride = value_cache_internal_tensor_assign_27_stride_0, update = value_states_159_cast_fp16, x = coreml_update_state_163)[name = string("value_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_27_cast_fp16, input = value_cache)[name = string("coreml_update_state_165_write_state")]; tensor coreml_update_state_165 = read_state(input = value_cache)[name = string("coreml_update_state_165")]; tensor var_9431_begin_0 = const()[name = string("op_9431_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9431_end_0 = const()[name = string("op_9431_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9431_end_mask_0 = const()[name = string("op_9431_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9431_cast_fp16 = slice_by_index(begin = var_9431_begin_0, end = var_9431_end_0, end_mask = var_9431_end_mask_0, x = coreml_update_state_164)[name = string("op_9431_cast_fp16")]; tensor tile_52 = const()[name = string("tile_52"), val = tensor([1, 1])]; int32 var_9434_axis_0 = const()[name = string("op_9434_axis_0"), val = int32(1)]; tensor var_9434_cast_fp16_0, tensor var_9434_cast_fp16_1 = split(axis = var_9434_axis_0, split_sizes = tile_52, x = var_9431_cast_fp16)[name = string("op_9434_cast_fp16")]; tensor var_9441_begin_0 = const()[name = string("op_9441_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9441_end_0 = const()[name = string("op_9441_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9441_end_mask_0 = const()[name = string("op_9441_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9441_cast_fp16 = slice_by_index(begin = var_9441_begin_0, end = var_9441_end_0, end_mask = var_9441_end_mask_0, x = coreml_update_state_165)[name = string("op_9441_cast_fp16")]; tensor tile_53 = const()[name = string("tile_53"), val = tensor([1, 1])]; int32 var_9444_axis_0 = const()[name = string("op_9444_axis_0"), val = int32(1)]; tensor var_9444_cast_fp16_0, tensor var_9444_cast_fp16_1 = split(axis = var_9444_axis_0, split_sizes = tile_53, x = var_9441_cast_fp16)[name = string("op_9444_cast_fp16")]; tensor var_9447_split_sizes_0 = const()[name = string("op_9447_split_sizes_0"), val = tensor([8, 8])]; int32 var_9447_axis_0 = const()[name = string("op_9447_axis_0"), val = int32(1)]; tensor var_9447_0, tensor var_9447_1 = split(axis = var_9447_axis_0, split_sizes = var_9447_split_sizes_0, x = query_states_159_cast_fp16)[name = string("op_9447")]; bool attn_weights_417_transpose_x_0 = const()[name = string("attn_weights_417_transpose_x_0"), val = bool(false)]; bool attn_weights_417_transpose_y_0 = const()[name = string("attn_weights_417_transpose_y_0"), val = bool(false)]; tensor attn_weights_417_cast_fp16 = matmul(transpose_x = attn_weights_417_transpose_x_0, transpose_y = attn_weights_417_transpose_y_0, x = var_9434_cast_fp16_0, y = var_9447_0)[name = string("attn_weights_417_cast_fp16")]; fp16 var_9450_to_fp16 = const()[name = string("op_9450_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_419_cast_fp16 = mul(x = attn_weights_417_cast_fp16, y = var_9450_to_fp16)[name = string("attn_weights_419_cast_fp16")]; tensor attn_weights_421_cast_fp16 = add(x = attn_weights_419_cast_fp16, y = attn_mask_1)[name = string("attn_weights_421_cast_fp16")]; int32 var_9454 = const()[name = string("op_9454"), val = int32(-2)]; tensor attn_weights_423_cast_fp16 = softmax(axis = var_9454, x = attn_weights_421_cast_fp16)[name = string("attn_weights_423_cast_fp16")]; bool var_9460_transpose_x_1 = const()[name = string("op_9460_transpose_x_1"), val = bool(true)]; bool var_9460_transpose_y_1 = const()[name = string("op_9460_transpose_y_1"), val = bool(false)]; tensor var_9460_cast_fp16 = matmul(transpose_x = var_9460_transpose_x_1, transpose_y = var_9460_transpose_y_1, x = attn_weights_423_cast_fp16, y = var_9444_cast_fp16_0)[name = string("op_9460_cast_fp16")]; bool attn_weights_425_transpose_x_0 = const()[name = string("attn_weights_425_transpose_x_0"), val = bool(false)]; bool attn_weights_425_transpose_y_0 = const()[name = string("attn_weights_425_transpose_y_0"), val = bool(false)]; tensor attn_weights_425_cast_fp16 = matmul(transpose_x = attn_weights_425_transpose_x_0, transpose_y = attn_weights_425_transpose_y_0, x = var_9434_cast_fp16_1, y = var_9447_1)[name = string("attn_weights_425_cast_fp16")]; fp16 var_9462_to_fp16 = const()[name = string("op_9462_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_427_cast_fp16 = mul(x = attn_weights_425_cast_fp16, y = var_9462_to_fp16)[name = string("attn_weights_427_cast_fp16")]; tensor attn_weights_429_cast_fp16 = add(x = attn_weights_427_cast_fp16, y = attn_mask_1)[name = string("attn_weights_429_cast_fp16")]; int32 var_9466 = const()[name = string("op_9466"), val = int32(-2)]; tensor attn_weights_431_cast_fp16 = softmax(axis = var_9466, x = attn_weights_429_cast_fp16)[name = string("attn_weights_431_cast_fp16")]; bool attn_output_209_transpose_x_1 = const()[name = string("attn_output_209_transpose_x_1"), val = bool(true)]; bool attn_output_209_transpose_y_1 = const()[name = string("attn_output_209_transpose_y_1"), val = bool(false)]; tensor attn_output_209_cast_fp16 = matmul(transpose_x = attn_output_209_transpose_x_1, transpose_y = attn_output_209_transpose_y_1, x = attn_weights_431_cast_fp16, y = var_9444_cast_fp16_1)[name = string("attn_output_209_cast_fp16")]; int32 var_9474 = const()[name = string("op_9474"), val = int32(1)]; bool attn_output_211_interleave_0 = const()[name = string("attn_output_211_interleave_0"), val = bool(false)]; tensor attn_output_211_cast_fp16 = concat(axis = var_9474, interleave = attn_output_211_interleave_0, values = (var_9460_cast_fp16, attn_output_209_cast_fp16))[name = string("attn_output_211_cast_fp16")]; tensor var_9478_perm_0 = const()[name = string("op_9478_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_323x = const()[name = string("concat_323x"), val = tensor([1, 2048, 1, -1])]; tensor var_9478_cast_fp16 = transpose(perm = var_9478_perm_0, x = attn_output_211_cast_fp16)[name = string("transpose_175")]; tensor attn_output_215_cast_fp16 = reshape(shape = concat_323x, x = var_9478_cast_fp16)[name = string("attn_output_215_cast_fp16")]; tensor hidden_states_263_strides_0 = const()[name = string("hidden_states_263_strides_0"), val = tensor([1, 1])]; string hidden_states_263_pad_type_0 = const()[name = string("hidden_states_263_pad_type_0"), val = string("valid")]; tensor hidden_states_263_pad_0 = const()[name = string("hidden_states_263_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_263_dilations_0 = const()[name = string("hidden_states_263_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_263_groups_0 = const()[name = string("hidden_states_263_groups_0"), val = int32(1)]; tensor hidden_states_263_cast_fp16 = conv(dilations = hidden_states_263_dilations_0, groups = hidden_states_263_groups_0, pad = hidden_states_263_pad_0, pad_type = hidden_states_263_pad_type_0, strides = hidden_states_263_strides_0, weight = layers_26_self_attn_o_proj_weight_cast_fp16, x = attn_output_215_cast_fp16)[name = string("hidden_states_263_cast_fp16")]; tensor hidden_states_265_cast_fp16 = add(x = hidden_states_259_cast_fp16, y = hidden_states_263_cast_fp16)[name = string("hidden_states_265_cast_fp16")]; fp16 const_268_promoted_to_fp16 = const()[name = string("const_268_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9511_cast_fp16 = mul(x = hidden_states_265_cast_fp16, y = const_268_promoted_to_fp16)[name = string("op_9511_cast_fp16")]; int32 var_9509 = const()[name = string("op_9509"), val = int32(1)]; bool doubled_213_interleave_0 = const()[name = string("doubled_213_interleave_0"), val = bool(false)]; tensor doubled_213_cast_fp16 = concat(axis = var_9509, interleave = doubled_213_interleave_0, values = (hidden_states_265_cast_fp16, var_9511_cast_fp16))[name = string("doubled_213_cast_fp16")]; tensor out_107_axes_0 = const()[name = string("out_107_axes_0"), val = tensor([1])]; tensor out_107_gamma_0_to_fp16 = const()[name = string("out_107_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567658624)))]; fp16 var_9521_to_fp16 = const()[name = string("op_9521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_107_cast_fp16 = layer_norm(axes = out_107_axes_0, epsilon = var_9521_to_fp16, gamma = out_107_gamma_0_to_fp16, x = doubled_213_cast_fp16)[name = string("out_107_cast_fp16")]; tensor var_9532_split_sizes_0 = const()[name = string("op_9532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9532_axis_0 = const()[name = string("op_9532_axis_0"), val = int32(1)]; tensor var_9532_cast_fp16_0, tensor var_9532_cast_fp16_1 = split(axis = var_9532_axis_0, split_sizes = var_9532_split_sizes_0, x = out_107_cast_fp16)[name = string("op_9532_cast_fp16")]; tensor input_53_strides_0 = const()[name = string("input_53_strides_0"), val = tensor([1, 1])]; string input_53_pad_type_0 = const()[name = string("input_53_pad_type_0"), val = string("valid")]; tensor input_53_pad_0 = const()[name = string("input_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_53_dilations_0 = const()[name = string("input_53_dilations_0"), val = tensor([1, 1])]; int32 input_53_groups_0 = const()[name = string("input_53_groups_0"), val = int32(1)]; tensor input_53_cast_fp16 = conv(dilations = input_53_dilations_0, groups = input_53_groups_0, pad = input_53_pad_0, pad_type = input_53_pad_type_0, strides = input_53_strides_0, weight = layers_26_mlp_gate_proj_weight_cast_fp16, x = var_9532_cast_fp16_0)[name = string("input_53_cast_fp16")]; tensor var_9549_cast_fp16 = silu(x = input_53_cast_fp16)[name = string("op_9549_cast_fp16")]; tensor layers_26_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567666880)))]; tensor var_9555_strides_0 = const()[name = string("op_9555_strides_0"), val = tensor([1, 1])]; string var_9555_pad_type_0 = const()[name = string("op_9555_pad_type_0"), val = string("valid")]; tensor var_9555_pad_0 = const()[name = string("op_9555_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9555_dilations_0 = const()[name = string("op_9555_dilations_0"), val = tensor([1, 1])]; int32 var_9555_groups_0 = const()[name = string("op_9555_groups_0"), val = int32(1)]; tensor var_9555_cast_fp16 = conv(dilations = var_9555_dilations_0, groups = var_9555_groups_0, pad = var_9555_pad_0, pad_type = var_9555_pad_type_0, strides = var_9555_strides_0, weight = layers_26_mlp_up_proj_weight_to_fp16, x = var_9532_cast_fp16_0)[name = string("op_9555_cast_fp16")]; tensor x_269_cast_fp16 = mul(x = var_9549_cast_fp16, y = var_9555_cast_fp16)[name = string("x_269_cast_fp16")]; tensor layers_26_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1592832768)))]; tensor hidden_states_267_strides_0 = const()[name = string("hidden_states_267_strides_0"), val = tensor([1, 1])]; string hidden_states_267_pad_type_0 = const()[name = string("hidden_states_267_pad_type_0"), val = string("valid")]; tensor hidden_states_267_pad_0 = const()[name = string("hidden_states_267_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_267_dilations_0 = const()[name = string("hidden_states_267_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_267_groups_0 = const()[name = string("hidden_states_267_groups_0"), val = int32(1)]; tensor hidden_states_267_cast_fp16 = conv(dilations = hidden_states_267_dilations_0, groups = hidden_states_267_groups_0, pad = hidden_states_267_pad_0, pad_type = hidden_states_267_pad_type_0, strides = hidden_states_267_strides_0, weight = layers_26_mlp_down_proj_weight_to_fp16, x = x_269_cast_fp16)[name = string("hidden_states_267_cast_fp16")]; tensor hidden_states_269_cast_fp16 = add(x = hidden_states_265_cast_fp16, y = hidden_states_267_cast_fp16)[name = string("hidden_states_269_cast_fp16")]; fp16 const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9573_cast_fp16 = mul(x = hidden_states_269_cast_fp16, y = const_270_promoted_to_fp16)[name = string("op_9573_cast_fp16")]; int32 var_9571 = const()[name = string("op_9571"), val = int32(1)]; bool doubled_217_interleave_0 = const()[name = string("doubled_217_interleave_0"), val = bool(false)]; tensor doubled_217_cast_fp16 = concat(axis = var_9571, interleave = doubled_217_interleave_0, values = (hidden_states_269_cast_fp16, var_9573_cast_fp16))[name = string("doubled_217_cast_fp16")]; tensor out_109_axes_0 = const()[name = string("out_109_axes_0"), val = tensor([1])]; tensor out_109_gamma_0_to_fp16 = const()[name = string("out_109_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1617998656)))]; fp16 var_9583_to_fp16 = const()[name = string("op_9583_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_109_cast_fp16 = layer_norm(axes = out_109_axes_0, epsilon = var_9583_to_fp16, gamma = out_109_gamma_0_to_fp16, x = doubled_217_cast_fp16)[name = string("out_109_cast_fp16")]; tensor var_9594_split_sizes_0 = const()[name = string("op_9594_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9594_axis_0 = const()[name = string("op_9594_axis_0"), val = int32(1)]; tensor var_9594_cast_fp16_0, tensor var_9594_cast_fp16_1 = split(axis = var_9594_axis_0, split_sizes = var_9594_split_sizes_0, x = out_109_cast_fp16)[name = string("op_9594_cast_fp16")]; tensor layers_27_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1618006912)))]; tensor query_states_163_strides_0 = const()[name = string("query_states_163_strides_0"), val = tensor([1, 1])]; string query_states_163_pad_type_0 = const()[name = string("query_states_163_pad_type_0"), val = string("valid")]; tensor query_states_163_pad_0 = const()[name = string("query_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_163_dilations_0 = const()[name = string("query_states_163_dilations_0"), val = tensor([1, 1])]; int32 query_states_163_groups_0 = const()[name = string("query_states_163_groups_0"), val = int32(1)]; tensor query_states_163_cast_fp16 = conv(dilations = query_states_163_dilations_0, groups = query_states_163_groups_0, pad = query_states_163_pad_0, pad_type = query_states_163_pad_type_0, strides = query_states_163_strides_0, weight = layers_27_self_attn_q_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("query_states_163_cast_fp16")]; tensor layers_27_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1626395584)))]; tensor key_states_271_strides_0 = const()[name = string("key_states_271_strides_0"), val = tensor([1, 1])]; string key_states_271_pad_type_0 = const()[name = string("key_states_271_pad_type_0"), val = string("valid")]; tensor key_states_271_pad_0 = const()[name = string("key_states_271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_271_dilations_0 = const()[name = string("key_states_271_dilations_0"), val = tensor([1, 1])]; int32 key_states_271_groups_0 = const()[name = string("key_states_271_groups_0"), val = int32(1)]; tensor key_states_271_cast_fp16 = conv(dilations = key_states_271_dilations_0, groups = key_states_271_groups_0, pad = key_states_271_pad_0, pad_type = key_states_271_pad_type_0, strides = key_states_271_strides_0, weight = layers_27_self_attn_k_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("key_states_271_cast_fp16")]; tensor layers_27_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1627444224)))]; tensor value_states_163_strides_0 = const()[name = string("value_states_163_strides_0"), val = tensor([1, 1])]; string value_states_163_pad_type_0 = const()[name = string("value_states_163_pad_type_0"), val = string("valid")]; tensor value_states_163_pad_0 = const()[name = string("value_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_163_dilations_0 = const()[name = string("value_states_163_dilations_0"), val = tensor([1, 1])]; int32 value_states_163_groups_0 = const()[name = string("value_states_163_groups_0"), val = int32(1)]; tensor value_states_163_cast_fp16 = conv(dilations = value_states_163_dilations_0, groups = value_states_163_groups_0, pad = value_states_163_pad_0, pad_type = value_states_163_pad_type_0, strides = value_states_163_strides_0, weight = layers_27_self_attn_v_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("value_states_163_cast_fp16")]; tensor concat_324x = const()[name = string("concat_324x"), val = tensor([1, 16, 128, -1])]; tensor x_271_cast_fp16 = reshape(shape = concat_324x, x = query_states_163_cast_fp16)[name = string("x_271_cast_fp16")]; tensor concat_325x = const()[name = string("concat_325x"), val = tensor([1, 2, 128, -1])]; tensor var_9651_cast_fp16 = reshape(shape = concat_325x, x = key_states_271_cast_fp16)[name = string("op_9651_cast_fp16")]; tensor concat_326x = const()[name = string("concat_326x"), val = tensor([1, 2, 128, -1])]; tensor var_9658_cast_fp16 = reshape(shape = concat_326x, x = value_states_163_cast_fp16)[name = string("op_9658_cast_fp16")]; tensor var_9662_cast_fp16 = mul(x = x_271_cast_fp16, y = var_869_cast_fp16)[name = string("op_9662_cast_fp16")]; tensor var_9663_split_sizes_0 = const()[name = string("op_9663_split_sizes_0"), val = tensor([64, 64])]; int32 var_9663_axis_0 = const()[name = string("op_9663_axis_0"), val = int32(-2)]; tensor var_9663_cast_fp16_0, tensor var_9663_cast_fp16_1 = split(axis = var_9663_axis_0, split_sizes = var_9663_split_sizes_0, x = x_271_cast_fp16)[name = string("op_9663_cast_fp16")]; fp16 const_272_promoted_to_fp16 = const()[name = string("const_272_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9665_cast_fp16 = mul(x = var_9663_cast_fp16_1, y = const_272_promoted_to_fp16)[name = string("op_9665_cast_fp16")]; int32 var_9667 = const()[name = string("op_9667"), val = int32(-2)]; bool var_9668_interleave_0 = const()[name = string("op_9668_interleave_0"), val = bool(false)]; tensor var_9668_cast_fp16 = concat(axis = var_9667, interleave = var_9668_interleave_0, values = (var_9665_cast_fp16, var_9663_cast_fp16_0))[name = string("op_9668_cast_fp16")]; tensor var_9669_cast_fp16 = mul(x = var_9668_cast_fp16, y = var_878_cast_fp16)[name = string("op_9669_cast_fp16")]; tensor query_states_165_cast_fp16 = add(x = var_9662_cast_fp16, y = var_9669_cast_fp16)[name = string("query_states_165_cast_fp16")]; tensor var_9675_cast_fp16 = mul(x = var_9651_cast_fp16, y = var_869_cast_fp16)[name = string("op_9675_cast_fp16")]; tensor var_9676_split_sizes_0 = const()[name = string("op_9676_split_sizes_0"), val = tensor([64, 64])]; int32 var_9676_axis_0 = const()[name = string("op_9676_axis_0"), val = int32(-2)]; tensor var_9676_cast_fp16_0, tensor var_9676_cast_fp16_1 = split(axis = var_9676_axis_0, split_sizes = var_9676_split_sizes_0, x = var_9651_cast_fp16)[name = string("op_9676_cast_fp16")]; fp16 const_273_promoted_to_fp16 = const()[name = string("const_273_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9678_cast_fp16 = mul(x = var_9676_cast_fp16_1, y = const_273_promoted_to_fp16)[name = string("op_9678_cast_fp16")]; int32 var_9680 = const()[name = string("op_9680"), val = int32(-2)]; bool var_9681_interleave_0 = const()[name = string("op_9681_interleave_0"), val = bool(false)]; tensor var_9681_cast_fp16 = concat(axis = var_9680, interleave = var_9681_interleave_0, values = (var_9678_cast_fp16, var_9676_cast_fp16_0))[name = string("op_9681_cast_fp16")]; tensor var_9682_cast_fp16 = mul(x = var_9681_cast_fp16, y = var_878_cast_fp16)[name = string("op_9682_cast_fp16")]; tensor key_states_275_cast_fp16 = add(x = var_9675_cast_fp16, y = var_9682_cast_fp16)[name = string("key_states_275_cast_fp16")]; tensor expand_dims_324 = const()[name = string("expand_dims_324"), val = tensor([27])]; tensor expand_dims_325 = const()[name = string("expand_dims_325"), val = tensor([0])]; tensor expand_dims_327 = const()[name = string("expand_dims_327"), val = tensor([0])]; int32 concat_329_axis_0 = const()[name = string("concat_329_axis_0"), val = int32(0)]; bool concat_329_interleave_0 = const()[name = string("concat_329_interleave_0"), val = bool(false)]; tensor concat_329 = concat(axis = concat_329_axis_0, interleave = concat_329_interleave_0, values = (expand_dims_324, expand_dims_325, position_id, expand_dims_327))[name = string("concat_329")]; tensor expand_dims_328 = const()[name = string("expand_dims_328"), val = tensor([28])]; tensor concat_330_values1_0 = const()[name = string("concat_330_values1_0"), val = tensor([0])]; tensor concat_330_values3_0 = const()[name = string("concat_330_values3_0"), val = tensor([0])]; int32 concat_330_axis_0 = const()[name = string("concat_330_axis_0"), val = int32(0)]; bool concat_330_interleave_0 = const()[name = string("concat_330_interleave_0"), val = bool(false)]; tensor concat_330 = concat(axis = concat_330_axis_0, interleave = concat_330_interleave_0, values = (expand_dims_328, concat_330_values1_0, cache_position_end, concat_330_values3_0))[name = string("concat_330")]; tensor key_states_277_perm_0 = const()[name = string("key_states_277_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_28_stride_0 = const()[name = string("key_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_277_cast_fp16 = transpose(perm = key_states_277_perm_0, x = key_states_275_cast_fp16)[name = string("transpose_174")]; tensor key_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = key_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = key_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_28_squeeze_mask_0, stride = key_cache_internal_tensor_assign_28_stride_0, update = key_states_277_cast_fp16, x = coreml_update_state_164)[name = string("key_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_28_cast_fp16, input = key_cache)[name = string("coreml_update_state_166_write_state")]; tensor coreml_update_state_166 = read_state(input = key_cache)[name = string("coreml_update_state_166")]; tensor value_states_165_perm_0 = const()[name = string("value_states_165_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_28_stride_0 = const()[name = string("value_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_165_cast_fp16 = transpose(perm = value_states_165_perm_0, x = var_9658_cast_fp16)[name = string("transpose_173")]; tensor value_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = value_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = value_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_28_squeeze_mask_0, stride = value_cache_internal_tensor_assign_28_stride_0, update = value_states_165_cast_fp16, x = coreml_update_state_165)[name = string("value_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_28_cast_fp16, input = value_cache)[name = string("coreml_update_state_167_write_state")]; tensor coreml_update_state_167 = read_state(input = value_cache)[name = string("coreml_update_state_167")]; tensor var_9752_begin_0 = const()[name = string("op_9752_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9752_end_0 = const()[name = string("op_9752_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9752_end_mask_0 = const()[name = string("op_9752_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9752_cast_fp16 = slice_by_index(begin = var_9752_begin_0, end = var_9752_end_0, end_mask = var_9752_end_mask_0, x = coreml_update_state_166)[name = string("op_9752_cast_fp16")]; tensor tile_54 = const()[name = string("tile_54"), val = tensor([1, 1])]; int32 var_9755_axis_0 = const()[name = string("op_9755_axis_0"), val = int32(1)]; tensor var_9755_cast_fp16_0, tensor var_9755_cast_fp16_1 = split(axis = var_9755_axis_0, split_sizes = tile_54, x = var_9752_cast_fp16)[name = string("op_9755_cast_fp16")]; tensor var_9762_begin_0 = const()[name = string("op_9762_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9762_end_0 = const()[name = string("op_9762_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9762_end_mask_0 = const()[name = string("op_9762_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9762_cast_fp16 = slice_by_index(begin = var_9762_begin_0, end = var_9762_end_0, end_mask = var_9762_end_mask_0, x = coreml_update_state_167)[name = string("op_9762_cast_fp16")]; tensor tile_55 = const()[name = string("tile_55"), val = tensor([1, 1])]; int32 var_9765_axis_0 = const()[name = string("op_9765_axis_0"), val = int32(1)]; tensor var_9765_cast_fp16_0, tensor var_9765_cast_fp16_1 = split(axis = var_9765_axis_0, split_sizes = tile_55, x = var_9762_cast_fp16)[name = string("op_9765_cast_fp16")]; tensor var_9768_split_sizes_0 = const()[name = string("op_9768_split_sizes_0"), val = tensor([8, 8])]; int32 var_9768_axis_0 = const()[name = string("op_9768_axis_0"), val = int32(1)]; tensor var_9768_0, tensor var_9768_1 = split(axis = var_9768_axis_0, split_sizes = var_9768_split_sizes_0, x = query_states_165_cast_fp16)[name = string("op_9768")]; bool attn_weights_433_transpose_x_0 = const()[name = string("attn_weights_433_transpose_x_0"), val = bool(false)]; bool attn_weights_433_transpose_y_0 = const()[name = string("attn_weights_433_transpose_y_0"), val = bool(false)]; tensor attn_weights_433_cast_fp16 = matmul(transpose_x = attn_weights_433_transpose_x_0, transpose_y = attn_weights_433_transpose_y_0, x = var_9755_cast_fp16_0, y = var_9768_0)[name = string("attn_weights_433_cast_fp16")]; fp16 var_9771_to_fp16 = const()[name = string("op_9771_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_435_cast_fp16 = mul(x = attn_weights_433_cast_fp16, y = var_9771_to_fp16)[name = string("attn_weights_435_cast_fp16")]; tensor attn_weights_437_cast_fp16 = add(x = attn_weights_435_cast_fp16, y = attn_mask_1)[name = string("attn_weights_437_cast_fp16")]; int32 var_9775 = const()[name = string("op_9775"), val = int32(-2)]; tensor attn_weights_439_cast_fp16 = softmax(axis = var_9775, x = attn_weights_437_cast_fp16)[name = string("attn_weights_439_cast_fp16")]; bool var_9781_transpose_x_1 = const()[name = string("op_9781_transpose_x_1"), val = bool(true)]; bool var_9781_transpose_y_1 = const()[name = string("op_9781_transpose_y_1"), val = bool(false)]; tensor var_9781_cast_fp16 = matmul(transpose_x = var_9781_transpose_x_1, transpose_y = var_9781_transpose_y_1, x = attn_weights_439_cast_fp16, y = var_9765_cast_fp16_0)[name = string("op_9781_cast_fp16")]; bool attn_weights_441_transpose_x_0 = const()[name = string("attn_weights_441_transpose_x_0"), val = bool(false)]; bool attn_weights_441_transpose_y_0 = const()[name = string("attn_weights_441_transpose_y_0"), val = bool(false)]; tensor attn_weights_441_cast_fp16 = matmul(transpose_x = attn_weights_441_transpose_x_0, transpose_y = attn_weights_441_transpose_y_0, x = var_9755_cast_fp16_1, y = var_9768_1)[name = string("attn_weights_441_cast_fp16")]; fp16 var_9783_to_fp16 = const()[name = string("op_9783_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_443_cast_fp16 = mul(x = attn_weights_441_cast_fp16, y = var_9783_to_fp16)[name = string("attn_weights_443_cast_fp16")]; tensor attn_weights_445_cast_fp16 = add(x = attn_weights_443_cast_fp16, y = attn_mask_1)[name = string("attn_weights_445_cast_fp16")]; int32 var_9787 = const()[name = string("op_9787"), val = int32(-2)]; tensor attn_weights_cast_fp16 = softmax(axis = var_9787, x = attn_weights_445_cast_fp16)[name = string("attn_weights_cast_fp16")]; bool attn_output_217_transpose_x_1 = const()[name = string("attn_output_217_transpose_x_1"), val = bool(true)]; bool attn_output_217_transpose_y_1 = const()[name = string("attn_output_217_transpose_y_1"), val = bool(false)]; tensor attn_output_217_cast_fp16 = matmul(transpose_x = attn_output_217_transpose_x_1, transpose_y = attn_output_217_transpose_y_1, x = attn_weights_cast_fp16, y = var_9765_cast_fp16_1)[name = string("attn_output_217_cast_fp16")]; int32 var_9795 = const()[name = string("op_9795"), val = int32(1)]; bool attn_output_219_interleave_0 = const()[name = string("attn_output_219_interleave_0"), val = bool(false)]; tensor attn_output_219_cast_fp16 = concat(axis = var_9795, interleave = attn_output_219_interleave_0, values = (var_9781_cast_fp16, attn_output_217_cast_fp16))[name = string("attn_output_219_cast_fp16")]; tensor var_9799_perm_0 = const()[name = string("op_9799_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_335x = const()[name = string("concat_335x"), val = tensor([1, 2048, 1, -1])]; tensor var_9799_cast_fp16 = transpose(perm = var_9799_perm_0, x = attn_output_219_cast_fp16)[name = string("transpose_172")]; tensor attn_output_cast_fp16 = reshape(shape = concat_335x, x = var_9799_cast_fp16)[name = string("attn_output_cast_fp16")]; tensor layers_27_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1628492864)))]; tensor hidden_states_273_strides_0 = const()[name = string("hidden_states_273_strides_0"), val = tensor([1, 1])]; string hidden_states_273_pad_type_0 = const()[name = string("hidden_states_273_pad_type_0"), val = string("valid")]; tensor hidden_states_273_pad_0 = const()[name = string("hidden_states_273_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_273_dilations_0 = const()[name = string("hidden_states_273_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_273_groups_0 = const()[name = string("hidden_states_273_groups_0"), val = int32(1)]; tensor hidden_states_273_cast_fp16 = conv(dilations = hidden_states_273_dilations_0, groups = hidden_states_273_groups_0, pad = hidden_states_273_pad_0, pad_type = hidden_states_273_pad_type_0, strides = hidden_states_273_strides_0, weight = layers_27_self_attn_o_proj_weight_to_fp16, x = attn_output_cast_fp16)[name = string("hidden_states_273_cast_fp16")]; tensor hidden_states_275_cast_fp16 = add(x = hidden_states_269_cast_fp16, y = hidden_states_273_cast_fp16)[name = string("hidden_states_275_cast_fp16")]; fp16 const_278_promoted_to_fp16 = const()[name = string("const_278_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9832_cast_fp16 = mul(x = hidden_states_275_cast_fp16, y = const_278_promoted_to_fp16)[name = string("op_9832_cast_fp16")]; int32 var_9830 = const()[name = string("op_9830"), val = int32(1)]; bool doubled_221_interleave_0 = const()[name = string("doubled_221_interleave_0"), val = bool(false)]; tensor doubled_221_cast_fp16 = concat(axis = var_9830, interleave = doubled_221_interleave_0, values = (hidden_states_275_cast_fp16, var_9832_cast_fp16))[name = string("doubled_221_cast_fp16")]; tensor out_111_axes_0 = const()[name = string("out_111_axes_0"), val = tensor([1])]; tensor out_111_gamma_0_to_fp16 = const()[name = string("out_111_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636881536)))]; fp16 var_9842_to_fp16 = const()[name = string("op_9842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_111_cast_fp16 = layer_norm(axes = out_111_axes_0, epsilon = var_9842_to_fp16, gamma = out_111_gamma_0_to_fp16, x = doubled_221_cast_fp16)[name = string("out_111_cast_fp16")]; tensor var_9853_split_sizes_0 = const()[name = string("op_9853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9853_axis_0 = const()[name = string("op_9853_axis_0"), val = int32(1)]; tensor var_9853_cast_fp16_0, tensor var_9853_cast_fp16_1 = split(axis = var_9853_axis_0, split_sizes = var_9853_split_sizes_0, x = out_111_cast_fp16)[name = string("op_9853_cast_fp16")]; tensor layers_27_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636889792)))]; tensor input_strides_0 = const()[name = string("input_strides_0"), val = tensor([1, 1])]; string input_pad_type_0 = const()[name = string("input_pad_type_0"), val = string("valid")]; tensor input_pad_0 = const()[name = string("input_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_dilations_0 = const()[name = string("input_dilations_0"), val = tensor([1, 1])]; int32 input_groups_0 = const()[name = string("input_groups_0"), val = int32(1)]; tensor input_cast_fp16 = conv(dilations = input_dilations_0, groups = input_groups_0, pad = input_pad_0, pad_type = input_pad_type_0, strides = input_strides_0, weight = layers_27_mlp_gate_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("input_cast_fp16")]; tensor var_9870_cast_fp16 = silu(x = input_cast_fp16)[name = string("op_9870_cast_fp16")]; tensor layers_27_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1662055680)))]; tensor var_9876_strides_0 = const()[name = string("op_9876_strides_0"), val = tensor([1, 1])]; string var_9876_pad_type_0 = const()[name = string("op_9876_pad_type_0"), val = string("valid")]; tensor var_9876_pad_0 = const()[name = string("op_9876_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9876_dilations_0 = const()[name = string("op_9876_dilations_0"), val = tensor([1, 1])]; int32 var_9876_groups_0 = const()[name = string("op_9876_groups_0"), val = int32(1)]; tensor var_9876_cast_fp16 = conv(dilations = var_9876_dilations_0, groups = var_9876_groups_0, pad = var_9876_pad_0, pad_type = var_9876_pad_type_0, strides = var_9876_strides_0, weight = layers_27_mlp_up_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("op_9876_cast_fp16")]; tensor x_cast_fp16 = mul(x = var_9870_cast_fp16, y = var_9876_cast_fp16)[name = string("x_cast_fp16")]; tensor layers_27_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1687221568)))]; tensor hidden_states_277_strides_0 = const()[name = string("hidden_states_277_strides_0"), val = tensor([1, 1])]; string hidden_states_277_pad_type_0 = const()[name = string("hidden_states_277_pad_type_0"), val = string("valid")]; tensor hidden_states_277_pad_0 = const()[name = string("hidden_states_277_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_277_dilations_0 = const()[name = string("hidden_states_277_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_277_groups_0 = const()[name = string("hidden_states_277_groups_0"), val = int32(1)]; tensor hidden_states_277_cast_fp16 = conv(dilations = hidden_states_277_dilations_0, groups = hidden_states_277_groups_0, pad = hidden_states_277_pad_0, pad_type = hidden_states_277_pad_type_0, strides = hidden_states_277_strides_0, weight = layers_27_mlp_down_proj_weight_to_fp16, x = x_cast_fp16)[name = string("hidden_states_277_cast_fp16")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_275_cast_fp16, y = hidden_states_277_cast_fp16)[name = string("hidden_states_cast_fp16")]; fp16 const_280_promoted_to_fp16 = const()[name = string("const_280_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9894_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_280_promoted_to_fp16)[name = string("op_9894_cast_fp16")]; int32 var_9892 = const()[name = string("op_9892"), val = int32(1)]; bool doubled_225_interleave_0 = const()[name = string("doubled_225_interleave_0"), val = bool(false)]; tensor doubled_225_cast_fp16 = concat(axis = var_9892, interleave = doubled_225_interleave_0, values = (hidden_states_cast_fp16, var_9894_cast_fp16))[name = string("doubled_225_cast_fp16")]; tensor out_axes_0 = const()[name = string("out_axes_0"), val = tensor([1])]; tensor out_gamma_0_to_fp16 = const()[name = string("out_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1712387456)))]; fp16 var_9904_to_fp16 = const()[name = string("op_9904_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_cast_fp16 = layer_norm(axes = out_axes_0, epsilon = var_9904_to_fp16, gamma = out_gamma_0_to_fp16, x = doubled_225_cast_fp16)[name = string("out_cast_fp16")]; tensor var_9915_split_sizes_0 = const()[name = string("op_9915_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9915_axis_0 = const()[name = string("op_9915_axis_0"), val = int32(1)]; tensor hidden_states, tensor var_9915_cast_fp16_1 = split(axis = var_9915_axis_0, split_sizes = var_9915_split_sizes_0, x = out_cast_fp16)[name = string("op_9915_cast_fp16")]; } -> (hidden_states); func main(tensor inputs_embeds, state> key_cache, tensor position_id, tensor position_index_seed, state> value_cache) { tensor layers_1_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(64))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524992))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(524416))))[name = string("layers_1_self_attn_v_proj_weight_cast_fp16")]; tensor layers_1_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(525312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13120640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13108288))))[name = string("layers_1_mlp_up_proj_weight_cast_fp16")]; tensor layers_2_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13126848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651776))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13651200))))[name = string("layers_2_self_attn_v_proj_weight_cast_fp16")]; tensor layers_2_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(13652096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26247424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26235072))))[name = string("layers_2_mlp_up_proj_weight_cast_fp16")]; tensor layers_3_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26253632))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778560))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26777984))))[name = string("layers_3_self_attn_v_proj_weight_cast_fp16")]; tensor layers_3_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(26778880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30977408))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30973248))))[name = string("layers_3_self_attn_o_proj_weight_cast_fp16")]; tensor layers_3_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(30979520))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43566656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43562496))))[name = string("layers_3_mlp_down_proj_weight_cast_fp16")]; tensor layers_4_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(43568768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093696))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44093120))))[name = string("layers_4_self_attn_v_proj_weight_cast_fp16")]; tensor layers_4_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(44094016))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48292544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48288384))))[name = string("layers_4_self_attn_o_proj_weight_cast_fp16")]; tensor layers_4_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(48294656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60877632))))[name = string("layers_4_mlp_gate_proj_weight_cast_fp16")]; tensor layers_4_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(60896192))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73491520))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73479168))))[name = string("layers_4_mlp_up_proj_weight_cast_fp16")]; tensor layers_4_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(73497728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86084864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86080704))))[name = string("layers_4_mlp_down_proj_weight_cast_fp16")]; tensor layers_5_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86086976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86611328))))[name = string("layers_5_self_attn_v_proj_weight_cast_fp16")]; tensor layers_5_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(86612224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90810752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90806592))))[name = string("layers_5_self_attn_o_proj_weight_cast_fp16")]; tensor layers_5_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(90812864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103408192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103395840))))[name = string("layers_5_mlp_up_proj_weight_cast_fp16")]; tensor layers_5_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(103414400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116001536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(115997376))))[name = string("layers_5_mlp_down_proj_weight_cast_fp16")]; tensor layers_6_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116003648))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528576))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528000))))[name = string("layers_6_self_attn_v_proj_weight_cast_fp16")]; tensor layers_6_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(116528896))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120727424))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120723264))))[name = string("layers_6_self_attn_o_proj_weight_cast_fp16")]; tensor layers_6_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(120729536))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133324864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133312512))))[name = string("layers_6_mlp_gate_proj_weight_cast_fp16")]; tensor layers_6_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(133331072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145926400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145914048))))[name = string("layers_6_mlp_up_proj_weight_cast_fp16")]; tensor layers_6_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(145932608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158519744))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158515584))))[name = string("layers_6_mlp_down_proj_weight_cast_fp16")]; tensor layers_7_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(158521856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159046208))))[name = string("layers_7_self_attn_v_proj_weight_cast_fp16")]; tensor layers_7_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(159047104))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163245632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163241472))))[name = string("layers_7_self_attn_o_proj_weight_cast_fp16")]; tensor layers_7_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(163247744))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175843072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175830720))))[name = string("layers_7_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(175849280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374208))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176373632))))[name = string("layers_8_self_attn_v_proj_weight_cast_fp16")]; tensor layers_8_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(176374528))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180573056))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180568896))))[name = string("layers_8_self_attn_o_proj_weight_cast_fp16")]; tensor layers_8_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(180575168))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193170496))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193158144))))[name = string("layers_8_mlp_gate_proj_weight_cast_fp16")]; tensor layers_8_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(193176704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205772032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205759680))))[name = string("layers_8_mlp_up_proj_weight_cast_fp16")]; tensor layers_8_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(205778240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218365376))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218361216))))[name = string("layers_8_mlp_down_proj_weight_cast_fp16")]; tensor layers_9_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218367488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218891840))))[name = string("layers_9_self_attn_v_proj_weight_cast_fp16")]; tensor layers_9_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(218892736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223091264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223087104))))[name = string("layers_9_self_attn_o_proj_weight_cast_fp16")]; tensor layers_9_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(223093376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235688704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235676352))))[name = string("layers_9_mlp_gate_proj_weight_cast_fp16")]; tensor layers_9_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(235694912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248290240))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248277888))))[name = string("layers_9_mlp_up_proj_weight_cast_fp16")]; tensor layers_9_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(248296448))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260883584))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260879424))))[name = string("layers_9_mlp_down_proj_weight_cast_fp16")]; tensor layers_10_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(260885696))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410048))))[name = string("layers_10_self_attn_v_proj_weight_cast_fp16")]; tensor layers_10_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(261410944))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265609472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265605312))))[name = string("layers_10_self_attn_o_proj_weight_cast_fp16")]; tensor layers_10_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(265611584))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278206912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278194560))))[name = string("layers_10_mlp_gate_proj_weight_cast_fp16")]; tensor layers_10_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(278213120))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290808448))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290796096))))[name = string("layers_10_mlp_up_proj_weight_cast_fp16")]; tensor layers_10_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(290814656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303401792))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303397632))))[name = string("layers_10_mlp_down_proj_weight_cast_fp16")]; tensor layers_11_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(303403904))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307602432))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307598272))))[name = string("layers_11_self_attn_q_proj_weight_cast_fp16")]; tensor layers_11_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(307604544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129472))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308128896))))[name = string("layers_11_self_attn_k_proj_weight_cast_fp16")]; tensor layers_11_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308129792))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654720))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308654144))))[name = string("layers_11_self_attn_v_proj_weight_cast_fp16")]; tensor layers_11_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(308655040))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312853568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312849408))))[name = string("layers_11_self_attn_o_proj_weight_cast_fp16")]; tensor layers_11_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(312855680))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325451008))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325438656))))[name = string("layers_11_mlp_gate_proj_weight_cast_fp16")]; tensor layers_11_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(325457216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338052544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338040192))))[name = string("layers_11_mlp_up_proj_weight_cast_fp16")]; tensor layers_11_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(338058752))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350645888))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350641728))))[name = string("layers_11_mlp_down_proj_weight_cast_fp16")]; tensor layers_12_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(350648000))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354846528))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354842368))))[name = string("layers_12_self_attn_q_proj_weight_cast_fp16")]; tensor layers_12_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(354848640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355372992))))[name = string("layers_12_self_attn_k_proj_weight_cast_fp16")]; tensor layers_12_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355373888))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898816))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355898240))))[name = string("layers_12_self_attn_v_proj_weight_cast_fp16")]; tensor layers_12_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(355899136))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360097664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360093504))))[name = string("layers_12_self_attn_o_proj_weight_cast_fp16")]; tensor layers_12_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(360099776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372695104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372682752))))[name = string("layers_12_mlp_gate_proj_weight_cast_fp16")]; tensor layers_12_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(372701312))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385296640))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385284288))))[name = string("layers_12_mlp_up_proj_weight_cast_fp16")]; tensor layers_12_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(385302848))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397889984))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397885824))))[name = string("layers_12_mlp_down_proj_weight_cast_fp16")]; tensor layers_13_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(397892096))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402090624))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402086464))))[name = string("layers_13_self_attn_q_proj_weight_cast_fp16")]; tensor layers_13_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402092736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617664))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617088))))[name = string("layers_13_self_attn_k_proj_weight_cast_fp16")]; tensor layers_13_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(402617984))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403142336))))[name = string("layers_13_self_attn_v_proj_weight_cast_fp16")]; tensor layers_13_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(403143232))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407341760))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407337600))))[name = string("layers_13_self_attn_o_proj_weight_cast_fp16")]; tensor layers_13_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(407343872))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419939200))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419926848))))[name = string("layers_13_mlp_gate_proj_weight_cast_fp16")]; tensor layers_13_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(419945408))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432532544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432528384))))[name = string("layers_13_mlp_down_proj_weight_cast_fp16")]; tensor layers_14_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(432534656))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436733184))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436729024))))[name = string("layers_14_self_attn_q_proj_weight_cast_fp16")]; tensor layers_14_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(436735296))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437259648))))[name = string("layers_14_self_attn_v_proj_weight_cast_fp16")]; tensor layers_14_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(437260544))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441459072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441454912))))[name = string("layers_14_self_attn_o_proj_weight_cast_fp16")]; tensor layers_14_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(441461184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454056512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454044160))))[name = string("layers_14_mlp_gate_proj_weight_cast_fp16")]; tensor layers_14_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(454062720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466658048))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466645696))))[name = string("layers_14_mlp_up_proj_weight_cast_fp16")]; tensor layers_14_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(466664256))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479251392))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479247232))))[name = string("layers_14_mlp_down_proj_weight_cast_fp16")]; tensor layers_15_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(479253504))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483452032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483447872))))[name = string("layers_15_self_attn_q_proj_weight_cast_fp16")]; tensor layers_15_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483454144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483978496))))[name = string("layers_15_self_attn_k_proj_weight_cast_fp16")]; tensor layers_15_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(483979392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504320))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484503744))))[name = string("layers_15_self_attn_v_proj_weight_cast_fp16")]; tensor layers_15_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(484504640))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488703168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488699008))))[name = string("layers_15_self_attn_o_proj_weight_cast_fp16")]; tensor layers_15_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(488705280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501300608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501288256))))[name = string("layers_15_mlp_gate_proj_weight_cast_fp16")]; tensor layers_15_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(501306816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513902144))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513889792))))[name = string("layers_15_mlp_up_proj_weight_cast_fp16")]; tensor layers_15_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(513908352))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526495488))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526491328))))[name = string("layers_15_mlp_down_proj_weight_cast_fp16")]; tensor layers_16_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(526497600))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530696128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530691968))))[name = string("layers_16_self_attn_q_proj_weight_cast_fp16")]; tensor layers_16_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(530698240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223168))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531222592))))[name = string("layers_16_self_attn_k_proj_weight_cast_fp16")]; tensor layers_16_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531223488))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748416))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531747840))))[name = string("layers_16_self_attn_v_proj_weight_cast_fp16")]; tensor layers_16_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(531748736))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535947264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535943104))))[name = string("layers_16_self_attn_o_proj_weight_cast_fp16")]; tensor layers_16_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(535949376))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548536512))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548532352))))[name = string("layers_16_mlp_down_proj_weight_cast_fp16")]; tensor layers_17_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(548538624))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552737152))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552732992))))[name = string("layers_17_self_attn_q_proj_weight_cast_fp16")]; tensor layers_17_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(552739264))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553263616))))[name = string("layers_17_self_attn_k_proj_weight_cast_fp16")]; tensor layers_17_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553264512))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553788864))))[name = string("layers_17_self_attn_v_proj_weight_cast_fp16")]; tensor layers_17_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(553789760))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557988288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557984128))))[name = string("layers_17_self_attn_o_proj_weight_cast_fp16")]; tensor layers_17_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(557990400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570585728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570573376))))[name = string("layers_17_mlp_gate_proj_weight_cast_fp16")]; tensor layers_17_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(570591936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583187264))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583174912))))[name = string("layers_17_mlp_up_proj_weight_cast_fp16")]; tensor layers_17_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(583193472))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595780608))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595776448))))[name = string("layers_17_mlp_down_proj_weight_cast_fp16")]; tensor layers_18_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(595782720))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599981248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599977088))))[name = string("layers_18_self_attn_q_proj_weight_cast_fp16")]; tensor layers_18_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(599983360))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508288))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600507712))))[name = string("layers_18_self_attn_k_proj_weight_cast_fp16")]; tensor layers_18_self_attn_v_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(600508608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033536))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601032960))))[name = string("layers_18_self_attn_v_proj_weight_cast_fp16")]; tensor layers_18_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(601033856))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605232384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605228224))))[name = string("layers_18_self_attn_o_proj_weight_cast_fp16")]; tensor layers_18_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(605234496))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617829824))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617817472))))[name = string("layers_18_mlp_gate_proj_weight_cast_fp16")]; tensor layers_18_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(617836032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630431360))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630419008))))[name = string("layers_18_mlp_up_proj_weight_cast_fp16")]; tensor layers_18_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(630437568))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643024704))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643020544))))[name = string("layers_18_mlp_down_proj_weight_cast_fp16")]; tensor layers_19_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(643026816))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647225344))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647221184))))[name = string("layers_19_self_attn_q_proj_weight_cast_fp16")]; tensor layers_19_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647227456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752384))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647751808))))[name = string("layers_19_self_attn_k_proj_weight_cast_fp16")]; tensor layers_19_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(647752704))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660348032))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660335680))))[name = string("layers_19_mlp_gate_proj_weight_cast_fp16")]; tensor layers_19_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(660354240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672949568))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672937216))))[name = string("layers_19_mlp_up_proj_weight_cast_fp16")]; tensor layers_19_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(672955776))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685542912))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685538752))))[name = string("layers_19_mlp_down_proj_weight_cast_fp16")]; tensor layers_20_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(685545024))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689743552))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689739392))))[name = string("layers_20_self_attn_q_proj_weight_cast_fp16")]; tensor layers_20_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(689745664))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270592))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270016))))[name = string("layers_20_self_attn_k_proj_weight_cast_fp16")]; tensor layers_20_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(690270912))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694469440))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694465280))))[name = string("layers_20_self_attn_o_proj_weight_cast_fp16")]; tensor layers_20_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(694471552))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707066880))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707054528))))[name = string("layers_20_mlp_gate_proj_weight_cast_fp16")]; tensor layers_20_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(707073088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719660224))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719656064))))[name = string("layers_20_mlp_down_proj_weight_cast_fp16")]; tensor layers_21_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(719662336))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723860864))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723856704))))[name = string("layers_21_self_attn_q_proj_weight_cast_fp16")]; tensor layers_21_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(723862976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387904))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724387328))))[name = string("layers_21_self_attn_k_proj_weight_cast_fp16")]; tensor layers_21_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(724388224))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728586752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728582592))))[name = string("layers_21_self_attn_o_proj_weight_cast_fp16")]; tensor layers_21_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(728588864))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741184192))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741171840))))[name = string("layers_21_mlp_gate_proj_weight_cast_fp16")]; tensor layers_21_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(741190400))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753785728))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753773376))))[name = string("layers_21_mlp_up_proj_weight_cast_fp16")]; tensor layers_21_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(753791936))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766379072))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766374912))))[name = string("layers_21_mlp_down_proj_weight_cast_fp16")]; tensor layers_22_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(766381184))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770579712))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770575552))))[name = string("layers_22_self_attn_q_proj_weight_cast_fp16")]; tensor layers_22_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(770581824))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106752))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771106176))))[name = string("layers_22_self_attn_k_proj_weight_cast_fp16")]; tensor layers_22_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(771107072))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783702400))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783690048))))[name = string("layers_22_mlp_gate_proj_weight_cast_fp16")]; tensor layers_22_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(783708608))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796303936))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796291584))))[name = string("layers_22_mlp_up_proj_weight_cast_fp16")]; tensor layers_22_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(796310144))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808897280))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808893120))))[name = string("layers_22_mlp_down_proj_weight_cast_fp16")]; tensor layers_23_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(808899392))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813097920))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813093760))))[name = string("layers_23_self_attn_q_proj_weight_cast_fp16")]; tensor layers_23_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813100032))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624960))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813624384))))[name = string("layers_23_self_attn_k_proj_weight_cast_fp16")]; tensor layers_23_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(813625280))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817823808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817819648))))[name = string("layers_23_self_attn_o_proj_weight_cast_fp16")]; tensor layers_23_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(817825920))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830421248))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830408896))))[name = string("layers_23_mlp_gate_proj_weight_cast_fp16")]; tensor layers_23_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(830427456))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843022784))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843010432))))[name = string("layers_23_mlp_up_proj_weight_cast_fp16")]; tensor layers_23_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(843028992))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855616128))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855611968))))[name = string("layers_23_mlp_down_proj_weight_cast_fp16")]; tensor layers_24_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(855618240))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859816768))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859812608))))[name = string("layers_24_self_attn_q_proj_weight_cast_fp16")]; tensor layers_24_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(859818880))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343808))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860343232))))[name = string("layers_24_self_attn_k_proj_weight_cast_fp16")]; tensor layers_24_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(860344128))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864542656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864538496))))[name = string("layers_24_self_attn_o_proj_weight_cast_fp16")]; tensor layers_24_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(864544768))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877140096))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877127744))))[name = string("layers_24_mlp_gate_proj_weight_cast_fp16")]; tensor layers_24_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(877146304))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889741632))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889729280))))[name = string("layers_24_mlp_up_proj_weight_cast_fp16")]; tensor layers_24_mlp_down_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(889747840))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902334976))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902330816))))[name = string("layers_24_mlp_down_proj_weight_cast_fp16")]; tensor layers_25_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(902337088))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906535616))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906531456))))[name = string("layers_25_self_attn_q_proj_weight_cast_fp16")]; tensor layers_25_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(906537728))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062656))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062080))))[name = string("layers_25_self_attn_k_proj_weight_cast_fp16")]; tensor layers_25_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(907062976))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911261504))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911257344))))[name = string("layers_25_self_attn_o_proj_weight_cast_fp16")]; tensor layers_25_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(911263616))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923858944))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923846592))))[name = string("layers_25_mlp_gate_proj_weight_cast_fp16")]; tensor layers_25_mlp_up_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(923865152))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936460480))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936448128))))[name = string("layers_25_mlp_up_proj_weight_cast_fp16")]; tensor layers_26_self_attn_q_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(936466688))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940665216))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940661056))))[name = string("layers_26_self_attn_q_proj_weight_cast_fp16")]; tensor layers_26_self_attn_k_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(940667328))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192256))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941191680))))[name = string("layers_26_self_attn_k_proj_weight_cast_fp16")]; tensor layers_26_self_attn_o_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(941192576))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945391104))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945386944))))[name = string("layers_26_self_attn_o_proj_weight_cast_fp16")]; tensor layers_26_mlp_gate_proj_weight_cast_fp16 = constexpr_blockwise_shift_scale(data = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(945393216))), offset = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957988544))), scale = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957976192))))[name = string("layers_26_mlp_gate_proj_weight_cast_fp16")]; int32 var_765 = const()[name = string("op_765"), val = int32(0)]; tensor var_766 = mul(x = position_index_seed, y = var_765)[name = string("op_766")]; int32 var_768 = const()[name = string("op_768"), val = int32(1)]; tensor ones = add(x = var_766, y = var_768)[name = string("ones")]; int32 var_770 = const()[name = string("op_770"), val = int32(0)]; bool var_772_exclusive_0 = const()[name = string("op_772_exclusive_0"), val = bool(false)]; bool var_772_reverse_0 = const()[name = string("op_772_reverse_0"), val = bool(false)]; tensor var_772 = cumsum(axis = var_770, exclusive = var_772_exclusive_0, reverse = var_772_reverse_0, x = ones)[name = string("op_772")]; int32 var_774 = const()[name = string("op_774"), val = int32(1)]; tensor position_offsets = sub(x = var_772, y = var_774)[name = string("position_offsets")]; tensor position_ids_1 = add(x = position_offsets, y = position_id)[name = string("position_ids_1")]; bool var_784_keep_dims_0 = const()[name = string("op_784_keep_dims_0"), val = bool(false)]; int32 var_784 = reduce_sum(keep_dims = var_784_keep_dims_0, x = ones)[name = string("op_784")]; int32 var_786 = const()[name = string("op_786"), val = int32(1)]; int32 offset = sub(x = var_784, y = var_786)[name = string("offset")]; tensor var_789 = add(x = position_id, y = offset)[name = string("op_789")]; int32 var_791 = const()[name = string("op_791"), val = int32(1)]; tensor cache_position_end = add(x = var_789, y = var_791)[name = string("cache_position_end")]; int32 greater_equal_0_y_0 = const()[name = string("greater_equal_0_y_0"), val = int32(0)]; tensor greater_equal_0 = greater_equal(x = position_ids_1, y = greater_equal_0_y_0)[name = string("greater_equal_0")]; int32 slice_by_index_0 = const()[name = string("slice_by_index_0"), val = int32(32768)]; tensor add_0 = add(x = position_ids_1, y = slice_by_index_0)[name = string("add_0")]; tensor select_0 = select(a = position_ids_1, b = add_0, cond = greater_equal_0)[name = string("select_0")]; tensor rope_emb_cos_cached_to_fp16 = const()[name = string("rope_emb_cos_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(957994752)))]; int32 cos_1_batch_dims_0 = const()[name = string("cos_1_batch_dims_0"), val = int32(0)]; bool cos_1_validate_indices_0 = const()[name = string("cos_1_validate_indices_0"), val = bool(false)]; int32 greater_equal_0_y_0_1 = const()[name = string("greater_equal_0_y_0_1"), val = int32(0)]; tensor greater_equal_0_1 = greater_equal(x = select_0, y = greater_equal_0_y_0_1)[name = string("greater_equal_0_1")]; int32 slice_by_index_0_1 = const()[name = string("slice_by_index_0_1"), val = int32(32768)]; tensor add_0_1 = add(x = select_0, y = slice_by_index_0_1)[name = string("add_0_1")]; tensor select_0_1 = select(a = select_0, b = add_0_1, cond = greater_equal_0_1)[name = string("select_0_1")]; int32 cos_1_cast_fp16_axis_0 = const()[name = string("cos_1_cast_fp16_axis_0"), val = int32(0)]; tensor cos_1_cast_fp16 = gather(axis = cos_1_cast_fp16_axis_0, batch_dims = cos_1_batch_dims_0, indices = select_0_1, validate_indices = cos_1_validate_indices_0, x = rope_emb_cos_cached_to_fp16)[name = string("cos_1_cast_fp16")]; tensor rope_emb_sin_cached_to_fp16 = const()[name = string("rope_emb_sin_cached_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(966383424)))]; int32 sin_1_batch_dims_0 = const()[name = string("sin_1_batch_dims_0"), val = int32(0)]; bool sin_1_validate_indices_0 = const()[name = string("sin_1_validate_indices_0"), val = bool(false)]; int32 sin_1_cast_fp16_axis_0 = const()[name = string("sin_1_cast_fp16_axis_0"), val = int32(0)]; tensor sin_1_cast_fp16 = gather(axis = sin_1_cast_fp16_axis_0, batch_dims = sin_1_batch_dims_0, indices = select_0_1, validate_indices = sin_1_validate_indices_0, x = rope_emb_sin_cached_to_fp16)[name = string("sin_1_cast_fp16")]; tensor var_865_perm_0 = const()[name = string("op_865_perm_0"), val = tensor([-1, -2])]; tensor var_867_axes_0 = const()[name = string("op_867_axes_0"), val = tensor([0])]; tensor var_865_cast_fp16 = transpose(perm = var_865_perm_0, x = cos_1_cast_fp16)[name = string("transpose_85")]; tensor var_867_cast_fp16 = expand_dims(axes = var_867_axes_0, x = var_865_cast_fp16)[name = string("op_867_cast_fp16")]; tensor var_869_axes_0 = const()[name = string("op_869_axes_0"), val = tensor([0])]; tensor var_869_cast_fp16 = expand_dims(axes = var_869_axes_0, x = var_867_cast_fp16)[name = string("op_869_cast_fp16")]; tensor var_874_perm_0 = const()[name = string("op_874_perm_0"), val = tensor([-1, -2])]; tensor var_876_axes_0 = const()[name = string("op_876_axes_0"), val = tensor([0])]; tensor var_874_cast_fp16 = transpose(perm = var_874_perm_0, x = sin_1_cast_fp16)[name = string("transpose_84")]; tensor var_876_cast_fp16 = expand_dims(axes = var_876_axes_0, x = var_874_cast_fp16)[name = string("op_876_cast_fp16")]; tensor var_878_axes_0 = const()[name = string("op_878_axes_0"), val = tensor([0])]; tensor var_878_cast_fp16 = expand_dims(axes = var_878_axes_0, x = var_876_cast_fp16)[name = string("op_878_cast_fp16")]; string position_ids_1_to_uint16_dtype_0 = const()[name = string("position_ids_1_to_uint16_dtype_0"), val = string("uint16")]; tensor causal_mask = const()[name = string("causal_mask"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(974772096)))]; int32 mask_axis_0 = const()[name = string("mask_axis_0"), val = int32(1)]; int32 mask_batch_dims_0 = const()[name = string("mask_batch_dims_0"), val = int32(0)]; bool mask_validate_indices_0 = const()[name = string("mask_validate_indices_0"), val = bool(false)]; tensor position_ids_1_to_uint16 = cast(dtype = position_ids_1_to_uint16_dtype_0, x = position_ids_1)[name = string("cast_1")]; tensor mask_cast_uint16 = gather(axis = mask_axis_0, batch_dims = mask_batch_dims_0, indices = position_ids_1_to_uint16, validate_indices = mask_validate_indices_0, x = causal_mask)[name = string("mask_cast_uint16")]; tensor var_895_axes_0 = const()[name = string("op_895_axes_0"), val = tensor([0])]; tensor var_895 = expand_dims(axes = var_895_axes_0, x = mask_cast_uint16)[name = string("op_895")]; tensor attn_mask_1_axes_0 = const()[name = string("attn_mask_1_axes_0"), val = tensor([0])]; tensor attn_mask_1 = expand_dims(axes = attn_mask_1_axes_0, x = var_895)[name = string("attn_mask_1")]; string inputs_embeds_to_fp16_dtype_0 = const()[name = string("inputs_embeds_to_fp16_dtype_0"), val = string("fp16")]; fp16 const_0_promoted_to_fp16 = const()[name = string("const_0_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor inputs_embeds_to_fp16 = cast(dtype = inputs_embeds_to_fp16_dtype_0, x = inputs_embeds)[name = string("cast_0")]; tensor var_906_cast_fp16 = mul(x = inputs_embeds_to_fp16, y = const_0_promoted_to_fp16)[name = string("op_906_cast_fp16")]; int32 var_904 = const()[name = string("op_904"), val = int32(1)]; bool doubled_1_interleave_0 = const()[name = string("doubled_1_interleave_0"), val = bool(false)]; tensor doubled_1_cast_fp16 = concat(axis = var_904, interleave = doubled_1_interleave_0, values = (inputs_embeds_to_fp16, var_906_cast_fp16))[name = string("doubled_1_cast_fp16")]; tensor out_1_axes_0 = const()[name = string("out_1_axes_0"), val = tensor([1])]; tensor out_1_gamma_0_to_fp16 = const()[name = string("out_1_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983160768)))]; fp16 var_916_to_fp16 = const()[name = string("op_916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_1_cast_fp16 = layer_norm(axes = out_1_axes_0, epsilon = var_916_to_fp16, gamma = out_1_gamma_0_to_fp16, x = doubled_1_cast_fp16)[name = string("out_1_cast_fp16")]; tensor var_927_split_sizes_0 = const()[name = string("op_927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_927_axis_0 = const()[name = string("op_927_axis_0"), val = int32(1)]; tensor var_927_cast_fp16_0, tensor var_927_cast_fp16_1 = split(axis = var_927_axis_0, split_sizes = var_927_split_sizes_0, x = out_1_cast_fp16)[name = string("op_927_cast_fp16")]; tensor layers_0_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(983169024)))]; tensor query_states_1_strides_0 = const()[name = string("query_states_1_strides_0"), val = tensor([1, 1])]; string query_states_1_pad_type_0 = const()[name = string("query_states_1_pad_type_0"), val = string("valid")]; tensor query_states_1_pad_0 = const()[name = string("query_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_1_dilations_0 = const()[name = string("query_states_1_dilations_0"), val = tensor([1, 1])]; int32 query_states_1_groups_0 = const()[name = string("query_states_1_groups_0"), val = int32(1)]; tensor query_states_1_cast_fp16 = conv(dilations = query_states_1_dilations_0, groups = query_states_1_groups_0, pad = query_states_1_pad_0, pad_type = query_states_1_pad_type_0, strides = query_states_1_strides_0, weight = layers_0_self_attn_q_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("query_states_1_cast_fp16")]; tensor layers_0_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(991557696)))]; tensor key_states_1_strides_0 = const()[name = string("key_states_1_strides_0"), val = tensor([1, 1])]; string key_states_1_pad_type_0 = const()[name = string("key_states_1_pad_type_0"), val = string("valid")]; tensor key_states_1_pad_0 = const()[name = string("key_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_1_dilations_0 = const()[name = string("key_states_1_dilations_0"), val = tensor([1, 1])]; int32 key_states_1_groups_0 = const()[name = string("key_states_1_groups_0"), val = int32(1)]; tensor key_states_1_cast_fp16 = conv(dilations = key_states_1_dilations_0, groups = key_states_1_groups_0, pad = key_states_1_pad_0, pad_type = key_states_1_pad_type_0, strides = key_states_1_strides_0, weight = layers_0_self_attn_k_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("key_states_1_cast_fp16")]; tensor layers_0_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(992606336)))]; tensor value_states_1_strides_0 = const()[name = string("value_states_1_strides_0"), val = tensor([1, 1])]; string value_states_1_pad_type_0 = const()[name = string("value_states_1_pad_type_0"), val = string("valid")]; tensor value_states_1_pad_0 = const()[name = string("value_states_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_1_dilations_0 = const()[name = string("value_states_1_dilations_0"), val = tensor([1, 1])]; int32 value_states_1_groups_0 = const()[name = string("value_states_1_groups_0"), val = int32(1)]; tensor value_states_1_cast_fp16 = conv(dilations = value_states_1_dilations_0, groups = value_states_1_groups_0, pad = value_states_1_pad_0, pad_type = value_states_1_pad_type_0, strides = value_states_1_strides_0, weight = layers_0_self_attn_v_proj_weight_to_fp16, x = var_927_cast_fp16_0)[name = string("value_states_1_cast_fp16")]; tensor concat_0x = const()[name = string("concat_0x"), val = tensor([1, 16, 128, -1])]; tensor x_1_cast_fp16 = reshape(shape = concat_0x, x = query_states_1_cast_fp16)[name = string("x_1_cast_fp16")]; tensor concat_1x = const()[name = string("concat_1x"), val = tensor([1, 2, 128, -1])]; tensor var_984_cast_fp16 = reshape(shape = concat_1x, x = key_states_1_cast_fp16)[name = string("op_984_cast_fp16")]; tensor concat_2x = const()[name = string("concat_2x"), val = tensor([1, 2, 128, -1])]; tensor var_991_cast_fp16 = reshape(shape = concat_2x, x = value_states_1_cast_fp16)[name = string("op_991_cast_fp16")]; tensor var_995_cast_fp16 = mul(x = x_1_cast_fp16, y = var_869_cast_fp16)[name = string("op_995_cast_fp16")]; tensor var_996_split_sizes_0 = const()[name = string("op_996_split_sizes_0"), val = tensor([64, 64])]; int32 var_996_axis_0 = const()[name = string("op_996_axis_0"), val = int32(-2)]; tensor var_996_cast_fp16_0, tensor var_996_cast_fp16_1 = split(axis = var_996_axis_0, split_sizes = var_996_split_sizes_0, x = x_1_cast_fp16)[name = string("op_996_cast_fp16")]; fp16 const_2_promoted_to_fp16 = const()[name = string("const_2_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_998_cast_fp16 = mul(x = var_996_cast_fp16_1, y = const_2_promoted_to_fp16)[name = string("op_998_cast_fp16")]; int32 var_1000 = const()[name = string("op_1000"), val = int32(-2)]; bool var_1001_interleave_0 = const()[name = string("op_1001_interleave_0"), val = bool(false)]; tensor var_1001_cast_fp16 = concat(axis = var_1000, interleave = var_1001_interleave_0, values = (var_998_cast_fp16, var_996_cast_fp16_0))[name = string("op_1001_cast_fp16")]; tensor var_1002_cast_fp16 = mul(x = var_1001_cast_fp16, y = var_878_cast_fp16)[name = string("op_1002_cast_fp16")]; tensor query_states_3_cast_fp16 = add(x = var_995_cast_fp16, y = var_1002_cast_fp16)[name = string("query_states_3_cast_fp16")]; tensor var_1008_cast_fp16 = mul(x = var_984_cast_fp16, y = var_869_cast_fp16)[name = string("op_1008_cast_fp16")]; tensor var_1009_split_sizes_0 = const()[name = string("op_1009_split_sizes_0"), val = tensor([64, 64])]; int32 var_1009_axis_0 = const()[name = string("op_1009_axis_0"), val = int32(-2)]; tensor var_1009_cast_fp16_0, tensor var_1009_cast_fp16_1 = split(axis = var_1009_axis_0, split_sizes = var_1009_split_sizes_0, x = var_984_cast_fp16)[name = string("op_1009_cast_fp16")]; fp16 const_3_promoted_to_fp16 = const()[name = string("const_3_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1011_cast_fp16 = mul(x = var_1009_cast_fp16_1, y = const_3_promoted_to_fp16)[name = string("op_1011_cast_fp16")]; int32 var_1013 = const()[name = string("op_1013"), val = int32(-2)]; bool var_1014_interleave_0 = const()[name = string("op_1014_interleave_0"), val = bool(false)]; tensor var_1014_cast_fp16 = concat(axis = var_1013, interleave = var_1014_interleave_0, values = (var_1011_cast_fp16, var_1009_cast_fp16_0))[name = string("op_1014_cast_fp16")]; tensor var_1015_cast_fp16 = mul(x = var_1014_cast_fp16, y = var_878_cast_fp16)[name = string("op_1015_cast_fp16")]; tensor key_states_5_cast_fp16 = add(x = var_1008_cast_fp16, y = var_1015_cast_fp16)[name = string("key_states_5_cast_fp16")]; tensor read_state_0 = read_state(input = key_cache)[name = string("read_state_0")]; tensor expand_dims_0 = const()[name = string("expand_dims_0"), val = tensor([0])]; tensor expand_dims_1 = const()[name = string("expand_dims_1"), val = tensor([0])]; tensor expand_dims_3 = const()[name = string("expand_dims_3"), val = tensor([0])]; int32 concat_5_axis_0 = const()[name = string("concat_5_axis_0"), val = int32(0)]; bool concat_5_interleave_0 = const()[name = string("concat_5_interleave_0"), val = bool(false)]; tensor concat_5 = concat(axis = concat_5_axis_0, interleave = concat_5_interleave_0, values = (expand_dims_0, expand_dims_1, position_id, expand_dims_3))[name = string("concat_5")]; tensor expand_dims_4 = const()[name = string("expand_dims_4"), val = tensor([1])]; tensor concat_6_values1_0 = const()[name = string("concat_6_values1_0"), val = tensor([0])]; tensor concat_6_values3_0 = const()[name = string("concat_6_values3_0"), val = tensor([0])]; int32 concat_6_axis_0 = const()[name = string("concat_6_axis_0"), val = int32(0)]; bool concat_6_interleave_0 = const()[name = string("concat_6_interleave_0"), val = bool(false)]; tensor concat_6 = concat(axis = concat_6_axis_0, interleave = concat_6_interleave_0, values = (expand_dims_4, concat_6_values1_0, cache_position_end, concat_6_values3_0))[name = string("concat_6")]; tensor key_states_7_perm_0 = const()[name = string("key_states_7_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_1_stride_0 = const()[name = string("key_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_7_cast_fp16 = transpose(perm = key_states_7_perm_0, x = key_states_5_cast_fp16)[name = string("transpose_83")]; tensor key_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = key_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = key_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_1_squeeze_mask_0, stride = key_cache_internal_tensor_assign_1_stride_0, update = key_states_7_cast_fp16, x = read_state_0)[name = string("key_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_1_cast_fp16, input = key_cache)[name = string("coreml_update_state_0_write_state")]; tensor coreml_update_state_0 = read_state(input = key_cache)[name = string("coreml_update_state_0")]; tensor read_state_1 = read_state(input = value_cache)[name = string("read_state_1")]; tensor value_states_3_perm_0 = const()[name = string("value_states_3_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_1_stride_0 = const()[name = string("value_cache_internal_tensor_assign_1_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_1_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_1_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_1_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_1_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_3_cast_fp16 = transpose(perm = value_states_3_perm_0, x = var_991_cast_fp16)[name = string("transpose_82")]; tensor value_cache_internal_tensor_assign_1_cast_fp16 = slice_update(begin = concat_5, begin_mask = value_cache_internal_tensor_assign_1_begin_mask_0, end = concat_6, end_mask = value_cache_internal_tensor_assign_1_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_1_squeeze_mask_0, stride = value_cache_internal_tensor_assign_1_stride_0, update = value_states_3_cast_fp16, x = read_state_1)[name = string("value_cache_internal_tensor_assign_1_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_1_cast_fp16, input = value_cache)[name = string("coreml_update_state_1_write_state")]; tensor coreml_update_state_1 = read_state(input = value_cache)[name = string("coreml_update_state_1")]; tensor var_1085_begin_0 = const()[name = string("op_1085_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1085_end_0 = const()[name = string("op_1085_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1085_end_mask_0 = const()[name = string("op_1085_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1085_cast_fp16 = slice_by_index(begin = var_1085_begin_0, end = var_1085_end_0, end_mask = var_1085_end_mask_0, x = coreml_update_state_0)[name = string("op_1085_cast_fp16")]; tensor tile_0 = const()[name = string("tile_0"), val = tensor([1, 1])]; int32 var_1088_axis_0 = const()[name = string("op_1088_axis_0"), val = int32(1)]; tensor var_1088_cast_fp16_0, tensor var_1088_cast_fp16_1 = split(axis = var_1088_axis_0, split_sizes = tile_0, x = var_1085_cast_fp16)[name = string("op_1088_cast_fp16")]; tensor var_1095_begin_0 = const()[name = string("op_1095_begin_0"), val = tensor([0, 0, 0, 0])]; tensor var_1095_end_0 = const()[name = string("op_1095_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_1095_end_mask_0 = const()[name = string("op_1095_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1095_cast_fp16 = slice_by_index(begin = var_1095_begin_0, end = var_1095_end_0, end_mask = var_1095_end_mask_0, x = coreml_update_state_1)[name = string("op_1095_cast_fp16")]; tensor tile_1 = const()[name = string("tile_1"), val = tensor([1, 1])]; int32 var_1098_axis_0 = const()[name = string("op_1098_axis_0"), val = int32(1)]; tensor var_1098_cast_fp16_0, tensor var_1098_cast_fp16_1 = split(axis = var_1098_axis_0, split_sizes = tile_1, x = var_1095_cast_fp16)[name = string("op_1098_cast_fp16")]; tensor var_1101_split_sizes_0 = const()[name = string("op_1101_split_sizes_0"), val = tensor([8, 8])]; int32 var_1101_axis_0 = const()[name = string("op_1101_axis_0"), val = int32(1)]; tensor var_1101_0, tensor var_1101_1 = split(axis = var_1101_axis_0, split_sizes = var_1101_split_sizes_0, x = query_states_3_cast_fp16)[name = string("op_1101")]; bool attn_weights_1_transpose_x_0 = const()[name = string("attn_weights_1_transpose_x_0"), val = bool(false)]; bool attn_weights_1_transpose_y_0 = const()[name = string("attn_weights_1_transpose_y_0"), val = bool(false)]; tensor attn_weights_1_cast_fp16 = matmul(transpose_x = attn_weights_1_transpose_x_0, transpose_y = attn_weights_1_transpose_y_0, x = var_1088_cast_fp16_0, y = var_1101_0)[name = string("attn_weights_1_cast_fp16")]; fp16 var_1104_to_fp16 = const()[name = string("op_1104_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_3_cast_fp16 = mul(x = attn_weights_1_cast_fp16, y = var_1104_to_fp16)[name = string("attn_weights_3_cast_fp16")]; tensor attn_weights_5_cast_fp16 = add(x = attn_weights_3_cast_fp16, y = attn_mask_1)[name = string("attn_weights_5_cast_fp16")]; int32 var_1108 = const()[name = string("op_1108"), val = int32(-2)]; tensor attn_weights_7_cast_fp16 = softmax(axis = var_1108, x = attn_weights_5_cast_fp16)[name = string("attn_weights_7_cast_fp16")]; bool var_1114_transpose_x_1 = const()[name = string("op_1114_transpose_x_1"), val = bool(true)]; bool var_1114_transpose_y_1 = const()[name = string("op_1114_transpose_y_1"), val = bool(false)]; tensor var_1114_cast_fp16 = matmul(transpose_x = var_1114_transpose_x_1, transpose_y = var_1114_transpose_y_1, x = attn_weights_7_cast_fp16, y = var_1098_cast_fp16_0)[name = string("op_1114_cast_fp16")]; bool attn_weights_9_transpose_x_0 = const()[name = string("attn_weights_9_transpose_x_0"), val = bool(false)]; bool attn_weights_9_transpose_y_0 = const()[name = string("attn_weights_9_transpose_y_0"), val = bool(false)]; tensor attn_weights_9_cast_fp16 = matmul(transpose_x = attn_weights_9_transpose_x_0, transpose_y = attn_weights_9_transpose_y_0, x = var_1088_cast_fp16_1, y = var_1101_1)[name = string("attn_weights_9_cast_fp16")]; fp16 var_1116_to_fp16 = const()[name = string("op_1116_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_11_cast_fp16 = mul(x = attn_weights_9_cast_fp16, y = var_1116_to_fp16)[name = string("attn_weights_11_cast_fp16")]; tensor attn_weights_13_cast_fp16 = add(x = attn_weights_11_cast_fp16, y = attn_mask_1)[name = string("attn_weights_13_cast_fp16")]; int32 var_1120 = const()[name = string("op_1120"), val = int32(-2)]; tensor attn_weights_15_cast_fp16 = softmax(axis = var_1120, x = attn_weights_13_cast_fp16)[name = string("attn_weights_15_cast_fp16")]; bool attn_output_1_transpose_x_1 = const()[name = string("attn_output_1_transpose_x_1"), val = bool(true)]; bool attn_output_1_transpose_y_1 = const()[name = string("attn_output_1_transpose_y_1"), val = bool(false)]; tensor attn_output_1_cast_fp16 = matmul(transpose_x = attn_output_1_transpose_x_1, transpose_y = attn_output_1_transpose_y_1, x = attn_weights_15_cast_fp16, y = var_1098_cast_fp16_1)[name = string("attn_output_1_cast_fp16")]; int32 var_1128 = const()[name = string("op_1128"), val = int32(1)]; bool attn_output_3_interleave_0 = const()[name = string("attn_output_3_interleave_0"), val = bool(false)]; tensor attn_output_3_cast_fp16 = concat(axis = var_1128, interleave = attn_output_3_interleave_0, values = (var_1114_cast_fp16, attn_output_1_cast_fp16))[name = string("attn_output_3_cast_fp16")]; tensor var_1132_perm_0 = const()[name = string("op_1132_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_11x = const()[name = string("concat_11x"), val = tensor([1, 2048, 1, -1])]; tensor var_1132_cast_fp16 = transpose(perm = var_1132_perm_0, x = attn_output_3_cast_fp16)[name = string("transpose_81")]; tensor attn_output_7_cast_fp16 = reshape(shape = concat_11x, x = var_1132_cast_fp16)[name = string("attn_output_7_cast_fp16")]; tensor layers_0_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_0_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(993654976)))]; tensor hidden_states_3_strides_0 = const()[name = string("hidden_states_3_strides_0"), val = tensor([1, 1])]; string hidden_states_3_pad_type_0 = const()[name = string("hidden_states_3_pad_type_0"), val = string("valid")]; tensor hidden_states_3_pad_0 = const()[name = string("hidden_states_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_3_dilations_0 = const()[name = string("hidden_states_3_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_3_groups_0 = const()[name = string("hidden_states_3_groups_0"), val = int32(1)]; tensor hidden_states_3_cast_fp16 = conv(dilations = hidden_states_3_dilations_0, groups = hidden_states_3_groups_0, pad = hidden_states_3_pad_0, pad_type = hidden_states_3_pad_type_0, strides = hidden_states_3_strides_0, weight = layers_0_self_attn_o_proj_weight_to_fp16, x = attn_output_7_cast_fp16)[name = string("hidden_states_3_cast_fp16")]; tensor hidden_states_5_cast_fp16 = add(x = inputs_embeds_to_fp16, y = hidden_states_3_cast_fp16)[name = string("hidden_states_5_cast_fp16")]; fp16 const_8_promoted_to_fp16 = const()[name = string("const_8_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1165_cast_fp16 = mul(x = hidden_states_5_cast_fp16, y = const_8_promoted_to_fp16)[name = string("op_1165_cast_fp16")]; int32 var_1163 = const()[name = string("op_1163"), val = int32(1)]; bool doubled_5_interleave_0 = const()[name = string("doubled_5_interleave_0"), val = bool(false)]; tensor doubled_5_cast_fp16 = concat(axis = var_1163, interleave = doubled_5_interleave_0, values = (hidden_states_5_cast_fp16, var_1165_cast_fp16))[name = string("doubled_5_cast_fp16")]; tensor out_3_axes_0 = const()[name = string("out_3_axes_0"), val = tensor([1])]; tensor out_3_gamma_0_to_fp16 = const()[name = string("out_3_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002043648)))]; fp16 var_1175_to_fp16 = const()[name = string("op_1175_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_3_cast_fp16 = layer_norm(axes = out_3_axes_0, epsilon = var_1175_to_fp16, gamma = out_3_gamma_0_to_fp16, x = doubled_5_cast_fp16)[name = string("out_3_cast_fp16")]; tensor var_1186_split_sizes_0 = const()[name = string("op_1186_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1186_axis_0 = const()[name = string("op_1186_axis_0"), val = int32(1)]; tensor var_1186_cast_fp16_0, tensor var_1186_cast_fp16_1 = split(axis = var_1186_axis_0, split_sizes = var_1186_split_sizes_0, x = out_3_cast_fp16)[name = string("op_1186_cast_fp16")]; tensor layers_0_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1002051904)))]; tensor input_1_strides_0 = const()[name = string("input_1_strides_0"), val = tensor([1, 1])]; string input_1_pad_type_0 = const()[name = string("input_1_pad_type_0"), val = string("valid")]; tensor input_1_pad_0 = const()[name = string("input_1_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_1_dilations_0 = const()[name = string("input_1_dilations_0"), val = tensor([1, 1])]; int32 input_1_groups_0 = const()[name = string("input_1_groups_0"), val = int32(1)]; tensor input_1_cast_fp16 = conv(dilations = input_1_dilations_0, groups = input_1_groups_0, pad = input_1_pad_0, pad_type = input_1_pad_type_0, strides = input_1_strides_0, weight = layers_0_mlp_gate_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("input_1_cast_fp16")]; tensor var_1203_cast_fp16 = silu(x = input_1_cast_fp16)[name = string("op_1203_cast_fp16")]; tensor layers_0_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1027217792)))]; tensor var_1209_strides_0 = const()[name = string("op_1209_strides_0"), val = tensor([1, 1])]; string var_1209_pad_type_0 = const()[name = string("op_1209_pad_type_0"), val = string("valid")]; tensor var_1209_pad_0 = const()[name = string("op_1209_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1209_dilations_0 = const()[name = string("op_1209_dilations_0"), val = tensor([1, 1])]; int32 var_1209_groups_0 = const()[name = string("op_1209_groups_0"), val = int32(1)]; tensor var_1209_cast_fp16 = conv(dilations = var_1209_dilations_0, groups = var_1209_groups_0, pad = var_1209_pad_0, pad_type = var_1209_pad_type_0, strides = var_1209_strides_0, weight = layers_0_mlp_up_proj_weight_to_fp16, x = var_1186_cast_fp16_0)[name = string("op_1209_cast_fp16")]; tensor x_9_cast_fp16 = mul(x = var_1203_cast_fp16, y = var_1209_cast_fp16)[name = string("x_9_cast_fp16")]; tensor layers_0_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_0_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1052383680)))]; tensor hidden_states_7_strides_0 = const()[name = string("hidden_states_7_strides_0"), val = tensor([1, 1])]; string hidden_states_7_pad_type_0 = const()[name = string("hidden_states_7_pad_type_0"), val = string("valid")]; tensor hidden_states_7_pad_0 = const()[name = string("hidden_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_7_dilations_0 = const()[name = string("hidden_states_7_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_7_groups_0 = const()[name = string("hidden_states_7_groups_0"), val = int32(1)]; tensor hidden_states_7_cast_fp16 = conv(dilations = hidden_states_7_dilations_0, groups = hidden_states_7_groups_0, pad = hidden_states_7_pad_0, pad_type = hidden_states_7_pad_type_0, strides = hidden_states_7_strides_0, weight = layers_0_mlp_down_proj_weight_to_fp16, x = x_9_cast_fp16)[name = string("hidden_states_7_cast_fp16")]; tensor hidden_states_9_cast_fp16 = add(x = hidden_states_5_cast_fp16, y = hidden_states_7_cast_fp16)[name = string("hidden_states_9_cast_fp16")]; fp16 const_10_promoted_to_fp16 = const()[name = string("const_10_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1227_cast_fp16 = mul(x = hidden_states_9_cast_fp16, y = const_10_promoted_to_fp16)[name = string("op_1227_cast_fp16")]; int32 var_1225 = const()[name = string("op_1225"), val = int32(1)]; bool doubled_9_interleave_0 = const()[name = string("doubled_9_interleave_0"), val = bool(false)]; tensor doubled_9_cast_fp16 = concat(axis = var_1225, interleave = doubled_9_interleave_0, values = (hidden_states_9_cast_fp16, var_1227_cast_fp16))[name = string("doubled_9_cast_fp16")]; tensor out_5_axes_0 = const()[name = string("out_5_axes_0"), val = tensor([1])]; tensor out_5_gamma_0_to_fp16 = const()[name = string("out_5_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077549568)))]; fp16 var_1237_to_fp16 = const()[name = string("op_1237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_5_cast_fp16 = layer_norm(axes = out_5_axes_0, epsilon = var_1237_to_fp16, gamma = out_5_gamma_0_to_fp16, x = doubled_9_cast_fp16)[name = string("out_5_cast_fp16")]; tensor var_1248_split_sizes_0 = const()[name = string("op_1248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1248_axis_0 = const()[name = string("op_1248_axis_0"), val = int32(1)]; tensor var_1248_cast_fp16_0, tensor var_1248_cast_fp16_1 = split(axis = var_1248_axis_0, split_sizes = var_1248_split_sizes_0, x = out_5_cast_fp16)[name = string("op_1248_cast_fp16")]; tensor layers_1_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1077557824)))]; tensor query_states_7_strides_0 = const()[name = string("query_states_7_strides_0"), val = tensor([1, 1])]; string query_states_7_pad_type_0 = const()[name = string("query_states_7_pad_type_0"), val = string("valid")]; tensor query_states_7_pad_0 = const()[name = string("query_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_7_dilations_0 = const()[name = string("query_states_7_dilations_0"), val = tensor([1, 1])]; int32 query_states_7_groups_0 = const()[name = string("query_states_7_groups_0"), val = int32(1)]; tensor query_states_7_cast_fp16 = conv(dilations = query_states_7_dilations_0, groups = query_states_7_groups_0, pad = query_states_7_pad_0, pad_type = query_states_7_pad_type_0, strides = query_states_7_strides_0, weight = layers_1_self_attn_q_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("query_states_7_cast_fp16")]; tensor layers_1_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1085946496)))]; tensor key_states_11_strides_0 = const()[name = string("key_states_11_strides_0"), val = tensor([1, 1])]; string key_states_11_pad_type_0 = const()[name = string("key_states_11_pad_type_0"), val = string("valid")]; tensor key_states_11_pad_0 = const()[name = string("key_states_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_11_dilations_0 = const()[name = string("key_states_11_dilations_0"), val = tensor([1, 1])]; int32 key_states_11_groups_0 = const()[name = string("key_states_11_groups_0"), val = int32(1)]; tensor key_states_11_cast_fp16 = conv(dilations = key_states_11_dilations_0, groups = key_states_11_groups_0, pad = key_states_11_pad_0, pad_type = key_states_11_pad_type_0, strides = key_states_11_strides_0, weight = layers_1_self_attn_k_proj_weight_to_fp16, x = var_1248_cast_fp16_0)[name = string("key_states_11_cast_fp16")]; tensor value_states_7_strides_0 = const()[name = string("value_states_7_strides_0"), val = tensor([1, 1])]; string value_states_7_pad_type_0 = const()[name = string("value_states_7_pad_type_0"), val = string("valid")]; tensor value_states_7_pad_0 = const()[name = string("value_states_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_7_dilations_0 = const()[name = string("value_states_7_dilations_0"), val = tensor([1, 1])]; int32 value_states_7_groups_0 = const()[name = string("value_states_7_groups_0"), val = int32(1)]; tensor value_states_7_cast_fp16 = conv(dilations = value_states_7_dilations_0, groups = value_states_7_groups_0, pad = value_states_7_pad_0, pad_type = value_states_7_pad_type_0, strides = value_states_7_strides_0, weight = layers_1_self_attn_v_proj_weight_cast_fp16, x = var_1248_cast_fp16_0)[name = string("value_states_7_cast_fp16")]; tensor concat_12x = const()[name = string("concat_12x"), val = tensor([1, 16, 128, -1])]; tensor x_11_cast_fp16 = reshape(shape = concat_12x, x = query_states_7_cast_fp16)[name = string("x_11_cast_fp16")]; tensor concat_13x = const()[name = string("concat_13x"), val = tensor([1, 2, 128, -1])]; tensor var_1305_cast_fp16 = reshape(shape = concat_13x, x = key_states_11_cast_fp16)[name = string("op_1305_cast_fp16")]; tensor concat_14x = const()[name = string("concat_14x"), val = tensor([1, 2, 128, -1])]; tensor var_1312_cast_fp16 = reshape(shape = concat_14x, x = value_states_7_cast_fp16)[name = string("op_1312_cast_fp16")]; tensor var_1316_cast_fp16 = mul(x = x_11_cast_fp16, y = var_869_cast_fp16)[name = string("op_1316_cast_fp16")]; tensor var_1317_split_sizes_0 = const()[name = string("op_1317_split_sizes_0"), val = tensor([64, 64])]; int32 var_1317_axis_0 = const()[name = string("op_1317_axis_0"), val = int32(-2)]; tensor var_1317_cast_fp16_0, tensor var_1317_cast_fp16_1 = split(axis = var_1317_axis_0, split_sizes = var_1317_split_sizes_0, x = x_11_cast_fp16)[name = string("op_1317_cast_fp16")]; fp16 const_12_promoted_to_fp16 = const()[name = string("const_12_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1319_cast_fp16 = mul(x = var_1317_cast_fp16_1, y = const_12_promoted_to_fp16)[name = string("op_1319_cast_fp16")]; int32 var_1321 = const()[name = string("op_1321"), val = int32(-2)]; bool var_1322_interleave_0 = const()[name = string("op_1322_interleave_0"), val = bool(false)]; tensor var_1322_cast_fp16 = concat(axis = var_1321, interleave = var_1322_interleave_0, values = (var_1319_cast_fp16, var_1317_cast_fp16_0))[name = string("op_1322_cast_fp16")]; tensor var_1323_cast_fp16 = mul(x = var_1322_cast_fp16, y = var_878_cast_fp16)[name = string("op_1323_cast_fp16")]; tensor query_states_9_cast_fp16 = add(x = var_1316_cast_fp16, y = var_1323_cast_fp16)[name = string("query_states_9_cast_fp16")]; tensor var_1329_cast_fp16 = mul(x = var_1305_cast_fp16, y = var_869_cast_fp16)[name = string("op_1329_cast_fp16")]; tensor var_1330_split_sizes_0 = const()[name = string("op_1330_split_sizes_0"), val = tensor([64, 64])]; int32 var_1330_axis_0 = const()[name = string("op_1330_axis_0"), val = int32(-2)]; tensor var_1330_cast_fp16_0, tensor var_1330_cast_fp16_1 = split(axis = var_1330_axis_0, split_sizes = var_1330_split_sizes_0, x = var_1305_cast_fp16)[name = string("op_1330_cast_fp16")]; fp16 const_13_promoted_to_fp16 = const()[name = string("const_13_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1332_cast_fp16 = mul(x = var_1330_cast_fp16_1, y = const_13_promoted_to_fp16)[name = string("op_1332_cast_fp16")]; int32 var_1334 = const()[name = string("op_1334"), val = int32(-2)]; bool var_1335_interleave_0 = const()[name = string("op_1335_interleave_0"), val = bool(false)]; tensor var_1335_cast_fp16 = concat(axis = var_1334, interleave = var_1335_interleave_0, values = (var_1332_cast_fp16, var_1330_cast_fp16_0))[name = string("op_1335_cast_fp16")]; tensor var_1336_cast_fp16 = mul(x = var_1335_cast_fp16, y = var_878_cast_fp16)[name = string("op_1336_cast_fp16")]; tensor key_states_15_cast_fp16 = add(x = var_1329_cast_fp16, y = var_1336_cast_fp16)[name = string("key_states_15_cast_fp16")]; tensor expand_dims_12 = const()[name = string("expand_dims_12"), val = tensor([1])]; tensor expand_dims_13 = const()[name = string("expand_dims_13"), val = tensor([0])]; tensor expand_dims_15 = const()[name = string("expand_dims_15"), val = tensor([0])]; int32 concat_17_axis_0 = const()[name = string("concat_17_axis_0"), val = int32(0)]; bool concat_17_interleave_0 = const()[name = string("concat_17_interleave_0"), val = bool(false)]; tensor concat_17 = concat(axis = concat_17_axis_0, interleave = concat_17_interleave_0, values = (expand_dims_12, expand_dims_13, position_id, expand_dims_15))[name = string("concat_17")]; tensor expand_dims_16 = const()[name = string("expand_dims_16"), val = tensor([2])]; tensor concat_18_values1_0 = const()[name = string("concat_18_values1_0"), val = tensor([0])]; tensor concat_18_values3_0 = const()[name = string("concat_18_values3_0"), val = tensor([0])]; int32 concat_18_axis_0 = const()[name = string("concat_18_axis_0"), val = int32(0)]; bool concat_18_interleave_0 = const()[name = string("concat_18_interleave_0"), val = bool(false)]; tensor concat_18 = concat(axis = concat_18_axis_0, interleave = concat_18_interleave_0, values = (expand_dims_16, concat_18_values1_0, cache_position_end, concat_18_values3_0))[name = string("concat_18")]; tensor key_states_17_perm_0 = const()[name = string("key_states_17_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_2_stride_0 = const()[name = string("key_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_17_cast_fp16 = transpose(perm = key_states_17_perm_0, x = key_states_15_cast_fp16)[name = string("transpose_80")]; tensor key_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = key_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = key_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_2_squeeze_mask_0, stride = key_cache_internal_tensor_assign_2_stride_0, update = key_states_17_cast_fp16, x = coreml_update_state_0)[name = string("key_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_2_cast_fp16, input = key_cache)[name = string("coreml_update_state_2_write_state")]; tensor coreml_update_state_2 = read_state(input = key_cache)[name = string("coreml_update_state_2")]; tensor value_states_9_perm_0 = const()[name = string("value_states_9_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_2_stride_0 = const()[name = string("value_cache_internal_tensor_assign_2_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_2_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_2_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_2_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_2_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_9_cast_fp16 = transpose(perm = value_states_9_perm_0, x = var_1312_cast_fp16)[name = string("transpose_79")]; tensor value_cache_internal_tensor_assign_2_cast_fp16 = slice_update(begin = concat_17, begin_mask = value_cache_internal_tensor_assign_2_begin_mask_0, end = concat_18, end_mask = value_cache_internal_tensor_assign_2_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_2_squeeze_mask_0, stride = value_cache_internal_tensor_assign_2_stride_0, update = value_states_9_cast_fp16, x = coreml_update_state_1)[name = string("value_cache_internal_tensor_assign_2_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_2_cast_fp16, input = value_cache)[name = string("coreml_update_state_3_write_state")]; tensor coreml_update_state_3 = read_state(input = value_cache)[name = string("coreml_update_state_3")]; tensor var_1406_begin_0 = const()[name = string("op_1406_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1406_end_0 = const()[name = string("op_1406_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1406_end_mask_0 = const()[name = string("op_1406_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1406_cast_fp16 = slice_by_index(begin = var_1406_begin_0, end = var_1406_end_0, end_mask = var_1406_end_mask_0, x = coreml_update_state_2)[name = string("op_1406_cast_fp16")]; tensor tile_2 = const()[name = string("tile_2"), val = tensor([1, 1])]; int32 var_1409_axis_0 = const()[name = string("op_1409_axis_0"), val = int32(1)]; tensor var_1409_cast_fp16_0, tensor var_1409_cast_fp16_1 = split(axis = var_1409_axis_0, split_sizes = tile_2, x = var_1406_cast_fp16)[name = string("op_1409_cast_fp16")]; tensor var_1416_begin_0 = const()[name = string("op_1416_begin_0"), val = tensor([1, 0, 0, 0])]; tensor var_1416_end_0 = const()[name = string("op_1416_end_0"), val = tensor([2, 2, 2048, 128])]; tensor var_1416_end_mask_0 = const()[name = string("op_1416_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1416_cast_fp16 = slice_by_index(begin = var_1416_begin_0, end = var_1416_end_0, end_mask = var_1416_end_mask_0, x = coreml_update_state_3)[name = string("op_1416_cast_fp16")]; tensor tile_3 = const()[name = string("tile_3"), val = tensor([1, 1])]; int32 var_1419_axis_0 = const()[name = string("op_1419_axis_0"), val = int32(1)]; tensor var_1419_cast_fp16_0, tensor var_1419_cast_fp16_1 = split(axis = var_1419_axis_0, split_sizes = tile_3, x = var_1416_cast_fp16)[name = string("op_1419_cast_fp16")]; tensor var_1422_split_sizes_0 = const()[name = string("op_1422_split_sizes_0"), val = tensor([8, 8])]; int32 var_1422_axis_0 = const()[name = string("op_1422_axis_0"), val = int32(1)]; tensor var_1422_0, tensor var_1422_1 = split(axis = var_1422_axis_0, split_sizes = var_1422_split_sizes_0, x = query_states_9_cast_fp16)[name = string("op_1422")]; bool attn_weights_17_transpose_x_0 = const()[name = string("attn_weights_17_transpose_x_0"), val = bool(false)]; bool attn_weights_17_transpose_y_0 = const()[name = string("attn_weights_17_transpose_y_0"), val = bool(false)]; tensor attn_weights_17_cast_fp16 = matmul(transpose_x = attn_weights_17_transpose_x_0, transpose_y = attn_weights_17_transpose_y_0, x = var_1409_cast_fp16_0, y = var_1422_0)[name = string("attn_weights_17_cast_fp16")]; fp16 var_1425_to_fp16 = const()[name = string("op_1425_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_19_cast_fp16 = mul(x = attn_weights_17_cast_fp16, y = var_1425_to_fp16)[name = string("attn_weights_19_cast_fp16")]; tensor attn_weights_21_cast_fp16 = add(x = attn_weights_19_cast_fp16, y = attn_mask_1)[name = string("attn_weights_21_cast_fp16")]; int32 var_1429 = const()[name = string("op_1429"), val = int32(-2)]; tensor attn_weights_23_cast_fp16 = softmax(axis = var_1429, x = attn_weights_21_cast_fp16)[name = string("attn_weights_23_cast_fp16")]; bool var_1435_transpose_x_1 = const()[name = string("op_1435_transpose_x_1"), val = bool(true)]; bool var_1435_transpose_y_1 = const()[name = string("op_1435_transpose_y_1"), val = bool(false)]; tensor var_1435_cast_fp16 = matmul(transpose_x = var_1435_transpose_x_1, transpose_y = var_1435_transpose_y_1, x = attn_weights_23_cast_fp16, y = var_1419_cast_fp16_0)[name = string("op_1435_cast_fp16")]; bool attn_weights_25_transpose_x_0 = const()[name = string("attn_weights_25_transpose_x_0"), val = bool(false)]; bool attn_weights_25_transpose_y_0 = const()[name = string("attn_weights_25_transpose_y_0"), val = bool(false)]; tensor attn_weights_25_cast_fp16 = matmul(transpose_x = attn_weights_25_transpose_x_0, transpose_y = attn_weights_25_transpose_y_0, x = var_1409_cast_fp16_1, y = var_1422_1)[name = string("attn_weights_25_cast_fp16")]; fp16 var_1437_to_fp16 = const()[name = string("op_1437_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_27_cast_fp16 = mul(x = attn_weights_25_cast_fp16, y = var_1437_to_fp16)[name = string("attn_weights_27_cast_fp16")]; tensor attn_weights_29_cast_fp16 = add(x = attn_weights_27_cast_fp16, y = attn_mask_1)[name = string("attn_weights_29_cast_fp16")]; int32 var_1441 = const()[name = string("op_1441"), val = int32(-2)]; tensor attn_weights_31_cast_fp16 = softmax(axis = var_1441, x = attn_weights_29_cast_fp16)[name = string("attn_weights_31_cast_fp16")]; bool attn_output_9_transpose_x_1 = const()[name = string("attn_output_9_transpose_x_1"), val = bool(true)]; bool attn_output_9_transpose_y_1 = const()[name = string("attn_output_9_transpose_y_1"), val = bool(false)]; tensor attn_output_9_cast_fp16 = matmul(transpose_x = attn_output_9_transpose_x_1, transpose_y = attn_output_9_transpose_y_1, x = attn_weights_31_cast_fp16, y = var_1419_cast_fp16_1)[name = string("attn_output_9_cast_fp16")]; int32 var_1449 = const()[name = string("op_1449"), val = int32(1)]; bool attn_output_11_interleave_0 = const()[name = string("attn_output_11_interleave_0"), val = bool(false)]; tensor attn_output_11_cast_fp16 = concat(axis = var_1449, interleave = attn_output_11_interleave_0, values = (var_1435_cast_fp16, attn_output_9_cast_fp16))[name = string("attn_output_11_cast_fp16")]; tensor var_1453_perm_0 = const()[name = string("op_1453_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_23x = const()[name = string("concat_23x"), val = tensor([1, 2048, 1, -1])]; tensor var_1453_cast_fp16 = transpose(perm = var_1453_perm_0, x = attn_output_11_cast_fp16)[name = string("transpose_78")]; tensor attn_output_15_cast_fp16 = reshape(shape = concat_23x, x = var_1453_cast_fp16)[name = string("attn_output_15_cast_fp16")]; tensor layers_1_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_1_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1086995136)))]; tensor hidden_states_13_strides_0 = const()[name = string("hidden_states_13_strides_0"), val = tensor([1, 1])]; string hidden_states_13_pad_type_0 = const()[name = string("hidden_states_13_pad_type_0"), val = string("valid")]; tensor hidden_states_13_pad_0 = const()[name = string("hidden_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_13_dilations_0 = const()[name = string("hidden_states_13_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_13_groups_0 = const()[name = string("hidden_states_13_groups_0"), val = int32(1)]; tensor hidden_states_13_cast_fp16 = conv(dilations = hidden_states_13_dilations_0, groups = hidden_states_13_groups_0, pad = hidden_states_13_pad_0, pad_type = hidden_states_13_pad_type_0, strides = hidden_states_13_strides_0, weight = layers_1_self_attn_o_proj_weight_to_fp16, x = attn_output_15_cast_fp16)[name = string("hidden_states_13_cast_fp16")]; tensor hidden_states_15_cast_fp16 = add(x = hidden_states_9_cast_fp16, y = hidden_states_13_cast_fp16)[name = string("hidden_states_15_cast_fp16")]; fp16 const_18_promoted_to_fp16 = const()[name = string("const_18_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1486_cast_fp16 = mul(x = hidden_states_15_cast_fp16, y = const_18_promoted_to_fp16)[name = string("op_1486_cast_fp16")]; int32 var_1484 = const()[name = string("op_1484"), val = int32(1)]; bool doubled_13_interleave_0 = const()[name = string("doubled_13_interleave_0"), val = bool(false)]; tensor doubled_13_cast_fp16 = concat(axis = var_1484, interleave = doubled_13_interleave_0, values = (hidden_states_15_cast_fp16, var_1486_cast_fp16))[name = string("doubled_13_cast_fp16")]; tensor out_7_axes_0 = const()[name = string("out_7_axes_0"), val = tensor([1])]; tensor out_7_gamma_0_to_fp16 = const()[name = string("out_7_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095383808)))]; fp16 var_1496_to_fp16 = const()[name = string("op_1496_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_7_cast_fp16 = layer_norm(axes = out_7_axes_0, epsilon = var_1496_to_fp16, gamma = out_7_gamma_0_to_fp16, x = doubled_13_cast_fp16)[name = string("out_7_cast_fp16")]; tensor var_1507_split_sizes_0 = const()[name = string("op_1507_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1507_axis_0 = const()[name = string("op_1507_axis_0"), val = int32(1)]; tensor var_1507_cast_fp16_0, tensor var_1507_cast_fp16_1 = split(axis = var_1507_axis_0, split_sizes = var_1507_split_sizes_0, x = out_7_cast_fp16)[name = string("op_1507_cast_fp16")]; tensor layers_1_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1095392064)))]; tensor input_3_strides_0 = const()[name = string("input_3_strides_0"), val = tensor([1, 1])]; string input_3_pad_type_0 = const()[name = string("input_3_pad_type_0"), val = string("valid")]; tensor input_3_pad_0 = const()[name = string("input_3_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_3_dilations_0 = const()[name = string("input_3_dilations_0"), val = tensor([1, 1])]; int32 input_3_groups_0 = const()[name = string("input_3_groups_0"), val = int32(1)]; tensor input_3_cast_fp16 = conv(dilations = input_3_dilations_0, groups = input_3_groups_0, pad = input_3_pad_0, pad_type = input_3_pad_type_0, strides = input_3_strides_0, weight = layers_1_mlp_gate_proj_weight_to_fp16, x = var_1507_cast_fp16_0)[name = string("input_3_cast_fp16")]; tensor var_1524_cast_fp16 = silu(x = input_3_cast_fp16)[name = string("op_1524_cast_fp16")]; tensor var_1530_strides_0 = const()[name = string("op_1530_strides_0"), val = tensor([1, 1])]; string var_1530_pad_type_0 = const()[name = string("op_1530_pad_type_0"), val = string("valid")]; tensor var_1530_pad_0 = const()[name = string("op_1530_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1530_dilations_0 = const()[name = string("op_1530_dilations_0"), val = tensor([1, 1])]; int32 var_1530_groups_0 = const()[name = string("op_1530_groups_0"), val = int32(1)]; tensor var_1530_cast_fp16 = conv(dilations = var_1530_dilations_0, groups = var_1530_groups_0, pad = var_1530_pad_0, pad_type = var_1530_pad_type_0, strides = var_1530_strides_0, weight = layers_1_mlp_up_proj_weight_cast_fp16, x = var_1507_cast_fp16_0)[name = string("op_1530_cast_fp16")]; tensor x_19_cast_fp16 = mul(x = var_1524_cast_fp16, y = var_1530_cast_fp16)[name = string("x_19_cast_fp16")]; tensor layers_1_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_1_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1120557952)))]; tensor hidden_states_17_strides_0 = const()[name = string("hidden_states_17_strides_0"), val = tensor([1, 1])]; string hidden_states_17_pad_type_0 = const()[name = string("hidden_states_17_pad_type_0"), val = string("valid")]; tensor hidden_states_17_pad_0 = const()[name = string("hidden_states_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_17_dilations_0 = const()[name = string("hidden_states_17_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_17_groups_0 = const()[name = string("hidden_states_17_groups_0"), val = int32(1)]; tensor hidden_states_17_cast_fp16 = conv(dilations = hidden_states_17_dilations_0, groups = hidden_states_17_groups_0, pad = hidden_states_17_pad_0, pad_type = hidden_states_17_pad_type_0, strides = hidden_states_17_strides_0, weight = layers_1_mlp_down_proj_weight_to_fp16, x = x_19_cast_fp16)[name = string("hidden_states_17_cast_fp16")]; tensor hidden_states_19_cast_fp16 = add(x = hidden_states_15_cast_fp16, y = hidden_states_17_cast_fp16)[name = string("hidden_states_19_cast_fp16")]; fp16 const_20_promoted_to_fp16 = const()[name = string("const_20_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1548_cast_fp16 = mul(x = hidden_states_19_cast_fp16, y = const_20_promoted_to_fp16)[name = string("op_1548_cast_fp16")]; int32 var_1546 = const()[name = string("op_1546"), val = int32(1)]; bool doubled_17_interleave_0 = const()[name = string("doubled_17_interleave_0"), val = bool(false)]; tensor doubled_17_cast_fp16 = concat(axis = var_1546, interleave = doubled_17_interleave_0, values = (hidden_states_19_cast_fp16, var_1548_cast_fp16))[name = string("doubled_17_cast_fp16")]; tensor out_9_axes_0 = const()[name = string("out_9_axes_0"), val = tensor([1])]; tensor out_9_gamma_0_to_fp16 = const()[name = string("out_9_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145723840)))]; fp16 var_1558_to_fp16 = const()[name = string("op_1558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_9_cast_fp16 = layer_norm(axes = out_9_axes_0, epsilon = var_1558_to_fp16, gamma = out_9_gamma_0_to_fp16, x = doubled_17_cast_fp16)[name = string("out_9_cast_fp16")]; tensor var_1569_split_sizes_0 = const()[name = string("op_1569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1569_axis_0 = const()[name = string("op_1569_axis_0"), val = int32(1)]; tensor var_1569_cast_fp16_0, tensor var_1569_cast_fp16_1 = split(axis = var_1569_axis_0, split_sizes = var_1569_split_sizes_0, x = out_9_cast_fp16)[name = string("op_1569_cast_fp16")]; tensor layers_2_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1145732096)))]; tensor query_states_13_strides_0 = const()[name = string("query_states_13_strides_0"), val = tensor([1, 1])]; string query_states_13_pad_type_0 = const()[name = string("query_states_13_pad_type_0"), val = string("valid")]; tensor query_states_13_pad_0 = const()[name = string("query_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_13_dilations_0 = const()[name = string("query_states_13_dilations_0"), val = tensor([1, 1])]; int32 query_states_13_groups_0 = const()[name = string("query_states_13_groups_0"), val = int32(1)]; tensor query_states_13_cast_fp16 = conv(dilations = query_states_13_dilations_0, groups = query_states_13_groups_0, pad = query_states_13_pad_0, pad_type = query_states_13_pad_type_0, strides = query_states_13_strides_0, weight = layers_2_self_attn_q_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("query_states_13_cast_fp16")]; tensor layers_2_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1154120768)))]; tensor key_states_21_strides_0 = const()[name = string("key_states_21_strides_0"), val = tensor([1, 1])]; string key_states_21_pad_type_0 = const()[name = string("key_states_21_pad_type_0"), val = string("valid")]; tensor key_states_21_pad_0 = const()[name = string("key_states_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_21_dilations_0 = const()[name = string("key_states_21_dilations_0"), val = tensor([1, 1])]; int32 key_states_21_groups_0 = const()[name = string("key_states_21_groups_0"), val = int32(1)]; tensor key_states_21_cast_fp16 = conv(dilations = key_states_21_dilations_0, groups = key_states_21_groups_0, pad = key_states_21_pad_0, pad_type = key_states_21_pad_type_0, strides = key_states_21_strides_0, weight = layers_2_self_attn_k_proj_weight_to_fp16, x = var_1569_cast_fp16_0)[name = string("key_states_21_cast_fp16")]; tensor value_states_13_strides_0 = const()[name = string("value_states_13_strides_0"), val = tensor([1, 1])]; string value_states_13_pad_type_0 = const()[name = string("value_states_13_pad_type_0"), val = string("valid")]; tensor value_states_13_pad_0 = const()[name = string("value_states_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_13_dilations_0 = const()[name = string("value_states_13_dilations_0"), val = tensor([1, 1])]; int32 value_states_13_groups_0 = const()[name = string("value_states_13_groups_0"), val = int32(1)]; tensor value_states_13_cast_fp16 = conv(dilations = value_states_13_dilations_0, groups = value_states_13_groups_0, pad = value_states_13_pad_0, pad_type = value_states_13_pad_type_0, strides = value_states_13_strides_0, weight = layers_2_self_attn_v_proj_weight_cast_fp16, x = var_1569_cast_fp16_0)[name = string("value_states_13_cast_fp16")]; tensor concat_24x = const()[name = string("concat_24x"), val = tensor([1, 16, 128, -1])]; tensor x_21_cast_fp16 = reshape(shape = concat_24x, x = query_states_13_cast_fp16)[name = string("x_21_cast_fp16")]; tensor concat_25x = const()[name = string("concat_25x"), val = tensor([1, 2, 128, -1])]; tensor var_1626_cast_fp16 = reshape(shape = concat_25x, x = key_states_21_cast_fp16)[name = string("op_1626_cast_fp16")]; tensor concat_26x = const()[name = string("concat_26x"), val = tensor([1, 2, 128, -1])]; tensor var_1633_cast_fp16 = reshape(shape = concat_26x, x = value_states_13_cast_fp16)[name = string("op_1633_cast_fp16")]; tensor var_1637_cast_fp16 = mul(x = x_21_cast_fp16, y = var_869_cast_fp16)[name = string("op_1637_cast_fp16")]; tensor var_1638_split_sizes_0 = const()[name = string("op_1638_split_sizes_0"), val = tensor([64, 64])]; int32 var_1638_axis_0 = const()[name = string("op_1638_axis_0"), val = int32(-2)]; tensor var_1638_cast_fp16_0, tensor var_1638_cast_fp16_1 = split(axis = var_1638_axis_0, split_sizes = var_1638_split_sizes_0, x = x_21_cast_fp16)[name = string("op_1638_cast_fp16")]; fp16 const_22_promoted_to_fp16 = const()[name = string("const_22_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1640_cast_fp16 = mul(x = var_1638_cast_fp16_1, y = const_22_promoted_to_fp16)[name = string("op_1640_cast_fp16")]; int32 var_1642 = const()[name = string("op_1642"), val = int32(-2)]; bool var_1643_interleave_0 = const()[name = string("op_1643_interleave_0"), val = bool(false)]; tensor var_1643_cast_fp16 = concat(axis = var_1642, interleave = var_1643_interleave_0, values = (var_1640_cast_fp16, var_1638_cast_fp16_0))[name = string("op_1643_cast_fp16")]; tensor var_1644_cast_fp16 = mul(x = var_1643_cast_fp16, y = var_878_cast_fp16)[name = string("op_1644_cast_fp16")]; tensor query_states_15_cast_fp16 = add(x = var_1637_cast_fp16, y = var_1644_cast_fp16)[name = string("query_states_15_cast_fp16")]; tensor var_1650_cast_fp16 = mul(x = var_1626_cast_fp16, y = var_869_cast_fp16)[name = string("op_1650_cast_fp16")]; tensor var_1651_split_sizes_0 = const()[name = string("op_1651_split_sizes_0"), val = tensor([64, 64])]; int32 var_1651_axis_0 = const()[name = string("op_1651_axis_0"), val = int32(-2)]; tensor var_1651_cast_fp16_0, tensor var_1651_cast_fp16_1 = split(axis = var_1651_axis_0, split_sizes = var_1651_split_sizes_0, x = var_1626_cast_fp16)[name = string("op_1651_cast_fp16")]; fp16 const_23_promoted_to_fp16 = const()[name = string("const_23_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1653_cast_fp16 = mul(x = var_1651_cast_fp16_1, y = const_23_promoted_to_fp16)[name = string("op_1653_cast_fp16")]; int32 var_1655 = const()[name = string("op_1655"), val = int32(-2)]; bool var_1656_interleave_0 = const()[name = string("op_1656_interleave_0"), val = bool(false)]; tensor var_1656_cast_fp16 = concat(axis = var_1655, interleave = var_1656_interleave_0, values = (var_1653_cast_fp16, var_1651_cast_fp16_0))[name = string("op_1656_cast_fp16")]; tensor var_1657_cast_fp16 = mul(x = var_1656_cast_fp16, y = var_878_cast_fp16)[name = string("op_1657_cast_fp16")]; tensor key_states_25_cast_fp16 = add(x = var_1650_cast_fp16, y = var_1657_cast_fp16)[name = string("key_states_25_cast_fp16")]; tensor expand_dims_24 = const()[name = string("expand_dims_24"), val = tensor([2])]; tensor expand_dims_25 = const()[name = string("expand_dims_25"), val = tensor([0])]; tensor expand_dims_27 = const()[name = string("expand_dims_27"), val = tensor([0])]; int32 concat_29_axis_0 = const()[name = string("concat_29_axis_0"), val = int32(0)]; bool concat_29_interleave_0 = const()[name = string("concat_29_interleave_0"), val = bool(false)]; tensor concat_29 = concat(axis = concat_29_axis_0, interleave = concat_29_interleave_0, values = (expand_dims_24, expand_dims_25, position_id, expand_dims_27))[name = string("concat_29")]; tensor expand_dims_28 = const()[name = string("expand_dims_28"), val = tensor([3])]; tensor concat_30_values1_0 = const()[name = string("concat_30_values1_0"), val = tensor([0])]; tensor concat_30_values3_0 = const()[name = string("concat_30_values3_0"), val = tensor([0])]; int32 concat_30_axis_0 = const()[name = string("concat_30_axis_0"), val = int32(0)]; bool concat_30_interleave_0 = const()[name = string("concat_30_interleave_0"), val = bool(false)]; tensor concat_30 = concat(axis = concat_30_axis_0, interleave = concat_30_interleave_0, values = (expand_dims_28, concat_30_values1_0, cache_position_end, concat_30_values3_0))[name = string("concat_30")]; tensor key_states_27_perm_0 = const()[name = string("key_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_3_stride_0 = const()[name = string("key_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_27_cast_fp16 = transpose(perm = key_states_27_perm_0, x = key_states_25_cast_fp16)[name = string("transpose_77")]; tensor key_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = key_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = key_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_3_squeeze_mask_0, stride = key_cache_internal_tensor_assign_3_stride_0, update = key_states_27_cast_fp16, x = coreml_update_state_2)[name = string("key_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_3_cast_fp16, input = key_cache)[name = string("coreml_update_state_4_write_state")]; tensor coreml_update_state_4 = read_state(input = key_cache)[name = string("coreml_update_state_4")]; tensor value_states_15_perm_0 = const()[name = string("value_states_15_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_3_stride_0 = const()[name = string("value_cache_internal_tensor_assign_3_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_3_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_3_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_3_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_3_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_15_cast_fp16 = transpose(perm = value_states_15_perm_0, x = var_1633_cast_fp16)[name = string("transpose_76")]; tensor value_cache_internal_tensor_assign_3_cast_fp16 = slice_update(begin = concat_29, begin_mask = value_cache_internal_tensor_assign_3_begin_mask_0, end = concat_30, end_mask = value_cache_internal_tensor_assign_3_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_3_squeeze_mask_0, stride = value_cache_internal_tensor_assign_3_stride_0, update = value_states_15_cast_fp16, x = coreml_update_state_3)[name = string("value_cache_internal_tensor_assign_3_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_3_cast_fp16, input = value_cache)[name = string("coreml_update_state_5_write_state")]; tensor coreml_update_state_5 = read_state(input = value_cache)[name = string("coreml_update_state_5")]; tensor var_1727_begin_0 = const()[name = string("op_1727_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1727_end_0 = const()[name = string("op_1727_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1727_end_mask_0 = const()[name = string("op_1727_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1727_cast_fp16 = slice_by_index(begin = var_1727_begin_0, end = var_1727_end_0, end_mask = var_1727_end_mask_0, x = coreml_update_state_4)[name = string("op_1727_cast_fp16")]; tensor tile_4 = const()[name = string("tile_4"), val = tensor([1, 1])]; int32 var_1730_axis_0 = const()[name = string("op_1730_axis_0"), val = int32(1)]; tensor var_1730_cast_fp16_0, tensor var_1730_cast_fp16_1 = split(axis = var_1730_axis_0, split_sizes = tile_4, x = var_1727_cast_fp16)[name = string("op_1730_cast_fp16")]; tensor var_1737_begin_0 = const()[name = string("op_1737_begin_0"), val = tensor([2, 0, 0, 0])]; tensor var_1737_end_0 = const()[name = string("op_1737_end_0"), val = tensor([3, 2, 2048, 128])]; tensor var_1737_end_mask_0 = const()[name = string("op_1737_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_1737_cast_fp16 = slice_by_index(begin = var_1737_begin_0, end = var_1737_end_0, end_mask = var_1737_end_mask_0, x = coreml_update_state_5)[name = string("op_1737_cast_fp16")]; tensor tile_5 = const()[name = string("tile_5"), val = tensor([1, 1])]; int32 var_1740_axis_0 = const()[name = string("op_1740_axis_0"), val = int32(1)]; tensor var_1740_cast_fp16_0, tensor var_1740_cast_fp16_1 = split(axis = var_1740_axis_0, split_sizes = tile_5, x = var_1737_cast_fp16)[name = string("op_1740_cast_fp16")]; tensor var_1743_split_sizes_0 = const()[name = string("op_1743_split_sizes_0"), val = tensor([8, 8])]; int32 var_1743_axis_0 = const()[name = string("op_1743_axis_0"), val = int32(1)]; tensor var_1743_0, tensor var_1743_1 = split(axis = var_1743_axis_0, split_sizes = var_1743_split_sizes_0, x = query_states_15_cast_fp16)[name = string("op_1743")]; bool attn_weights_33_transpose_x_0 = const()[name = string("attn_weights_33_transpose_x_0"), val = bool(false)]; bool attn_weights_33_transpose_y_0 = const()[name = string("attn_weights_33_transpose_y_0"), val = bool(false)]; tensor attn_weights_33_cast_fp16 = matmul(transpose_x = attn_weights_33_transpose_x_0, transpose_y = attn_weights_33_transpose_y_0, x = var_1730_cast_fp16_0, y = var_1743_0)[name = string("attn_weights_33_cast_fp16")]; fp16 var_1746_to_fp16 = const()[name = string("op_1746_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_35_cast_fp16 = mul(x = attn_weights_33_cast_fp16, y = var_1746_to_fp16)[name = string("attn_weights_35_cast_fp16")]; tensor attn_weights_37_cast_fp16 = add(x = attn_weights_35_cast_fp16, y = attn_mask_1)[name = string("attn_weights_37_cast_fp16")]; int32 var_1750 = const()[name = string("op_1750"), val = int32(-2)]; tensor attn_weights_39_cast_fp16 = softmax(axis = var_1750, x = attn_weights_37_cast_fp16)[name = string("attn_weights_39_cast_fp16")]; bool var_1756_transpose_x_1 = const()[name = string("op_1756_transpose_x_1"), val = bool(true)]; bool var_1756_transpose_y_1 = const()[name = string("op_1756_transpose_y_1"), val = bool(false)]; tensor var_1756_cast_fp16 = matmul(transpose_x = var_1756_transpose_x_1, transpose_y = var_1756_transpose_y_1, x = attn_weights_39_cast_fp16, y = var_1740_cast_fp16_0)[name = string("op_1756_cast_fp16")]; bool attn_weights_41_transpose_x_0 = const()[name = string("attn_weights_41_transpose_x_0"), val = bool(false)]; bool attn_weights_41_transpose_y_0 = const()[name = string("attn_weights_41_transpose_y_0"), val = bool(false)]; tensor attn_weights_41_cast_fp16 = matmul(transpose_x = attn_weights_41_transpose_x_0, transpose_y = attn_weights_41_transpose_y_0, x = var_1730_cast_fp16_1, y = var_1743_1)[name = string("attn_weights_41_cast_fp16")]; fp16 var_1758_to_fp16 = const()[name = string("op_1758_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_43_cast_fp16 = mul(x = attn_weights_41_cast_fp16, y = var_1758_to_fp16)[name = string("attn_weights_43_cast_fp16")]; tensor attn_weights_45_cast_fp16 = add(x = attn_weights_43_cast_fp16, y = attn_mask_1)[name = string("attn_weights_45_cast_fp16")]; int32 var_1762 = const()[name = string("op_1762"), val = int32(-2)]; tensor attn_weights_47_cast_fp16 = softmax(axis = var_1762, x = attn_weights_45_cast_fp16)[name = string("attn_weights_47_cast_fp16")]; bool attn_output_17_transpose_x_1 = const()[name = string("attn_output_17_transpose_x_1"), val = bool(true)]; bool attn_output_17_transpose_y_1 = const()[name = string("attn_output_17_transpose_y_1"), val = bool(false)]; tensor attn_output_17_cast_fp16 = matmul(transpose_x = attn_output_17_transpose_x_1, transpose_y = attn_output_17_transpose_y_1, x = attn_weights_47_cast_fp16, y = var_1740_cast_fp16_1)[name = string("attn_output_17_cast_fp16")]; int32 var_1770 = const()[name = string("op_1770"), val = int32(1)]; bool attn_output_19_interleave_0 = const()[name = string("attn_output_19_interleave_0"), val = bool(false)]; tensor attn_output_19_cast_fp16 = concat(axis = var_1770, interleave = attn_output_19_interleave_0, values = (var_1756_cast_fp16, attn_output_17_cast_fp16))[name = string("attn_output_19_cast_fp16")]; tensor var_1774_perm_0 = const()[name = string("op_1774_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_35x = const()[name = string("concat_35x"), val = tensor([1, 2048, 1, -1])]; tensor var_1774_cast_fp16 = transpose(perm = var_1774_perm_0, x = attn_output_19_cast_fp16)[name = string("transpose_75")]; tensor attn_output_23_cast_fp16 = reshape(shape = concat_35x, x = var_1774_cast_fp16)[name = string("attn_output_23_cast_fp16")]; tensor layers_2_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_2_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1155169408)))]; tensor hidden_states_23_strides_0 = const()[name = string("hidden_states_23_strides_0"), val = tensor([1, 1])]; string hidden_states_23_pad_type_0 = const()[name = string("hidden_states_23_pad_type_0"), val = string("valid")]; tensor hidden_states_23_pad_0 = const()[name = string("hidden_states_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_23_dilations_0 = const()[name = string("hidden_states_23_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_23_groups_0 = const()[name = string("hidden_states_23_groups_0"), val = int32(1)]; tensor hidden_states_23_cast_fp16 = conv(dilations = hidden_states_23_dilations_0, groups = hidden_states_23_groups_0, pad = hidden_states_23_pad_0, pad_type = hidden_states_23_pad_type_0, strides = hidden_states_23_strides_0, weight = layers_2_self_attn_o_proj_weight_to_fp16, x = attn_output_23_cast_fp16)[name = string("hidden_states_23_cast_fp16")]; tensor hidden_states_25_cast_fp16 = add(x = hidden_states_19_cast_fp16, y = hidden_states_23_cast_fp16)[name = string("hidden_states_25_cast_fp16")]; fp16 const_28_promoted_to_fp16 = const()[name = string("const_28_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1807_cast_fp16 = mul(x = hidden_states_25_cast_fp16, y = const_28_promoted_to_fp16)[name = string("op_1807_cast_fp16")]; int32 var_1805 = const()[name = string("op_1805"), val = int32(1)]; bool doubled_21_interleave_0 = const()[name = string("doubled_21_interleave_0"), val = bool(false)]; tensor doubled_21_cast_fp16 = concat(axis = var_1805, interleave = doubled_21_interleave_0, values = (hidden_states_25_cast_fp16, var_1807_cast_fp16))[name = string("doubled_21_cast_fp16")]; tensor out_11_axes_0 = const()[name = string("out_11_axes_0"), val = tensor([1])]; tensor out_11_gamma_0_to_fp16 = const()[name = string("out_11_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163558080)))]; fp16 var_1817_to_fp16 = const()[name = string("op_1817_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_11_cast_fp16 = layer_norm(axes = out_11_axes_0, epsilon = var_1817_to_fp16, gamma = out_11_gamma_0_to_fp16, x = doubled_21_cast_fp16)[name = string("out_11_cast_fp16")]; tensor var_1828_split_sizes_0 = const()[name = string("op_1828_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1828_axis_0 = const()[name = string("op_1828_axis_0"), val = int32(1)]; tensor var_1828_cast_fp16_0, tensor var_1828_cast_fp16_1 = split(axis = var_1828_axis_0, split_sizes = var_1828_split_sizes_0, x = out_11_cast_fp16)[name = string("op_1828_cast_fp16")]; tensor layers_2_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1163566336)))]; tensor input_5_strides_0 = const()[name = string("input_5_strides_0"), val = tensor([1, 1])]; string input_5_pad_type_0 = const()[name = string("input_5_pad_type_0"), val = string("valid")]; tensor input_5_pad_0 = const()[name = string("input_5_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_5_dilations_0 = const()[name = string("input_5_dilations_0"), val = tensor([1, 1])]; int32 input_5_groups_0 = const()[name = string("input_5_groups_0"), val = int32(1)]; tensor input_5_cast_fp16 = conv(dilations = input_5_dilations_0, groups = input_5_groups_0, pad = input_5_pad_0, pad_type = input_5_pad_type_0, strides = input_5_strides_0, weight = layers_2_mlp_gate_proj_weight_to_fp16, x = var_1828_cast_fp16_0)[name = string("input_5_cast_fp16")]; tensor var_1845_cast_fp16 = silu(x = input_5_cast_fp16)[name = string("op_1845_cast_fp16")]; tensor var_1851_strides_0 = const()[name = string("op_1851_strides_0"), val = tensor([1, 1])]; string var_1851_pad_type_0 = const()[name = string("op_1851_pad_type_0"), val = string("valid")]; tensor var_1851_pad_0 = const()[name = string("op_1851_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_1851_dilations_0 = const()[name = string("op_1851_dilations_0"), val = tensor([1, 1])]; int32 var_1851_groups_0 = const()[name = string("op_1851_groups_0"), val = int32(1)]; tensor var_1851_cast_fp16 = conv(dilations = var_1851_dilations_0, groups = var_1851_groups_0, pad = var_1851_pad_0, pad_type = var_1851_pad_type_0, strides = var_1851_strides_0, weight = layers_2_mlp_up_proj_weight_cast_fp16, x = var_1828_cast_fp16_0)[name = string("op_1851_cast_fp16")]; tensor x_29_cast_fp16 = mul(x = var_1845_cast_fp16, y = var_1851_cast_fp16)[name = string("x_29_cast_fp16")]; tensor layers_2_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_2_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1188732224)))]; tensor hidden_states_27_strides_0 = const()[name = string("hidden_states_27_strides_0"), val = tensor([1, 1])]; string hidden_states_27_pad_type_0 = const()[name = string("hidden_states_27_pad_type_0"), val = string("valid")]; tensor hidden_states_27_pad_0 = const()[name = string("hidden_states_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_27_dilations_0 = const()[name = string("hidden_states_27_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_27_groups_0 = const()[name = string("hidden_states_27_groups_0"), val = int32(1)]; tensor hidden_states_27_cast_fp16 = conv(dilations = hidden_states_27_dilations_0, groups = hidden_states_27_groups_0, pad = hidden_states_27_pad_0, pad_type = hidden_states_27_pad_type_0, strides = hidden_states_27_strides_0, weight = layers_2_mlp_down_proj_weight_to_fp16, x = x_29_cast_fp16)[name = string("hidden_states_27_cast_fp16")]; tensor hidden_states_29_cast_fp16 = add(x = hidden_states_25_cast_fp16, y = hidden_states_27_cast_fp16)[name = string("hidden_states_29_cast_fp16")]; fp16 const_30_promoted_to_fp16 = const()[name = string("const_30_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1869_cast_fp16 = mul(x = hidden_states_29_cast_fp16, y = const_30_promoted_to_fp16)[name = string("op_1869_cast_fp16")]; int32 var_1867 = const()[name = string("op_1867"), val = int32(1)]; bool doubled_25_interleave_0 = const()[name = string("doubled_25_interleave_0"), val = bool(false)]; tensor doubled_25_cast_fp16 = concat(axis = var_1867, interleave = doubled_25_interleave_0, values = (hidden_states_29_cast_fp16, var_1869_cast_fp16))[name = string("doubled_25_cast_fp16")]; tensor out_13_axes_0 = const()[name = string("out_13_axes_0"), val = tensor([1])]; tensor out_13_gamma_0_to_fp16 = const()[name = string("out_13_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213898112)))]; fp16 var_1879_to_fp16 = const()[name = string("op_1879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_13_cast_fp16 = layer_norm(axes = out_13_axes_0, epsilon = var_1879_to_fp16, gamma = out_13_gamma_0_to_fp16, x = doubled_25_cast_fp16)[name = string("out_13_cast_fp16")]; tensor var_1890_split_sizes_0 = const()[name = string("op_1890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_1890_axis_0 = const()[name = string("op_1890_axis_0"), val = int32(1)]; tensor var_1890_cast_fp16_0, tensor var_1890_cast_fp16_1 = split(axis = var_1890_axis_0, split_sizes = var_1890_split_sizes_0, x = out_13_cast_fp16)[name = string("op_1890_cast_fp16")]; tensor layers_3_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1213906368)))]; tensor query_states_19_strides_0 = const()[name = string("query_states_19_strides_0"), val = tensor([1, 1])]; string query_states_19_pad_type_0 = const()[name = string("query_states_19_pad_type_0"), val = string("valid")]; tensor query_states_19_pad_0 = const()[name = string("query_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_19_dilations_0 = const()[name = string("query_states_19_dilations_0"), val = tensor([1, 1])]; int32 query_states_19_groups_0 = const()[name = string("query_states_19_groups_0"), val = int32(1)]; tensor query_states_19_cast_fp16 = conv(dilations = query_states_19_dilations_0, groups = query_states_19_groups_0, pad = query_states_19_pad_0, pad_type = query_states_19_pad_type_0, strides = query_states_19_strides_0, weight = layers_3_self_attn_q_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("query_states_19_cast_fp16")]; tensor layers_3_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_3_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1222295040)))]; tensor key_states_31_strides_0 = const()[name = string("key_states_31_strides_0"), val = tensor([1, 1])]; string key_states_31_pad_type_0 = const()[name = string("key_states_31_pad_type_0"), val = string("valid")]; tensor key_states_31_pad_0 = const()[name = string("key_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_31_dilations_0 = const()[name = string("key_states_31_dilations_0"), val = tensor([1, 1])]; int32 key_states_31_groups_0 = const()[name = string("key_states_31_groups_0"), val = int32(1)]; tensor key_states_31_cast_fp16 = conv(dilations = key_states_31_dilations_0, groups = key_states_31_groups_0, pad = key_states_31_pad_0, pad_type = key_states_31_pad_type_0, strides = key_states_31_strides_0, weight = layers_3_self_attn_k_proj_weight_to_fp16, x = var_1890_cast_fp16_0)[name = string("key_states_31_cast_fp16")]; tensor value_states_19_strides_0 = const()[name = string("value_states_19_strides_0"), val = tensor([1, 1])]; string value_states_19_pad_type_0 = const()[name = string("value_states_19_pad_type_0"), val = string("valid")]; tensor value_states_19_pad_0 = const()[name = string("value_states_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_19_dilations_0 = const()[name = string("value_states_19_dilations_0"), val = tensor([1, 1])]; int32 value_states_19_groups_0 = const()[name = string("value_states_19_groups_0"), val = int32(1)]; tensor value_states_19_cast_fp16 = conv(dilations = value_states_19_dilations_0, groups = value_states_19_groups_0, pad = value_states_19_pad_0, pad_type = value_states_19_pad_type_0, strides = value_states_19_strides_0, weight = layers_3_self_attn_v_proj_weight_cast_fp16, x = var_1890_cast_fp16_0)[name = string("value_states_19_cast_fp16")]; tensor concat_36x = const()[name = string("concat_36x"), val = tensor([1, 16, 128, -1])]; tensor x_31_cast_fp16 = reshape(shape = concat_36x, x = query_states_19_cast_fp16)[name = string("x_31_cast_fp16")]; tensor concat_37x = const()[name = string("concat_37x"), val = tensor([1, 2, 128, -1])]; tensor var_1947_cast_fp16 = reshape(shape = concat_37x, x = key_states_31_cast_fp16)[name = string("op_1947_cast_fp16")]; tensor concat_38x = const()[name = string("concat_38x"), val = tensor([1, 2, 128, -1])]; tensor var_1954_cast_fp16 = reshape(shape = concat_38x, x = value_states_19_cast_fp16)[name = string("op_1954_cast_fp16")]; tensor var_1958_cast_fp16 = mul(x = x_31_cast_fp16, y = var_869_cast_fp16)[name = string("op_1958_cast_fp16")]; tensor var_1959_split_sizes_0 = const()[name = string("op_1959_split_sizes_0"), val = tensor([64, 64])]; int32 var_1959_axis_0 = const()[name = string("op_1959_axis_0"), val = int32(-2)]; tensor var_1959_cast_fp16_0, tensor var_1959_cast_fp16_1 = split(axis = var_1959_axis_0, split_sizes = var_1959_split_sizes_0, x = x_31_cast_fp16)[name = string("op_1959_cast_fp16")]; fp16 const_32_promoted_to_fp16 = const()[name = string("const_32_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1961_cast_fp16 = mul(x = var_1959_cast_fp16_1, y = const_32_promoted_to_fp16)[name = string("op_1961_cast_fp16")]; int32 var_1963 = const()[name = string("op_1963"), val = int32(-2)]; bool var_1964_interleave_0 = const()[name = string("op_1964_interleave_0"), val = bool(false)]; tensor var_1964_cast_fp16 = concat(axis = var_1963, interleave = var_1964_interleave_0, values = (var_1961_cast_fp16, var_1959_cast_fp16_0))[name = string("op_1964_cast_fp16")]; tensor var_1965_cast_fp16 = mul(x = var_1964_cast_fp16, y = var_878_cast_fp16)[name = string("op_1965_cast_fp16")]; tensor query_states_21_cast_fp16 = add(x = var_1958_cast_fp16, y = var_1965_cast_fp16)[name = string("query_states_21_cast_fp16")]; tensor var_1971_cast_fp16 = mul(x = var_1947_cast_fp16, y = var_869_cast_fp16)[name = string("op_1971_cast_fp16")]; tensor var_1972_split_sizes_0 = const()[name = string("op_1972_split_sizes_0"), val = tensor([64, 64])]; int32 var_1972_axis_0 = const()[name = string("op_1972_axis_0"), val = int32(-2)]; tensor var_1972_cast_fp16_0, tensor var_1972_cast_fp16_1 = split(axis = var_1972_axis_0, split_sizes = var_1972_split_sizes_0, x = var_1947_cast_fp16)[name = string("op_1972_cast_fp16")]; fp16 const_33_promoted_to_fp16 = const()[name = string("const_33_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_1974_cast_fp16 = mul(x = var_1972_cast_fp16_1, y = const_33_promoted_to_fp16)[name = string("op_1974_cast_fp16")]; int32 var_1976 = const()[name = string("op_1976"), val = int32(-2)]; bool var_1977_interleave_0 = const()[name = string("op_1977_interleave_0"), val = bool(false)]; tensor var_1977_cast_fp16 = concat(axis = var_1976, interleave = var_1977_interleave_0, values = (var_1974_cast_fp16, var_1972_cast_fp16_0))[name = string("op_1977_cast_fp16")]; tensor var_1978_cast_fp16 = mul(x = var_1977_cast_fp16, y = var_878_cast_fp16)[name = string("op_1978_cast_fp16")]; tensor key_states_35_cast_fp16 = add(x = var_1971_cast_fp16, y = var_1978_cast_fp16)[name = string("key_states_35_cast_fp16")]; tensor expand_dims_36 = const()[name = string("expand_dims_36"), val = tensor([3])]; tensor expand_dims_37 = const()[name = string("expand_dims_37"), val = tensor([0])]; tensor expand_dims_39 = const()[name = string("expand_dims_39"), val = tensor([0])]; int32 concat_41_axis_0 = const()[name = string("concat_41_axis_0"), val = int32(0)]; bool concat_41_interleave_0 = const()[name = string("concat_41_interleave_0"), val = bool(false)]; tensor concat_41 = concat(axis = concat_41_axis_0, interleave = concat_41_interleave_0, values = (expand_dims_36, expand_dims_37, position_id, expand_dims_39))[name = string("concat_41")]; tensor expand_dims_40 = const()[name = string("expand_dims_40"), val = tensor([4])]; tensor concat_42_values1_0 = const()[name = string("concat_42_values1_0"), val = tensor([0])]; tensor concat_42_values3_0 = const()[name = string("concat_42_values3_0"), val = tensor([0])]; int32 concat_42_axis_0 = const()[name = string("concat_42_axis_0"), val = int32(0)]; bool concat_42_interleave_0 = const()[name = string("concat_42_interleave_0"), val = bool(false)]; tensor concat_42 = concat(axis = concat_42_axis_0, interleave = concat_42_interleave_0, values = (expand_dims_40, concat_42_values1_0, cache_position_end, concat_42_values3_0))[name = string("concat_42")]; tensor key_states_37_perm_0 = const()[name = string("key_states_37_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_4_stride_0 = const()[name = string("key_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_37_cast_fp16 = transpose(perm = key_states_37_perm_0, x = key_states_35_cast_fp16)[name = string("transpose_74")]; tensor key_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = key_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = key_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_4_squeeze_mask_0, stride = key_cache_internal_tensor_assign_4_stride_0, update = key_states_37_cast_fp16, x = coreml_update_state_4)[name = string("key_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_4_cast_fp16, input = key_cache)[name = string("coreml_update_state_6_write_state")]; tensor coreml_update_state_6 = read_state(input = key_cache)[name = string("coreml_update_state_6")]; tensor value_states_21_perm_0 = const()[name = string("value_states_21_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_4_stride_0 = const()[name = string("value_cache_internal_tensor_assign_4_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_4_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_4_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_4_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_4_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_21_cast_fp16 = transpose(perm = value_states_21_perm_0, x = var_1954_cast_fp16)[name = string("transpose_73")]; tensor value_cache_internal_tensor_assign_4_cast_fp16 = slice_update(begin = concat_41, begin_mask = value_cache_internal_tensor_assign_4_begin_mask_0, end = concat_42, end_mask = value_cache_internal_tensor_assign_4_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_4_squeeze_mask_0, stride = value_cache_internal_tensor_assign_4_stride_0, update = value_states_21_cast_fp16, x = coreml_update_state_5)[name = string("value_cache_internal_tensor_assign_4_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_4_cast_fp16, input = value_cache)[name = string("coreml_update_state_7_write_state")]; tensor coreml_update_state_7 = read_state(input = value_cache)[name = string("coreml_update_state_7")]; tensor var_2048_begin_0 = const()[name = string("op_2048_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2048_end_0 = const()[name = string("op_2048_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2048_end_mask_0 = const()[name = string("op_2048_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2048_cast_fp16 = slice_by_index(begin = var_2048_begin_0, end = var_2048_end_0, end_mask = var_2048_end_mask_0, x = coreml_update_state_6)[name = string("op_2048_cast_fp16")]; tensor tile_6 = const()[name = string("tile_6"), val = tensor([1, 1])]; int32 var_2051_axis_0 = const()[name = string("op_2051_axis_0"), val = int32(1)]; tensor var_2051_cast_fp16_0, tensor var_2051_cast_fp16_1 = split(axis = var_2051_axis_0, split_sizes = tile_6, x = var_2048_cast_fp16)[name = string("op_2051_cast_fp16")]; tensor var_2058_begin_0 = const()[name = string("op_2058_begin_0"), val = tensor([3, 0, 0, 0])]; tensor var_2058_end_0 = const()[name = string("op_2058_end_0"), val = tensor([4, 2, 2048, 128])]; tensor var_2058_end_mask_0 = const()[name = string("op_2058_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2058_cast_fp16 = slice_by_index(begin = var_2058_begin_0, end = var_2058_end_0, end_mask = var_2058_end_mask_0, x = coreml_update_state_7)[name = string("op_2058_cast_fp16")]; tensor tile_7 = const()[name = string("tile_7"), val = tensor([1, 1])]; int32 var_2061_axis_0 = const()[name = string("op_2061_axis_0"), val = int32(1)]; tensor var_2061_cast_fp16_0, tensor var_2061_cast_fp16_1 = split(axis = var_2061_axis_0, split_sizes = tile_7, x = var_2058_cast_fp16)[name = string("op_2061_cast_fp16")]; tensor var_2064_split_sizes_0 = const()[name = string("op_2064_split_sizes_0"), val = tensor([8, 8])]; int32 var_2064_axis_0 = const()[name = string("op_2064_axis_0"), val = int32(1)]; tensor var_2064_0, tensor var_2064_1 = split(axis = var_2064_axis_0, split_sizes = var_2064_split_sizes_0, x = query_states_21_cast_fp16)[name = string("op_2064")]; bool attn_weights_49_transpose_x_0 = const()[name = string("attn_weights_49_transpose_x_0"), val = bool(false)]; bool attn_weights_49_transpose_y_0 = const()[name = string("attn_weights_49_transpose_y_0"), val = bool(false)]; tensor attn_weights_49_cast_fp16 = matmul(transpose_x = attn_weights_49_transpose_x_0, transpose_y = attn_weights_49_transpose_y_0, x = var_2051_cast_fp16_0, y = var_2064_0)[name = string("attn_weights_49_cast_fp16")]; fp16 var_2067_to_fp16 = const()[name = string("op_2067_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_51_cast_fp16 = mul(x = attn_weights_49_cast_fp16, y = var_2067_to_fp16)[name = string("attn_weights_51_cast_fp16")]; tensor attn_weights_53_cast_fp16 = add(x = attn_weights_51_cast_fp16, y = attn_mask_1)[name = string("attn_weights_53_cast_fp16")]; int32 var_2071 = const()[name = string("op_2071"), val = int32(-2)]; tensor attn_weights_55_cast_fp16 = softmax(axis = var_2071, x = attn_weights_53_cast_fp16)[name = string("attn_weights_55_cast_fp16")]; bool var_2077_transpose_x_1 = const()[name = string("op_2077_transpose_x_1"), val = bool(true)]; bool var_2077_transpose_y_1 = const()[name = string("op_2077_transpose_y_1"), val = bool(false)]; tensor var_2077_cast_fp16 = matmul(transpose_x = var_2077_transpose_x_1, transpose_y = var_2077_transpose_y_1, x = attn_weights_55_cast_fp16, y = var_2061_cast_fp16_0)[name = string("op_2077_cast_fp16")]; bool attn_weights_57_transpose_x_0 = const()[name = string("attn_weights_57_transpose_x_0"), val = bool(false)]; bool attn_weights_57_transpose_y_0 = const()[name = string("attn_weights_57_transpose_y_0"), val = bool(false)]; tensor attn_weights_57_cast_fp16 = matmul(transpose_x = attn_weights_57_transpose_x_0, transpose_y = attn_weights_57_transpose_y_0, x = var_2051_cast_fp16_1, y = var_2064_1)[name = string("attn_weights_57_cast_fp16")]; fp16 var_2079_to_fp16 = const()[name = string("op_2079_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_59_cast_fp16 = mul(x = attn_weights_57_cast_fp16, y = var_2079_to_fp16)[name = string("attn_weights_59_cast_fp16")]; tensor attn_weights_61_cast_fp16 = add(x = attn_weights_59_cast_fp16, y = attn_mask_1)[name = string("attn_weights_61_cast_fp16")]; int32 var_2083 = const()[name = string("op_2083"), val = int32(-2)]; tensor attn_weights_63_cast_fp16 = softmax(axis = var_2083, x = attn_weights_61_cast_fp16)[name = string("attn_weights_63_cast_fp16")]; bool attn_output_25_transpose_x_1 = const()[name = string("attn_output_25_transpose_x_1"), val = bool(true)]; bool attn_output_25_transpose_y_1 = const()[name = string("attn_output_25_transpose_y_1"), val = bool(false)]; tensor attn_output_25_cast_fp16 = matmul(transpose_x = attn_output_25_transpose_x_1, transpose_y = attn_output_25_transpose_y_1, x = attn_weights_63_cast_fp16, y = var_2061_cast_fp16_1)[name = string("attn_output_25_cast_fp16")]; int32 var_2091 = const()[name = string("op_2091"), val = int32(1)]; bool attn_output_27_interleave_0 = const()[name = string("attn_output_27_interleave_0"), val = bool(false)]; tensor attn_output_27_cast_fp16 = concat(axis = var_2091, interleave = attn_output_27_interleave_0, values = (var_2077_cast_fp16, attn_output_25_cast_fp16))[name = string("attn_output_27_cast_fp16")]; tensor var_2095_perm_0 = const()[name = string("op_2095_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_47x = const()[name = string("concat_47x"), val = tensor([1, 2048, 1, -1])]; tensor var_2095_cast_fp16 = transpose(perm = var_2095_perm_0, x = attn_output_27_cast_fp16)[name = string("transpose_72")]; tensor attn_output_31_cast_fp16 = reshape(shape = concat_47x, x = var_2095_cast_fp16)[name = string("attn_output_31_cast_fp16")]; tensor hidden_states_33_strides_0 = const()[name = string("hidden_states_33_strides_0"), val = tensor([1, 1])]; string hidden_states_33_pad_type_0 = const()[name = string("hidden_states_33_pad_type_0"), val = string("valid")]; tensor hidden_states_33_pad_0 = const()[name = string("hidden_states_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_33_dilations_0 = const()[name = string("hidden_states_33_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_33_groups_0 = const()[name = string("hidden_states_33_groups_0"), val = int32(1)]; tensor hidden_states_33_cast_fp16 = conv(dilations = hidden_states_33_dilations_0, groups = hidden_states_33_groups_0, pad = hidden_states_33_pad_0, pad_type = hidden_states_33_pad_type_0, strides = hidden_states_33_strides_0, weight = layers_3_self_attn_o_proj_weight_cast_fp16, x = attn_output_31_cast_fp16)[name = string("hidden_states_33_cast_fp16")]; tensor hidden_states_35_cast_fp16 = add(x = hidden_states_29_cast_fp16, y = hidden_states_33_cast_fp16)[name = string("hidden_states_35_cast_fp16")]; fp16 const_38_promoted_to_fp16 = const()[name = string("const_38_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2128_cast_fp16 = mul(x = hidden_states_35_cast_fp16, y = const_38_promoted_to_fp16)[name = string("op_2128_cast_fp16")]; int32 var_2126 = const()[name = string("op_2126"), val = int32(1)]; bool doubled_29_interleave_0 = const()[name = string("doubled_29_interleave_0"), val = bool(false)]; tensor doubled_29_cast_fp16 = concat(axis = var_2126, interleave = doubled_29_interleave_0, values = (hidden_states_35_cast_fp16, var_2128_cast_fp16))[name = string("doubled_29_cast_fp16")]; tensor out_15_axes_0 = const()[name = string("out_15_axes_0"), val = tensor([1])]; tensor out_15_gamma_0_to_fp16 = const()[name = string("out_15_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223343680)))]; fp16 var_2138_to_fp16 = const()[name = string("op_2138_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_15_cast_fp16 = layer_norm(axes = out_15_axes_0, epsilon = var_2138_to_fp16, gamma = out_15_gamma_0_to_fp16, x = doubled_29_cast_fp16)[name = string("out_15_cast_fp16")]; tensor var_2149_split_sizes_0 = const()[name = string("op_2149_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2149_axis_0 = const()[name = string("op_2149_axis_0"), val = int32(1)]; tensor var_2149_cast_fp16_0, tensor var_2149_cast_fp16_1 = split(axis = var_2149_axis_0, split_sizes = var_2149_split_sizes_0, x = out_15_cast_fp16)[name = string("op_2149_cast_fp16")]; tensor layers_3_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1223351936)))]; tensor input_7_strides_0 = const()[name = string("input_7_strides_0"), val = tensor([1, 1])]; string input_7_pad_type_0 = const()[name = string("input_7_pad_type_0"), val = string("valid")]; tensor input_7_pad_0 = const()[name = string("input_7_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_7_dilations_0 = const()[name = string("input_7_dilations_0"), val = tensor([1, 1])]; int32 input_7_groups_0 = const()[name = string("input_7_groups_0"), val = int32(1)]; tensor input_7_cast_fp16 = conv(dilations = input_7_dilations_0, groups = input_7_groups_0, pad = input_7_pad_0, pad_type = input_7_pad_type_0, strides = input_7_strides_0, weight = layers_3_mlp_gate_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("input_7_cast_fp16")]; tensor var_2166_cast_fp16 = silu(x = input_7_cast_fp16)[name = string("op_2166_cast_fp16")]; tensor layers_3_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_3_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1248517824)))]; tensor var_2172_strides_0 = const()[name = string("op_2172_strides_0"), val = tensor([1, 1])]; string var_2172_pad_type_0 = const()[name = string("op_2172_pad_type_0"), val = string("valid")]; tensor var_2172_pad_0 = const()[name = string("op_2172_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2172_dilations_0 = const()[name = string("op_2172_dilations_0"), val = tensor([1, 1])]; int32 var_2172_groups_0 = const()[name = string("op_2172_groups_0"), val = int32(1)]; tensor var_2172_cast_fp16 = conv(dilations = var_2172_dilations_0, groups = var_2172_groups_0, pad = var_2172_pad_0, pad_type = var_2172_pad_type_0, strides = var_2172_strides_0, weight = layers_3_mlp_up_proj_weight_to_fp16, x = var_2149_cast_fp16_0)[name = string("op_2172_cast_fp16")]; tensor x_39_cast_fp16 = mul(x = var_2166_cast_fp16, y = var_2172_cast_fp16)[name = string("x_39_cast_fp16")]; tensor hidden_states_37_strides_0 = const()[name = string("hidden_states_37_strides_0"), val = tensor([1, 1])]; string hidden_states_37_pad_type_0 = const()[name = string("hidden_states_37_pad_type_0"), val = string("valid")]; tensor hidden_states_37_pad_0 = const()[name = string("hidden_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_37_dilations_0 = const()[name = string("hidden_states_37_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_37_groups_0 = const()[name = string("hidden_states_37_groups_0"), val = int32(1)]; tensor hidden_states_37_cast_fp16 = conv(dilations = hidden_states_37_dilations_0, groups = hidden_states_37_groups_0, pad = hidden_states_37_pad_0, pad_type = hidden_states_37_pad_type_0, strides = hidden_states_37_strides_0, weight = layers_3_mlp_down_proj_weight_cast_fp16, x = x_39_cast_fp16)[name = string("hidden_states_37_cast_fp16")]; tensor hidden_states_39_cast_fp16 = add(x = hidden_states_35_cast_fp16, y = hidden_states_37_cast_fp16)[name = string("hidden_states_39_cast_fp16")]; fp16 const_40_promoted_to_fp16 = const()[name = string("const_40_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2190_cast_fp16 = mul(x = hidden_states_39_cast_fp16, y = const_40_promoted_to_fp16)[name = string("op_2190_cast_fp16")]; int32 var_2188 = const()[name = string("op_2188"), val = int32(1)]; bool doubled_33_interleave_0 = const()[name = string("doubled_33_interleave_0"), val = bool(false)]; tensor doubled_33_cast_fp16 = concat(axis = var_2188, interleave = doubled_33_interleave_0, values = (hidden_states_39_cast_fp16, var_2190_cast_fp16))[name = string("doubled_33_cast_fp16")]; tensor out_17_axes_0 = const()[name = string("out_17_axes_0"), val = tensor([1])]; tensor out_17_gamma_0_to_fp16 = const()[name = string("out_17_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273683712)))]; fp16 var_2200_to_fp16 = const()[name = string("op_2200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_17_cast_fp16 = layer_norm(axes = out_17_axes_0, epsilon = var_2200_to_fp16, gamma = out_17_gamma_0_to_fp16, x = doubled_33_cast_fp16)[name = string("out_17_cast_fp16")]; tensor var_2211_split_sizes_0 = const()[name = string("op_2211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2211_axis_0 = const()[name = string("op_2211_axis_0"), val = int32(1)]; tensor var_2211_cast_fp16_0, tensor var_2211_cast_fp16_1 = split(axis = var_2211_axis_0, split_sizes = var_2211_split_sizes_0, x = out_17_cast_fp16)[name = string("op_2211_cast_fp16")]; tensor layers_4_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1273691968)))]; tensor query_states_25_strides_0 = const()[name = string("query_states_25_strides_0"), val = tensor([1, 1])]; string query_states_25_pad_type_0 = const()[name = string("query_states_25_pad_type_0"), val = string("valid")]; tensor query_states_25_pad_0 = const()[name = string("query_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_25_dilations_0 = const()[name = string("query_states_25_dilations_0"), val = tensor([1, 1])]; int32 query_states_25_groups_0 = const()[name = string("query_states_25_groups_0"), val = int32(1)]; tensor query_states_25_cast_fp16 = conv(dilations = query_states_25_dilations_0, groups = query_states_25_groups_0, pad = query_states_25_pad_0, pad_type = query_states_25_pad_type_0, strides = query_states_25_strides_0, weight = layers_4_self_attn_q_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("query_states_25_cast_fp16")]; tensor layers_4_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_4_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1282080640)))]; tensor key_states_41_strides_0 = const()[name = string("key_states_41_strides_0"), val = tensor([1, 1])]; string key_states_41_pad_type_0 = const()[name = string("key_states_41_pad_type_0"), val = string("valid")]; tensor key_states_41_pad_0 = const()[name = string("key_states_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_41_dilations_0 = const()[name = string("key_states_41_dilations_0"), val = tensor([1, 1])]; int32 key_states_41_groups_0 = const()[name = string("key_states_41_groups_0"), val = int32(1)]; tensor key_states_41_cast_fp16 = conv(dilations = key_states_41_dilations_0, groups = key_states_41_groups_0, pad = key_states_41_pad_0, pad_type = key_states_41_pad_type_0, strides = key_states_41_strides_0, weight = layers_4_self_attn_k_proj_weight_to_fp16, x = var_2211_cast_fp16_0)[name = string("key_states_41_cast_fp16")]; tensor value_states_25_strides_0 = const()[name = string("value_states_25_strides_0"), val = tensor([1, 1])]; string value_states_25_pad_type_0 = const()[name = string("value_states_25_pad_type_0"), val = string("valid")]; tensor value_states_25_pad_0 = const()[name = string("value_states_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_25_dilations_0 = const()[name = string("value_states_25_dilations_0"), val = tensor([1, 1])]; int32 value_states_25_groups_0 = const()[name = string("value_states_25_groups_0"), val = int32(1)]; tensor value_states_25_cast_fp16 = conv(dilations = value_states_25_dilations_0, groups = value_states_25_groups_0, pad = value_states_25_pad_0, pad_type = value_states_25_pad_type_0, strides = value_states_25_strides_0, weight = layers_4_self_attn_v_proj_weight_cast_fp16, x = var_2211_cast_fp16_0)[name = string("value_states_25_cast_fp16")]; tensor concat_48x = const()[name = string("concat_48x"), val = tensor([1, 16, 128, -1])]; tensor x_41_cast_fp16 = reshape(shape = concat_48x, x = query_states_25_cast_fp16)[name = string("x_41_cast_fp16")]; tensor concat_49x = const()[name = string("concat_49x"), val = tensor([1, 2, 128, -1])]; tensor var_2268_cast_fp16 = reshape(shape = concat_49x, x = key_states_41_cast_fp16)[name = string("op_2268_cast_fp16")]; tensor concat_50x = const()[name = string("concat_50x"), val = tensor([1, 2, 128, -1])]; tensor var_2275_cast_fp16 = reshape(shape = concat_50x, x = value_states_25_cast_fp16)[name = string("op_2275_cast_fp16")]; tensor var_2279_cast_fp16 = mul(x = x_41_cast_fp16, y = var_869_cast_fp16)[name = string("op_2279_cast_fp16")]; tensor var_2280_split_sizes_0 = const()[name = string("op_2280_split_sizes_0"), val = tensor([64, 64])]; int32 var_2280_axis_0 = const()[name = string("op_2280_axis_0"), val = int32(-2)]; tensor var_2280_cast_fp16_0, tensor var_2280_cast_fp16_1 = split(axis = var_2280_axis_0, split_sizes = var_2280_split_sizes_0, x = x_41_cast_fp16)[name = string("op_2280_cast_fp16")]; fp16 const_42_promoted_to_fp16 = const()[name = string("const_42_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2282_cast_fp16 = mul(x = var_2280_cast_fp16_1, y = const_42_promoted_to_fp16)[name = string("op_2282_cast_fp16")]; int32 var_2284 = const()[name = string("op_2284"), val = int32(-2)]; bool var_2285_interleave_0 = const()[name = string("op_2285_interleave_0"), val = bool(false)]; tensor var_2285_cast_fp16 = concat(axis = var_2284, interleave = var_2285_interleave_0, values = (var_2282_cast_fp16, var_2280_cast_fp16_0))[name = string("op_2285_cast_fp16")]; tensor var_2286_cast_fp16 = mul(x = var_2285_cast_fp16, y = var_878_cast_fp16)[name = string("op_2286_cast_fp16")]; tensor query_states_27_cast_fp16 = add(x = var_2279_cast_fp16, y = var_2286_cast_fp16)[name = string("query_states_27_cast_fp16")]; tensor var_2292_cast_fp16 = mul(x = var_2268_cast_fp16, y = var_869_cast_fp16)[name = string("op_2292_cast_fp16")]; tensor var_2293_split_sizes_0 = const()[name = string("op_2293_split_sizes_0"), val = tensor([64, 64])]; int32 var_2293_axis_0 = const()[name = string("op_2293_axis_0"), val = int32(-2)]; tensor var_2293_cast_fp16_0, tensor var_2293_cast_fp16_1 = split(axis = var_2293_axis_0, split_sizes = var_2293_split_sizes_0, x = var_2268_cast_fp16)[name = string("op_2293_cast_fp16")]; fp16 const_43_promoted_to_fp16 = const()[name = string("const_43_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2295_cast_fp16 = mul(x = var_2293_cast_fp16_1, y = const_43_promoted_to_fp16)[name = string("op_2295_cast_fp16")]; int32 var_2297 = const()[name = string("op_2297"), val = int32(-2)]; bool var_2298_interleave_0 = const()[name = string("op_2298_interleave_0"), val = bool(false)]; tensor var_2298_cast_fp16 = concat(axis = var_2297, interleave = var_2298_interleave_0, values = (var_2295_cast_fp16, var_2293_cast_fp16_0))[name = string("op_2298_cast_fp16")]; tensor var_2299_cast_fp16 = mul(x = var_2298_cast_fp16, y = var_878_cast_fp16)[name = string("op_2299_cast_fp16")]; tensor key_states_45_cast_fp16 = add(x = var_2292_cast_fp16, y = var_2299_cast_fp16)[name = string("key_states_45_cast_fp16")]; tensor expand_dims_48 = const()[name = string("expand_dims_48"), val = tensor([4])]; tensor expand_dims_49 = const()[name = string("expand_dims_49"), val = tensor([0])]; tensor expand_dims_51 = const()[name = string("expand_dims_51"), val = tensor([0])]; int32 concat_53_axis_0 = const()[name = string("concat_53_axis_0"), val = int32(0)]; bool concat_53_interleave_0 = const()[name = string("concat_53_interleave_0"), val = bool(false)]; tensor concat_53 = concat(axis = concat_53_axis_0, interleave = concat_53_interleave_0, values = (expand_dims_48, expand_dims_49, position_id, expand_dims_51))[name = string("concat_53")]; tensor expand_dims_52 = const()[name = string("expand_dims_52"), val = tensor([5])]; tensor concat_54_values1_0 = const()[name = string("concat_54_values1_0"), val = tensor([0])]; tensor concat_54_values3_0 = const()[name = string("concat_54_values3_0"), val = tensor([0])]; int32 concat_54_axis_0 = const()[name = string("concat_54_axis_0"), val = int32(0)]; bool concat_54_interleave_0 = const()[name = string("concat_54_interleave_0"), val = bool(false)]; tensor concat_54 = concat(axis = concat_54_axis_0, interleave = concat_54_interleave_0, values = (expand_dims_52, concat_54_values1_0, cache_position_end, concat_54_values3_0))[name = string("concat_54")]; tensor key_states_47_perm_0 = const()[name = string("key_states_47_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_5_stride_0 = const()[name = string("key_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_47_cast_fp16 = transpose(perm = key_states_47_perm_0, x = key_states_45_cast_fp16)[name = string("transpose_71")]; tensor key_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = key_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = key_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_5_squeeze_mask_0, stride = key_cache_internal_tensor_assign_5_stride_0, update = key_states_47_cast_fp16, x = coreml_update_state_6)[name = string("key_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_5_cast_fp16, input = key_cache)[name = string("coreml_update_state_8_write_state")]; tensor coreml_update_state_8 = read_state(input = key_cache)[name = string("coreml_update_state_8")]; tensor value_states_27_perm_0 = const()[name = string("value_states_27_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_5_stride_0 = const()[name = string("value_cache_internal_tensor_assign_5_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_5_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_5_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_5_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_5_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_27_cast_fp16 = transpose(perm = value_states_27_perm_0, x = var_2275_cast_fp16)[name = string("transpose_70")]; tensor value_cache_internal_tensor_assign_5_cast_fp16 = slice_update(begin = concat_53, begin_mask = value_cache_internal_tensor_assign_5_begin_mask_0, end = concat_54, end_mask = value_cache_internal_tensor_assign_5_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_5_squeeze_mask_0, stride = value_cache_internal_tensor_assign_5_stride_0, update = value_states_27_cast_fp16, x = coreml_update_state_7)[name = string("value_cache_internal_tensor_assign_5_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_5_cast_fp16, input = value_cache)[name = string("coreml_update_state_9_write_state")]; tensor coreml_update_state_9 = read_state(input = value_cache)[name = string("coreml_update_state_9")]; tensor var_2369_begin_0 = const()[name = string("op_2369_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2369_end_0 = const()[name = string("op_2369_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2369_end_mask_0 = const()[name = string("op_2369_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2369_cast_fp16 = slice_by_index(begin = var_2369_begin_0, end = var_2369_end_0, end_mask = var_2369_end_mask_0, x = coreml_update_state_8)[name = string("op_2369_cast_fp16")]; tensor tile_8 = const()[name = string("tile_8"), val = tensor([1, 1])]; int32 var_2372_axis_0 = const()[name = string("op_2372_axis_0"), val = int32(1)]; tensor var_2372_cast_fp16_0, tensor var_2372_cast_fp16_1 = split(axis = var_2372_axis_0, split_sizes = tile_8, x = var_2369_cast_fp16)[name = string("op_2372_cast_fp16")]; tensor var_2379_begin_0 = const()[name = string("op_2379_begin_0"), val = tensor([4, 0, 0, 0])]; tensor var_2379_end_0 = const()[name = string("op_2379_end_0"), val = tensor([5, 2, 2048, 128])]; tensor var_2379_end_mask_0 = const()[name = string("op_2379_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2379_cast_fp16 = slice_by_index(begin = var_2379_begin_0, end = var_2379_end_0, end_mask = var_2379_end_mask_0, x = coreml_update_state_9)[name = string("op_2379_cast_fp16")]; tensor tile_9 = const()[name = string("tile_9"), val = tensor([1, 1])]; int32 var_2382_axis_0 = const()[name = string("op_2382_axis_0"), val = int32(1)]; tensor var_2382_cast_fp16_0, tensor var_2382_cast_fp16_1 = split(axis = var_2382_axis_0, split_sizes = tile_9, x = var_2379_cast_fp16)[name = string("op_2382_cast_fp16")]; tensor var_2385_split_sizes_0 = const()[name = string("op_2385_split_sizes_0"), val = tensor([8, 8])]; int32 var_2385_axis_0 = const()[name = string("op_2385_axis_0"), val = int32(1)]; tensor var_2385_0, tensor var_2385_1 = split(axis = var_2385_axis_0, split_sizes = var_2385_split_sizes_0, x = query_states_27_cast_fp16)[name = string("op_2385")]; bool attn_weights_65_transpose_x_0 = const()[name = string("attn_weights_65_transpose_x_0"), val = bool(false)]; bool attn_weights_65_transpose_y_0 = const()[name = string("attn_weights_65_transpose_y_0"), val = bool(false)]; tensor attn_weights_65_cast_fp16 = matmul(transpose_x = attn_weights_65_transpose_x_0, transpose_y = attn_weights_65_transpose_y_0, x = var_2372_cast_fp16_0, y = var_2385_0)[name = string("attn_weights_65_cast_fp16")]; fp16 var_2388_to_fp16 = const()[name = string("op_2388_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_67_cast_fp16 = mul(x = attn_weights_65_cast_fp16, y = var_2388_to_fp16)[name = string("attn_weights_67_cast_fp16")]; tensor attn_weights_69_cast_fp16 = add(x = attn_weights_67_cast_fp16, y = attn_mask_1)[name = string("attn_weights_69_cast_fp16")]; int32 var_2392 = const()[name = string("op_2392"), val = int32(-2)]; tensor attn_weights_71_cast_fp16 = softmax(axis = var_2392, x = attn_weights_69_cast_fp16)[name = string("attn_weights_71_cast_fp16")]; bool var_2398_transpose_x_1 = const()[name = string("op_2398_transpose_x_1"), val = bool(true)]; bool var_2398_transpose_y_1 = const()[name = string("op_2398_transpose_y_1"), val = bool(false)]; tensor var_2398_cast_fp16 = matmul(transpose_x = var_2398_transpose_x_1, transpose_y = var_2398_transpose_y_1, x = attn_weights_71_cast_fp16, y = var_2382_cast_fp16_0)[name = string("op_2398_cast_fp16")]; bool attn_weights_73_transpose_x_0 = const()[name = string("attn_weights_73_transpose_x_0"), val = bool(false)]; bool attn_weights_73_transpose_y_0 = const()[name = string("attn_weights_73_transpose_y_0"), val = bool(false)]; tensor attn_weights_73_cast_fp16 = matmul(transpose_x = attn_weights_73_transpose_x_0, transpose_y = attn_weights_73_transpose_y_0, x = var_2372_cast_fp16_1, y = var_2385_1)[name = string("attn_weights_73_cast_fp16")]; fp16 var_2400_to_fp16 = const()[name = string("op_2400_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_75_cast_fp16 = mul(x = attn_weights_73_cast_fp16, y = var_2400_to_fp16)[name = string("attn_weights_75_cast_fp16")]; tensor attn_weights_77_cast_fp16 = add(x = attn_weights_75_cast_fp16, y = attn_mask_1)[name = string("attn_weights_77_cast_fp16")]; int32 var_2404 = const()[name = string("op_2404"), val = int32(-2)]; tensor attn_weights_79_cast_fp16 = softmax(axis = var_2404, x = attn_weights_77_cast_fp16)[name = string("attn_weights_79_cast_fp16")]; bool attn_output_33_transpose_x_1 = const()[name = string("attn_output_33_transpose_x_1"), val = bool(true)]; bool attn_output_33_transpose_y_1 = const()[name = string("attn_output_33_transpose_y_1"), val = bool(false)]; tensor attn_output_33_cast_fp16 = matmul(transpose_x = attn_output_33_transpose_x_1, transpose_y = attn_output_33_transpose_y_1, x = attn_weights_79_cast_fp16, y = var_2382_cast_fp16_1)[name = string("attn_output_33_cast_fp16")]; int32 var_2412 = const()[name = string("op_2412"), val = int32(1)]; bool attn_output_35_interleave_0 = const()[name = string("attn_output_35_interleave_0"), val = bool(false)]; tensor attn_output_35_cast_fp16 = concat(axis = var_2412, interleave = attn_output_35_interleave_0, values = (var_2398_cast_fp16, attn_output_33_cast_fp16))[name = string("attn_output_35_cast_fp16")]; tensor var_2416_perm_0 = const()[name = string("op_2416_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_59x = const()[name = string("concat_59x"), val = tensor([1, 2048, 1, -1])]; tensor var_2416_cast_fp16 = transpose(perm = var_2416_perm_0, x = attn_output_35_cast_fp16)[name = string("transpose_69")]; tensor attn_output_39_cast_fp16 = reshape(shape = concat_59x, x = var_2416_cast_fp16)[name = string("attn_output_39_cast_fp16")]; tensor hidden_states_43_strides_0 = const()[name = string("hidden_states_43_strides_0"), val = tensor([1, 1])]; string hidden_states_43_pad_type_0 = const()[name = string("hidden_states_43_pad_type_0"), val = string("valid")]; tensor hidden_states_43_pad_0 = const()[name = string("hidden_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_43_dilations_0 = const()[name = string("hidden_states_43_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_43_groups_0 = const()[name = string("hidden_states_43_groups_0"), val = int32(1)]; tensor hidden_states_43_cast_fp16 = conv(dilations = hidden_states_43_dilations_0, groups = hidden_states_43_groups_0, pad = hidden_states_43_pad_0, pad_type = hidden_states_43_pad_type_0, strides = hidden_states_43_strides_0, weight = layers_4_self_attn_o_proj_weight_cast_fp16, x = attn_output_39_cast_fp16)[name = string("hidden_states_43_cast_fp16")]; tensor hidden_states_45_cast_fp16 = add(x = hidden_states_39_cast_fp16, y = hidden_states_43_cast_fp16)[name = string("hidden_states_45_cast_fp16")]; fp16 const_48_promoted_to_fp16 = const()[name = string("const_48_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2449_cast_fp16 = mul(x = hidden_states_45_cast_fp16, y = const_48_promoted_to_fp16)[name = string("op_2449_cast_fp16")]; int32 var_2447 = const()[name = string("op_2447"), val = int32(1)]; bool doubled_37_interleave_0 = const()[name = string("doubled_37_interleave_0"), val = bool(false)]; tensor doubled_37_cast_fp16 = concat(axis = var_2447, interleave = doubled_37_interleave_0, values = (hidden_states_45_cast_fp16, var_2449_cast_fp16))[name = string("doubled_37_cast_fp16")]; tensor out_19_axes_0 = const()[name = string("out_19_axes_0"), val = tensor([1])]; tensor out_19_gamma_0_to_fp16 = const()[name = string("out_19_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283129280)))]; fp16 var_2459_to_fp16 = const()[name = string("op_2459_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_19_cast_fp16 = layer_norm(axes = out_19_axes_0, epsilon = var_2459_to_fp16, gamma = out_19_gamma_0_to_fp16, x = doubled_37_cast_fp16)[name = string("out_19_cast_fp16")]; tensor var_2470_split_sizes_0 = const()[name = string("op_2470_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2470_axis_0 = const()[name = string("op_2470_axis_0"), val = int32(1)]; tensor var_2470_cast_fp16_0, tensor var_2470_cast_fp16_1 = split(axis = var_2470_axis_0, split_sizes = var_2470_split_sizes_0, x = out_19_cast_fp16)[name = string("op_2470_cast_fp16")]; tensor input_9_strides_0 = const()[name = string("input_9_strides_0"), val = tensor([1, 1])]; string input_9_pad_type_0 = const()[name = string("input_9_pad_type_0"), val = string("valid")]; tensor input_9_pad_0 = const()[name = string("input_9_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_9_dilations_0 = const()[name = string("input_9_dilations_0"), val = tensor([1, 1])]; int32 input_9_groups_0 = const()[name = string("input_9_groups_0"), val = int32(1)]; tensor input_9_cast_fp16 = conv(dilations = input_9_dilations_0, groups = input_9_groups_0, pad = input_9_pad_0, pad_type = input_9_pad_type_0, strides = input_9_strides_0, weight = layers_4_mlp_gate_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("input_9_cast_fp16")]; tensor var_2487_cast_fp16 = silu(x = input_9_cast_fp16)[name = string("op_2487_cast_fp16")]; tensor var_2493_strides_0 = const()[name = string("op_2493_strides_0"), val = tensor([1, 1])]; string var_2493_pad_type_0 = const()[name = string("op_2493_pad_type_0"), val = string("valid")]; tensor var_2493_pad_0 = const()[name = string("op_2493_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2493_dilations_0 = const()[name = string("op_2493_dilations_0"), val = tensor([1, 1])]; int32 var_2493_groups_0 = const()[name = string("op_2493_groups_0"), val = int32(1)]; tensor var_2493_cast_fp16 = conv(dilations = var_2493_dilations_0, groups = var_2493_groups_0, pad = var_2493_pad_0, pad_type = var_2493_pad_type_0, strides = var_2493_strides_0, weight = layers_4_mlp_up_proj_weight_cast_fp16, x = var_2470_cast_fp16_0)[name = string("op_2493_cast_fp16")]; tensor x_49_cast_fp16 = mul(x = var_2487_cast_fp16, y = var_2493_cast_fp16)[name = string("x_49_cast_fp16")]; tensor hidden_states_47_strides_0 = const()[name = string("hidden_states_47_strides_0"), val = tensor([1, 1])]; string hidden_states_47_pad_type_0 = const()[name = string("hidden_states_47_pad_type_0"), val = string("valid")]; tensor hidden_states_47_pad_0 = const()[name = string("hidden_states_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_47_dilations_0 = const()[name = string("hidden_states_47_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_47_groups_0 = const()[name = string("hidden_states_47_groups_0"), val = int32(1)]; tensor hidden_states_47_cast_fp16 = conv(dilations = hidden_states_47_dilations_0, groups = hidden_states_47_groups_0, pad = hidden_states_47_pad_0, pad_type = hidden_states_47_pad_type_0, strides = hidden_states_47_strides_0, weight = layers_4_mlp_down_proj_weight_cast_fp16, x = x_49_cast_fp16)[name = string("hidden_states_47_cast_fp16")]; tensor hidden_states_49_cast_fp16 = add(x = hidden_states_45_cast_fp16, y = hidden_states_47_cast_fp16)[name = string("hidden_states_49_cast_fp16")]; fp16 const_50_promoted_to_fp16 = const()[name = string("const_50_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2511_cast_fp16 = mul(x = hidden_states_49_cast_fp16, y = const_50_promoted_to_fp16)[name = string("op_2511_cast_fp16")]; int32 var_2509 = const()[name = string("op_2509"), val = int32(1)]; bool doubled_41_interleave_0 = const()[name = string("doubled_41_interleave_0"), val = bool(false)]; tensor doubled_41_cast_fp16 = concat(axis = var_2509, interleave = doubled_41_interleave_0, values = (hidden_states_49_cast_fp16, var_2511_cast_fp16))[name = string("doubled_41_cast_fp16")]; tensor out_21_axes_0 = const()[name = string("out_21_axes_0"), val = tensor([1])]; tensor out_21_gamma_0_to_fp16 = const()[name = string("out_21_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283137536)))]; fp16 var_2521_to_fp16 = const()[name = string("op_2521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_21_cast_fp16 = layer_norm(axes = out_21_axes_0, epsilon = var_2521_to_fp16, gamma = out_21_gamma_0_to_fp16, x = doubled_41_cast_fp16)[name = string("out_21_cast_fp16")]; tensor var_2532_split_sizes_0 = const()[name = string("op_2532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2532_axis_0 = const()[name = string("op_2532_axis_0"), val = int32(1)]; tensor var_2532_cast_fp16_0, tensor var_2532_cast_fp16_1 = split(axis = var_2532_axis_0, split_sizes = var_2532_split_sizes_0, x = out_21_cast_fp16)[name = string("op_2532_cast_fp16")]; tensor layers_5_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1283145792)))]; tensor query_states_31_strides_0 = const()[name = string("query_states_31_strides_0"), val = tensor([1, 1])]; string query_states_31_pad_type_0 = const()[name = string("query_states_31_pad_type_0"), val = string("valid")]; tensor query_states_31_pad_0 = const()[name = string("query_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_31_dilations_0 = const()[name = string("query_states_31_dilations_0"), val = tensor([1, 1])]; int32 query_states_31_groups_0 = const()[name = string("query_states_31_groups_0"), val = int32(1)]; tensor query_states_31_cast_fp16 = conv(dilations = query_states_31_dilations_0, groups = query_states_31_groups_0, pad = query_states_31_pad_0, pad_type = query_states_31_pad_type_0, strides = query_states_31_strides_0, weight = layers_5_self_attn_q_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("query_states_31_cast_fp16")]; tensor layers_5_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_5_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1291534464)))]; tensor key_states_51_strides_0 = const()[name = string("key_states_51_strides_0"), val = tensor([1, 1])]; string key_states_51_pad_type_0 = const()[name = string("key_states_51_pad_type_0"), val = string("valid")]; tensor key_states_51_pad_0 = const()[name = string("key_states_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_51_dilations_0 = const()[name = string("key_states_51_dilations_0"), val = tensor([1, 1])]; int32 key_states_51_groups_0 = const()[name = string("key_states_51_groups_0"), val = int32(1)]; tensor key_states_51_cast_fp16 = conv(dilations = key_states_51_dilations_0, groups = key_states_51_groups_0, pad = key_states_51_pad_0, pad_type = key_states_51_pad_type_0, strides = key_states_51_strides_0, weight = layers_5_self_attn_k_proj_weight_to_fp16, x = var_2532_cast_fp16_0)[name = string("key_states_51_cast_fp16")]; tensor value_states_31_strides_0 = const()[name = string("value_states_31_strides_0"), val = tensor([1, 1])]; string value_states_31_pad_type_0 = const()[name = string("value_states_31_pad_type_0"), val = string("valid")]; tensor value_states_31_pad_0 = const()[name = string("value_states_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_31_dilations_0 = const()[name = string("value_states_31_dilations_0"), val = tensor([1, 1])]; int32 value_states_31_groups_0 = const()[name = string("value_states_31_groups_0"), val = int32(1)]; tensor value_states_31_cast_fp16 = conv(dilations = value_states_31_dilations_0, groups = value_states_31_groups_0, pad = value_states_31_pad_0, pad_type = value_states_31_pad_type_0, strides = value_states_31_strides_0, weight = layers_5_self_attn_v_proj_weight_cast_fp16, x = var_2532_cast_fp16_0)[name = string("value_states_31_cast_fp16")]; tensor concat_60x = const()[name = string("concat_60x"), val = tensor([1, 16, 128, -1])]; tensor x_51_cast_fp16 = reshape(shape = concat_60x, x = query_states_31_cast_fp16)[name = string("x_51_cast_fp16")]; tensor concat_61x = const()[name = string("concat_61x"), val = tensor([1, 2, 128, -1])]; tensor var_2589_cast_fp16 = reshape(shape = concat_61x, x = key_states_51_cast_fp16)[name = string("op_2589_cast_fp16")]; tensor concat_62x = const()[name = string("concat_62x"), val = tensor([1, 2, 128, -1])]; tensor var_2596_cast_fp16 = reshape(shape = concat_62x, x = value_states_31_cast_fp16)[name = string("op_2596_cast_fp16")]; tensor var_2600_cast_fp16 = mul(x = x_51_cast_fp16, y = var_869_cast_fp16)[name = string("op_2600_cast_fp16")]; tensor var_2601_split_sizes_0 = const()[name = string("op_2601_split_sizes_0"), val = tensor([64, 64])]; int32 var_2601_axis_0 = const()[name = string("op_2601_axis_0"), val = int32(-2)]; tensor var_2601_cast_fp16_0, tensor var_2601_cast_fp16_1 = split(axis = var_2601_axis_0, split_sizes = var_2601_split_sizes_0, x = x_51_cast_fp16)[name = string("op_2601_cast_fp16")]; fp16 const_52_promoted_to_fp16 = const()[name = string("const_52_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2603_cast_fp16 = mul(x = var_2601_cast_fp16_1, y = const_52_promoted_to_fp16)[name = string("op_2603_cast_fp16")]; int32 var_2605 = const()[name = string("op_2605"), val = int32(-2)]; bool var_2606_interleave_0 = const()[name = string("op_2606_interleave_0"), val = bool(false)]; tensor var_2606_cast_fp16 = concat(axis = var_2605, interleave = var_2606_interleave_0, values = (var_2603_cast_fp16, var_2601_cast_fp16_0))[name = string("op_2606_cast_fp16")]; tensor var_2607_cast_fp16 = mul(x = var_2606_cast_fp16, y = var_878_cast_fp16)[name = string("op_2607_cast_fp16")]; tensor query_states_33_cast_fp16 = add(x = var_2600_cast_fp16, y = var_2607_cast_fp16)[name = string("query_states_33_cast_fp16")]; tensor var_2613_cast_fp16 = mul(x = var_2589_cast_fp16, y = var_869_cast_fp16)[name = string("op_2613_cast_fp16")]; tensor var_2614_split_sizes_0 = const()[name = string("op_2614_split_sizes_0"), val = tensor([64, 64])]; int32 var_2614_axis_0 = const()[name = string("op_2614_axis_0"), val = int32(-2)]; tensor var_2614_cast_fp16_0, tensor var_2614_cast_fp16_1 = split(axis = var_2614_axis_0, split_sizes = var_2614_split_sizes_0, x = var_2589_cast_fp16)[name = string("op_2614_cast_fp16")]; fp16 const_53_promoted_to_fp16 = const()[name = string("const_53_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2616_cast_fp16 = mul(x = var_2614_cast_fp16_1, y = const_53_promoted_to_fp16)[name = string("op_2616_cast_fp16")]; int32 var_2618 = const()[name = string("op_2618"), val = int32(-2)]; bool var_2619_interleave_0 = const()[name = string("op_2619_interleave_0"), val = bool(false)]; tensor var_2619_cast_fp16 = concat(axis = var_2618, interleave = var_2619_interleave_0, values = (var_2616_cast_fp16, var_2614_cast_fp16_0))[name = string("op_2619_cast_fp16")]; tensor var_2620_cast_fp16 = mul(x = var_2619_cast_fp16, y = var_878_cast_fp16)[name = string("op_2620_cast_fp16")]; tensor key_states_55_cast_fp16 = add(x = var_2613_cast_fp16, y = var_2620_cast_fp16)[name = string("key_states_55_cast_fp16")]; tensor expand_dims_60 = const()[name = string("expand_dims_60"), val = tensor([5])]; tensor expand_dims_61 = const()[name = string("expand_dims_61"), val = tensor([0])]; tensor expand_dims_63 = const()[name = string("expand_dims_63"), val = tensor([0])]; int32 concat_65_axis_0 = const()[name = string("concat_65_axis_0"), val = int32(0)]; bool concat_65_interleave_0 = const()[name = string("concat_65_interleave_0"), val = bool(false)]; tensor concat_65 = concat(axis = concat_65_axis_0, interleave = concat_65_interleave_0, values = (expand_dims_60, expand_dims_61, position_id, expand_dims_63))[name = string("concat_65")]; tensor expand_dims_64 = const()[name = string("expand_dims_64"), val = tensor([6])]; tensor concat_66_values1_0 = const()[name = string("concat_66_values1_0"), val = tensor([0])]; tensor concat_66_values3_0 = const()[name = string("concat_66_values3_0"), val = tensor([0])]; int32 concat_66_axis_0 = const()[name = string("concat_66_axis_0"), val = int32(0)]; bool concat_66_interleave_0 = const()[name = string("concat_66_interleave_0"), val = bool(false)]; tensor concat_66 = concat(axis = concat_66_axis_0, interleave = concat_66_interleave_0, values = (expand_dims_64, concat_66_values1_0, cache_position_end, concat_66_values3_0))[name = string("concat_66")]; tensor key_states_57_perm_0 = const()[name = string("key_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_6_stride_0 = const()[name = string("key_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_57_cast_fp16 = transpose(perm = key_states_57_perm_0, x = key_states_55_cast_fp16)[name = string("transpose_68")]; tensor key_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = key_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = key_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_6_squeeze_mask_0, stride = key_cache_internal_tensor_assign_6_stride_0, update = key_states_57_cast_fp16, x = coreml_update_state_8)[name = string("key_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_6_cast_fp16, input = key_cache)[name = string("coreml_update_state_10_write_state")]; tensor coreml_update_state_10 = read_state(input = key_cache)[name = string("coreml_update_state_10")]; tensor value_states_33_perm_0 = const()[name = string("value_states_33_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_6_stride_0 = const()[name = string("value_cache_internal_tensor_assign_6_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_6_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_6_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_6_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_6_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_33_cast_fp16 = transpose(perm = value_states_33_perm_0, x = var_2596_cast_fp16)[name = string("transpose_67")]; tensor value_cache_internal_tensor_assign_6_cast_fp16 = slice_update(begin = concat_65, begin_mask = value_cache_internal_tensor_assign_6_begin_mask_0, end = concat_66, end_mask = value_cache_internal_tensor_assign_6_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_6_squeeze_mask_0, stride = value_cache_internal_tensor_assign_6_stride_0, update = value_states_33_cast_fp16, x = coreml_update_state_9)[name = string("value_cache_internal_tensor_assign_6_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_6_cast_fp16, input = value_cache)[name = string("coreml_update_state_11_write_state")]; tensor coreml_update_state_11 = read_state(input = value_cache)[name = string("coreml_update_state_11")]; tensor var_2690_begin_0 = const()[name = string("op_2690_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2690_end_0 = const()[name = string("op_2690_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2690_end_mask_0 = const()[name = string("op_2690_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2690_cast_fp16 = slice_by_index(begin = var_2690_begin_0, end = var_2690_end_0, end_mask = var_2690_end_mask_0, x = coreml_update_state_10)[name = string("op_2690_cast_fp16")]; tensor tile_10 = const()[name = string("tile_10"), val = tensor([1, 1])]; int32 var_2693_axis_0 = const()[name = string("op_2693_axis_0"), val = int32(1)]; tensor var_2693_cast_fp16_0, tensor var_2693_cast_fp16_1 = split(axis = var_2693_axis_0, split_sizes = tile_10, x = var_2690_cast_fp16)[name = string("op_2693_cast_fp16")]; tensor var_2700_begin_0 = const()[name = string("op_2700_begin_0"), val = tensor([5, 0, 0, 0])]; tensor var_2700_end_0 = const()[name = string("op_2700_end_0"), val = tensor([6, 2, 2048, 128])]; tensor var_2700_end_mask_0 = const()[name = string("op_2700_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_2700_cast_fp16 = slice_by_index(begin = var_2700_begin_0, end = var_2700_end_0, end_mask = var_2700_end_mask_0, x = coreml_update_state_11)[name = string("op_2700_cast_fp16")]; tensor tile_11 = const()[name = string("tile_11"), val = tensor([1, 1])]; int32 var_2703_axis_0 = const()[name = string("op_2703_axis_0"), val = int32(1)]; tensor var_2703_cast_fp16_0, tensor var_2703_cast_fp16_1 = split(axis = var_2703_axis_0, split_sizes = tile_11, x = var_2700_cast_fp16)[name = string("op_2703_cast_fp16")]; tensor var_2706_split_sizes_0 = const()[name = string("op_2706_split_sizes_0"), val = tensor([8, 8])]; int32 var_2706_axis_0 = const()[name = string("op_2706_axis_0"), val = int32(1)]; tensor var_2706_0, tensor var_2706_1 = split(axis = var_2706_axis_0, split_sizes = var_2706_split_sizes_0, x = query_states_33_cast_fp16)[name = string("op_2706")]; bool attn_weights_81_transpose_x_0 = const()[name = string("attn_weights_81_transpose_x_0"), val = bool(false)]; bool attn_weights_81_transpose_y_0 = const()[name = string("attn_weights_81_transpose_y_0"), val = bool(false)]; tensor attn_weights_81_cast_fp16 = matmul(transpose_x = attn_weights_81_transpose_x_0, transpose_y = attn_weights_81_transpose_y_0, x = var_2693_cast_fp16_0, y = var_2706_0)[name = string("attn_weights_81_cast_fp16")]; fp16 var_2709_to_fp16 = const()[name = string("op_2709_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_83_cast_fp16 = mul(x = attn_weights_81_cast_fp16, y = var_2709_to_fp16)[name = string("attn_weights_83_cast_fp16")]; tensor attn_weights_85_cast_fp16 = add(x = attn_weights_83_cast_fp16, y = attn_mask_1)[name = string("attn_weights_85_cast_fp16")]; int32 var_2713 = const()[name = string("op_2713"), val = int32(-2)]; tensor attn_weights_87_cast_fp16 = softmax(axis = var_2713, x = attn_weights_85_cast_fp16)[name = string("attn_weights_87_cast_fp16")]; bool var_2719_transpose_x_1 = const()[name = string("op_2719_transpose_x_1"), val = bool(true)]; bool var_2719_transpose_y_1 = const()[name = string("op_2719_transpose_y_1"), val = bool(false)]; tensor var_2719_cast_fp16 = matmul(transpose_x = var_2719_transpose_x_1, transpose_y = var_2719_transpose_y_1, x = attn_weights_87_cast_fp16, y = var_2703_cast_fp16_0)[name = string("op_2719_cast_fp16")]; bool attn_weights_89_transpose_x_0 = const()[name = string("attn_weights_89_transpose_x_0"), val = bool(false)]; bool attn_weights_89_transpose_y_0 = const()[name = string("attn_weights_89_transpose_y_0"), val = bool(false)]; tensor attn_weights_89_cast_fp16 = matmul(transpose_x = attn_weights_89_transpose_x_0, transpose_y = attn_weights_89_transpose_y_0, x = var_2693_cast_fp16_1, y = var_2706_1)[name = string("attn_weights_89_cast_fp16")]; fp16 var_2721_to_fp16 = const()[name = string("op_2721_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_91_cast_fp16 = mul(x = attn_weights_89_cast_fp16, y = var_2721_to_fp16)[name = string("attn_weights_91_cast_fp16")]; tensor attn_weights_93_cast_fp16 = add(x = attn_weights_91_cast_fp16, y = attn_mask_1)[name = string("attn_weights_93_cast_fp16")]; int32 var_2725 = const()[name = string("op_2725"), val = int32(-2)]; tensor attn_weights_95_cast_fp16 = softmax(axis = var_2725, x = attn_weights_93_cast_fp16)[name = string("attn_weights_95_cast_fp16")]; bool attn_output_41_transpose_x_1 = const()[name = string("attn_output_41_transpose_x_1"), val = bool(true)]; bool attn_output_41_transpose_y_1 = const()[name = string("attn_output_41_transpose_y_1"), val = bool(false)]; tensor attn_output_41_cast_fp16 = matmul(transpose_x = attn_output_41_transpose_x_1, transpose_y = attn_output_41_transpose_y_1, x = attn_weights_95_cast_fp16, y = var_2703_cast_fp16_1)[name = string("attn_output_41_cast_fp16")]; int32 var_2733 = const()[name = string("op_2733"), val = int32(1)]; bool attn_output_43_interleave_0 = const()[name = string("attn_output_43_interleave_0"), val = bool(false)]; tensor attn_output_43_cast_fp16 = concat(axis = var_2733, interleave = attn_output_43_interleave_0, values = (var_2719_cast_fp16, attn_output_41_cast_fp16))[name = string("attn_output_43_cast_fp16")]; tensor var_2737_perm_0 = const()[name = string("op_2737_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_71x = const()[name = string("concat_71x"), val = tensor([1, 2048, 1, -1])]; tensor var_2737_cast_fp16 = transpose(perm = var_2737_perm_0, x = attn_output_43_cast_fp16)[name = string("transpose_66")]; tensor attn_output_47_cast_fp16 = reshape(shape = concat_71x, x = var_2737_cast_fp16)[name = string("attn_output_47_cast_fp16")]; tensor hidden_states_53_strides_0 = const()[name = string("hidden_states_53_strides_0"), val = tensor([1, 1])]; string hidden_states_53_pad_type_0 = const()[name = string("hidden_states_53_pad_type_0"), val = string("valid")]; tensor hidden_states_53_pad_0 = const()[name = string("hidden_states_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_53_dilations_0 = const()[name = string("hidden_states_53_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_53_groups_0 = const()[name = string("hidden_states_53_groups_0"), val = int32(1)]; tensor hidden_states_53_cast_fp16 = conv(dilations = hidden_states_53_dilations_0, groups = hidden_states_53_groups_0, pad = hidden_states_53_pad_0, pad_type = hidden_states_53_pad_type_0, strides = hidden_states_53_strides_0, weight = layers_5_self_attn_o_proj_weight_cast_fp16, x = attn_output_47_cast_fp16)[name = string("hidden_states_53_cast_fp16")]; tensor hidden_states_55_cast_fp16 = add(x = hidden_states_49_cast_fp16, y = hidden_states_53_cast_fp16)[name = string("hidden_states_55_cast_fp16")]; fp16 const_58_promoted_to_fp16 = const()[name = string("const_58_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2770_cast_fp16 = mul(x = hidden_states_55_cast_fp16, y = const_58_promoted_to_fp16)[name = string("op_2770_cast_fp16")]; int32 var_2768 = const()[name = string("op_2768"), val = int32(1)]; bool doubled_45_interleave_0 = const()[name = string("doubled_45_interleave_0"), val = bool(false)]; tensor doubled_45_cast_fp16 = concat(axis = var_2768, interleave = doubled_45_interleave_0, values = (hidden_states_55_cast_fp16, var_2770_cast_fp16))[name = string("doubled_45_cast_fp16")]; tensor out_23_axes_0 = const()[name = string("out_23_axes_0"), val = tensor([1])]; tensor out_23_gamma_0_to_fp16 = const()[name = string("out_23_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292583104)))]; fp16 var_2780_to_fp16 = const()[name = string("op_2780_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_23_cast_fp16 = layer_norm(axes = out_23_axes_0, epsilon = var_2780_to_fp16, gamma = out_23_gamma_0_to_fp16, x = doubled_45_cast_fp16)[name = string("out_23_cast_fp16")]; tensor var_2791_split_sizes_0 = const()[name = string("op_2791_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2791_axis_0 = const()[name = string("op_2791_axis_0"), val = int32(1)]; tensor var_2791_cast_fp16_0, tensor var_2791_cast_fp16_1 = split(axis = var_2791_axis_0, split_sizes = var_2791_split_sizes_0, x = out_23_cast_fp16)[name = string("op_2791_cast_fp16")]; tensor layers_5_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_5_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1292591360)))]; tensor input_11_strides_0 = const()[name = string("input_11_strides_0"), val = tensor([1, 1])]; string input_11_pad_type_0 = const()[name = string("input_11_pad_type_0"), val = string("valid")]; tensor input_11_pad_0 = const()[name = string("input_11_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_11_dilations_0 = const()[name = string("input_11_dilations_0"), val = tensor([1, 1])]; int32 input_11_groups_0 = const()[name = string("input_11_groups_0"), val = int32(1)]; tensor input_11_cast_fp16 = conv(dilations = input_11_dilations_0, groups = input_11_groups_0, pad = input_11_pad_0, pad_type = input_11_pad_type_0, strides = input_11_strides_0, weight = layers_5_mlp_gate_proj_weight_to_fp16, x = var_2791_cast_fp16_0)[name = string("input_11_cast_fp16")]; tensor var_2808_cast_fp16 = silu(x = input_11_cast_fp16)[name = string("op_2808_cast_fp16")]; tensor var_2814_strides_0 = const()[name = string("op_2814_strides_0"), val = tensor([1, 1])]; string var_2814_pad_type_0 = const()[name = string("op_2814_pad_type_0"), val = string("valid")]; tensor var_2814_pad_0 = const()[name = string("op_2814_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_2814_dilations_0 = const()[name = string("op_2814_dilations_0"), val = tensor([1, 1])]; int32 var_2814_groups_0 = const()[name = string("op_2814_groups_0"), val = int32(1)]; tensor var_2814_cast_fp16 = conv(dilations = var_2814_dilations_0, groups = var_2814_groups_0, pad = var_2814_pad_0, pad_type = var_2814_pad_type_0, strides = var_2814_strides_0, weight = layers_5_mlp_up_proj_weight_cast_fp16, x = var_2791_cast_fp16_0)[name = string("op_2814_cast_fp16")]; tensor x_59_cast_fp16 = mul(x = var_2808_cast_fp16, y = var_2814_cast_fp16)[name = string("x_59_cast_fp16")]; tensor hidden_states_57_strides_0 = const()[name = string("hidden_states_57_strides_0"), val = tensor([1, 1])]; string hidden_states_57_pad_type_0 = const()[name = string("hidden_states_57_pad_type_0"), val = string("valid")]; tensor hidden_states_57_pad_0 = const()[name = string("hidden_states_57_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_57_dilations_0 = const()[name = string("hidden_states_57_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_57_groups_0 = const()[name = string("hidden_states_57_groups_0"), val = int32(1)]; tensor hidden_states_57_cast_fp16 = conv(dilations = hidden_states_57_dilations_0, groups = hidden_states_57_groups_0, pad = hidden_states_57_pad_0, pad_type = hidden_states_57_pad_type_0, strides = hidden_states_57_strides_0, weight = layers_5_mlp_down_proj_weight_cast_fp16, x = x_59_cast_fp16)[name = string("hidden_states_57_cast_fp16")]; tensor hidden_states_59_cast_fp16 = add(x = hidden_states_55_cast_fp16, y = hidden_states_57_cast_fp16)[name = string("hidden_states_59_cast_fp16")]; fp16 const_60_promoted_to_fp16 = const()[name = string("const_60_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2832_cast_fp16 = mul(x = hidden_states_59_cast_fp16, y = const_60_promoted_to_fp16)[name = string("op_2832_cast_fp16")]; int32 var_2830 = const()[name = string("op_2830"), val = int32(1)]; bool doubled_49_interleave_0 = const()[name = string("doubled_49_interleave_0"), val = bool(false)]; tensor doubled_49_cast_fp16 = concat(axis = var_2830, interleave = doubled_49_interleave_0, values = (hidden_states_59_cast_fp16, var_2832_cast_fp16))[name = string("doubled_49_cast_fp16")]; tensor out_25_axes_0 = const()[name = string("out_25_axes_0"), val = tensor([1])]; tensor out_25_gamma_0_to_fp16 = const()[name = string("out_25_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317757248)))]; fp16 var_2842_to_fp16 = const()[name = string("op_2842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_25_cast_fp16 = layer_norm(axes = out_25_axes_0, epsilon = var_2842_to_fp16, gamma = out_25_gamma_0_to_fp16, x = doubled_49_cast_fp16)[name = string("out_25_cast_fp16")]; tensor var_2853_split_sizes_0 = const()[name = string("op_2853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_2853_axis_0 = const()[name = string("op_2853_axis_0"), val = int32(1)]; tensor var_2853_cast_fp16_0, tensor var_2853_cast_fp16_1 = split(axis = var_2853_axis_0, split_sizes = var_2853_split_sizes_0, x = out_25_cast_fp16)[name = string("op_2853_cast_fp16")]; tensor layers_6_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1317765504)))]; tensor query_states_37_strides_0 = const()[name = string("query_states_37_strides_0"), val = tensor([1, 1])]; string query_states_37_pad_type_0 = const()[name = string("query_states_37_pad_type_0"), val = string("valid")]; tensor query_states_37_pad_0 = const()[name = string("query_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_37_dilations_0 = const()[name = string("query_states_37_dilations_0"), val = tensor([1, 1])]; int32 query_states_37_groups_0 = const()[name = string("query_states_37_groups_0"), val = int32(1)]; tensor query_states_37_cast_fp16 = conv(dilations = query_states_37_dilations_0, groups = query_states_37_groups_0, pad = query_states_37_pad_0, pad_type = query_states_37_pad_type_0, strides = query_states_37_strides_0, weight = layers_6_self_attn_q_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("query_states_37_cast_fp16")]; tensor layers_6_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_6_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1326154176)))]; tensor key_states_61_strides_0 = const()[name = string("key_states_61_strides_0"), val = tensor([1, 1])]; string key_states_61_pad_type_0 = const()[name = string("key_states_61_pad_type_0"), val = string("valid")]; tensor key_states_61_pad_0 = const()[name = string("key_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_61_dilations_0 = const()[name = string("key_states_61_dilations_0"), val = tensor([1, 1])]; int32 key_states_61_groups_0 = const()[name = string("key_states_61_groups_0"), val = int32(1)]; tensor key_states_61_cast_fp16 = conv(dilations = key_states_61_dilations_0, groups = key_states_61_groups_0, pad = key_states_61_pad_0, pad_type = key_states_61_pad_type_0, strides = key_states_61_strides_0, weight = layers_6_self_attn_k_proj_weight_to_fp16, x = var_2853_cast_fp16_0)[name = string("key_states_61_cast_fp16")]; tensor value_states_37_strides_0 = const()[name = string("value_states_37_strides_0"), val = tensor([1, 1])]; string value_states_37_pad_type_0 = const()[name = string("value_states_37_pad_type_0"), val = string("valid")]; tensor value_states_37_pad_0 = const()[name = string("value_states_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_37_dilations_0 = const()[name = string("value_states_37_dilations_0"), val = tensor([1, 1])]; int32 value_states_37_groups_0 = const()[name = string("value_states_37_groups_0"), val = int32(1)]; tensor value_states_37_cast_fp16 = conv(dilations = value_states_37_dilations_0, groups = value_states_37_groups_0, pad = value_states_37_pad_0, pad_type = value_states_37_pad_type_0, strides = value_states_37_strides_0, weight = layers_6_self_attn_v_proj_weight_cast_fp16, x = var_2853_cast_fp16_0)[name = string("value_states_37_cast_fp16")]; tensor concat_72x = const()[name = string("concat_72x"), val = tensor([1, 16, 128, -1])]; tensor x_61_cast_fp16 = reshape(shape = concat_72x, x = query_states_37_cast_fp16)[name = string("x_61_cast_fp16")]; tensor concat_73x = const()[name = string("concat_73x"), val = tensor([1, 2, 128, -1])]; tensor var_2910_cast_fp16 = reshape(shape = concat_73x, x = key_states_61_cast_fp16)[name = string("op_2910_cast_fp16")]; tensor concat_74x = const()[name = string("concat_74x"), val = tensor([1, 2, 128, -1])]; tensor var_2917_cast_fp16 = reshape(shape = concat_74x, x = value_states_37_cast_fp16)[name = string("op_2917_cast_fp16")]; tensor var_2921_cast_fp16 = mul(x = x_61_cast_fp16, y = var_869_cast_fp16)[name = string("op_2921_cast_fp16")]; tensor var_2922_split_sizes_0 = const()[name = string("op_2922_split_sizes_0"), val = tensor([64, 64])]; int32 var_2922_axis_0 = const()[name = string("op_2922_axis_0"), val = int32(-2)]; tensor var_2922_cast_fp16_0, tensor var_2922_cast_fp16_1 = split(axis = var_2922_axis_0, split_sizes = var_2922_split_sizes_0, x = x_61_cast_fp16)[name = string("op_2922_cast_fp16")]; fp16 const_62_promoted_to_fp16 = const()[name = string("const_62_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2924_cast_fp16 = mul(x = var_2922_cast_fp16_1, y = const_62_promoted_to_fp16)[name = string("op_2924_cast_fp16")]; int32 var_2926 = const()[name = string("op_2926"), val = int32(-2)]; bool var_2927_interleave_0 = const()[name = string("op_2927_interleave_0"), val = bool(false)]; tensor var_2927_cast_fp16 = concat(axis = var_2926, interleave = var_2927_interleave_0, values = (var_2924_cast_fp16, var_2922_cast_fp16_0))[name = string("op_2927_cast_fp16")]; tensor var_2928_cast_fp16 = mul(x = var_2927_cast_fp16, y = var_878_cast_fp16)[name = string("op_2928_cast_fp16")]; tensor query_states_39_cast_fp16 = add(x = var_2921_cast_fp16, y = var_2928_cast_fp16)[name = string("query_states_39_cast_fp16")]; tensor var_2934_cast_fp16 = mul(x = var_2910_cast_fp16, y = var_869_cast_fp16)[name = string("op_2934_cast_fp16")]; tensor var_2935_split_sizes_0 = const()[name = string("op_2935_split_sizes_0"), val = tensor([64, 64])]; int32 var_2935_axis_0 = const()[name = string("op_2935_axis_0"), val = int32(-2)]; tensor var_2935_cast_fp16_0, tensor var_2935_cast_fp16_1 = split(axis = var_2935_axis_0, split_sizes = var_2935_split_sizes_0, x = var_2910_cast_fp16)[name = string("op_2935_cast_fp16")]; fp16 const_63_promoted_to_fp16 = const()[name = string("const_63_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_2937_cast_fp16 = mul(x = var_2935_cast_fp16_1, y = const_63_promoted_to_fp16)[name = string("op_2937_cast_fp16")]; int32 var_2939 = const()[name = string("op_2939"), val = int32(-2)]; bool var_2940_interleave_0 = const()[name = string("op_2940_interleave_0"), val = bool(false)]; tensor var_2940_cast_fp16 = concat(axis = var_2939, interleave = var_2940_interleave_0, values = (var_2937_cast_fp16, var_2935_cast_fp16_0))[name = string("op_2940_cast_fp16")]; tensor var_2941_cast_fp16 = mul(x = var_2940_cast_fp16, y = var_878_cast_fp16)[name = string("op_2941_cast_fp16")]; tensor key_states_65_cast_fp16 = add(x = var_2934_cast_fp16, y = var_2941_cast_fp16)[name = string("key_states_65_cast_fp16")]; tensor expand_dims_72 = const()[name = string("expand_dims_72"), val = tensor([6])]; tensor expand_dims_73 = const()[name = string("expand_dims_73"), val = tensor([0])]; tensor expand_dims_75 = const()[name = string("expand_dims_75"), val = tensor([0])]; int32 concat_77_axis_0 = const()[name = string("concat_77_axis_0"), val = int32(0)]; bool concat_77_interleave_0 = const()[name = string("concat_77_interleave_0"), val = bool(false)]; tensor concat_77 = concat(axis = concat_77_axis_0, interleave = concat_77_interleave_0, values = (expand_dims_72, expand_dims_73, position_id, expand_dims_75))[name = string("concat_77")]; tensor expand_dims_76 = const()[name = string("expand_dims_76"), val = tensor([7])]; tensor concat_78_values1_0 = const()[name = string("concat_78_values1_0"), val = tensor([0])]; tensor concat_78_values3_0 = const()[name = string("concat_78_values3_0"), val = tensor([0])]; int32 concat_78_axis_0 = const()[name = string("concat_78_axis_0"), val = int32(0)]; bool concat_78_interleave_0 = const()[name = string("concat_78_interleave_0"), val = bool(false)]; tensor concat_78 = concat(axis = concat_78_axis_0, interleave = concat_78_interleave_0, values = (expand_dims_76, concat_78_values1_0, cache_position_end, concat_78_values3_0))[name = string("concat_78")]; tensor key_states_67_perm_0 = const()[name = string("key_states_67_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_7_stride_0 = const()[name = string("key_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_67_cast_fp16 = transpose(perm = key_states_67_perm_0, x = key_states_65_cast_fp16)[name = string("transpose_65")]; tensor key_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = key_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = key_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_7_squeeze_mask_0, stride = key_cache_internal_tensor_assign_7_stride_0, update = key_states_67_cast_fp16, x = coreml_update_state_10)[name = string("key_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_7_cast_fp16, input = key_cache)[name = string("coreml_update_state_12_write_state")]; tensor coreml_update_state_12 = read_state(input = key_cache)[name = string("coreml_update_state_12")]; tensor value_states_39_perm_0 = const()[name = string("value_states_39_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_7_stride_0 = const()[name = string("value_cache_internal_tensor_assign_7_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_7_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_7_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_7_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_7_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_39_cast_fp16 = transpose(perm = value_states_39_perm_0, x = var_2917_cast_fp16)[name = string("transpose_64")]; tensor value_cache_internal_tensor_assign_7_cast_fp16 = slice_update(begin = concat_77, begin_mask = value_cache_internal_tensor_assign_7_begin_mask_0, end = concat_78, end_mask = value_cache_internal_tensor_assign_7_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_7_squeeze_mask_0, stride = value_cache_internal_tensor_assign_7_stride_0, update = value_states_39_cast_fp16, x = coreml_update_state_11)[name = string("value_cache_internal_tensor_assign_7_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_7_cast_fp16, input = value_cache)[name = string("coreml_update_state_13_write_state")]; tensor coreml_update_state_13 = read_state(input = value_cache)[name = string("coreml_update_state_13")]; tensor var_3011_begin_0 = const()[name = string("op_3011_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3011_end_0 = const()[name = string("op_3011_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3011_end_mask_0 = const()[name = string("op_3011_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3011_cast_fp16 = slice_by_index(begin = var_3011_begin_0, end = var_3011_end_0, end_mask = var_3011_end_mask_0, x = coreml_update_state_12)[name = string("op_3011_cast_fp16")]; tensor tile_12 = const()[name = string("tile_12"), val = tensor([1, 1])]; int32 var_3014_axis_0 = const()[name = string("op_3014_axis_0"), val = int32(1)]; tensor var_3014_cast_fp16_0, tensor var_3014_cast_fp16_1 = split(axis = var_3014_axis_0, split_sizes = tile_12, x = var_3011_cast_fp16)[name = string("op_3014_cast_fp16")]; tensor var_3021_begin_0 = const()[name = string("op_3021_begin_0"), val = tensor([6, 0, 0, 0])]; tensor var_3021_end_0 = const()[name = string("op_3021_end_0"), val = tensor([7, 2, 2048, 128])]; tensor var_3021_end_mask_0 = const()[name = string("op_3021_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3021_cast_fp16 = slice_by_index(begin = var_3021_begin_0, end = var_3021_end_0, end_mask = var_3021_end_mask_0, x = coreml_update_state_13)[name = string("op_3021_cast_fp16")]; tensor tile_13 = const()[name = string("tile_13"), val = tensor([1, 1])]; int32 var_3024_axis_0 = const()[name = string("op_3024_axis_0"), val = int32(1)]; tensor var_3024_cast_fp16_0, tensor var_3024_cast_fp16_1 = split(axis = var_3024_axis_0, split_sizes = tile_13, x = var_3021_cast_fp16)[name = string("op_3024_cast_fp16")]; tensor var_3027_split_sizes_0 = const()[name = string("op_3027_split_sizes_0"), val = tensor([8, 8])]; int32 var_3027_axis_0 = const()[name = string("op_3027_axis_0"), val = int32(1)]; tensor var_3027_0, tensor var_3027_1 = split(axis = var_3027_axis_0, split_sizes = var_3027_split_sizes_0, x = query_states_39_cast_fp16)[name = string("op_3027")]; bool attn_weights_97_transpose_x_0 = const()[name = string("attn_weights_97_transpose_x_0"), val = bool(false)]; bool attn_weights_97_transpose_y_0 = const()[name = string("attn_weights_97_transpose_y_0"), val = bool(false)]; tensor attn_weights_97_cast_fp16 = matmul(transpose_x = attn_weights_97_transpose_x_0, transpose_y = attn_weights_97_transpose_y_0, x = var_3014_cast_fp16_0, y = var_3027_0)[name = string("attn_weights_97_cast_fp16")]; fp16 var_3030_to_fp16 = const()[name = string("op_3030_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_99_cast_fp16 = mul(x = attn_weights_97_cast_fp16, y = var_3030_to_fp16)[name = string("attn_weights_99_cast_fp16")]; tensor attn_weights_101_cast_fp16 = add(x = attn_weights_99_cast_fp16, y = attn_mask_1)[name = string("attn_weights_101_cast_fp16")]; int32 var_3034 = const()[name = string("op_3034"), val = int32(-2)]; tensor attn_weights_103_cast_fp16 = softmax(axis = var_3034, x = attn_weights_101_cast_fp16)[name = string("attn_weights_103_cast_fp16")]; bool var_3040_transpose_x_1 = const()[name = string("op_3040_transpose_x_1"), val = bool(true)]; bool var_3040_transpose_y_1 = const()[name = string("op_3040_transpose_y_1"), val = bool(false)]; tensor var_3040_cast_fp16 = matmul(transpose_x = var_3040_transpose_x_1, transpose_y = var_3040_transpose_y_1, x = attn_weights_103_cast_fp16, y = var_3024_cast_fp16_0)[name = string("op_3040_cast_fp16")]; bool attn_weights_105_transpose_x_0 = const()[name = string("attn_weights_105_transpose_x_0"), val = bool(false)]; bool attn_weights_105_transpose_y_0 = const()[name = string("attn_weights_105_transpose_y_0"), val = bool(false)]; tensor attn_weights_105_cast_fp16 = matmul(transpose_x = attn_weights_105_transpose_x_0, transpose_y = attn_weights_105_transpose_y_0, x = var_3014_cast_fp16_1, y = var_3027_1)[name = string("attn_weights_105_cast_fp16")]; fp16 var_3042_to_fp16 = const()[name = string("op_3042_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_107_cast_fp16 = mul(x = attn_weights_105_cast_fp16, y = var_3042_to_fp16)[name = string("attn_weights_107_cast_fp16")]; tensor attn_weights_109_cast_fp16 = add(x = attn_weights_107_cast_fp16, y = attn_mask_1)[name = string("attn_weights_109_cast_fp16")]; int32 var_3046 = const()[name = string("op_3046"), val = int32(-2)]; tensor attn_weights_111_cast_fp16 = softmax(axis = var_3046, x = attn_weights_109_cast_fp16)[name = string("attn_weights_111_cast_fp16")]; bool attn_output_49_transpose_x_1 = const()[name = string("attn_output_49_transpose_x_1"), val = bool(true)]; bool attn_output_49_transpose_y_1 = const()[name = string("attn_output_49_transpose_y_1"), val = bool(false)]; tensor attn_output_49_cast_fp16 = matmul(transpose_x = attn_output_49_transpose_x_1, transpose_y = attn_output_49_transpose_y_1, x = attn_weights_111_cast_fp16, y = var_3024_cast_fp16_1)[name = string("attn_output_49_cast_fp16")]; int32 var_3054 = const()[name = string("op_3054"), val = int32(1)]; bool attn_output_51_interleave_0 = const()[name = string("attn_output_51_interleave_0"), val = bool(false)]; tensor attn_output_51_cast_fp16 = concat(axis = var_3054, interleave = attn_output_51_interleave_0, values = (var_3040_cast_fp16, attn_output_49_cast_fp16))[name = string("attn_output_51_cast_fp16")]; tensor var_3058_perm_0 = const()[name = string("op_3058_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_83x = const()[name = string("concat_83x"), val = tensor([1, 2048, 1, -1])]; tensor var_3058_cast_fp16 = transpose(perm = var_3058_perm_0, x = attn_output_51_cast_fp16)[name = string("transpose_63")]; tensor attn_output_55_cast_fp16 = reshape(shape = concat_83x, x = var_3058_cast_fp16)[name = string("attn_output_55_cast_fp16")]; tensor hidden_states_63_strides_0 = const()[name = string("hidden_states_63_strides_0"), val = tensor([1, 1])]; string hidden_states_63_pad_type_0 = const()[name = string("hidden_states_63_pad_type_0"), val = string("valid")]; tensor hidden_states_63_pad_0 = const()[name = string("hidden_states_63_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_63_dilations_0 = const()[name = string("hidden_states_63_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_63_groups_0 = const()[name = string("hidden_states_63_groups_0"), val = int32(1)]; tensor hidden_states_63_cast_fp16 = conv(dilations = hidden_states_63_dilations_0, groups = hidden_states_63_groups_0, pad = hidden_states_63_pad_0, pad_type = hidden_states_63_pad_type_0, strides = hidden_states_63_strides_0, weight = layers_6_self_attn_o_proj_weight_cast_fp16, x = attn_output_55_cast_fp16)[name = string("hidden_states_63_cast_fp16")]; tensor hidden_states_65_cast_fp16 = add(x = hidden_states_59_cast_fp16, y = hidden_states_63_cast_fp16)[name = string("hidden_states_65_cast_fp16")]; fp16 const_68_promoted_to_fp16 = const()[name = string("const_68_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3091_cast_fp16 = mul(x = hidden_states_65_cast_fp16, y = const_68_promoted_to_fp16)[name = string("op_3091_cast_fp16")]; int32 var_3089 = const()[name = string("op_3089"), val = int32(1)]; bool doubled_53_interleave_0 = const()[name = string("doubled_53_interleave_0"), val = bool(false)]; tensor doubled_53_cast_fp16 = concat(axis = var_3089, interleave = doubled_53_interleave_0, values = (hidden_states_65_cast_fp16, var_3091_cast_fp16))[name = string("doubled_53_cast_fp16")]; tensor out_27_axes_0 = const()[name = string("out_27_axes_0"), val = tensor([1])]; tensor out_27_gamma_0_to_fp16 = const()[name = string("out_27_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327202816)))]; fp16 var_3101_to_fp16 = const()[name = string("op_3101_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_27_cast_fp16 = layer_norm(axes = out_27_axes_0, epsilon = var_3101_to_fp16, gamma = out_27_gamma_0_to_fp16, x = doubled_53_cast_fp16)[name = string("out_27_cast_fp16")]; tensor var_3112_split_sizes_0 = const()[name = string("op_3112_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3112_axis_0 = const()[name = string("op_3112_axis_0"), val = int32(1)]; tensor var_3112_cast_fp16_0, tensor var_3112_cast_fp16_1 = split(axis = var_3112_axis_0, split_sizes = var_3112_split_sizes_0, x = out_27_cast_fp16)[name = string("op_3112_cast_fp16")]; tensor input_13_strides_0 = const()[name = string("input_13_strides_0"), val = tensor([1, 1])]; string input_13_pad_type_0 = const()[name = string("input_13_pad_type_0"), val = string("valid")]; tensor input_13_pad_0 = const()[name = string("input_13_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_13_dilations_0 = const()[name = string("input_13_dilations_0"), val = tensor([1, 1])]; int32 input_13_groups_0 = const()[name = string("input_13_groups_0"), val = int32(1)]; tensor input_13_cast_fp16 = conv(dilations = input_13_dilations_0, groups = input_13_groups_0, pad = input_13_pad_0, pad_type = input_13_pad_type_0, strides = input_13_strides_0, weight = layers_6_mlp_gate_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("input_13_cast_fp16")]; tensor var_3129_cast_fp16 = silu(x = input_13_cast_fp16)[name = string("op_3129_cast_fp16")]; tensor var_3135_strides_0 = const()[name = string("op_3135_strides_0"), val = tensor([1, 1])]; string var_3135_pad_type_0 = const()[name = string("op_3135_pad_type_0"), val = string("valid")]; tensor var_3135_pad_0 = const()[name = string("op_3135_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3135_dilations_0 = const()[name = string("op_3135_dilations_0"), val = tensor([1, 1])]; int32 var_3135_groups_0 = const()[name = string("op_3135_groups_0"), val = int32(1)]; tensor var_3135_cast_fp16 = conv(dilations = var_3135_dilations_0, groups = var_3135_groups_0, pad = var_3135_pad_0, pad_type = var_3135_pad_type_0, strides = var_3135_strides_0, weight = layers_6_mlp_up_proj_weight_cast_fp16, x = var_3112_cast_fp16_0)[name = string("op_3135_cast_fp16")]; tensor x_69_cast_fp16 = mul(x = var_3129_cast_fp16, y = var_3135_cast_fp16)[name = string("x_69_cast_fp16")]; tensor hidden_states_67_strides_0 = const()[name = string("hidden_states_67_strides_0"), val = tensor([1, 1])]; string hidden_states_67_pad_type_0 = const()[name = string("hidden_states_67_pad_type_0"), val = string("valid")]; tensor hidden_states_67_pad_0 = const()[name = string("hidden_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_67_dilations_0 = const()[name = string("hidden_states_67_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_67_groups_0 = const()[name = string("hidden_states_67_groups_0"), val = int32(1)]; tensor hidden_states_67_cast_fp16 = conv(dilations = hidden_states_67_dilations_0, groups = hidden_states_67_groups_0, pad = hidden_states_67_pad_0, pad_type = hidden_states_67_pad_type_0, strides = hidden_states_67_strides_0, weight = layers_6_mlp_down_proj_weight_cast_fp16, x = x_69_cast_fp16)[name = string("hidden_states_67_cast_fp16")]; tensor hidden_states_69_cast_fp16 = add(x = hidden_states_65_cast_fp16, y = hidden_states_67_cast_fp16)[name = string("hidden_states_69_cast_fp16")]; fp16 const_70_promoted_to_fp16 = const()[name = string("const_70_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3153_cast_fp16 = mul(x = hidden_states_69_cast_fp16, y = const_70_promoted_to_fp16)[name = string("op_3153_cast_fp16")]; int32 var_3151 = const()[name = string("op_3151"), val = int32(1)]; bool doubled_57_interleave_0 = const()[name = string("doubled_57_interleave_0"), val = bool(false)]; tensor doubled_57_cast_fp16 = concat(axis = var_3151, interleave = doubled_57_interleave_0, values = (hidden_states_69_cast_fp16, var_3153_cast_fp16))[name = string("doubled_57_cast_fp16")]; tensor out_29_axes_0 = const()[name = string("out_29_axes_0"), val = tensor([1])]; tensor out_29_gamma_0_to_fp16 = const()[name = string("out_29_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327211072)))]; fp16 var_3163_to_fp16 = const()[name = string("op_3163_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_29_cast_fp16 = layer_norm(axes = out_29_axes_0, epsilon = var_3163_to_fp16, gamma = out_29_gamma_0_to_fp16, x = doubled_57_cast_fp16)[name = string("out_29_cast_fp16")]; tensor var_3174_split_sizes_0 = const()[name = string("op_3174_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3174_axis_0 = const()[name = string("op_3174_axis_0"), val = int32(1)]; tensor var_3174_cast_fp16_0, tensor var_3174_cast_fp16_1 = split(axis = var_3174_axis_0, split_sizes = var_3174_split_sizes_0, x = out_29_cast_fp16)[name = string("op_3174_cast_fp16")]; tensor layers_7_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1327219328)))]; tensor query_states_43_strides_0 = const()[name = string("query_states_43_strides_0"), val = tensor([1, 1])]; string query_states_43_pad_type_0 = const()[name = string("query_states_43_pad_type_0"), val = string("valid")]; tensor query_states_43_pad_0 = const()[name = string("query_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_43_dilations_0 = const()[name = string("query_states_43_dilations_0"), val = tensor([1, 1])]; int32 query_states_43_groups_0 = const()[name = string("query_states_43_groups_0"), val = int32(1)]; tensor query_states_43_cast_fp16 = conv(dilations = query_states_43_dilations_0, groups = query_states_43_groups_0, pad = query_states_43_pad_0, pad_type = query_states_43_pad_type_0, strides = query_states_43_strides_0, weight = layers_7_self_attn_q_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("query_states_43_cast_fp16")]; tensor layers_7_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_7_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1335608000)))]; tensor key_states_71_strides_0 = const()[name = string("key_states_71_strides_0"), val = tensor([1, 1])]; string key_states_71_pad_type_0 = const()[name = string("key_states_71_pad_type_0"), val = string("valid")]; tensor key_states_71_pad_0 = const()[name = string("key_states_71_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_71_dilations_0 = const()[name = string("key_states_71_dilations_0"), val = tensor([1, 1])]; int32 key_states_71_groups_0 = const()[name = string("key_states_71_groups_0"), val = int32(1)]; tensor key_states_71_cast_fp16 = conv(dilations = key_states_71_dilations_0, groups = key_states_71_groups_0, pad = key_states_71_pad_0, pad_type = key_states_71_pad_type_0, strides = key_states_71_strides_0, weight = layers_7_self_attn_k_proj_weight_to_fp16, x = var_3174_cast_fp16_0)[name = string("key_states_71_cast_fp16")]; tensor value_states_43_strides_0 = const()[name = string("value_states_43_strides_0"), val = tensor([1, 1])]; string value_states_43_pad_type_0 = const()[name = string("value_states_43_pad_type_0"), val = string("valid")]; tensor value_states_43_pad_0 = const()[name = string("value_states_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_43_dilations_0 = const()[name = string("value_states_43_dilations_0"), val = tensor([1, 1])]; int32 value_states_43_groups_0 = const()[name = string("value_states_43_groups_0"), val = int32(1)]; tensor value_states_43_cast_fp16 = conv(dilations = value_states_43_dilations_0, groups = value_states_43_groups_0, pad = value_states_43_pad_0, pad_type = value_states_43_pad_type_0, strides = value_states_43_strides_0, weight = layers_7_self_attn_v_proj_weight_cast_fp16, x = var_3174_cast_fp16_0)[name = string("value_states_43_cast_fp16")]; tensor concat_84x = const()[name = string("concat_84x"), val = tensor([1, 16, 128, -1])]; tensor x_71_cast_fp16 = reshape(shape = concat_84x, x = query_states_43_cast_fp16)[name = string("x_71_cast_fp16")]; tensor concat_85x = const()[name = string("concat_85x"), val = tensor([1, 2, 128, -1])]; tensor var_3231_cast_fp16 = reshape(shape = concat_85x, x = key_states_71_cast_fp16)[name = string("op_3231_cast_fp16")]; tensor concat_86x = const()[name = string("concat_86x"), val = tensor([1, 2, 128, -1])]; tensor var_3238_cast_fp16 = reshape(shape = concat_86x, x = value_states_43_cast_fp16)[name = string("op_3238_cast_fp16")]; tensor var_3242_cast_fp16 = mul(x = x_71_cast_fp16, y = var_869_cast_fp16)[name = string("op_3242_cast_fp16")]; tensor var_3243_split_sizes_0 = const()[name = string("op_3243_split_sizes_0"), val = tensor([64, 64])]; int32 var_3243_axis_0 = const()[name = string("op_3243_axis_0"), val = int32(-2)]; tensor var_3243_cast_fp16_0, tensor var_3243_cast_fp16_1 = split(axis = var_3243_axis_0, split_sizes = var_3243_split_sizes_0, x = x_71_cast_fp16)[name = string("op_3243_cast_fp16")]; fp16 const_72_promoted_to_fp16 = const()[name = string("const_72_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3245_cast_fp16 = mul(x = var_3243_cast_fp16_1, y = const_72_promoted_to_fp16)[name = string("op_3245_cast_fp16")]; int32 var_3247 = const()[name = string("op_3247"), val = int32(-2)]; bool var_3248_interleave_0 = const()[name = string("op_3248_interleave_0"), val = bool(false)]; tensor var_3248_cast_fp16 = concat(axis = var_3247, interleave = var_3248_interleave_0, values = (var_3245_cast_fp16, var_3243_cast_fp16_0))[name = string("op_3248_cast_fp16")]; tensor var_3249_cast_fp16 = mul(x = var_3248_cast_fp16, y = var_878_cast_fp16)[name = string("op_3249_cast_fp16")]; tensor query_states_45_cast_fp16 = add(x = var_3242_cast_fp16, y = var_3249_cast_fp16)[name = string("query_states_45_cast_fp16")]; tensor var_3255_cast_fp16 = mul(x = var_3231_cast_fp16, y = var_869_cast_fp16)[name = string("op_3255_cast_fp16")]; tensor var_3256_split_sizes_0 = const()[name = string("op_3256_split_sizes_0"), val = tensor([64, 64])]; int32 var_3256_axis_0 = const()[name = string("op_3256_axis_0"), val = int32(-2)]; tensor var_3256_cast_fp16_0, tensor var_3256_cast_fp16_1 = split(axis = var_3256_axis_0, split_sizes = var_3256_split_sizes_0, x = var_3231_cast_fp16)[name = string("op_3256_cast_fp16")]; fp16 const_73_promoted_to_fp16 = const()[name = string("const_73_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3258_cast_fp16 = mul(x = var_3256_cast_fp16_1, y = const_73_promoted_to_fp16)[name = string("op_3258_cast_fp16")]; int32 var_3260 = const()[name = string("op_3260"), val = int32(-2)]; bool var_3261_interleave_0 = const()[name = string("op_3261_interleave_0"), val = bool(false)]; tensor var_3261_cast_fp16 = concat(axis = var_3260, interleave = var_3261_interleave_0, values = (var_3258_cast_fp16, var_3256_cast_fp16_0))[name = string("op_3261_cast_fp16")]; tensor var_3262_cast_fp16 = mul(x = var_3261_cast_fp16, y = var_878_cast_fp16)[name = string("op_3262_cast_fp16")]; tensor key_states_75_cast_fp16 = add(x = var_3255_cast_fp16, y = var_3262_cast_fp16)[name = string("key_states_75_cast_fp16")]; tensor expand_dims_84 = const()[name = string("expand_dims_84"), val = tensor([7])]; tensor expand_dims_85 = const()[name = string("expand_dims_85"), val = tensor([0])]; tensor expand_dims_87 = const()[name = string("expand_dims_87"), val = tensor([0])]; int32 concat_89_axis_0 = const()[name = string("concat_89_axis_0"), val = int32(0)]; bool concat_89_interleave_0 = const()[name = string("concat_89_interleave_0"), val = bool(false)]; tensor concat_89 = concat(axis = concat_89_axis_0, interleave = concat_89_interleave_0, values = (expand_dims_84, expand_dims_85, position_id, expand_dims_87))[name = string("concat_89")]; tensor expand_dims_88 = const()[name = string("expand_dims_88"), val = tensor([8])]; tensor concat_90_values1_0 = const()[name = string("concat_90_values1_0"), val = tensor([0])]; tensor concat_90_values3_0 = const()[name = string("concat_90_values3_0"), val = tensor([0])]; int32 concat_90_axis_0 = const()[name = string("concat_90_axis_0"), val = int32(0)]; bool concat_90_interleave_0 = const()[name = string("concat_90_interleave_0"), val = bool(false)]; tensor concat_90 = concat(axis = concat_90_axis_0, interleave = concat_90_interleave_0, values = (expand_dims_88, concat_90_values1_0, cache_position_end, concat_90_values3_0))[name = string("concat_90")]; tensor key_states_77_perm_0 = const()[name = string("key_states_77_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_8_stride_0 = const()[name = string("key_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_77_cast_fp16 = transpose(perm = key_states_77_perm_0, x = key_states_75_cast_fp16)[name = string("transpose_62")]; tensor key_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = key_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = key_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_8_squeeze_mask_0, stride = key_cache_internal_tensor_assign_8_stride_0, update = key_states_77_cast_fp16, x = coreml_update_state_12)[name = string("key_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_8_cast_fp16, input = key_cache)[name = string("coreml_update_state_14_write_state")]; tensor coreml_update_state_14 = read_state(input = key_cache)[name = string("coreml_update_state_14")]; tensor value_states_45_perm_0 = const()[name = string("value_states_45_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_8_stride_0 = const()[name = string("value_cache_internal_tensor_assign_8_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_8_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_8_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_8_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_8_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_45_cast_fp16 = transpose(perm = value_states_45_perm_0, x = var_3238_cast_fp16)[name = string("transpose_61")]; tensor value_cache_internal_tensor_assign_8_cast_fp16 = slice_update(begin = concat_89, begin_mask = value_cache_internal_tensor_assign_8_begin_mask_0, end = concat_90, end_mask = value_cache_internal_tensor_assign_8_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_8_squeeze_mask_0, stride = value_cache_internal_tensor_assign_8_stride_0, update = value_states_45_cast_fp16, x = coreml_update_state_13)[name = string("value_cache_internal_tensor_assign_8_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_8_cast_fp16, input = value_cache)[name = string("coreml_update_state_15_write_state")]; tensor coreml_update_state_15 = read_state(input = value_cache)[name = string("coreml_update_state_15")]; tensor var_3332_begin_0 = const()[name = string("op_3332_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3332_end_0 = const()[name = string("op_3332_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3332_end_mask_0 = const()[name = string("op_3332_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3332_cast_fp16 = slice_by_index(begin = var_3332_begin_0, end = var_3332_end_0, end_mask = var_3332_end_mask_0, x = coreml_update_state_14)[name = string("op_3332_cast_fp16")]; tensor tile_14 = const()[name = string("tile_14"), val = tensor([1, 1])]; int32 var_3335_axis_0 = const()[name = string("op_3335_axis_0"), val = int32(1)]; tensor var_3335_cast_fp16_0, tensor var_3335_cast_fp16_1 = split(axis = var_3335_axis_0, split_sizes = tile_14, x = var_3332_cast_fp16)[name = string("op_3335_cast_fp16")]; tensor var_3342_begin_0 = const()[name = string("op_3342_begin_0"), val = tensor([7, 0, 0, 0])]; tensor var_3342_end_0 = const()[name = string("op_3342_end_0"), val = tensor([8, 2, 2048, 128])]; tensor var_3342_end_mask_0 = const()[name = string("op_3342_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3342_cast_fp16 = slice_by_index(begin = var_3342_begin_0, end = var_3342_end_0, end_mask = var_3342_end_mask_0, x = coreml_update_state_15)[name = string("op_3342_cast_fp16")]; tensor tile_15 = const()[name = string("tile_15"), val = tensor([1, 1])]; int32 var_3345_axis_0 = const()[name = string("op_3345_axis_0"), val = int32(1)]; tensor var_3345_cast_fp16_0, tensor var_3345_cast_fp16_1 = split(axis = var_3345_axis_0, split_sizes = tile_15, x = var_3342_cast_fp16)[name = string("op_3345_cast_fp16")]; tensor var_3348_split_sizes_0 = const()[name = string("op_3348_split_sizes_0"), val = tensor([8, 8])]; int32 var_3348_axis_0 = const()[name = string("op_3348_axis_0"), val = int32(1)]; tensor var_3348_0, tensor var_3348_1 = split(axis = var_3348_axis_0, split_sizes = var_3348_split_sizes_0, x = query_states_45_cast_fp16)[name = string("op_3348")]; bool attn_weights_113_transpose_x_0 = const()[name = string("attn_weights_113_transpose_x_0"), val = bool(false)]; bool attn_weights_113_transpose_y_0 = const()[name = string("attn_weights_113_transpose_y_0"), val = bool(false)]; tensor attn_weights_113_cast_fp16 = matmul(transpose_x = attn_weights_113_transpose_x_0, transpose_y = attn_weights_113_transpose_y_0, x = var_3335_cast_fp16_0, y = var_3348_0)[name = string("attn_weights_113_cast_fp16")]; fp16 var_3351_to_fp16 = const()[name = string("op_3351_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_115_cast_fp16 = mul(x = attn_weights_113_cast_fp16, y = var_3351_to_fp16)[name = string("attn_weights_115_cast_fp16")]; tensor attn_weights_117_cast_fp16 = add(x = attn_weights_115_cast_fp16, y = attn_mask_1)[name = string("attn_weights_117_cast_fp16")]; int32 var_3355 = const()[name = string("op_3355"), val = int32(-2)]; tensor attn_weights_119_cast_fp16 = softmax(axis = var_3355, x = attn_weights_117_cast_fp16)[name = string("attn_weights_119_cast_fp16")]; bool var_3361_transpose_x_1 = const()[name = string("op_3361_transpose_x_1"), val = bool(true)]; bool var_3361_transpose_y_1 = const()[name = string("op_3361_transpose_y_1"), val = bool(false)]; tensor var_3361_cast_fp16 = matmul(transpose_x = var_3361_transpose_x_1, transpose_y = var_3361_transpose_y_1, x = attn_weights_119_cast_fp16, y = var_3345_cast_fp16_0)[name = string("op_3361_cast_fp16")]; bool attn_weights_121_transpose_x_0 = const()[name = string("attn_weights_121_transpose_x_0"), val = bool(false)]; bool attn_weights_121_transpose_y_0 = const()[name = string("attn_weights_121_transpose_y_0"), val = bool(false)]; tensor attn_weights_121_cast_fp16 = matmul(transpose_x = attn_weights_121_transpose_x_0, transpose_y = attn_weights_121_transpose_y_0, x = var_3335_cast_fp16_1, y = var_3348_1)[name = string("attn_weights_121_cast_fp16")]; fp16 var_3363_to_fp16 = const()[name = string("op_3363_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_123_cast_fp16 = mul(x = attn_weights_121_cast_fp16, y = var_3363_to_fp16)[name = string("attn_weights_123_cast_fp16")]; tensor attn_weights_125_cast_fp16 = add(x = attn_weights_123_cast_fp16, y = attn_mask_1)[name = string("attn_weights_125_cast_fp16")]; int32 var_3367 = const()[name = string("op_3367"), val = int32(-2)]; tensor attn_weights_127_cast_fp16 = softmax(axis = var_3367, x = attn_weights_125_cast_fp16)[name = string("attn_weights_127_cast_fp16")]; bool attn_output_57_transpose_x_1 = const()[name = string("attn_output_57_transpose_x_1"), val = bool(true)]; bool attn_output_57_transpose_y_1 = const()[name = string("attn_output_57_transpose_y_1"), val = bool(false)]; tensor attn_output_57_cast_fp16 = matmul(transpose_x = attn_output_57_transpose_x_1, transpose_y = attn_output_57_transpose_y_1, x = attn_weights_127_cast_fp16, y = var_3345_cast_fp16_1)[name = string("attn_output_57_cast_fp16")]; int32 var_3375 = const()[name = string("op_3375"), val = int32(1)]; bool attn_output_59_interleave_0 = const()[name = string("attn_output_59_interleave_0"), val = bool(false)]; tensor attn_output_59_cast_fp16 = concat(axis = var_3375, interleave = attn_output_59_interleave_0, values = (var_3361_cast_fp16, attn_output_57_cast_fp16))[name = string("attn_output_59_cast_fp16")]; tensor var_3379_perm_0 = const()[name = string("op_3379_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_95x = const()[name = string("concat_95x"), val = tensor([1, 2048, 1, -1])]; tensor var_3379_cast_fp16 = transpose(perm = var_3379_perm_0, x = attn_output_59_cast_fp16)[name = string("transpose_60")]; tensor attn_output_63_cast_fp16 = reshape(shape = concat_95x, x = var_3379_cast_fp16)[name = string("attn_output_63_cast_fp16")]; tensor hidden_states_73_strides_0 = const()[name = string("hidden_states_73_strides_0"), val = tensor([1, 1])]; string hidden_states_73_pad_type_0 = const()[name = string("hidden_states_73_pad_type_0"), val = string("valid")]; tensor hidden_states_73_pad_0 = const()[name = string("hidden_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_73_dilations_0 = const()[name = string("hidden_states_73_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_73_groups_0 = const()[name = string("hidden_states_73_groups_0"), val = int32(1)]; tensor hidden_states_73_cast_fp16 = conv(dilations = hidden_states_73_dilations_0, groups = hidden_states_73_groups_0, pad = hidden_states_73_pad_0, pad_type = hidden_states_73_pad_type_0, strides = hidden_states_73_strides_0, weight = layers_7_self_attn_o_proj_weight_cast_fp16, x = attn_output_63_cast_fp16)[name = string("hidden_states_73_cast_fp16")]; tensor hidden_states_75_cast_fp16 = add(x = hidden_states_69_cast_fp16, y = hidden_states_73_cast_fp16)[name = string("hidden_states_75_cast_fp16")]; fp16 const_78_promoted_to_fp16 = const()[name = string("const_78_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3412_cast_fp16 = mul(x = hidden_states_75_cast_fp16, y = const_78_promoted_to_fp16)[name = string("op_3412_cast_fp16")]; int32 var_3410 = const()[name = string("op_3410"), val = int32(1)]; bool doubled_61_interleave_0 = const()[name = string("doubled_61_interleave_0"), val = bool(false)]; tensor doubled_61_cast_fp16 = concat(axis = var_3410, interleave = doubled_61_interleave_0, values = (hidden_states_75_cast_fp16, var_3412_cast_fp16))[name = string("doubled_61_cast_fp16")]; tensor out_31_axes_0 = const()[name = string("out_31_axes_0"), val = tensor([1])]; tensor out_31_gamma_0_to_fp16 = const()[name = string("out_31_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336656640)))]; fp16 var_3422_to_fp16 = const()[name = string("op_3422_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_31_cast_fp16 = layer_norm(axes = out_31_axes_0, epsilon = var_3422_to_fp16, gamma = out_31_gamma_0_to_fp16, x = doubled_61_cast_fp16)[name = string("out_31_cast_fp16")]; tensor var_3433_split_sizes_0 = const()[name = string("op_3433_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3433_axis_0 = const()[name = string("op_3433_axis_0"), val = int32(1)]; tensor var_3433_cast_fp16_0, tensor var_3433_cast_fp16_1 = split(axis = var_3433_axis_0, split_sizes = var_3433_split_sizes_0, x = out_31_cast_fp16)[name = string("op_3433_cast_fp16")]; tensor input_15_strides_0 = const()[name = string("input_15_strides_0"), val = tensor([1, 1])]; string input_15_pad_type_0 = const()[name = string("input_15_pad_type_0"), val = string("valid")]; tensor input_15_pad_0 = const()[name = string("input_15_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_15_dilations_0 = const()[name = string("input_15_dilations_0"), val = tensor([1, 1])]; int32 input_15_groups_0 = const()[name = string("input_15_groups_0"), val = int32(1)]; tensor input_15_cast_fp16 = conv(dilations = input_15_dilations_0, groups = input_15_groups_0, pad = input_15_pad_0, pad_type = input_15_pad_type_0, strides = input_15_strides_0, weight = layers_7_mlp_gate_proj_weight_cast_fp16, x = var_3433_cast_fp16_0)[name = string("input_15_cast_fp16")]; tensor var_3450_cast_fp16 = silu(x = input_15_cast_fp16)[name = string("op_3450_cast_fp16")]; tensor layers_7_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1336664896)))]; tensor var_3456_strides_0 = const()[name = string("op_3456_strides_0"), val = tensor([1, 1])]; string var_3456_pad_type_0 = const()[name = string("op_3456_pad_type_0"), val = string("valid")]; tensor var_3456_pad_0 = const()[name = string("op_3456_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3456_dilations_0 = const()[name = string("op_3456_dilations_0"), val = tensor([1, 1])]; int32 var_3456_groups_0 = const()[name = string("op_3456_groups_0"), val = int32(1)]; tensor var_3456_cast_fp16 = conv(dilations = var_3456_dilations_0, groups = var_3456_groups_0, pad = var_3456_pad_0, pad_type = var_3456_pad_type_0, strides = var_3456_strides_0, weight = layers_7_mlp_up_proj_weight_to_fp16, x = var_3433_cast_fp16_0)[name = string("op_3456_cast_fp16")]; tensor x_79_cast_fp16 = mul(x = var_3450_cast_fp16, y = var_3456_cast_fp16)[name = string("x_79_cast_fp16")]; tensor layers_7_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_7_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1361830784)))]; tensor hidden_states_77_strides_0 = const()[name = string("hidden_states_77_strides_0"), val = tensor([1, 1])]; string hidden_states_77_pad_type_0 = const()[name = string("hidden_states_77_pad_type_0"), val = string("valid")]; tensor hidden_states_77_pad_0 = const()[name = string("hidden_states_77_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_77_dilations_0 = const()[name = string("hidden_states_77_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_77_groups_0 = const()[name = string("hidden_states_77_groups_0"), val = int32(1)]; tensor hidden_states_77_cast_fp16 = conv(dilations = hidden_states_77_dilations_0, groups = hidden_states_77_groups_0, pad = hidden_states_77_pad_0, pad_type = hidden_states_77_pad_type_0, strides = hidden_states_77_strides_0, weight = layers_7_mlp_down_proj_weight_to_fp16, x = x_79_cast_fp16)[name = string("hidden_states_77_cast_fp16")]; tensor hidden_states_79_cast_fp16 = add(x = hidden_states_75_cast_fp16, y = hidden_states_77_cast_fp16)[name = string("hidden_states_79_cast_fp16")]; fp16 const_80_promoted_to_fp16 = const()[name = string("const_80_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3474_cast_fp16 = mul(x = hidden_states_79_cast_fp16, y = const_80_promoted_to_fp16)[name = string("op_3474_cast_fp16")]; int32 var_3472 = const()[name = string("op_3472"), val = int32(1)]; bool doubled_65_interleave_0 = const()[name = string("doubled_65_interleave_0"), val = bool(false)]; tensor doubled_65_cast_fp16 = concat(axis = var_3472, interleave = doubled_65_interleave_0, values = (hidden_states_79_cast_fp16, var_3474_cast_fp16))[name = string("doubled_65_cast_fp16")]; tensor out_33_axes_0 = const()[name = string("out_33_axes_0"), val = tensor([1])]; tensor out_33_gamma_0_to_fp16 = const()[name = string("out_33_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1386996672)))]; fp16 var_3484_to_fp16 = const()[name = string("op_3484_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_33_cast_fp16 = layer_norm(axes = out_33_axes_0, epsilon = var_3484_to_fp16, gamma = out_33_gamma_0_to_fp16, x = doubled_65_cast_fp16)[name = string("out_33_cast_fp16")]; tensor var_3495_split_sizes_0 = const()[name = string("op_3495_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3495_axis_0 = const()[name = string("op_3495_axis_0"), val = int32(1)]; tensor var_3495_cast_fp16_0, tensor var_3495_cast_fp16_1 = split(axis = var_3495_axis_0, split_sizes = var_3495_split_sizes_0, x = out_33_cast_fp16)[name = string("op_3495_cast_fp16")]; tensor layers_8_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1387004928)))]; tensor query_states_49_strides_0 = const()[name = string("query_states_49_strides_0"), val = tensor([1, 1])]; string query_states_49_pad_type_0 = const()[name = string("query_states_49_pad_type_0"), val = string("valid")]; tensor query_states_49_pad_0 = const()[name = string("query_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_49_dilations_0 = const()[name = string("query_states_49_dilations_0"), val = tensor([1, 1])]; int32 query_states_49_groups_0 = const()[name = string("query_states_49_groups_0"), val = int32(1)]; tensor query_states_49_cast_fp16 = conv(dilations = query_states_49_dilations_0, groups = query_states_49_groups_0, pad = query_states_49_pad_0, pad_type = query_states_49_pad_type_0, strides = query_states_49_strides_0, weight = layers_8_self_attn_q_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("query_states_49_cast_fp16")]; tensor layers_8_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_8_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1395393600)))]; tensor key_states_81_strides_0 = const()[name = string("key_states_81_strides_0"), val = tensor([1, 1])]; string key_states_81_pad_type_0 = const()[name = string("key_states_81_pad_type_0"), val = string("valid")]; tensor key_states_81_pad_0 = const()[name = string("key_states_81_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_81_dilations_0 = const()[name = string("key_states_81_dilations_0"), val = tensor([1, 1])]; int32 key_states_81_groups_0 = const()[name = string("key_states_81_groups_0"), val = int32(1)]; tensor key_states_81_cast_fp16 = conv(dilations = key_states_81_dilations_0, groups = key_states_81_groups_0, pad = key_states_81_pad_0, pad_type = key_states_81_pad_type_0, strides = key_states_81_strides_0, weight = layers_8_self_attn_k_proj_weight_to_fp16, x = var_3495_cast_fp16_0)[name = string("key_states_81_cast_fp16")]; tensor value_states_49_strides_0 = const()[name = string("value_states_49_strides_0"), val = tensor([1, 1])]; string value_states_49_pad_type_0 = const()[name = string("value_states_49_pad_type_0"), val = string("valid")]; tensor value_states_49_pad_0 = const()[name = string("value_states_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_49_dilations_0 = const()[name = string("value_states_49_dilations_0"), val = tensor([1, 1])]; int32 value_states_49_groups_0 = const()[name = string("value_states_49_groups_0"), val = int32(1)]; tensor value_states_49_cast_fp16 = conv(dilations = value_states_49_dilations_0, groups = value_states_49_groups_0, pad = value_states_49_pad_0, pad_type = value_states_49_pad_type_0, strides = value_states_49_strides_0, weight = layers_8_self_attn_v_proj_weight_cast_fp16, x = var_3495_cast_fp16_0)[name = string("value_states_49_cast_fp16")]; tensor concat_96x = const()[name = string("concat_96x"), val = tensor([1, 16, 128, -1])]; tensor x_81_cast_fp16 = reshape(shape = concat_96x, x = query_states_49_cast_fp16)[name = string("x_81_cast_fp16")]; tensor concat_97x = const()[name = string("concat_97x"), val = tensor([1, 2, 128, -1])]; tensor var_3552_cast_fp16 = reshape(shape = concat_97x, x = key_states_81_cast_fp16)[name = string("op_3552_cast_fp16")]; tensor concat_98x = const()[name = string("concat_98x"), val = tensor([1, 2, 128, -1])]; tensor var_3559_cast_fp16 = reshape(shape = concat_98x, x = value_states_49_cast_fp16)[name = string("op_3559_cast_fp16")]; tensor var_3563_cast_fp16 = mul(x = x_81_cast_fp16, y = var_869_cast_fp16)[name = string("op_3563_cast_fp16")]; tensor var_3564_split_sizes_0 = const()[name = string("op_3564_split_sizes_0"), val = tensor([64, 64])]; int32 var_3564_axis_0 = const()[name = string("op_3564_axis_0"), val = int32(-2)]; tensor var_3564_cast_fp16_0, tensor var_3564_cast_fp16_1 = split(axis = var_3564_axis_0, split_sizes = var_3564_split_sizes_0, x = x_81_cast_fp16)[name = string("op_3564_cast_fp16")]; fp16 const_82_promoted_to_fp16 = const()[name = string("const_82_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3566_cast_fp16 = mul(x = var_3564_cast_fp16_1, y = const_82_promoted_to_fp16)[name = string("op_3566_cast_fp16")]; int32 var_3568 = const()[name = string("op_3568"), val = int32(-2)]; bool var_3569_interleave_0 = const()[name = string("op_3569_interleave_0"), val = bool(false)]; tensor var_3569_cast_fp16 = concat(axis = var_3568, interleave = var_3569_interleave_0, values = (var_3566_cast_fp16, var_3564_cast_fp16_0))[name = string("op_3569_cast_fp16")]; tensor var_3570_cast_fp16 = mul(x = var_3569_cast_fp16, y = var_878_cast_fp16)[name = string("op_3570_cast_fp16")]; tensor query_states_51_cast_fp16 = add(x = var_3563_cast_fp16, y = var_3570_cast_fp16)[name = string("query_states_51_cast_fp16")]; tensor var_3576_cast_fp16 = mul(x = var_3552_cast_fp16, y = var_869_cast_fp16)[name = string("op_3576_cast_fp16")]; tensor var_3577_split_sizes_0 = const()[name = string("op_3577_split_sizes_0"), val = tensor([64, 64])]; int32 var_3577_axis_0 = const()[name = string("op_3577_axis_0"), val = int32(-2)]; tensor var_3577_cast_fp16_0, tensor var_3577_cast_fp16_1 = split(axis = var_3577_axis_0, split_sizes = var_3577_split_sizes_0, x = var_3552_cast_fp16)[name = string("op_3577_cast_fp16")]; fp16 const_83_promoted_to_fp16 = const()[name = string("const_83_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3579_cast_fp16 = mul(x = var_3577_cast_fp16_1, y = const_83_promoted_to_fp16)[name = string("op_3579_cast_fp16")]; int32 var_3581 = const()[name = string("op_3581"), val = int32(-2)]; bool var_3582_interleave_0 = const()[name = string("op_3582_interleave_0"), val = bool(false)]; tensor var_3582_cast_fp16 = concat(axis = var_3581, interleave = var_3582_interleave_0, values = (var_3579_cast_fp16, var_3577_cast_fp16_0))[name = string("op_3582_cast_fp16")]; tensor var_3583_cast_fp16 = mul(x = var_3582_cast_fp16, y = var_878_cast_fp16)[name = string("op_3583_cast_fp16")]; tensor key_states_85_cast_fp16 = add(x = var_3576_cast_fp16, y = var_3583_cast_fp16)[name = string("key_states_85_cast_fp16")]; tensor expand_dims_96 = const()[name = string("expand_dims_96"), val = tensor([8])]; tensor expand_dims_97 = const()[name = string("expand_dims_97"), val = tensor([0])]; tensor expand_dims_99 = const()[name = string("expand_dims_99"), val = tensor([0])]; int32 concat_101_axis_0 = const()[name = string("concat_101_axis_0"), val = int32(0)]; bool concat_101_interleave_0 = const()[name = string("concat_101_interleave_0"), val = bool(false)]; tensor concat_101 = concat(axis = concat_101_axis_0, interleave = concat_101_interleave_0, values = (expand_dims_96, expand_dims_97, position_id, expand_dims_99))[name = string("concat_101")]; tensor expand_dims_100 = const()[name = string("expand_dims_100"), val = tensor([9])]; tensor concat_102_values1_0 = const()[name = string("concat_102_values1_0"), val = tensor([0])]; tensor concat_102_values3_0 = const()[name = string("concat_102_values3_0"), val = tensor([0])]; int32 concat_102_axis_0 = const()[name = string("concat_102_axis_0"), val = int32(0)]; bool concat_102_interleave_0 = const()[name = string("concat_102_interleave_0"), val = bool(false)]; tensor concat_102 = concat(axis = concat_102_axis_0, interleave = concat_102_interleave_0, values = (expand_dims_100, concat_102_values1_0, cache_position_end, concat_102_values3_0))[name = string("concat_102")]; tensor key_states_87_perm_0 = const()[name = string("key_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_9_stride_0 = const()[name = string("key_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_87_cast_fp16 = transpose(perm = key_states_87_perm_0, x = key_states_85_cast_fp16)[name = string("transpose_59")]; tensor key_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = key_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = key_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_9_squeeze_mask_0, stride = key_cache_internal_tensor_assign_9_stride_0, update = key_states_87_cast_fp16, x = coreml_update_state_14)[name = string("key_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_9_cast_fp16, input = key_cache)[name = string("coreml_update_state_16_write_state")]; tensor coreml_update_state_16 = read_state(input = key_cache)[name = string("coreml_update_state_16")]; tensor value_states_51_perm_0 = const()[name = string("value_states_51_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_9_stride_0 = const()[name = string("value_cache_internal_tensor_assign_9_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_9_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_9_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_9_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_9_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_51_cast_fp16 = transpose(perm = value_states_51_perm_0, x = var_3559_cast_fp16)[name = string("transpose_58")]; tensor value_cache_internal_tensor_assign_9_cast_fp16 = slice_update(begin = concat_101, begin_mask = value_cache_internal_tensor_assign_9_begin_mask_0, end = concat_102, end_mask = value_cache_internal_tensor_assign_9_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_9_squeeze_mask_0, stride = value_cache_internal_tensor_assign_9_stride_0, update = value_states_51_cast_fp16, x = coreml_update_state_15)[name = string("value_cache_internal_tensor_assign_9_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_9_cast_fp16, input = value_cache)[name = string("coreml_update_state_17_write_state")]; tensor coreml_update_state_17 = read_state(input = value_cache)[name = string("coreml_update_state_17")]; tensor var_3653_begin_0 = const()[name = string("op_3653_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3653_end_0 = const()[name = string("op_3653_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3653_end_mask_0 = const()[name = string("op_3653_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3653_cast_fp16 = slice_by_index(begin = var_3653_begin_0, end = var_3653_end_0, end_mask = var_3653_end_mask_0, x = coreml_update_state_16)[name = string("op_3653_cast_fp16")]; tensor tile_16 = const()[name = string("tile_16"), val = tensor([1, 1])]; int32 var_3656_axis_0 = const()[name = string("op_3656_axis_0"), val = int32(1)]; tensor var_3656_cast_fp16_0, tensor var_3656_cast_fp16_1 = split(axis = var_3656_axis_0, split_sizes = tile_16, x = var_3653_cast_fp16)[name = string("op_3656_cast_fp16")]; tensor var_3663_begin_0 = const()[name = string("op_3663_begin_0"), val = tensor([8, 0, 0, 0])]; tensor var_3663_end_0 = const()[name = string("op_3663_end_0"), val = tensor([9, 2, 2048, 128])]; tensor var_3663_end_mask_0 = const()[name = string("op_3663_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3663_cast_fp16 = slice_by_index(begin = var_3663_begin_0, end = var_3663_end_0, end_mask = var_3663_end_mask_0, x = coreml_update_state_17)[name = string("op_3663_cast_fp16")]; tensor tile_17 = const()[name = string("tile_17"), val = tensor([1, 1])]; int32 var_3666_axis_0 = const()[name = string("op_3666_axis_0"), val = int32(1)]; tensor var_3666_cast_fp16_0, tensor var_3666_cast_fp16_1 = split(axis = var_3666_axis_0, split_sizes = tile_17, x = var_3663_cast_fp16)[name = string("op_3666_cast_fp16")]; tensor var_3669_split_sizes_0 = const()[name = string("op_3669_split_sizes_0"), val = tensor([8, 8])]; int32 var_3669_axis_0 = const()[name = string("op_3669_axis_0"), val = int32(1)]; tensor var_3669_0, tensor var_3669_1 = split(axis = var_3669_axis_0, split_sizes = var_3669_split_sizes_0, x = query_states_51_cast_fp16)[name = string("op_3669")]; bool attn_weights_129_transpose_x_0 = const()[name = string("attn_weights_129_transpose_x_0"), val = bool(false)]; bool attn_weights_129_transpose_y_0 = const()[name = string("attn_weights_129_transpose_y_0"), val = bool(false)]; tensor attn_weights_129_cast_fp16 = matmul(transpose_x = attn_weights_129_transpose_x_0, transpose_y = attn_weights_129_transpose_y_0, x = var_3656_cast_fp16_0, y = var_3669_0)[name = string("attn_weights_129_cast_fp16")]; fp16 var_3672_to_fp16 = const()[name = string("op_3672_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_131_cast_fp16 = mul(x = attn_weights_129_cast_fp16, y = var_3672_to_fp16)[name = string("attn_weights_131_cast_fp16")]; tensor attn_weights_133_cast_fp16 = add(x = attn_weights_131_cast_fp16, y = attn_mask_1)[name = string("attn_weights_133_cast_fp16")]; int32 var_3676 = const()[name = string("op_3676"), val = int32(-2)]; tensor attn_weights_135_cast_fp16 = softmax(axis = var_3676, x = attn_weights_133_cast_fp16)[name = string("attn_weights_135_cast_fp16")]; bool var_3682_transpose_x_1 = const()[name = string("op_3682_transpose_x_1"), val = bool(true)]; bool var_3682_transpose_y_1 = const()[name = string("op_3682_transpose_y_1"), val = bool(false)]; tensor var_3682_cast_fp16 = matmul(transpose_x = var_3682_transpose_x_1, transpose_y = var_3682_transpose_y_1, x = attn_weights_135_cast_fp16, y = var_3666_cast_fp16_0)[name = string("op_3682_cast_fp16")]; bool attn_weights_137_transpose_x_0 = const()[name = string("attn_weights_137_transpose_x_0"), val = bool(false)]; bool attn_weights_137_transpose_y_0 = const()[name = string("attn_weights_137_transpose_y_0"), val = bool(false)]; tensor attn_weights_137_cast_fp16 = matmul(transpose_x = attn_weights_137_transpose_x_0, transpose_y = attn_weights_137_transpose_y_0, x = var_3656_cast_fp16_1, y = var_3669_1)[name = string("attn_weights_137_cast_fp16")]; fp16 var_3684_to_fp16 = const()[name = string("op_3684_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_139_cast_fp16 = mul(x = attn_weights_137_cast_fp16, y = var_3684_to_fp16)[name = string("attn_weights_139_cast_fp16")]; tensor attn_weights_141_cast_fp16 = add(x = attn_weights_139_cast_fp16, y = attn_mask_1)[name = string("attn_weights_141_cast_fp16")]; int32 var_3688 = const()[name = string("op_3688"), val = int32(-2)]; tensor attn_weights_143_cast_fp16 = softmax(axis = var_3688, x = attn_weights_141_cast_fp16)[name = string("attn_weights_143_cast_fp16")]; bool attn_output_65_transpose_x_1 = const()[name = string("attn_output_65_transpose_x_1"), val = bool(true)]; bool attn_output_65_transpose_y_1 = const()[name = string("attn_output_65_transpose_y_1"), val = bool(false)]; tensor attn_output_65_cast_fp16 = matmul(transpose_x = attn_output_65_transpose_x_1, transpose_y = attn_output_65_transpose_y_1, x = attn_weights_143_cast_fp16, y = var_3666_cast_fp16_1)[name = string("attn_output_65_cast_fp16")]; int32 var_3696 = const()[name = string("op_3696"), val = int32(1)]; bool attn_output_67_interleave_0 = const()[name = string("attn_output_67_interleave_0"), val = bool(false)]; tensor attn_output_67_cast_fp16 = concat(axis = var_3696, interleave = attn_output_67_interleave_0, values = (var_3682_cast_fp16, attn_output_65_cast_fp16))[name = string("attn_output_67_cast_fp16")]; tensor var_3700_perm_0 = const()[name = string("op_3700_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_107x = const()[name = string("concat_107x"), val = tensor([1, 2048, 1, -1])]; tensor var_3700_cast_fp16 = transpose(perm = var_3700_perm_0, x = attn_output_67_cast_fp16)[name = string("transpose_57")]; tensor attn_output_71_cast_fp16 = reshape(shape = concat_107x, x = var_3700_cast_fp16)[name = string("attn_output_71_cast_fp16")]; tensor hidden_states_83_strides_0 = const()[name = string("hidden_states_83_strides_0"), val = tensor([1, 1])]; string hidden_states_83_pad_type_0 = const()[name = string("hidden_states_83_pad_type_0"), val = string("valid")]; tensor hidden_states_83_pad_0 = const()[name = string("hidden_states_83_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_83_dilations_0 = const()[name = string("hidden_states_83_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_83_groups_0 = const()[name = string("hidden_states_83_groups_0"), val = int32(1)]; tensor hidden_states_83_cast_fp16 = conv(dilations = hidden_states_83_dilations_0, groups = hidden_states_83_groups_0, pad = hidden_states_83_pad_0, pad_type = hidden_states_83_pad_type_0, strides = hidden_states_83_strides_0, weight = layers_8_self_attn_o_proj_weight_cast_fp16, x = attn_output_71_cast_fp16)[name = string("hidden_states_83_cast_fp16")]; tensor hidden_states_85_cast_fp16 = add(x = hidden_states_79_cast_fp16, y = hidden_states_83_cast_fp16)[name = string("hidden_states_85_cast_fp16")]; fp16 const_88_promoted_to_fp16 = const()[name = string("const_88_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3733_cast_fp16 = mul(x = hidden_states_85_cast_fp16, y = const_88_promoted_to_fp16)[name = string("op_3733_cast_fp16")]; int32 var_3731 = const()[name = string("op_3731"), val = int32(1)]; bool doubled_69_interleave_0 = const()[name = string("doubled_69_interleave_0"), val = bool(false)]; tensor doubled_69_cast_fp16 = concat(axis = var_3731, interleave = doubled_69_interleave_0, values = (hidden_states_85_cast_fp16, var_3733_cast_fp16))[name = string("doubled_69_cast_fp16")]; tensor out_35_axes_0 = const()[name = string("out_35_axes_0"), val = tensor([1])]; tensor out_35_gamma_0_to_fp16 = const()[name = string("out_35_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396442240)))]; fp16 var_3743_to_fp16 = const()[name = string("op_3743_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_35_cast_fp16 = layer_norm(axes = out_35_axes_0, epsilon = var_3743_to_fp16, gamma = out_35_gamma_0_to_fp16, x = doubled_69_cast_fp16)[name = string("out_35_cast_fp16")]; tensor var_3754_split_sizes_0 = const()[name = string("op_3754_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3754_axis_0 = const()[name = string("op_3754_axis_0"), val = int32(1)]; tensor var_3754_cast_fp16_0, tensor var_3754_cast_fp16_1 = split(axis = var_3754_axis_0, split_sizes = var_3754_split_sizes_0, x = out_35_cast_fp16)[name = string("op_3754_cast_fp16")]; tensor input_17_strides_0 = const()[name = string("input_17_strides_0"), val = tensor([1, 1])]; string input_17_pad_type_0 = const()[name = string("input_17_pad_type_0"), val = string("valid")]; tensor input_17_pad_0 = const()[name = string("input_17_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_17_dilations_0 = const()[name = string("input_17_dilations_0"), val = tensor([1, 1])]; int32 input_17_groups_0 = const()[name = string("input_17_groups_0"), val = int32(1)]; tensor input_17_cast_fp16 = conv(dilations = input_17_dilations_0, groups = input_17_groups_0, pad = input_17_pad_0, pad_type = input_17_pad_type_0, strides = input_17_strides_0, weight = layers_8_mlp_gate_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("input_17_cast_fp16")]; tensor var_3771_cast_fp16 = silu(x = input_17_cast_fp16)[name = string("op_3771_cast_fp16")]; tensor var_3777_strides_0 = const()[name = string("op_3777_strides_0"), val = tensor([1, 1])]; string var_3777_pad_type_0 = const()[name = string("op_3777_pad_type_0"), val = string("valid")]; tensor var_3777_pad_0 = const()[name = string("op_3777_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_3777_dilations_0 = const()[name = string("op_3777_dilations_0"), val = tensor([1, 1])]; int32 var_3777_groups_0 = const()[name = string("op_3777_groups_0"), val = int32(1)]; tensor var_3777_cast_fp16 = conv(dilations = var_3777_dilations_0, groups = var_3777_groups_0, pad = var_3777_pad_0, pad_type = var_3777_pad_type_0, strides = var_3777_strides_0, weight = layers_8_mlp_up_proj_weight_cast_fp16, x = var_3754_cast_fp16_0)[name = string("op_3777_cast_fp16")]; tensor x_89_cast_fp16 = mul(x = var_3771_cast_fp16, y = var_3777_cast_fp16)[name = string("x_89_cast_fp16")]; tensor hidden_states_87_strides_0 = const()[name = string("hidden_states_87_strides_0"), val = tensor([1, 1])]; string hidden_states_87_pad_type_0 = const()[name = string("hidden_states_87_pad_type_0"), val = string("valid")]; tensor hidden_states_87_pad_0 = const()[name = string("hidden_states_87_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_87_dilations_0 = const()[name = string("hidden_states_87_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_87_groups_0 = const()[name = string("hidden_states_87_groups_0"), val = int32(1)]; tensor hidden_states_87_cast_fp16 = conv(dilations = hidden_states_87_dilations_0, groups = hidden_states_87_groups_0, pad = hidden_states_87_pad_0, pad_type = hidden_states_87_pad_type_0, strides = hidden_states_87_strides_0, weight = layers_8_mlp_down_proj_weight_cast_fp16, x = x_89_cast_fp16)[name = string("hidden_states_87_cast_fp16")]; tensor hidden_states_89_cast_fp16 = add(x = hidden_states_85_cast_fp16, y = hidden_states_87_cast_fp16)[name = string("hidden_states_89_cast_fp16")]; fp16 const_90_promoted_to_fp16 = const()[name = string("const_90_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3795_cast_fp16 = mul(x = hidden_states_89_cast_fp16, y = const_90_promoted_to_fp16)[name = string("op_3795_cast_fp16")]; int32 var_3793 = const()[name = string("op_3793"), val = int32(1)]; bool doubled_73_interleave_0 = const()[name = string("doubled_73_interleave_0"), val = bool(false)]; tensor doubled_73_cast_fp16 = concat(axis = var_3793, interleave = doubled_73_interleave_0, values = (hidden_states_89_cast_fp16, var_3795_cast_fp16))[name = string("doubled_73_cast_fp16")]; tensor out_37_axes_0 = const()[name = string("out_37_axes_0"), val = tensor([1])]; tensor out_37_gamma_0_to_fp16 = const()[name = string("out_37_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396450496)))]; fp16 var_3805_to_fp16 = const()[name = string("op_3805_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_37_cast_fp16 = layer_norm(axes = out_37_axes_0, epsilon = var_3805_to_fp16, gamma = out_37_gamma_0_to_fp16, x = doubled_73_cast_fp16)[name = string("out_37_cast_fp16")]; tensor var_3816_split_sizes_0 = const()[name = string("op_3816_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_3816_axis_0 = const()[name = string("op_3816_axis_0"), val = int32(1)]; tensor var_3816_cast_fp16_0, tensor var_3816_cast_fp16_1 = split(axis = var_3816_axis_0, split_sizes = var_3816_split_sizes_0, x = out_37_cast_fp16)[name = string("op_3816_cast_fp16")]; tensor layers_9_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1396458752)))]; tensor query_states_55_strides_0 = const()[name = string("query_states_55_strides_0"), val = tensor([1, 1])]; string query_states_55_pad_type_0 = const()[name = string("query_states_55_pad_type_0"), val = string("valid")]; tensor query_states_55_pad_0 = const()[name = string("query_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_55_dilations_0 = const()[name = string("query_states_55_dilations_0"), val = tensor([1, 1])]; int32 query_states_55_groups_0 = const()[name = string("query_states_55_groups_0"), val = int32(1)]; tensor query_states_55_cast_fp16 = conv(dilations = query_states_55_dilations_0, groups = query_states_55_groups_0, pad = query_states_55_pad_0, pad_type = query_states_55_pad_type_0, strides = query_states_55_strides_0, weight = layers_9_self_attn_q_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("query_states_55_cast_fp16")]; tensor layers_9_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_9_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1404847424)))]; tensor key_states_91_strides_0 = const()[name = string("key_states_91_strides_0"), val = tensor([1, 1])]; string key_states_91_pad_type_0 = const()[name = string("key_states_91_pad_type_0"), val = string("valid")]; tensor key_states_91_pad_0 = const()[name = string("key_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_91_dilations_0 = const()[name = string("key_states_91_dilations_0"), val = tensor([1, 1])]; int32 key_states_91_groups_0 = const()[name = string("key_states_91_groups_0"), val = int32(1)]; tensor key_states_91_cast_fp16 = conv(dilations = key_states_91_dilations_0, groups = key_states_91_groups_0, pad = key_states_91_pad_0, pad_type = key_states_91_pad_type_0, strides = key_states_91_strides_0, weight = layers_9_self_attn_k_proj_weight_to_fp16, x = var_3816_cast_fp16_0)[name = string("key_states_91_cast_fp16")]; tensor value_states_55_strides_0 = const()[name = string("value_states_55_strides_0"), val = tensor([1, 1])]; string value_states_55_pad_type_0 = const()[name = string("value_states_55_pad_type_0"), val = string("valid")]; tensor value_states_55_pad_0 = const()[name = string("value_states_55_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_55_dilations_0 = const()[name = string("value_states_55_dilations_0"), val = tensor([1, 1])]; int32 value_states_55_groups_0 = const()[name = string("value_states_55_groups_0"), val = int32(1)]; tensor value_states_55_cast_fp16 = conv(dilations = value_states_55_dilations_0, groups = value_states_55_groups_0, pad = value_states_55_pad_0, pad_type = value_states_55_pad_type_0, strides = value_states_55_strides_0, weight = layers_9_self_attn_v_proj_weight_cast_fp16, x = var_3816_cast_fp16_0)[name = string("value_states_55_cast_fp16")]; tensor concat_108x = const()[name = string("concat_108x"), val = tensor([1, 16, 128, -1])]; tensor x_91_cast_fp16 = reshape(shape = concat_108x, x = query_states_55_cast_fp16)[name = string("x_91_cast_fp16")]; tensor concat_109x = const()[name = string("concat_109x"), val = tensor([1, 2, 128, -1])]; tensor var_3873_cast_fp16 = reshape(shape = concat_109x, x = key_states_91_cast_fp16)[name = string("op_3873_cast_fp16")]; tensor concat_110x = const()[name = string("concat_110x"), val = tensor([1, 2, 128, -1])]; tensor var_3880_cast_fp16 = reshape(shape = concat_110x, x = value_states_55_cast_fp16)[name = string("op_3880_cast_fp16")]; tensor var_3884_cast_fp16 = mul(x = x_91_cast_fp16, y = var_869_cast_fp16)[name = string("op_3884_cast_fp16")]; tensor var_3885_split_sizes_0 = const()[name = string("op_3885_split_sizes_0"), val = tensor([64, 64])]; int32 var_3885_axis_0 = const()[name = string("op_3885_axis_0"), val = int32(-2)]; tensor var_3885_cast_fp16_0, tensor var_3885_cast_fp16_1 = split(axis = var_3885_axis_0, split_sizes = var_3885_split_sizes_0, x = x_91_cast_fp16)[name = string("op_3885_cast_fp16")]; fp16 const_92_promoted_to_fp16 = const()[name = string("const_92_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3887_cast_fp16 = mul(x = var_3885_cast_fp16_1, y = const_92_promoted_to_fp16)[name = string("op_3887_cast_fp16")]; int32 var_3889 = const()[name = string("op_3889"), val = int32(-2)]; bool var_3890_interleave_0 = const()[name = string("op_3890_interleave_0"), val = bool(false)]; tensor var_3890_cast_fp16 = concat(axis = var_3889, interleave = var_3890_interleave_0, values = (var_3887_cast_fp16, var_3885_cast_fp16_0))[name = string("op_3890_cast_fp16")]; tensor var_3891_cast_fp16 = mul(x = var_3890_cast_fp16, y = var_878_cast_fp16)[name = string("op_3891_cast_fp16")]; tensor query_states_57_cast_fp16 = add(x = var_3884_cast_fp16, y = var_3891_cast_fp16)[name = string("query_states_57_cast_fp16")]; tensor var_3897_cast_fp16 = mul(x = var_3873_cast_fp16, y = var_869_cast_fp16)[name = string("op_3897_cast_fp16")]; tensor var_3898_split_sizes_0 = const()[name = string("op_3898_split_sizes_0"), val = tensor([64, 64])]; int32 var_3898_axis_0 = const()[name = string("op_3898_axis_0"), val = int32(-2)]; tensor var_3898_cast_fp16_0, tensor var_3898_cast_fp16_1 = split(axis = var_3898_axis_0, split_sizes = var_3898_split_sizes_0, x = var_3873_cast_fp16)[name = string("op_3898_cast_fp16")]; fp16 const_93_promoted_to_fp16 = const()[name = string("const_93_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_3900_cast_fp16 = mul(x = var_3898_cast_fp16_1, y = const_93_promoted_to_fp16)[name = string("op_3900_cast_fp16")]; int32 var_3902 = const()[name = string("op_3902"), val = int32(-2)]; bool var_3903_interleave_0 = const()[name = string("op_3903_interleave_0"), val = bool(false)]; tensor var_3903_cast_fp16 = concat(axis = var_3902, interleave = var_3903_interleave_0, values = (var_3900_cast_fp16, var_3898_cast_fp16_0))[name = string("op_3903_cast_fp16")]; tensor var_3904_cast_fp16 = mul(x = var_3903_cast_fp16, y = var_878_cast_fp16)[name = string("op_3904_cast_fp16")]; tensor key_states_95_cast_fp16 = add(x = var_3897_cast_fp16, y = var_3904_cast_fp16)[name = string("key_states_95_cast_fp16")]; tensor expand_dims_108 = const()[name = string("expand_dims_108"), val = tensor([9])]; tensor expand_dims_109 = const()[name = string("expand_dims_109"), val = tensor([0])]; tensor expand_dims_111 = const()[name = string("expand_dims_111"), val = tensor([0])]; int32 concat_113_axis_0 = const()[name = string("concat_113_axis_0"), val = int32(0)]; bool concat_113_interleave_0 = const()[name = string("concat_113_interleave_0"), val = bool(false)]; tensor concat_113 = concat(axis = concat_113_axis_0, interleave = concat_113_interleave_0, values = (expand_dims_108, expand_dims_109, position_id, expand_dims_111))[name = string("concat_113")]; tensor expand_dims_112 = const()[name = string("expand_dims_112"), val = tensor([10])]; tensor concat_114_values1_0 = const()[name = string("concat_114_values1_0"), val = tensor([0])]; tensor concat_114_values3_0 = const()[name = string("concat_114_values3_0"), val = tensor([0])]; int32 concat_114_axis_0 = const()[name = string("concat_114_axis_0"), val = int32(0)]; bool concat_114_interleave_0 = const()[name = string("concat_114_interleave_0"), val = bool(false)]; tensor concat_114 = concat(axis = concat_114_axis_0, interleave = concat_114_interleave_0, values = (expand_dims_112, concat_114_values1_0, cache_position_end, concat_114_values3_0))[name = string("concat_114")]; tensor key_states_97_perm_0 = const()[name = string("key_states_97_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_10_stride_0 = const()[name = string("key_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_97_cast_fp16 = transpose(perm = key_states_97_perm_0, x = key_states_95_cast_fp16)[name = string("transpose_56")]; tensor key_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = key_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = key_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_10_squeeze_mask_0, stride = key_cache_internal_tensor_assign_10_stride_0, update = key_states_97_cast_fp16, x = coreml_update_state_16)[name = string("key_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_10_cast_fp16, input = key_cache)[name = string("coreml_update_state_18_write_state")]; tensor coreml_update_state_18 = read_state(input = key_cache)[name = string("coreml_update_state_18")]; tensor value_states_57_perm_0 = const()[name = string("value_states_57_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_10_stride_0 = const()[name = string("value_cache_internal_tensor_assign_10_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_10_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_10_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_10_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_10_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_57_cast_fp16 = transpose(perm = value_states_57_perm_0, x = var_3880_cast_fp16)[name = string("transpose_55")]; tensor value_cache_internal_tensor_assign_10_cast_fp16 = slice_update(begin = concat_113, begin_mask = value_cache_internal_tensor_assign_10_begin_mask_0, end = concat_114, end_mask = value_cache_internal_tensor_assign_10_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_10_squeeze_mask_0, stride = value_cache_internal_tensor_assign_10_stride_0, update = value_states_57_cast_fp16, x = coreml_update_state_17)[name = string("value_cache_internal_tensor_assign_10_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_10_cast_fp16, input = value_cache)[name = string("coreml_update_state_19_write_state")]; tensor coreml_update_state_19 = read_state(input = value_cache)[name = string("coreml_update_state_19")]; tensor var_3974_begin_0 = const()[name = string("op_3974_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3974_end_0 = const()[name = string("op_3974_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3974_end_mask_0 = const()[name = string("op_3974_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3974_cast_fp16 = slice_by_index(begin = var_3974_begin_0, end = var_3974_end_0, end_mask = var_3974_end_mask_0, x = coreml_update_state_18)[name = string("op_3974_cast_fp16")]; tensor tile_18 = const()[name = string("tile_18"), val = tensor([1, 1])]; int32 var_3977_axis_0 = const()[name = string("op_3977_axis_0"), val = int32(1)]; tensor var_3977_cast_fp16_0, tensor var_3977_cast_fp16_1 = split(axis = var_3977_axis_0, split_sizes = tile_18, x = var_3974_cast_fp16)[name = string("op_3977_cast_fp16")]; tensor var_3984_begin_0 = const()[name = string("op_3984_begin_0"), val = tensor([9, 0, 0, 0])]; tensor var_3984_end_0 = const()[name = string("op_3984_end_0"), val = tensor([10, 2, 2048, 128])]; tensor var_3984_end_mask_0 = const()[name = string("op_3984_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_3984_cast_fp16 = slice_by_index(begin = var_3984_begin_0, end = var_3984_end_0, end_mask = var_3984_end_mask_0, x = coreml_update_state_19)[name = string("op_3984_cast_fp16")]; tensor tile_19 = const()[name = string("tile_19"), val = tensor([1, 1])]; int32 var_3987_axis_0 = const()[name = string("op_3987_axis_0"), val = int32(1)]; tensor var_3987_cast_fp16_0, tensor var_3987_cast_fp16_1 = split(axis = var_3987_axis_0, split_sizes = tile_19, x = var_3984_cast_fp16)[name = string("op_3987_cast_fp16")]; tensor var_3990_split_sizes_0 = const()[name = string("op_3990_split_sizes_0"), val = tensor([8, 8])]; int32 var_3990_axis_0 = const()[name = string("op_3990_axis_0"), val = int32(1)]; tensor var_3990_0, tensor var_3990_1 = split(axis = var_3990_axis_0, split_sizes = var_3990_split_sizes_0, x = query_states_57_cast_fp16)[name = string("op_3990")]; bool attn_weights_145_transpose_x_0 = const()[name = string("attn_weights_145_transpose_x_0"), val = bool(false)]; bool attn_weights_145_transpose_y_0 = const()[name = string("attn_weights_145_transpose_y_0"), val = bool(false)]; tensor attn_weights_145_cast_fp16 = matmul(transpose_x = attn_weights_145_transpose_x_0, transpose_y = attn_weights_145_transpose_y_0, x = var_3977_cast_fp16_0, y = var_3990_0)[name = string("attn_weights_145_cast_fp16")]; fp16 var_3993_to_fp16 = const()[name = string("op_3993_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_147_cast_fp16 = mul(x = attn_weights_145_cast_fp16, y = var_3993_to_fp16)[name = string("attn_weights_147_cast_fp16")]; tensor attn_weights_149_cast_fp16 = add(x = attn_weights_147_cast_fp16, y = attn_mask_1)[name = string("attn_weights_149_cast_fp16")]; int32 var_3997 = const()[name = string("op_3997"), val = int32(-2)]; tensor attn_weights_151_cast_fp16 = softmax(axis = var_3997, x = attn_weights_149_cast_fp16)[name = string("attn_weights_151_cast_fp16")]; bool var_4003_transpose_x_1 = const()[name = string("op_4003_transpose_x_1"), val = bool(true)]; bool var_4003_transpose_y_1 = const()[name = string("op_4003_transpose_y_1"), val = bool(false)]; tensor var_4003_cast_fp16 = matmul(transpose_x = var_4003_transpose_x_1, transpose_y = var_4003_transpose_y_1, x = attn_weights_151_cast_fp16, y = var_3987_cast_fp16_0)[name = string("op_4003_cast_fp16")]; bool attn_weights_153_transpose_x_0 = const()[name = string("attn_weights_153_transpose_x_0"), val = bool(false)]; bool attn_weights_153_transpose_y_0 = const()[name = string("attn_weights_153_transpose_y_0"), val = bool(false)]; tensor attn_weights_153_cast_fp16 = matmul(transpose_x = attn_weights_153_transpose_x_0, transpose_y = attn_weights_153_transpose_y_0, x = var_3977_cast_fp16_1, y = var_3990_1)[name = string("attn_weights_153_cast_fp16")]; fp16 var_4005_to_fp16 = const()[name = string("op_4005_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_155_cast_fp16 = mul(x = attn_weights_153_cast_fp16, y = var_4005_to_fp16)[name = string("attn_weights_155_cast_fp16")]; tensor attn_weights_157_cast_fp16 = add(x = attn_weights_155_cast_fp16, y = attn_mask_1)[name = string("attn_weights_157_cast_fp16")]; int32 var_4009 = const()[name = string("op_4009"), val = int32(-2)]; tensor attn_weights_159_cast_fp16 = softmax(axis = var_4009, x = attn_weights_157_cast_fp16)[name = string("attn_weights_159_cast_fp16")]; bool attn_output_73_transpose_x_1 = const()[name = string("attn_output_73_transpose_x_1"), val = bool(true)]; bool attn_output_73_transpose_y_1 = const()[name = string("attn_output_73_transpose_y_1"), val = bool(false)]; tensor attn_output_73_cast_fp16 = matmul(transpose_x = attn_output_73_transpose_x_1, transpose_y = attn_output_73_transpose_y_1, x = attn_weights_159_cast_fp16, y = var_3987_cast_fp16_1)[name = string("attn_output_73_cast_fp16")]; int32 var_4017 = const()[name = string("op_4017"), val = int32(1)]; bool attn_output_75_interleave_0 = const()[name = string("attn_output_75_interleave_0"), val = bool(false)]; tensor attn_output_75_cast_fp16 = concat(axis = var_4017, interleave = attn_output_75_interleave_0, values = (var_4003_cast_fp16, attn_output_73_cast_fp16))[name = string("attn_output_75_cast_fp16")]; tensor var_4021_perm_0 = const()[name = string("op_4021_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_119x = const()[name = string("concat_119x"), val = tensor([1, 2048, 1, -1])]; tensor var_4021_cast_fp16 = transpose(perm = var_4021_perm_0, x = attn_output_75_cast_fp16)[name = string("transpose_54")]; tensor attn_output_79_cast_fp16 = reshape(shape = concat_119x, x = var_4021_cast_fp16)[name = string("attn_output_79_cast_fp16")]; tensor hidden_states_93_strides_0 = const()[name = string("hidden_states_93_strides_0"), val = tensor([1, 1])]; string hidden_states_93_pad_type_0 = const()[name = string("hidden_states_93_pad_type_0"), val = string("valid")]; tensor hidden_states_93_pad_0 = const()[name = string("hidden_states_93_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_93_dilations_0 = const()[name = string("hidden_states_93_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_93_groups_0 = const()[name = string("hidden_states_93_groups_0"), val = int32(1)]; tensor hidden_states_93_cast_fp16 = conv(dilations = hidden_states_93_dilations_0, groups = hidden_states_93_groups_0, pad = hidden_states_93_pad_0, pad_type = hidden_states_93_pad_type_0, strides = hidden_states_93_strides_0, weight = layers_9_self_attn_o_proj_weight_cast_fp16, x = attn_output_79_cast_fp16)[name = string("hidden_states_93_cast_fp16")]; tensor hidden_states_95_cast_fp16 = add(x = hidden_states_89_cast_fp16, y = hidden_states_93_cast_fp16)[name = string("hidden_states_95_cast_fp16")]; fp16 const_98_promoted_to_fp16 = const()[name = string("const_98_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4054_cast_fp16 = mul(x = hidden_states_95_cast_fp16, y = const_98_promoted_to_fp16)[name = string("op_4054_cast_fp16")]; int32 var_4052 = const()[name = string("op_4052"), val = int32(1)]; bool doubled_77_interleave_0 = const()[name = string("doubled_77_interleave_0"), val = bool(false)]; tensor doubled_77_cast_fp16 = concat(axis = var_4052, interleave = doubled_77_interleave_0, values = (hidden_states_95_cast_fp16, var_4054_cast_fp16))[name = string("doubled_77_cast_fp16")]; tensor out_39_axes_0 = const()[name = string("out_39_axes_0"), val = tensor([1])]; tensor out_39_gamma_0_to_fp16 = const()[name = string("out_39_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405896064)))]; fp16 var_4064_to_fp16 = const()[name = string("op_4064_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_39_cast_fp16 = layer_norm(axes = out_39_axes_0, epsilon = var_4064_to_fp16, gamma = out_39_gamma_0_to_fp16, x = doubled_77_cast_fp16)[name = string("out_39_cast_fp16")]; tensor var_4075_split_sizes_0 = const()[name = string("op_4075_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4075_axis_0 = const()[name = string("op_4075_axis_0"), val = int32(1)]; tensor var_4075_cast_fp16_0, tensor var_4075_cast_fp16_1 = split(axis = var_4075_axis_0, split_sizes = var_4075_split_sizes_0, x = out_39_cast_fp16)[name = string("op_4075_cast_fp16")]; tensor input_19_strides_0 = const()[name = string("input_19_strides_0"), val = tensor([1, 1])]; string input_19_pad_type_0 = const()[name = string("input_19_pad_type_0"), val = string("valid")]; tensor input_19_pad_0 = const()[name = string("input_19_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_19_dilations_0 = const()[name = string("input_19_dilations_0"), val = tensor([1, 1])]; int32 input_19_groups_0 = const()[name = string("input_19_groups_0"), val = int32(1)]; tensor input_19_cast_fp16 = conv(dilations = input_19_dilations_0, groups = input_19_groups_0, pad = input_19_pad_0, pad_type = input_19_pad_type_0, strides = input_19_strides_0, weight = layers_9_mlp_gate_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("input_19_cast_fp16")]; tensor var_4092_cast_fp16 = silu(x = input_19_cast_fp16)[name = string("op_4092_cast_fp16")]; tensor var_4098_strides_0 = const()[name = string("op_4098_strides_0"), val = tensor([1, 1])]; string var_4098_pad_type_0 = const()[name = string("op_4098_pad_type_0"), val = string("valid")]; tensor var_4098_pad_0 = const()[name = string("op_4098_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4098_dilations_0 = const()[name = string("op_4098_dilations_0"), val = tensor([1, 1])]; int32 var_4098_groups_0 = const()[name = string("op_4098_groups_0"), val = int32(1)]; tensor var_4098_cast_fp16 = conv(dilations = var_4098_dilations_0, groups = var_4098_groups_0, pad = var_4098_pad_0, pad_type = var_4098_pad_type_0, strides = var_4098_strides_0, weight = layers_9_mlp_up_proj_weight_cast_fp16, x = var_4075_cast_fp16_0)[name = string("op_4098_cast_fp16")]; tensor x_99_cast_fp16 = mul(x = var_4092_cast_fp16, y = var_4098_cast_fp16)[name = string("x_99_cast_fp16")]; tensor hidden_states_97_strides_0 = const()[name = string("hidden_states_97_strides_0"), val = tensor([1, 1])]; string hidden_states_97_pad_type_0 = const()[name = string("hidden_states_97_pad_type_0"), val = string("valid")]; tensor hidden_states_97_pad_0 = const()[name = string("hidden_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_97_dilations_0 = const()[name = string("hidden_states_97_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_97_groups_0 = const()[name = string("hidden_states_97_groups_0"), val = int32(1)]; tensor hidden_states_97_cast_fp16 = conv(dilations = hidden_states_97_dilations_0, groups = hidden_states_97_groups_0, pad = hidden_states_97_pad_0, pad_type = hidden_states_97_pad_type_0, strides = hidden_states_97_strides_0, weight = layers_9_mlp_down_proj_weight_cast_fp16, x = x_99_cast_fp16)[name = string("hidden_states_97_cast_fp16")]; tensor hidden_states_99_cast_fp16 = add(x = hidden_states_95_cast_fp16, y = hidden_states_97_cast_fp16)[name = string("hidden_states_99_cast_fp16")]; fp16 const_100_promoted_to_fp16 = const()[name = string("const_100_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4116_cast_fp16 = mul(x = hidden_states_99_cast_fp16, y = const_100_promoted_to_fp16)[name = string("op_4116_cast_fp16")]; int32 var_4114 = const()[name = string("op_4114"), val = int32(1)]; bool doubled_81_interleave_0 = const()[name = string("doubled_81_interleave_0"), val = bool(false)]; tensor doubled_81_cast_fp16 = concat(axis = var_4114, interleave = doubled_81_interleave_0, values = (hidden_states_99_cast_fp16, var_4116_cast_fp16))[name = string("doubled_81_cast_fp16")]; tensor out_41_axes_0 = const()[name = string("out_41_axes_0"), val = tensor([1])]; tensor out_41_gamma_0_to_fp16 = const()[name = string("out_41_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405904320)))]; fp16 var_4126_to_fp16 = const()[name = string("op_4126_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_41_cast_fp16 = layer_norm(axes = out_41_axes_0, epsilon = var_4126_to_fp16, gamma = out_41_gamma_0_to_fp16, x = doubled_81_cast_fp16)[name = string("out_41_cast_fp16")]; tensor var_4137_split_sizes_0 = const()[name = string("op_4137_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4137_axis_0 = const()[name = string("op_4137_axis_0"), val = int32(1)]; tensor var_4137_cast_fp16_0, tensor var_4137_cast_fp16_1 = split(axis = var_4137_axis_0, split_sizes = var_4137_split_sizes_0, x = out_41_cast_fp16)[name = string("op_4137_cast_fp16")]; tensor layers_10_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1405912576)))]; tensor query_states_61_strides_0 = const()[name = string("query_states_61_strides_0"), val = tensor([1, 1])]; string query_states_61_pad_type_0 = const()[name = string("query_states_61_pad_type_0"), val = string("valid")]; tensor query_states_61_pad_0 = const()[name = string("query_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_61_dilations_0 = const()[name = string("query_states_61_dilations_0"), val = tensor([1, 1])]; int32 query_states_61_groups_0 = const()[name = string("query_states_61_groups_0"), val = int32(1)]; tensor query_states_61_cast_fp16 = conv(dilations = query_states_61_dilations_0, groups = query_states_61_groups_0, pad = query_states_61_pad_0, pad_type = query_states_61_pad_type_0, strides = query_states_61_strides_0, weight = layers_10_self_attn_q_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("query_states_61_cast_fp16")]; tensor layers_10_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_10_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1414301248)))]; tensor key_states_101_strides_0 = const()[name = string("key_states_101_strides_0"), val = tensor([1, 1])]; string key_states_101_pad_type_0 = const()[name = string("key_states_101_pad_type_0"), val = string("valid")]; tensor key_states_101_pad_0 = const()[name = string("key_states_101_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_101_dilations_0 = const()[name = string("key_states_101_dilations_0"), val = tensor([1, 1])]; int32 key_states_101_groups_0 = const()[name = string("key_states_101_groups_0"), val = int32(1)]; tensor key_states_101_cast_fp16 = conv(dilations = key_states_101_dilations_0, groups = key_states_101_groups_0, pad = key_states_101_pad_0, pad_type = key_states_101_pad_type_0, strides = key_states_101_strides_0, weight = layers_10_self_attn_k_proj_weight_to_fp16, x = var_4137_cast_fp16_0)[name = string("key_states_101_cast_fp16")]; tensor value_states_61_strides_0 = const()[name = string("value_states_61_strides_0"), val = tensor([1, 1])]; string value_states_61_pad_type_0 = const()[name = string("value_states_61_pad_type_0"), val = string("valid")]; tensor value_states_61_pad_0 = const()[name = string("value_states_61_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_61_dilations_0 = const()[name = string("value_states_61_dilations_0"), val = tensor([1, 1])]; int32 value_states_61_groups_0 = const()[name = string("value_states_61_groups_0"), val = int32(1)]; tensor value_states_61_cast_fp16 = conv(dilations = value_states_61_dilations_0, groups = value_states_61_groups_0, pad = value_states_61_pad_0, pad_type = value_states_61_pad_type_0, strides = value_states_61_strides_0, weight = layers_10_self_attn_v_proj_weight_cast_fp16, x = var_4137_cast_fp16_0)[name = string("value_states_61_cast_fp16")]; tensor concat_120x = const()[name = string("concat_120x"), val = tensor([1, 16, 128, -1])]; tensor x_101_cast_fp16 = reshape(shape = concat_120x, x = query_states_61_cast_fp16)[name = string("x_101_cast_fp16")]; tensor concat_121x = const()[name = string("concat_121x"), val = tensor([1, 2, 128, -1])]; tensor var_4194_cast_fp16 = reshape(shape = concat_121x, x = key_states_101_cast_fp16)[name = string("op_4194_cast_fp16")]; tensor concat_122x = const()[name = string("concat_122x"), val = tensor([1, 2, 128, -1])]; tensor var_4201_cast_fp16 = reshape(shape = concat_122x, x = value_states_61_cast_fp16)[name = string("op_4201_cast_fp16")]; tensor var_4205_cast_fp16 = mul(x = x_101_cast_fp16, y = var_869_cast_fp16)[name = string("op_4205_cast_fp16")]; tensor var_4206_split_sizes_0 = const()[name = string("op_4206_split_sizes_0"), val = tensor([64, 64])]; int32 var_4206_axis_0 = const()[name = string("op_4206_axis_0"), val = int32(-2)]; tensor var_4206_cast_fp16_0, tensor var_4206_cast_fp16_1 = split(axis = var_4206_axis_0, split_sizes = var_4206_split_sizes_0, x = x_101_cast_fp16)[name = string("op_4206_cast_fp16")]; fp16 const_102_promoted_to_fp16 = const()[name = string("const_102_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4208_cast_fp16 = mul(x = var_4206_cast_fp16_1, y = const_102_promoted_to_fp16)[name = string("op_4208_cast_fp16")]; int32 var_4210 = const()[name = string("op_4210"), val = int32(-2)]; bool var_4211_interleave_0 = const()[name = string("op_4211_interleave_0"), val = bool(false)]; tensor var_4211_cast_fp16 = concat(axis = var_4210, interleave = var_4211_interleave_0, values = (var_4208_cast_fp16, var_4206_cast_fp16_0))[name = string("op_4211_cast_fp16")]; tensor var_4212_cast_fp16 = mul(x = var_4211_cast_fp16, y = var_878_cast_fp16)[name = string("op_4212_cast_fp16")]; tensor query_states_63_cast_fp16 = add(x = var_4205_cast_fp16, y = var_4212_cast_fp16)[name = string("query_states_63_cast_fp16")]; tensor var_4218_cast_fp16 = mul(x = var_4194_cast_fp16, y = var_869_cast_fp16)[name = string("op_4218_cast_fp16")]; tensor var_4219_split_sizes_0 = const()[name = string("op_4219_split_sizes_0"), val = tensor([64, 64])]; int32 var_4219_axis_0 = const()[name = string("op_4219_axis_0"), val = int32(-2)]; tensor var_4219_cast_fp16_0, tensor var_4219_cast_fp16_1 = split(axis = var_4219_axis_0, split_sizes = var_4219_split_sizes_0, x = var_4194_cast_fp16)[name = string("op_4219_cast_fp16")]; fp16 const_103_promoted_to_fp16 = const()[name = string("const_103_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4221_cast_fp16 = mul(x = var_4219_cast_fp16_1, y = const_103_promoted_to_fp16)[name = string("op_4221_cast_fp16")]; int32 var_4223 = const()[name = string("op_4223"), val = int32(-2)]; bool var_4224_interleave_0 = const()[name = string("op_4224_interleave_0"), val = bool(false)]; tensor var_4224_cast_fp16 = concat(axis = var_4223, interleave = var_4224_interleave_0, values = (var_4221_cast_fp16, var_4219_cast_fp16_0))[name = string("op_4224_cast_fp16")]; tensor var_4225_cast_fp16 = mul(x = var_4224_cast_fp16, y = var_878_cast_fp16)[name = string("op_4225_cast_fp16")]; tensor key_states_105_cast_fp16 = add(x = var_4218_cast_fp16, y = var_4225_cast_fp16)[name = string("key_states_105_cast_fp16")]; tensor expand_dims_120 = const()[name = string("expand_dims_120"), val = tensor([10])]; tensor expand_dims_121 = const()[name = string("expand_dims_121"), val = tensor([0])]; tensor expand_dims_123 = const()[name = string("expand_dims_123"), val = tensor([0])]; int32 concat_125_axis_0 = const()[name = string("concat_125_axis_0"), val = int32(0)]; bool concat_125_interleave_0 = const()[name = string("concat_125_interleave_0"), val = bool(false)]; tensor concat_125 = concat(axis = concat_125_axis_0, interleave = concat_125_interleave_0, values = (expand_dims_120, expand_dims_121, position_id, expand_dims_123))[name = string("concat_125")]; tensor expand_dims_124 = const()[name = string("expand_dims_124"), val = tensor([11])]; tensor concat_126_values1_0 = const()[name = string("concat_126_values1_0"), val = tensor([0])]; tensor concat_126_values3_0 = const()[name = string("concat_126_values3_0"), val = tensor([0])]; int32 concat_126_axis_0 = const()[name = string("concat_126_axis_0"), val = int32(0)]; bool concat_126_interleave_0 = const()[name = string("concat_126_interleave_0"), val = bool(false)]; tensor concat_126 = concat(axis = concat_126_axis_0, interleave = concat_126_interleave_0, values = (expand_dims_124, concat_126_values1_0, cache_position_end, concat_126_values3_0))[name = string("concat_126")]; tensor key_states_107_perm_0 = const()[name = string("key_states_107_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_11_stride_0 = const()[name = string("key_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_107_cast_fp16 = transpose(perm = key_states_107_perm_0, x = key_states_105_cast_fp16)[name = string("transpose_53")]; tensor key_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = key_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = key_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_11_squeeze_mask_0, stride = key_cache_internal_tensor_assign_11_stride_0, update = key_states_107_cast_fp16, x = coreml_update_state_18)[name = string("key_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_11_cast_fp16, input = key_cache)[name = string("coreml_update_state_20_write_state")]; tensor coreml_update_state_20 = read_state(input = key_cache)[name = string("coreml_update_state_20")]; tensor value_states_63_perm_0 = const()[name = string("value_states_63_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_11_stride_0 = const()[name = string("value_cache_internal_tensor_assign_11_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_11_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_11_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_11_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_11_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_63_cast_fp16 = transpose(perm = value_states_63_perm_0, x = var_4201_cast_fp16)[name = string("transpose_52")]; tensor value_cache_internal_tensor_assign_11_cast_fp16 = slice_update(begin = concat_125, begin_mask = value_cache_internal_tensor_assign_11_begin_mask_0, end = concat_126, end_mask = value_cache_internal_tensor_assign_11_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_11_squeeze_mask_0, stride = value_cache_internal_tensor_assign_11_stride_0, update = value_states_63_cast_fp16, x = coreml_update_state_19)[name = string("value_cache_internal_tensor_assign_11_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_11_cast_fp16, input = value_cache)[name = string("coreml_update_state_21_write_state")]; tensor coreml_update_state_21 = read_state(input = value_cache)[name = string("coreml_update_state_21")]; tensor var_4295_begin_0 = const()[name = string("op_4295_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4295_end_0 = const()[name = string("op_4295_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4295_end_mask_0 = const()[name = string("op_4295_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4295_cast_fp16 = slice_by_index(begin = var_4295_begin_0, end = var_4295_end_0, end_mask = var_4295_end_mask_0, x = coreml_update_state_20)[name = string("op_4295_cast_fp16")]; tensor tile_20 = const()[name = string("tile_20"), val = tensor([1, 1])]; int32 var_4298_axis_0 = const()[name = string("op_4298_axis_0"), val = int32(1)]; tensor var_4298_cast_fp16_0, tensor var_4298_cast_fp16_1 = split(axis = var_4298_axis_0, split_sizes = tile_20, x = var_4295_cast_fp16)[name = string("op_4298_cast_fp16")]; tensor var_4305_begin_0 = const()[name = string("op_4305_begin_0"), val = tensor([10, 0, 0, 0])]; tensor var_4305_end_0 = const()[name = string("op_4305_end_0"), val = tensor([11, 2, 2048, 128])]; tensor var_4305_end_mask_0 = const()[name = string("op_4305_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4305_cast_fp16 = slice_by_index(begin = var_4305_begin_0, end = var_4305_end_0, end_mask = var_4305_end_mask_0, x = coreml_update_state_21)[name = string("op_4305_cast_fp16")]; tensor tile_21 = const()[name = string("tile_21"), val = tensor([1, 1])]; int32 var_4308_axis_0 = const()[name = string("op_4308_axis_0"), val = int32(1)]; tensor var_4308_cast_fp16_0, tensor var_4308_cast_fp16_1 = split(axis = var_4308_axis_0, split_sizes = tile_21, x = var_4305_cast_fp16)[name = string("op_4308_cast_fp16")]; tensor var_4311_split_sizes_0 = const()[name = string("op_4311_split_sizes_0"), val = tensor([8, 8])]; int32 var_4311_axis_0 = const()[name = string("op_4311_axis_0"), val = int32(1)]; tensor var_4311_0, tensor var_4311_1 = split(axis = var_4311_axis_0, split_sizes = var_4311_split_sizes_0, x = query_states_63_cast_fp16)[name = string("op_4311")]; bool attn_weights_161_transpose_x_0 = const()[name = string("attn_weights_161_transpose_x_0"), val = bool(false)]; bool attn_weights_161_transpose_y_0 = const()[name = string("attn_weights_161_transpose_y_0"), val = bool(false)]; tensor attn_weights_161_cast_fp16 = matmul(transpose_x = attn_weights_161_transpose_x_0, transpose_y = attn_weights_161_transpose_y_0, x = var_4298_cast_fp16_0, y = var_4311_0)[name = string("attn_weights_161_cast_fp16")]; fp16 var_4314_to_fp16 = const()[name = string("op_4314_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_163_cast_fp16 = mul(x = attn_weights_161_cast_fp16, y = var_4314_to_fp16)[name = string("attn_weights_163_cast_fp16")]; tensor attn_weights_165_cast_fp16 = add(x = attn_weights_163_cast_fp16, y = attn_mask_1)[name = string("attn_weights_165_cast_fp16")]; int32 var_4318 = const()[name = string("op_4318"), val = int32(-2)]; tensor attn_weights_167_cast_fp16 = softmax(axis = var_4318, x = attn_weights_165_cast_fp16)[name = string("attn_weights_167_cast_fp16")]; bool var_4324_transpose_x_1 = const()[name = string("op_4324_transpose_x_1"), val = bool(true)]; bool var_4324_transpose_y_1 = const()[name = string("op_4324_transpose_y_1"), val = bool(false)]; tensor var_4324_cast_fp16 = matmul(transpose_x = var_4324_transpose_x_1, transpose_y = var_4324_transpose_y_1, x = attn_weights_167_cast_fp16, y = var_4308_cast_fp16_0)[name = string("op_4324_cast_fp16")]; bool attn_weights_169_transpose_x_0 = const()[name = string("attn_weights_169_transpose_x_0"), val = bool(false)]; bool attn_weights_169_transpose_y_0 = const()[name = string("attn_weights_169_transpose_y_0"), val = bool(false)]; tensor attn_weights_169_cast_fp16 = matmul(transpose_x = attn_weights_169_transpose_x_0, transpose_y = attn_weights_169_transpose_y_0, x = var_4298_cast_fp16_1, y = var_4311_1)[name = string("attn_weights_169_cast_fp16")]; fp16 var_4326_to_fp16 = const()[name = string("op_4326_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_171_cast_fp16 = mul(x = attn_weights_169_cast_fp16, y = var_4326_to_fp16)[name = string("attn_weights_171_cast_fp16")]; tensor attn_weights_173_cast_fp16 = add(x = attn_weights_171_cast_fp16, y = attn_mask_1)[name = string("attn_weights_173_cast_fp16")]; int32 var_4330 = const()[name = string("op_4330"), val = int32(-2)]; tensor attn_weights_175_cast_fp16 = softmax(axis = var_4330, x = attn_weights_173_cast_fp16)[name = string("attn_weights_175_cast_fp16")]; bool attn_output_81_transpose_x_1 = const()[name = string("attn_output_81_transpose_x_1"), val = bool(true)]; bool attn_output_81_transpose_y_1 = const()[name = string("attn_output_81_transpose_y_1"), val = bool(false)]; tensor attn_output_81_cast_fp16 = matmul(transpose_x = attn_output_81_transpose_x_1, transpose_y = attn_output_81_transpose_y_1, x = attn_weights_175_cast_fp16, y = var_4308_cast_fp16_1)[name = string("attn_output_81_cast_fp16")]; int32 var_4338 = const()[name = string("op_4338"), val = int32(1)]; bool attn_output_83_interleave_0 = const()[name = string("attn_output_83_interleave_0"), val = bool(false)]; tensor attn_output_83_cast_fp16 = concat(axis = var_4338, interleave = attn_output_83_interleave_0, values = (var_4324_cast_fp16, attn_output_81_cast_fp16))[name = string("attn_output_83_cast_fp16")]; tensor var_4342_perm_0 = const()[name = string("op_4342_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_131x = const()[name = string("concat_131x"), val = tensor([1, 2048, 1, -1])]; tensor var_4342_cast_fp16 = transpose(perm = var_4342_perm_0, x = attn_output_83_cast_fp16)[name = string("transpose_51")]; tensor attn_output_87_cast_fp16 = reshape(shape = concat_131x, x = var_4342_cast_fp16)[name = string("attn_output_87_cast_fp16")]; tensor hidden_states_103_strides_0 = const()[name = string("hidden_states_103_strides_0"), val = tensor([1, 1])]; string hidden_states_103_pad_type_0 = const()[name = string("hidden_states_103_pad_type_0"), val = string("valid")]; tensor hidden_states_103_pad_0 = const()[name = string("hidden_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_103_dilations_0 = const()[name = string("hidden_states_103_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_103_groups_0 = const()[name = string("hidden_states_103_groups_0"), val = int32(1)]; tensor hidden_states_103_cast_fp16 = conv(dilations = hidden_states_103_dilations_0, groups = hidden_states_103_groups_0, pad = hidden_states_103_pad_0, pad_type = hidden_states_103_pad_type_0, strides = hidden_states_103_strides_0, weight = layers_10_self_attn_o_proj_weight_cast_fp16, x = attn_output_87_cast_fp16)[name = string("hidden_states_103_cast_fp16")]; tensor hidden_states_105_cast_fp16 = add(x = hidden_states_99_cast_fp16, y = hidden_states_103_cast_fp16)[name = string("hidden_states_105_cast_fp16")]; fp16 const_108_promoted_to_fp16 = const()[name = string("const_108_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4375_cast_fp16 = mul(x = hidden_states_105_cast_fp16, y = const_108_promoted_to_fp16)[name = string("op_4375_cast_fp16")]; int32 var_4373 = const()[name = string("op_4373"), val = int32(1)]; bool doubled_85_interleave_0 = const()[name = string("doubled_85_interleave_0"), val = bool(false)]; tensor doubled_85_cast_fp16 = concat(axis = var_4373, interleave = doubled_85_interleave_0, values = (hidden_states_105_cast_fp16, var_4375_cast_fp16))[name = string("doubled_85_cast_fp16")]; tensor out_43_axes_0 = const()[name = string("out_43_axes_0"), val = tensor([1])]; tensor out_43_gamma_0_to_fp16 = const()[name = string("out_43_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415349888)))]; fp16 var_4385_to_fp16 = const()[name = string("op_4385_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_43_cast_fp16 = layer_norm(axes = out_43_axes_0, epsilon = var_4385_to_fp16, gamma = out_43_gamma_0_to_fp16, x = doubled_85_cast_fp16)[name = string("out_43_cast_fp16")]; tensor var_4396_split_sizes_0 = const()[name = string("op_4396_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4396_axis_0 = const()[name = string("op_4396_axis_0"), val = int32(1)]; tensor var_4396_cast_fp16_0, tensor var_4396_cast_fp16_1 = split(axis = var_4396_axis_0, split_sizes = var_4396_split_sizes_0, x = out_43_cast_fp16)[name = string("op_4396_cast_fp16")]; tensor input_21_strides_0 = const()[name = string("input_21_strides_0"), val = tensor([1, 1])]; string input_21_pad_type_0 = const()[name = string("input_21_pad_type_0"), val = string("valid")]; tensor input_21_pad_0 = const()[name = string("input_21_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_21_dilations_0 = const()[name = string("input_21_dilations_0"), val = tensor([1, 1])]; int32 input_21_groups_0 = const()[name = string("input_21_groups_0"), val = int32(1)]; tensor input_21_cast_fp16 = conv(dilations = input_21_dilations_0, groups = input_21_groups_0, pad = input_21_pad_0, pad_type = input_21_pad_type_0, strides = input_21_strides_0, weight = layers_10_mlp_gate_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("input_21_cast_fp16")]; tensor var_4413_cast_fp16 = silu(x = input_21_cast_fp16)[name = string("op_4413_cast_fp16")]; tensor var_4419_strides_0 = const()[name = string("op_4419_strides_0"), val = tensor([1, 1])]; string var_4419_pad_type_0 = const()[name = string("op_4419_pad_type_0"), val = string("valid")]; tensor var_4419_pad_0 = const()[name = string("op_4419_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4419_dilations_0 = const()[name = string("op_4419_dilations_0"), val = tensor([1, 1])]; int32 var_4419_groups_0 = const()[name = string("op_4419_groups_0"), val = int32(1)]; tensor var_4419_cast_fp16 = conv(dilations = var_4419_dilations_0, groups = var_4419_groups_0, pad = var_4419_pad_0, pad_type = var_4419_pad_type_0, strides = var_4419_strides_0, weight = layers_10_mlp_up_proj_weight_cast_fp16, x = var_4396_cast_fp16_0)[name = string("op_4419_cast_fp16")]; tensor x_109_cast_fp16 = mul(x = var_4413_cast_fp16, y = var_4419_cast_fp16)[name = string("x_109_cast_fp16")]; tensor hidden_states_107_strides_0 = const()[name = string("hidden_states_107_strides_0"), val = tensor([1, 1])]; string hidden_states_107_pad_type_0 = const()[name = string("hidden_states_107_pad_type_0"), val = string("valid")]; tensor hidden_states_107_pad_0 = const()[name = string("hidden_states_107_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_107_dilations_0 = const()[name = string("hidden_states_107_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_107_groups_0 = const()[name = string("hidden_states_107_groups_0"), val = int32(1)]; tensor hidden_states_107_cast_fp16 = conv(dilations = hidden_states_107_dilations_0, groups = hidden_states_107_groups_0, pad = hidden_states_107_pad_0, pad_type = hidden_states_107_pad_type_0, strides = hidden_states_107_strides_0, weight = layers_10_mlp_down_proj_weight_cast_fp16, x = x_109_cast_fp16)[name = string("hidden_states_107_cast_fp16")]; tensor hidden_states_109_cast_fp16 = add(x = hidden_states_105_cast_fp16, y = hidden_states_107_cast_fp16)[name = string("hidden_states_109_cast_fp16")]; fp16 const_110_promoted_to_fp16 = const()[name = string("const_110_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4437_cast_fp16 = mul(x = hidden_states_109_cast_fp16, y = const_110_promoted_to_fp16)[name = string("op_4437_cast_fp16")]; int32 var_4435 = const()[name = string("op_4435"), val = int32(1)]; bool doubled_89_interleave_0 = const()[name = string("doubled_89_interleave_0"), val = bool(false)]; tensor doubled_89_cast_fp16 = concat(axis = var_4435, interleave = doubled_89_interleave_0, values = (hidden_states_109_cast_fp16, var_4437_cast_fp16))[name = string("doubled_89_cast_fp16")]; tensor out_45_axes_0 = const()[name = string("out_45_axes_0"), val = tensor([1])]; tensor out_45_gamma_0_to_fp16 = const()[name = string("out_45_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415358144)))]; fp16 var_4447_to_fp16 = const()[name = string("op_4447_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_45_cast_fp16 = layer_norm(axes = out_45_axes_0, epsilon = var_4447_to_fp16, gamma = out_45_gamma_0_to_fp16, x = doubled_89_cast_fp16)[name = string("out_45_cast_fp16")]; tensor var_4458_split_sizes_0 = const()[name = string("op_4458_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4458_axis_0 = const()[name = string("op_4458_axis_0"), val = int32(1)]; tensor var_4458_cast_fp16_0, tensor var_4458_cast_fp16_1 = split(axis = var_4458_axis_0, split_sizes = var_4458_split_sizes_0, x = out_45_cast_fp16)[name = string("op_4458_cast_fp16")]; tensor query_states_67_strides_0 = const()[name = string("query_states_67_strides_0"), val = tensor([1, 1])]; string query_states_67_pad_type_0 = const()[name = string("query_states_67_pad_type_0"), val = string("valid")]; tensor query_states_67_pad_0 = const()[name = string("query_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_67_dilations_0 = const()[name = string("query_states_67_dilations_0"), val = tensor([1, 1])]; int32 query_states_67_groups_0 = const()[name = string("query_states_67_groups_0"), val = int32(1)]; tensor query_states_67_cast_fp16 = conv(dilations = query_states_67_dilations_0, groups = query_states_67_groups_0, pad = query_states_67_pad_0, pad_type = query_states_67_pad_type_0, strides = query_states_67_strides_0, weight = layers_11_self_attn_q_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("query_states_67_cast_fp16")]; tensor key_states_111_strides_0 = const()[name = string("key_states_111_strides_0"), val = tensor([1, 1])]; string key_states_111_pad_type_0 = const()[name = string("key_states_111_pad_type_0"), val = string("valid")]; tensor key_states_111_pad_0 = const()[name = string("key_states_111_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_111_dilations_0 = const()[name = string("key_states_111_dilations_0"), val = tensor([1, 1])]; int32 key_states_111_groups_0 = const()[name = string("key_states_111_groups_0"), val = int32(1)]; tensor key_states_111_cast_fp16 = conv(dilations = key_states_111_dilations_0, groups = key_states_111_groups_0, pad = key_states_111_pad_0, pad_type = key_states_111_pad_type_0, strides = key_states_111_strides_0, weight = layers_11_self_attn_k_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("key_states_111_cast_fp16")]; tensor value_states_67_strides_0 = const()[name = string("value_states_67_strides_0"), val = tensor([1, 1])]; string value_states_67_pad_type_0 = const()[name = string("value_states_67_pad_type_0"), val = string("valid")]; tensor value_states_67_pad_0 = const()[name = string("value_states_67_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_67_dilations_0 = const()[name = string("value_states_67_dilations_0"), val = tensor([1, 1])]; int32 value_states_67_groups_0 = const()[name = string("value_states_67_groups_0"), val = int32(1)]; tensor value_states_67_cast_fp16 = conv(dilations = value_states_67_dilations_0, groups = value_states_67_groups_0, pad = value_states_67_pad_0, pad_type = value_states_67_pad_type_0, strides = value_states_67_strides_0, weight = layers_11_self_attn_v_proj_weight_cast_fp16, x = var_4458_cast_fp16_0)[name = string("value_states_67_cast_fp16")]; tensor concat_132x = const()[name = string("concat_132x"), val = tensor([1, 16, 128, -1])]; tensor x_111_cast_fp16 = reshape(shape = concat_132x, x = query_states_67_cast_fp16)[name = string("x_111_cast_fp16")]; tensor concat_133x = const()[name = string("concat_133x"), val = tensor([1, 2, 128, -1])]; tensor var_4515_cast_fp16 = reshape(shape = concat_133x, x = key_states_111_cast_fp16)[name = string("op_4515_cast_fp16")]; tensor concat_134x = const()[name = string("concat_134x"), val = tensor([1, 2, 128, -1])]; tensor var_4522_cast_fp16 = reshape(shape = concat_134x, x = value_states_67_cast_fp16)[name = string("op_4522_cast_fp16")]; tensor var_4526_cast_fp16 = mul(x = x_111_cast_fp16, y = var_869_cast_fp16)[name = string("op_4526_cast_fp16")]; tensor var_4527_split_sizes_0 = const()[name = string("op_4527_split_sizes_0"), val = tensor([64, 64])]; int32 var_4527_axis_0 = const()[name = string("op_4527_axis_0"), val = int32(-2)]; tensor var_4527_cast_fp16_0, tensor var_4527_cast_fp16_1 = split(axis = var_4527_axis_0, split_sizes = var_4527_split_sizes_0, x = x_111_cast_fp16)[name = string("op_4527_cast_fp16")]; fp16 const_112_promoted_to_fp16 = const()[name = string("const_112_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4529_cast_fp16 = mul(x = var_4527_cast_fp16_1, y = const_112_promoted_to_fp16)[name = string("op_4529_cast_fp16")]; int32 var_4531 = const()[name = string("op_4531"), val = int32(-2)]; bool var_4532_interleave_0 = const()[name = string("op_4532_interleave_0"), val = bool(false)]; tensor var_4532_cast_fp16 = concat(axis = var_4531, interleave = var_4532_interleave_0, values = (var_4529_cast_fp16, var_4527_cast_fp16_0))[name = string("op_4532_cast_fp16")]; tensor var_4533_cast_fp16 = mul(x = var_4532_cast_fp16, y = var_878_cast_fp16)[name = string("op_4533_cast_fp16")]; tensor query_states_69_cast_fp16 = add(x = var_4526_cast_fp16, y = var_4533_cast_fp16)[name = string("query_states_69_cast_fp16")]; tensor var_4539_cast_fp16 = mul(x = var_4515_cast_fp16, y = var_869_cast_fp16)[name = string("op_4539_cast_fp16")]; tensor var_4540_split_sizes_0 = const()[name = string("op_4540_split_sizes_0"), val = tensor([64, 64])]; int32 var_4540_axis_0 = const()[name = string("op_4540_axis_0"), val = int32(-2)]; tensor var_4540_cast_fp16_0, tensor var_4540_cast_fp16_1 = split(axis = var_4540_axis_0, split_sizes = var_4540_split_sizes_0, x = var_4515_cast_fp16)[name = string("op_4540_cast_fp16")]; fp16 const_113_promoted_to_fp16 = const()[name = string("const_113_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4542_cast_fp16 = mul(x = var_4540_cast_fp16_1, y = const_113_promoted_to_fp16)[name = string("op_4542_cast_fp16")]; int32 var_4544 = const()[name = string("op_4544"), val = int32(-2)]; bool var_4545_interleave_0 = const()[name = string("op_4545_interleave_0"), val = bool(false)]; tensor var_4545_cast_fp16 = concat(axis = var_4544, interleave = var_4545_interleave_0, values = (var_4542_cast_fp16, var_4540_cast_fp16_0))[name = string("op_4545_cast_fp16")]; tensor var_4546_cast_fp16 = mul(x = var_4545_cast_fp16, y = var_878_cast_fp16)[name = string("op_4546_cast_fp16")]; tensor key_states_115_cast_fp16 = add(x = var_4539_cast_fp16, y = var_4546_cast_fp16)[name = string("key_states_115_cast_fp16")]; tensor expand_dims_132 = const()[name = string("expand_dims_132"), val = tensor([11])]; tensor expand_dims_133 = const()[name = string("expand_dims_133"), val = tensor([0])]; tensor expand_dims_135 = const()[name = string("expand_dims_135"), val = tensor([0])]; int32 concat_137_axis_0 = const()[name = string("concat_137_axis_0"), val = int32(0)]; bool concat_137_interleave_0 = const()[name = string("concat_137_interleave_0"), val = bool(false)]; tensor concat_137 = concat(axis = concat_137_axis_0, interleave = concat_137_interleave_0, values = (expand_dims_132, expand_dims_133, position_id, expand_dims_135))[name = string("concat_137")]; tensor expand_dims_136 = const()[name = string("expand_dims_136"), val = tensor([12])]; tensor concat_138_values1_0 = const()[name = string("concat_138_values1_0"), val = tensor([0])]; tensor concat_138_values3_0 = const()[name = string("concat_138_values3_0"), val = tensor([0])]; int32 concat_138_axis_0 = const()[name = string("concat_138_axis_0"), val = int32(0)]; bool concat_138_interleave_0 = const()[name = string("concat_138_interleave_0"), val = bool(false)]; tensor concat_138 = concat(axis = concat_138_axis_0, interleave = concat_138_interleave_0, values = (expand_dims_136, concat_138_values1_0, cache_position_end, concat_138_values3_0))[name = string("concat_138")]; tensor key_states_117_perm_0 = const()[name = string("key_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_12_stride_0 = const()[name = string("key_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_117_cast_fp16 = transpose(perm = key_states_117_perm_0, x = key_states_115_cast_fp16)[name = string("transpose_50")]; tensor key_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = key_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = key_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_12_squeeze_mask_0, stride = key_cache_internal_tensor_assign_12_stride_0, update = key_states_117_cast_fp16, x = coreml_update_state_20)[name = string("key_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_12_cast_fp16, input = key_cache)[name = string("coreml_update_state_22_write_state")]; tensor coreml_update_state_22 = read_state(input = key_cache)[name = string("coreml_update_state_22")]; tensor value_states_69_perm_0 = const()[name = string("value_states_69_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_12_stride_0 = const()[name = string("value_cache_internal_tensor_assign_12_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_12_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_12_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_12_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_12_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_69_cast_fp16 = transpose(perm = value_states_69_perm_0, x = var_4522_cast_fp16)[name = string("transpose_49")]; tensor value_cache_internal_tensor_assign_12_cast_fp16 = slice_update(begin = concat_137, begin_mask = value_cache_internal_tensor_assign_12_begin_mask_0, end = concat_138, end_mask = value_cache_internal_tensor_assign_12_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_12_squeeze_mask_0, stride = value_cache_internal_tensor_assign_12_stride_0, update = value_states_69_cast_fp16, x = coreml_update_state_21)[name = string("value_cache_internal_tensor_assign_12_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_12_cast_fp16, input = value_cache)[name = string("coreml_update_state_23_write_state")]; tensor coreml_update_state_23 = read_state(input = value_cache)[name = string("coreml_update_state_23")]; tensor var_4616_begin_0 = const()[name = string("op_4616_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4616_end_0 = const()[name = string("op_4616_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4616_end_mask_0 = const()[name = string("op_4616_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4616_cast_fp16 = slice_by_index(begin = var_4616_begin_0, end = var_4616_end_0, end_mask = var_4616_end_mask_0, x = coreml_update_state_22)[name = string("op_4616_cast_fp16")]; tensor tile_22 = const()[name = string("tile_22"), val = tensor([1, 1])]; int32 var_4619_axis_0 = const()[name = string("op_4619_axis_0"), val = int32(1)]; tensor var_4619_cast_fp16_0, tensor var_4619_cast_fp16_1 = split(axis = var_4619_axis_0, split_sizes = tile_22, x = var_4616_cast_fp16)[name = string("op_4619_cast_fp16")]; tensor var_4626_begin_0 = const()[name = string("op_4626_begin_0"), val = tensor([11, 0, 0, 0])]; tensor var_4626_end_0 = const()[name = string("op_4626_end_0"), val = tensor([12, 2, 2048, 128])]; tensor var_4626_end_mask_0 = const()[name = string("op_4626_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4626_cast_fp16 = slice_by_index(begin = var_4626_begin_0, end = var_4626_end_0, end_mask = var_4626_end_mask_0, x = coreml_update_state_23)[name = string("op_4626_cast_fp16")]; tensor tile_23 = const()[name = string("tile_23"), val = tensor([1, 1])]; int32 var_4629_axis_0 = const()[name = string("op_4629_axis_0"), val = int32(1)]; tensor var_4629_cast_fp16_0, tensor var_4629_cast_fp16_1 = split(axis = var_4629_axis_0, split_sizes = tile_23, x = var_4626_cast_fp16)[name = string("op_4629_cast_fp16")]; tensor var_4632_split_sizes_0 = const()[name = string("op_4632_split_sizes_0"), val = tensor([8, 8])]; int32 var_4632_axis_0 = const()[name = string("op_4632_axis_0"), val = int32(1)]; tensor var_4632_0, tensor var_4632_1 = split(axis = var_4632_axis_0, split_sizes = var_4632_split_sizes_0, x = query_states_69_cast_fp16)[name = string("op_4632")]; bool attn_weights_177_transpose_x_0 = const()[name = string("attn_weights_177_transpose_x_0"), val = bool(false)]; bool attn_weights_177_transpose_y_0 = const()[name = string("attn_weights_177_transpose_y_0"), val = bool(false)]; tensor attn_weights_177_cast_fp16 = matmul(transpose_x = attn_weights_177_transpose_x_0, transpose_y = attn_weights_177_transpose_y_0, x = var_4619_cast_fp16_0, y = var_4632_0)[name = string("attn_weights_177_cast_fp16")]; fp16 var_4635_to_fp16 = const()[name = string("op_4635_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_179_cast_fp16 = mul(x = attn_weights_177_cast_fp16, y = var_4635_to_fp16)[name = string("attn_weights_179_cast_fp16")]; tensor attn_weights_181_cast_fp16 = add(x = attn_weights_179_cast_fp16, y = attn_mask_1)[name = string("attn_weights_181_cast_fp16")]; int32 var_4639 = const()[name = string("op_4639"), val = int32(-2)]; tensor attn_weights_183_cast_fp16 = softmax(axis = var_4639, x = attn_weights_181_cast_fp16)[name = string("attn_weights_183_cast_fp16")]; bool var_4645_transpose_x_1 = const()[name = string("op_4645_transpose_x_1"), val = bool(true)]; bool var_4645_transpose_y_1 = const()[name = string("op_4645_transpose_y_1"), val = bool(false)]; tensor var_4645_cast_fp16 = matmul(transpose_x = var_4645_transpose_x_1, transpose_y = var_4645_transpose_y_1, x = attn_weights_183_cast_fp16, y = var_4629_cast_fp16_0)[name = string("op_4645_cast_fp16")]; bool attn_weights_185_transpose_x_0 = const()[name = string("attn_weights_185_transpose_x_0"), val = bool(false)]; bool attn_weights_185_transpose_y_0 = const()[name = string("attn_weights_185_transpose_y_0"), val = bool(false)]; tensor attn_weights_185_cast_fp16 = matmul(transpose_x = attn_weights_185_transpose_x_0, transpose_y = attn_weights_185_transpose_y_0, x = var_4619_cast_fp16_1, y = var_4632_1)[name = string("attn_weights_185_cast_fp16")]; fp16 var_4647_to_fp16 = const()[name = string("op_4647_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_187_cast_fp16 = mul(x = attn_weights_185_cast_fp16, y = var_4647_to_fp16)[name = string("attn_weights_187_cast_fp16")]; tensor attn_weights_189_cast_fp16 = add(x = attn_weights_187_cast_fp16, y = attn_mask_1)[name = string("attn_weights_189_cast_fp16")]; int32 var_4651 = const()[name = string("op_4651"), val = int32(-2)]; tensor attn_weights_191_cast_fp16 = softmax(axis = var_4651, x = attn_weights_189_cast_fp16)[name = string("attn_weights_191_cast_fp16")]; bool attn_output_89_transpose_x_1 = const()[name = string("attn_output_89_transpose_x_1"), val = bool(true)]; bool attn_output_89_transpose_y_1 = const()[name = string("attn_output_89_transpose_y_1"), val = bool(false)]; tensor attn_output_89_cast_fp16 = matmul(transpose_x = attn_output_89_transpose_x_1, transpose_y = attn_output_89_transpose_y_1, x = attn_weights_191_cast_fp16, y = var_4629_cast_fp16_1)[name = string("attn_output_89_cast_fp16")]; int32 var_4659 = const()[name = string("op_4659"), val = int32(1)]; bool attn_output_91_interleave_0 = const()[name = string("attn_output_91_interleave_0"), val = bool(false)]; tensor attn_output_91_cast_fp16 = concat(axis = var_4659, interleave = attn_output_91_interleave_0, values = (var_4645_cast_fp16, attn_output_89_cast_fp16))[name = string("attn_output_91_cast_fp16")]; tensor var_4663_perm_0 = const()[name = string("op_4663_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_143x = const()[name = string("concat_143x"), val = tensor([1, 2048, 1, -1])]; tensor var_4663_cast_fp16 = transpose(perm = var_4663_perm_0, x = attn_output_91_cast_fp16)[name = string("transpose_48")]; tensor attn_output_95_cast_fp16 = reshape(shape = concat_143x, x = var_4663_cast_fp16)[name = string("attn_output_95_cast_fp16")]; tensor hidden_states_113_strides_0 = const()[name = string("hidden_states_113_strides_0"), val = tensor([1, 1])]; string hidden_states_113_pad_type_0 = const()[name = string("hidden_states_113_pad_type_0"), val = string("valid")]; tensor hidden_states_113_pad_0 = const()[name = string("hidden_states_113_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_113_dilations_0 = const()[name = string("hidden_states_113_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_113_groups_0 = const()[name = string("hidden_states_113_groups_0"), val = int32(1)]; tensor hidden_states_113_cast_fp16 = conv(dilations = hidden_states_113_dilations_0, groups = hidden_states_113_groups_0, pad = hidden_states_113_pad_0, pad_type = hidden_states_113_pad_type_0, strides = hidden_states_113_strides_0, weight = layers_11_self_attn_o_proj_weight_cast_fp16, x = attn_output_95_cast_fp16)[name = string("hidden_states_113_cast_fp16")]; tensor hidden_states_115_cast_fp16 = add(x = hidden_states_109_cast_fp16, y = hidden_states_113_cast_fp16)[name = string("hidden_states_115_cast_fp16")]; fp16 const_118_promoted_to_fp16 = const()[name = string("const_118_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4696_cast_fp16 = mul(x = hidden_states_115_cast_fp16, y = const_118_promoted_to_fp16)[name = string("op_4696_cast_fp16")]; int32 var_4694 = const()[name = string("op_4694"), val = int32(1)]; bool doubled_93_interleave_0 = const()[name = string("doubled_93_interleave_0"), val = bool(false)]; tensor doubled_93_cast_fp16 = concat(axis = var_4694, interleave = doubled_93_interleave_0, values = (hidden_states_115_cast_fp16, var_4696_cast_fp16))[name = string("doubled_93_cast_fp16")]; tensor out_47_axes_0 = const()[name = string("out_47_axes_0"), val = tensor([1])]; tensor out_47_gamma_0_to_fp16 = const()[name = string("out_47_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415366400)))]; fp16 var_4706_to_fp16 = const()[name = string("op_4706_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_47_cast_fp16 = layer_norm(axes = out_47_axes_0, epsilon = var_4706_to_fp16, gamma = out_47_gamma_0_to_fp16, x = doubled_93_cast_fp16)[name = string("out_47_cast_fp16")]; tensor var_4717_split_sizes_0 = const()[name = string("op_4717_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4717_axis_0 = const()[name = string("op_4717_axis_0"), val = int32(1)]; tensor var_4717_cast_fp16_0, tensor var_4717_cast_fp16_1 = split(axis = var_4717_axis_0, split_sizes = var_4717_split_sizes_0, x = out_47_cast_fp16)[name = string("op_4717_cast_fp16")]; tensor input_23_strides_0 = const()[name = string("input_23_strides_0"), val = tensor([1, 1])]; string input_23_pad_type_0 = const()[name = string("input_23_pad_type_0"), val = string("valid")]; tensor input_23_pad_0 = const()[name = string("input_23_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_23_dilations_0 = const()[name = string("input_23_dilations_0"), val = tensor([1, 1])]; int32 input_23_groups_0 = const()[name = string("input_23_groups_0"), val = int32(1)]; tensor input_23_cast_fp16 = conv(dilations = input_23_dilations_0, groups = input_23_groups_0, pad = input_23_pad_0, pad_type = input_23_pad_type_0, strides = input_23_strides_0, weight = layers_11_mlp_gate_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("input_23_cast_fp16")]; tensor var_4734_cast_fp16 = silu(x = input_23_cast_fp16)[name = string("op_4734_cast_fp16")]; tensor var_4740_strides_0 = const()[name = string("op_4740_strides_0"), val = tensor([1, 1])]; string var_4740_pad_type_0 = const()[name = string("op_4740_pad_type_0"), val = string("valid")]; tensor var_4740_pad_0 = const()[name = string("op_4740_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_4740_dilations_0 = const()[name = string("op_4740_dilations_0"), val = tensor([1, 1])]; int32 var_4740_groups_0 = const()[name = string("op_4740_groups_0"), val = int32(1)]; tensor var_4740_cast_fp16 = conv(dilations = var_4740_dilations_0, groups = var_4740_groups_0, pad = var_4740_pad_0, pad_type = var_4740_pad_type_0, strides = var_4740_strides_0, weight = layers_11_mlp_up_proj_weight_cast_fp16, x = var_4717_cast_fp16_0)[name = string("op_4740_cast_fp16")]; tensor x_119_cast_fp16 = mul(x = var_4734_cast_fp16, y = var_4740_cast_fp16)[name = string("x_119_cast_fp16")]; tensor hidden_states_117_strides_0 = const()[name = string("hidden_states_117_strides_0"), val = tensor([1, 1])]; string hidden_states_117_pad_type_0 = const()[name = string("hidden_states_117_pad_type_0"), val = string("valid")]; tensor hidden_states_117_pad_0 = const()[name = string("hidden_states_117_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_117_dilations_0 = const()[name = string("hidden_states_117_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_117_groups_0 = const()[name = string("hidden_states_117_groups_0"), val = int32(1)]; tensor hidden_states_117_cast_fp16 = conv(dilations = hidden_states_117_dilations_0, groups = hidden_states_117_groups_0, pad = hidden_states_117_pad_0, pad_type = hidden_states_117_pad_type_0, strides = hidden_states_117_strides_0, weight = layers_11_mlp_down_proj_weight_cast_fp16, x = x_119_cast_fp16)[name = string("hidden_states_117_cast_fp16")]; tensor hidden_states_119_cast_fp16 = add(x = hidden_states_115_cast_fp16, y = hidden_states_117_cast_fp16)[name = string("hidden_states_119_cast_fp16")]; fp16 const_120_promoted_to_fp16 = const()[name = string("const_120_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4758_cast_fp16 = mul(x = hidden_states_119_cast_fp16, y = const_120_promoted_to_fp16)[name = string("op_4758_cast_fp16")]; int32 var_4756 = const()[name = string("op_4756"), val = int32(1)]; bool doubled_97_interleave_0 = const()[name = string("doubled_97_interleave_0"), val = bool(false)]; tensor doubled_97_cast_fp16 = concat(axis = var_4756, interleave = doubled_97_interleave_0, values = (hidden_states_119_cast_fp16, var_4758_cast_fp16))[name = string("doubled_97_cast_fp16")]; tensor out_49_axes_0 = const()[name = string("out_49_axes_0"), val = tensor([1])]; tensor out_49_gamma_0_to_fp16 = const()[name = string("out_49_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415374656)))]; fp16 var_4768_to_fp16 = const()[name = string("op_4768_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_49_cast_fp16 = layer_norm(axes = out_49_axes_0, epsilon = var_4768_to_fp16, gamma = out_49_gamma_0_to_fp16, x = doubled_97_cast_fp16)[name = string("out_49_cast_fp16")]; tensor var_4779_split_sizes_0 = const()[name = string("op_4779_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_4779_axis_0 = const()[name = string("op_4779_axis_0"), val = int32(1)]; tensor var_4779_cast_fp16_0, tensor var_4779_cast_fp16_1 = split(axis = var_4779_axis_0, split_sizes = var_4779_split_sizes_0, x = out_49_cast_fp16)[name = string("op_4779_cast_fp16")]; tensor query_states_73_strides_0 = const()[name = string("query_states_73_strides_0"), val = tensor([1, 1])]; string query_states_73_pad_type_0 = const()[name = string("query_states_73_pad_type_0"), val = string("valid")]; tensor query_states_73_pad_0 = const()[name = string("query_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_73_dilations_0 = const()[name = string("query_states_73_dilations_0"), val = tensor([1, 1])]; int32 query_states_73_groups_0 = const()[name = string("query_states_73_groups_0"), val = int32(1)]; tensor query_states_73_cast_fp16 = conv(dilations = query_states_73_dilations_0, groups = query_states_73_groups_0, pad = query_states_73_pad_0, pad_type = query_states_73_pad_type_0, strides = query_states_73_strides_0, weight = layers_12_self_attn_q_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("query_states_73_cast_fp16")]; tensor key_states_121_strides_0 = const()[name = string("key_states_121_strides_0"), val = tensor([1, 1])]; string key_states_121_pad_type_0 = const()[name = string("key_states_121_pad_type_0"), val = string("valid")]; tensor key_states_121_pad_0 = const()[name = string("key_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_121_dilations_0 = const()[name = string("key_states_121_dilations_0"), val = tensor([1, 1])]; int32 key_states_121_groups_0 = const()[name = string("key_states_121_groups_0"), val = int32(1)]; tensor key_states_121_cast_fp16 = conv(dilations = key_states_121_dilations_0, groups = key_states_121_groups_0, pad = key_states_121_pad_0, pad_type = key_states_121_pad_type_0, strides = key_states_121_strides_0, weight = layers_12_self_attn_k_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("key_states_121_cast_fp16")]; tensor value_states_73_strides_0 = const()[name = string("value_states_73_strides_0"), val = tensor([1, 1])]; string value_states_73_pad_type_0 = const()[name = string("value_states_73_pad_type_0"), val = string("valid")]; tensor value_states_73_pad_0 = const()[name = string("value_states_73_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_73_dilations_0 = const()[name = string("value_states_73_dilations_0"), val = tensor([1, 1])]; int32 value_states_73_groups_0 = const()[name = string("value_states_73_groups_0"), val = int32(1)]; tensor value_states_73_cast_fp16 = conv(dilations = value_states_73_dilations_0, groups = value_states_73_groups_0, pad = value_states_73_pad_0, pad_type = value_states_73_pad_type_0, strides = value_states_73_strides_0, weight = layers_12_self_attn_v_proj_weight_cast_fp16, x = var_4779_cast_fp16_0)[name = string("value_states_73_cast_fp16")]; tensor concat_144x = const()[name = string("concat_144x"), val = tensor([1, 16, 128, -1])]; tensor x_121_cast_fp16 = reshape(shape = concat_144x, x = query_states_73_cast_fp16)[name = string("x_121_cast_fp16")]; tensor concat_145x = const()[name = string("concat_145x"), val = tensor([1, 2, 128, -1])]; tensor var_4836_cast_fp16 = reshape(shape = concat_145x, x = key_states_121_cast_fp16)[name = string("op_4836_cast_fp16")]; tensor concat_146x = const()[name = string("concat_146x"), val = tensor([1, 2, 128, -1])]; tensor var_4843_cast_fp16 = reshape(shape = concat_146x, x = value_states_73_cast_fp16)[name = string("op_4843_cast_fp16")]; tensor var_4847_cast_fp16 = mul(x = x_121_cast_fp16, y = var_869_cast_fp16)[name = string("op_4847_cast_fp16")]; tensor var_4848_split_sizes_0 = const()[name = string("op_4848_split_sizes_0"), val = tensor([64, 64])]; int32 var_4848_axis_0 = const()[name = string("op_4848_axis_0"), val = int32(-2)]; tensor var_4848_cast_fp16_0, tensor var_4848_cast_fp16_1 = split(axis = var_4848_axis_0, split_sizes = var_4848_split_sizes_0, x = x_121_cast_fp16)[name = string("op_4848_cast_fp16")]; fp16 const_122_promoted_to_fp16 = const()[name = string("const_122_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4850_cast_fp16 = mul(x = var_4848_cast_fp16_1, y = const_122_promoted_to_fp16)[name = string("op_4850_cast_fp16")]; int32 var_4852 = const()[name = string("op_4852"), val = int32(-2)]; bool var_4853_interleave_0 = const()[name = string("op_4853_interleave_0"), val = bool(false)]; tensor var_4853_cast_fp16 = concat(axis = var_4852, interleave = var_4853_interleave_0, values = (var_4850_cast_fp16, var_4848_cast_fp16_0))[name = string("op_4853_cast_fp16")]; tensor var_4854_cast_fp16 = mul(x = var_4853_cast_fp16, y = var_878_cast_fp16)[name = string("op_4854_cast_fp16")]; tensor query_states_75_cast_fp16 = add(x = var_4847_cast_fp16, y = var_4854_cast_fp16)[name = string("query_states_75_cast_fp16")]; tensor var_4860_cast_fp16 = mul(x = var_4836_cast_fp16, y = var_869_cast_fp16)[name = string("op_4860_cast_fp16")]; tensor var_4861_split_sizes_0 = const()[name = string("op_4861_split_sizes_0"), val = tensor([64, 64])]; int32 var_4861_axis_0 = const()[name = string("op_4861_axis_0"), val = int32(-2)]; tensor var_4861_cast_fp16_0, tensor var_4861_cast_fp16_1 = split(axis = var_4861_axis_0, split_sizes = var_4861_split_sizes_0, x = var_4836_cast_fp16)[name = string("op_4861_cast_fp16")]; fp16 const_123_promoted_to_fp16 = const()[name = string("const_123_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_4863_cast_fp16 = mul(x = var_4861_cast_fp16_1, y = const_123_promoted_to_fp16)[name = string("op_4863_cast_fp16")]; int32 var_4865 = const()[name = string("op_4865"), val = int32(-2)]; bool var_4866_interleave_0 = const()[name = string("op_4866_interleave_0"), val = bool(false)]; tensor var_4866_cast_fp16 = concat(axis = var_4865, interleave = var_4866_interleave_0, values = (var_4863_cast_fp16, var_4861_cast_fp16_0))[name = string("op_4866_cast_fp16")]; tensor var_4867_cast_fp16 = mul(x = var_4866_cast_fp16, y = var_878_cast_fp16)[name = string("op_4867_cast_fp16")]; tensor key_states_125_cast_fp16 = add(x = var_4860_cast_fp16, y = var_4867_cast_fp16)[name = string("key_states_125_cast_fp16")]; tensor expand_dims_144 = const()[name = string("expand_dims_144"), val = tensor([12])]; tensor expand_dims_145 = const()[name = string("expand_dims_145"), val = tensor([0])]; tensor expand_dims_147 = const()[name = string("expand_dims_147"), val = tensor([0])]; int32 concat_149_axis_0 = const()[name = string("concat_149_axis_0"), val = int32(0)]; bool concat_149_interleave_0 = const()[name = string("concat_149_interleave_0"), val = bool(false)]; tensor concat_149 = concat(axis = concat_149_axis_0, interleave = concat_149_interleave_0, values = (expand_dims_144, expand_dims_145, position_id, expand_dims_147))[name = string("concat_149")]; tensor expand_dims_148 = const()[name = string("expand_dims_148"), val = tensor([13])]; tensor concat_150_values1_0 = const()[name = string("concat_150_values1_0"), val = tensor([0])]; tensor concat_150_values3_0 = const()[name = string("concat_150_values3_0"), val = tensor([0])]; int32 concat_150_axis_0 = const()[name = string("concat_150_axis_0"), val = int32(0)]; bool concat_150_interleave_0 = const()[name = string("concat_150_interleave_0"), val = bool(false)]; tensor concat_150 = concat(axis = concat_150_axis_0, interleave = concat_150_interleave_0, values = (expand_dims_148, concat_150_values1_0, cache_position_end, concat_150_values3_0))[name = string("concat_150")]; tensor key_states_127_perm_0 = const()[name = string("key_states_127_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_13_stride_0 = const()[name = string("key_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_127_cast_fp16 = transpose(perm = key_states_127_perm_0, x = key_states_125_cast_fp16)[name = string("transpose_47")]; tensor key_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = key_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = key_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_13_squeeze_mask_0, stride = key_cache_internal_tensor_assign_13_stride_0, update = key_states_127_cast_fp16, x = coreml_update_state_22)[name = string("key_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_13_cast_fp16, input = key_cache)[name = string("coreml_update_state_24_write_state")]; tensor coreml_update_state_24 = read_state(input = key_cache)[name = string("coreml_update_state_24")]; tensor value_states_75_perm_0 = const()[name = string("value_states_75_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_13_stride_0 = const()[name = string("value_cache_internal_tensor_assign_13_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_13_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_13_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_13_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_13_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_75_cast_fp16 = transpose(perm = value_states_75_perm_0, x = var_4843_cast_fp16)[name = string("transpose_46")]; tensor value_cache_internal_tensor_assign_13_cast_fp16 = slice_update(begin = concat_149, begin_mask = value_cache_internal_tensor_assign_13_begin_mask_0, end = concat_150, end_mask = value_cache_internal_tensor_assign_13_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_13_squeeze_mask_0, stride = value_cache_internal_tensor_assign_13_stride_0, update = value_states_75_cast_fp16, x = coreml_update_state_23)[name = string("value_cache_internal_tensor_assign_13_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_13_cast_fp16, input = value_cache)[name = string("coreml_update_state_25_write_state")]; tensor coreml_update_state_25 = read_state(input = value_cache)[name = string("coreml_update_state_25")]; tensor var_4937_begin_0 = const()[name = string("op_4937_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4937_end_0 = const()[name = string("op_4937_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4937_end_mask_0 = const()[name = string("op_4937_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4937_cast_fp16 = slice_by_index(begin = var_4937_begin_0, end = var_4937_end_0, end_mask = var_4937_end_mask_0, x = coreml_update_state_24)[name = string("op_4937_cast_fp16")]; tensor tile_24 = const()[name = string("tile_24"), val = tensor([1, 1])]; int32 var_4940_axis_0 = const()[name = string("op_4940_axis_0"), val = int32(1)]; tensor var_4940_cast_fp16_0, tensor var_4940_cast_fp16_1 = split(axis = var_4940_axis_0, split_sizes = tile_24, x = var_4937_cast_fp16)[name = string("op_4940_cast_fp16")]; tensor var_4947_begin_0 = const()[name = string("op_4947_begin_0"), val = tensor([12, 0, 0, 0])]; tensor var_4947_end_0 = const()[name = string("op_4947_end_0"), val = tensor([13, 2, 2048, 128])]; tensor var_4947_end_mask_0 = const()[name = string("op_4947_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_4947_cast_fp16 = slice_by_index(begin = var_4947_begin_0, end = var_4947_end_0, end_mask = var_4947_end_mask_0, x = coreml_update_state_25)[name = string("op_4947_cast_fp16")]; tensor tile_25 = const()[name = string("tile_25"), val = tensor([1, 1])]; int32 var_4950_axis_0 = const()[name = string("op_4950_axis_0"), val = int32(1)]; tensor var_4950_cast_fp16_0, tensor var_4950_cast_fp16_1 = split(axis = var_4950_axis_0, split_sizes = tile_25, x = var_4947_cast_fp16)[name = string("op_4950_cast_fp16")]; tensor var_4953_split_sizes_0 = const()[name = string("op_4953_split_sizes_0"), val = tensor([8, 8])]; int32 var_4953_axis_0 = const()[name = string("op_4953_axis_0"), val = int32(1)]; tensor var_4953_0, tensor var_4953_1 = split(axis = var_4953_axis_0, split_sizes = var_4953_split_sizes_0, x = query_states_75_cast_fp16)[name = string("op_4953")]; bool attn_weights_193_transpose_x_0 = const()[name = string("attn_weights_193_transpose_x_0"), val = bool(false)]; bool attn_weights_193_transpose_y_0 = const()[name = string("attn_weights_193_transpose_y_0"), val = bool(false)]; tensor attn_weights_193_cast_fp16 = matmul(transpose_x = attn_weights_193_transpose_x_0, transpose_y = attn_weights_193_transpose_y_0, x = var_4940_cast_fp16_0, y = var_4953_0)[name = string("attn_weights_193_cast_fp16")]; fp16 var_4956_to_fp16 = const()[name = string("op_4956_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_195_cast_fp16 = mul(x = attn_weights_193_cast_fp16, y = var_4956_to_fp16)[name = string("attn_weights_195_cast_fp16")]; tensor attn_weights_197_cast_fp16 = add(x = attn_weights_195_cast_fp16, y = attn_mask_1)[name = string("attn_weights_197_cast_fp16")]; int32 var_4960 = const()[name = string("op_4960"), val = int32(-2)]; tensor attn_weights_199_cast_fp16 = softmax(axis = var_4960, x = attn_weights_197_cast_fp16)[name = string("attn_weights_199_cast_fp16")]; bool var_4966_transpose_x_1 = const()[name = string("op_4966_transpose_x_1"), val = bool(true)]; bool var_4966_transpose_y_1 = const()[name = string("op_4966_transpose_y_1"), val = bool(false)]; tensor var_4966_cast_fp16 = matmul(transpose_x = var_4966_transpose_x_1, transpose_y = var_4966_transpose_y_1, x = attn_weights_199_cast_fp16, y = var_4950_cast_fp16_0)[name = string("op_4966_cast_fp16")]; bool attn_weights_201_transpose_x_0 = const()[name = string("attn_weights_201_transpose_x_0"), val = bool(false)]; bool attn_weights_201_transpose_y_0 = const()[name = string("attn_weights_201_transpose_y_0"), val = bool(false)]; tensor attn_weights_201_cast_fp16 = matmul(transpose_x = attn_weights_201_transpose_x_0, transpose_y = attn_weights_201_transpose_y_0, x = var_4940_cast_fp16_1, y = var_4953_1)[name = string("attn_weights_201_cast_fp16")]; fp16 var_4968_to_fp16 = const()[name = string("op_4968_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_203_cast_fp16 = mul(x = attn_weights_201_cast_fp16, y = var_4968_to_fp16)[name = string("attn_weights_203_cast_fp16")]; tensor attn_weights_205_cast_fp16 = add(x = attn_weights_203_cast_fp16, y = attn_mask_1)[name = string("attn_weights_205_cast_fp16")]; int32 var_4972 = const()[name = string("op_4972"), val = int32(-2)]; tensor attn_weights_207_cast_fp16 = softmax(axis = var_4972, x = attn_weights_205_cast_fp16)[name = string("attn_weights_207_cast_fp16")]; bool attn_output_97_transpose_x_1 = const()[name = string("attn_output_97_transpose_x_1"), val = bool(true)]; bool attn_output_97_transpose_y_1 = const()[name = string("attn_output_97_transpose_y_1"), val = bool(false)]; tensor attn_output_97_cast_fp16 = matmul(transpose_x = attn_output_97_transpose_x_1, transpose_y = attn_output_97_transpose_y_1, x = attn_weights_207_cast_fp16, y = var_4950_cast_fp16_1)[name = string("attn_output_97_cast_fp16")]; int32 var_4980 = const()[name = string("op_4980"), val = int32(1)]; bool attn_output_99_interleave_0 = const()[name = string("attn_output_99_interleave_0"), val = bool(false)]; tensor attn_output_99_cast_fp16 = concat(axis = var_4980, interleave = attn_output_99_interleave_0, values = (var_4966_cast_fp16, attn_output_97_cast_fp16))[name = string("attn_output_99_cast_fp16")]; tensor var_4984_perm_0 = const()[name = string("op_4984_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_155x = const()[name = string("concat_155x"), val = tensor([1, 2048, 1, -1])]; tensor var_4984_cast_fp16 = transpose(perm = var_4984_perm_0, x = attn_output_99_cast_fp16)[name = string("transpose_45")]; tensor attn_output_103_cast_fp16 = reshape(shape = concat_155x, x = var_4984_cast_fp16)[name = string("attn_output_103_cast_fp16")]; tensor hidden_states_123_strides_0 = const()[name = string("hidden_states_123_strides_0"), val = tensor([1, 1])]; string hidden_states_123_pad_type_0 = const()[name = string("hidden_states_123_pad_type_0"), val = string("valid")]; tensor hidden_states_123_pad_0 = const()[name = string("hidden_states_123_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_123_dilations_0 = const()[name = string("hidden_states_123_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_123_groups_0 = const()[name = string("hidden_states_123_groups_0"), val = int32(1)]; tensor hidden_states_123_cast_fp16 = conv(dilations = hidden_states_123_dilations_0, groups = hidden_states_123_groups_0, pad = hidden_states_123_pad_0, pad_type = hidden_states_123_pad_type_0, strides = hidden_states_123_strides_0, weight = layers_12_self_attn_o_proj_weight_cast_fp16, x = attn_output_103_cast_fp16)[name = string("hidden_states_123_cast_fp16")]; tensor hidden_states_125_cast_fp16 = add(x = hidden_states_119_cast_fp16, y = hidden_states_123_cast_fp16)[name = string("hidden_states_125_cast_fp16")]; fp16 const_128_promoted_to_fp16 = const()[name = string("const_128_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5017_cast_fp16 = mul(x = hidden_states_125_cast_fp16, y = const_128_promoted_to_fp16)[name = string("op_5017_cast_fp16")]; int32 var_5015 = const()[name = string("op_5015"), val = int32(1)]; bool doubled_101_interleave_0 = const()[name = string("doubled_101_interleave_0"), val = bool(false)]; tensor doubled_101_cast_fp16 = concat(axis = var_5015, interleave = doubled_101_interleave_0, values = (hidden_states_125_cast_fp16, var_5017_cast_fp16))[name = string("doubled_101_cast_fp16")]; tensor out_51_axes_0 = const()[name = string("out_51_axes_0"), val = tensor([1])]; tensor out_51_gamma_0_to_fp16 = const()[name = string("out_51_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415382912)))]; fp16 var_5027_to_fp16 = const()[name = string("op_5027_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_51_cast_fp16 = layer_norm(axes = out_51_axes_0, epsilon = var_5027_to_fp16, gamma = out_51_gamma_0_to_fp16, x = doubled_101_cast_fp16)[name = string("out_51_cast_fp16")]; tensor var_5038_split_sizes_0 = const()[name = string("op_5038_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5038_axis_0 = const()[name = string("op_5038_axis_0"), val = int32(1)]; tensor var_5038_cast_fp16_0, tensor var_5038_cast_fp16_1 = split(axis = var_5038_axis_0, split_sizes = var_5038_split_sizes_0, x = out_51_cast_fp16)[name = string("op_5038_cast_fp16")]; tensor input_25_strides_0 = const()[name = string("input_25_strides_0"), val = tensor([1, 1])]; string input_25_pad_type_0 = const()[name = string("input_25_pad_type_0"), val = string("valid")]; tensor input_25_pad_0 = const()[name = string("input_25_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_25_dilations_0 = const()[name = string("input_25_dilations_0"), val = tensor([1, 1])]; int32 input_25_groups_0 = const()[name = string("input_25_groups_0"), val = int32(1)]; tensor input_25_cast_fp16 = conv(dilations = input_25_dilations_0, groups = input_25_groups_0, pad = input_25_pad_0, pad_type = input_25_pad_type_0, strides = input_25_strides_0, weight = layers_12_mlp_gate_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("input_25_cast_fp16")]; tensor var_5055_cast_fp16 = silu(x = input_25_cast_fp16)[name = string("op_5055_cast_fp16")]; tensor var_5061_strides_0 = const()[name = string("op_5061_strides_0"), val = tensor([1, 1])]; string var_5061_pad_type_0 = const()[name = string("op_5061_pad_type_0"), val = string("valid")]; tensor var_5061_pad_0 = const()[name = string("op_5061_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5061_dilations_0 = const()[name = string("op_5061_dilations_0"), val = tensor([1, 1])]; int32 var_5061_groups_0 = const()[name = string("op_5061_groups_0"), val = int32(1)]; tensor var_5061_cast_fp16 = conv(dilations = var_5061_dilations_0, groups = var_5061_groups_0, pad = var_5061_pad_0, pad_type = var_5061_pad_type_0, strides = var_5061_strides_0, weight = layers_12_mlp_up_proj_weight_cast_fp16, x = var_5038_cast_fp16_0)[name = string("op_5061_cast_fp16")]; tensor x_129_cast_fp16 = mul(x = var_5055_cast_fp16, y = var_5061_cast_fp16)[name = string("x_129_cast_fp16")]; tensor hidden_states_127_strides_0 = const()[name = string("hidden_states_127_strides_0"), val = tensor([1, 1])]; string hidden_states_127_pad_type_0 = const()[name = string("hidden_states_127_pad_type_0"), val = string("valid")]; tensor hidden_states_127_pad_0 = const()[name = string("hidden_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_127_dilations_0 = const()[name = string("hidden_states_127_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_127_groups_0 = const()[name = string("hidden_states_127_groups_0"), val = int32(1)]; tensor hidden_states_127_cast_fp16 = conv(dilations = hidden_states_127_dilations_0, groups = hidden_states_127_groups_0, pad = hidden_states_127_pad_0, pad_type = hidden_states_127_pad_type_0, strides = hidden_states_127_strides_0, weight = layers_12_mlp_down_proj_weight_cast_fp16, x = x_129_cast_fp16)[name = string("hidden_states_127_cast_fp16")]; tensor hidden_states_129_cast_fp16 = add(x = hidden_states_125_cast_fp16, y = hidden_states_127_cast_fp16)[name = string("hidden_states_129_cast_fp16")]; fp16 const_130_promoted_to_fp16 = const()[name = string("const_130_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5079_cast_fp16 = mul(x = hidden_states_129_cast_fp16, y = const_130_promoted_to_fp16)[name = string("op_5079_cast_fp16")]; int32 var_5077 = const()[name = string("op_5077"), val = int32(1)]; bool doubled_105_interleave_0 = const()[name = string("doubled_105_interleave_0"), val = bool(false)]; tensor doubled_105_cast_fp16 = concat(axis = var_5077, interleave = doubled_105_interleave_0, values = (hidden_states_129_cast_fp16, var_5079_cast_fp16))[name = string("doubled_105_cast_fp16")]; tensor out_53_axes_0 = const()[name = string("out_53_axes_0"), val = tensor([1])]; tensor out_53_gamma_0_to_fp16 = const()[name = string("out_53_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415391168)))]; fp16 var_5089_to_fp16 = const()[name = string("op_5089_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_53_cast_fp16 = layer_norm(axes = out_53_axes_0, epsilon = var_5089_to_fp16, gamma = out_53_gamma_0_to_fp16, x = doubled_105_cast_fp16)[name = string("out_53_cast_fp16")]; tensor var_5100_split_sizes_0 = const()[name = string("op_5100_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5100_axis_0 = const()[name = string("op_5100_axis_0"), val = int32(1)]; tensor var_5100_cast_fp16_0, tensor var_5100_cast_fp16_1 = split(axis = var_5100_axis_0, split_sizes = var_5100_split_sizes_0, x = out_53_cast_fp16)[name = string("op_5100_cast_fp16")]; tensor query_states_79_strides_0 = const()[name = string("query_states_79_strides_0"), val = tensor([1, 1])]; string query_states_79_pad_type_0 = const()[name = string("query_states_79_pad_type_0"), val = string("valid")]; tensor query_states_79_pad_0 = const()[name = string("query_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_79_dilations_0 = const()[name = string("query_states_79_dilations_0"), val = tensor([1, 1])]; int32 query_states_79_groups_0 = const()[name = string("query_states_79_groups_0"), val = int32(1)]; tensor query_states_79_cast_fp16 = conv(dilations = query_states_79_dilations_0, groups = query_states_79_groups_0, pad = query_states_79_pad_0, pad_type = query_states_79_pad_type_0, strides = query_states_79_strides_0, weight = layers_13_self_attn_q_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("query_states_79_cast_fp16")]; tensor key_states_131_strides_0 = const()[name = string("key_states_131_strides_0"), val = tensor([1, 1])]; string key_states_131_pad_type_0 = const()[name = string("key_states_131_pad_type_0"), val = string("valid")]; tensor key_states_131_pad_0 = const()[name = string("key_states_131_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_131_dilations_0 = const()[name = string("key_states_131_dilations_0"), val = tensor([1, 1])]; int32 key_states_131_groups_0 = const()[name = string("key_states_131_groups_0"), val = int32(1)]; tensor key_states_131_cast_fp16 = conv(dilations = key_states_131_dilations_0, groups = key_states_131_groups_0, pad = key_states_131_pad_0, pad_type = key_states_131_pad_type_0, strides = key_states_131_strides_0, weight = layers_13_self_attn_k_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("key_states_131_cast_fp16")]; tensor value_states_79_strides_0 = const()[name = string("value_states_79_strides_0"), val = tensor([1, 1])]; string value_states_79_pad_type_0 = const()[name = string("value_states_79_pad_type_0"), val = string("valid")]; tensor value_states_79_pad_0 = const()[name = string("value_states_79_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_79_dilations_0 = const()[name = string("value_states_79_dilations_0"), val = tensor([1, 1])]; int32 value_states_79_groups_0 = const()[name = string("value_states_79_groups_0"), val = int32(1)]; tensor value_states_79_cast_fp16 = conv(dilations = value_states_79_dilations_0, groups = value_states_79_groups_0, pad = value_states_79_pad_0, pad_type = value_states_79_pad_type_0, strides = value_states_79_strides_0, weight = layers_13_self_attn_v_proj_weight_cast_fp16, x = var_5100_cast_fp16_0)[name = string("value_states_79_cast_fp16")]; tensor concat_156x = const()[name = string("concat_156x"), val = tensor([1, 16, 128, -1])]; tensor x_131_cast_fp16 = reshape(shape = concat_156x, x = query_states_79_cast_fp16)[name = string("x_131_cast_fp16")]; tensor concat_157x = const()[name = string("concat_157x"), val = tensor([1, 2, 128, -1])]; tensor var_5157_cast_fp16 = reshape(shape = concat_157x, x = key_states_131_cast_fp16)[name = string("op_5157_cast_fp16")]; tensor concat_158x = const()[name = string("concat_158x"), val = tensor([1, 2, 128, -1])]; tensor var_5164_cast_fp16 = reshape(shape = concat_158x, x = value_states_79_cast_fp16)[name = string("op_5164_cast_fp16")]; tensor var_5168_cast_fp16 = mul(x = x_131_cast_fp16, y = var_869_cast_fp16)[name = string("op_5168_cast_fp16")]; tensor var_5169_split_sizes_0 = const()[name = string("op_5169_split_sizes_0"), val = tensor([64, 64])]; int32 var_5169_axis_0 = const()[name = string("op_5169_axis_0"), val = int32(-2)]; tensor var_5169_cast_fp16_0, tensor var_5169_cast_fp16_1 = split(axis = var_5169_axis_0, split_sizes = var_5169_split_sizes_0, x = x_131_cast_fp16)[name = string("op_5169_cast_fp16")]; fp16 const_132_promoted_to_fp16 = const()[name = string("const_132_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5171_cast_fp16 = mul(x = var_5169_cast_fp16_1, y = const_132_promoted_to_fp16)[name = string("op_5171_cast_fp16")]; int32 var_5173 = const()[name = string("op_5173"), val = int32(-2)]; bool var_5174_interleave_0 = const()[name = string("op_5174_interleave_0"), val = bool(false)]; tensor var_5174_cast_fp16 = concat(axis = var_5173, interleave = var_5174_interleave_0, values = (var_5171_cast_fp16, var_5169_cast_fp16_0))[name = string("op_5174_cast_fp16")]; tensor var_5175_cast_fp16 = mul(x = var_5174_cast_fp16, y = var_878_cast_fp16)[name = string("op_5175_cast_fp16")]; tensor query_states_81_cast_fp16 = add(x = var_5168_cast_fp16, y = var_5175_cast_fp16)[name = string("query_states_81_cast_fp16")]; tensor var_5181_cast_fp16 = mul(x = var_5157_cast_fp16, y = var_869_cast_fp16)[name = string("op_5181_cast_fp16")]; tensor var_5182_split_sizes_0 = const()[name = string("op_5182_split_sizes_0"), val = tensor([64, 64])]; int32 var_5182_axis_0 = const()[name = string("op_5182_axis_0"), val = int32(-2)]; tensor var_5182_cast_fp16_0, tensor var_5182_cast_fp16_1 = split(axis = var_5182_axis_0, split_sizes = var_5182_split_sizes_0, x = var_5157_cast_fp16)[name = string("op_5182_cast_fp16")]; fp16 const_133_promoted_to_fp16 = const()[name = string("const_133_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5184_cast_fp16 = mul(x = var_5182_cast_fp16_1, y = const_133_promoted_to_fp16)[name = string("op_5184_cast_fp16")]; int32 var_5186 = const()[name = string("op_5186"), val = int32(-2)]; bool var_5187_interleave_0 = const()[name = string("op_5187_interleave_0"), val = bool(false)]; tensor var_5187_cast_fp16 = concat(axis = var_5186, interleave = var_5187_interleave_0, values = (var_5184_cast_fp16, var_5182_cast_fp16_0))[name = string("op_5187_cast_fp16")]; tensor var_5188_cast_fp16 = mul(x = var_5187_cast_fp16, y = var_878_cast_fp16)[name = string("op_5188_cast_fp16")]; tensor key_states_135_cast_fp16 = add(x = var_5181_cast_fp16, y = var_5188_cast_fp16)[name = string("key_states_135_cast_fp16")]; tensor expand_dims_156 = const()[name = string("expand_dims_156"), val = tensor([13])]; tensor expand_dims_157 = const()[name = string("expand_dims_157"), val = tensor([0])]; tensor expand_dims_159 = const()[name = string("expand_dims_159"), val = tensor([0])]; int32 concat_161_axis_0 = const()[name = string("concat_161_axis_0"), val = int32(0)]; bool concat_161_interleave_0 = const()[name = string("concat_161_interleave_0"), val = bool(false)]; tensor concat_161 = concat(axis = concat_161_axis_0, interleave = concat_161_interleave_0, values = (expand_dims_156, expand_dims_157, position_id, expand_dims_159))[name = string("concat_161")]; tensor expand_dims_160 = const()[name = string("expand_dims_160"), val = tensor([14])]; tensor concat_162_values1_0 = const()[name = string("concat_162_values1_0"), val = tensor([0])]; tensor concat_162_values3_0 = const()[name = string("concat_162_values3_0"), val = tensor([0])]; int32 concat_162_axis_0 = const()[name = string("concat_162_axis_0"), val = int32(0)]; bool concat_162_interleave_0 = const()[name = string("concat_162_interleave_0"), val = bool(false)]; tensor concat_162 = concat(axis = concat_162_axis_0, interleave = concat_162_interleave_0, values = (expand_dims_160, concat_162_values1_0, cache_position_end, concat_162_values3_0))[name = string("concat_162")]; tensor key_states_137_perm_0 = const()[name = string("key_states_137_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_14_stride_0 = const()[name = string("key_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_137_cast_fp16 = transpose(perm = key_states_137_perm_0, x = key_states_135_cast_fp16)[name = string("transpose_44")]; tensor key_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = key_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = key_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_14_squeeze_mask_0, stride = key_cache_internal_tensor_assign_14_stride_0, update = key_states_137_cast_fp16, x = coreml_update_state_24)[name = string("key_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_14_cast_fp16, input = key_cache)[name = string("coreml_update_state_26_write_state")]; tensor coreml_update_state_26 = read_state(input = key_cache)[name = string("coreml_update_state_26")]; tensor value_states_81_perm_0 = const()[name = string("value_states_81_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_14_stride_0 = const()[name = string("value_cache_internal_tensor_assign_14_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_14_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_14_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_14_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_14_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_81_cast_fp16 = transpose(perm = value_states_81_perm_0, x = var_5164_cast_fp16)[name = string("transpose_43")]; tensor value_cache_internal_tensor_assign_14_cast_fp16 = slice_update(begin = concat_161, begin_mask = value_cache_internal_tensor_assign_14_begin_mask_0, end = concat_162, end_mask = value_cache_internal_tensor_assign_14_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_14_squeeze_mask_0, stride = value_cache_internal_tensor_assign_14_stride_0, update = value_states_81_cast_fp16, x = coreml_update_state_25)[name = string("value_cache_internal_tensor_assign_14_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_14_cast_fp16, input = value_cache)[name = string("coreml_update_state_27_write_state")]; tensor coreml_update_state_27 = read_state(input = value_cache)[name = string("coreml_update_state_27")]; tensor var_5258_begin_0 = const()[name = string("op_5258_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5258_end_0 = const()[name = string("op_5258_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5258_end_mask_0 = const()[name = string("op_5258_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5258_cast_fp16 = slice_by_index(begin = var_5258_begin_0, end = var_5258_end_0, end_mask = var_5258_end_mask_0, x = coreml_update_state_26)[name = string("op_5258_cast_fp16")]; tensor tile_26 = const()[name = string("tile_26"), val = tensor([1, 1])]; int32 var_5261_axis_0 = const()[name = string("op_5261_axis_0"), val = int32(1)]; tensor var_5261_cast_fp16_0, tensor var_5261_cast_fp16_1 = split(axis = var_5261_axis_0, split_sizes = tile_26, x = var_5258_cast_fp16)[name = string("op_5261_cast_fp16")]; tensor var_5268_begin_0 = const()[name = string("op_5268_begin_0"), val = tensor([13, 0, 0, 0])]; tensor var_5268_end_0 = const()[name = string("op_5268_end_0"), val = tensor([14, 2, 2048, 128])]; tensor var_5268_end_mask_0 = const()[name = string("op_5268_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5268_cast_fp16 = slice_by_index(begin = var_5268_begin_0, end = var_5268_end_0, end_mask = var_5268_end_mask_0, x = coreml_update_state_27)[name = string("op_5268_cast_fp16")]; tensor tile_27 = const()[name = string("tile_27"), val = tensor([1, 1])]; int32 var_5271_axis_0 = const()[name = string("op_5271_axis_0"), val = int32(1)]; tensor var_5271_cast_fp16_0, tensor var_5271_cast_fp16_1 = split(axis = var_5271_axis_0, split_sizes = tile_27, x = var_5268_cast_fp16)[name = string("op_5271_cast_fp16")]; tensor var_5274_split_sizes_0 = const()[name = string("op_5274_split_sizes_0"), val = tensor([8, 8])]; int32 var_5274_axis_0 = const()[name = string("op_5274_axis_0"), val = int32(1)]; tensor var_5274_0, tensor var_5274_1 = split(axis = var_5274_axis_0, split_sizes = var_5274_split_sizes_0, x = query_states_81_cast_fp16)[name = string("op_5274")]; bool attn_weights_209_transpose_x_0 = const()[name = string("attn_weights_209_transpose_x_0"), val = bool(false)]; bool attn_weights_209_transpose_y_0 = const()[name = string("attn_weights_209_transpose_y_0"), val = bool(false)]; tensor attn_weights_209_cast_fp16 = matmul(transpose_x = attn_weights_209_transpose_x_0, transpose_y = attn_weights_209_transpose_y_0, x = var_5261_cast_fp16_0, y = var_5274_0)[name = string("attn_weights_209_cast_fp16")]; fp16 var_5277_to_fp16 = const()[name = string("op_5277_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_211_cast_fp16 = mul(x = attn_weights_209_cast_fp16, y = var_5277_to_fp16)[name = string("attn_weights_211_cast_fp16")]; tensor attn_weights_213_cast_fp16 = add(x = attn_weights_211_cast_fp16, y = attn_mask_1)[name = string("attn_weights_213_cast_fp16")]; int32 var_5281 = const()[name = string("op_5281"), val = int32(-2)]; tensor attn_weights_215_cast_fp16 = softmax(axis = var_5281, x = attn_weights_213_cast_fp16)[name = string("attn_weights_215_cast_fp16")]; bool var_5287_transpose_x_1 = const()[name = string("op_5287_transpose_x_1"), val = bool(true)]; bool var_5287_transpose_y_1 = const()[name = string("op_5287_transpose_y_1"), val = bool(false)]; tensor var_5287_cast_fp16 = matmul(transpose_x = var_5287_transpose_x_1, transpose_y = var_5287_transpose_y_1, x = attn_weights_215_cast_fp16, y = var_5271_cast_fp16_0)[name = string("op_5287_cast_fp16")]; bool attn_weights_217_transpose_x_0 = const()[name = string("attn_weights_217_transpose_x_0"), val = bool(false)]; bool attn_weights_217_transpose_y_0 = const()[name = string("attn_weights_217_transpose_y_0"), val = bool(false)]; tensor attn_weights_217_cast_fp16 = matmul(transpose_x = attn_weights_217_transpose_x_0, transpose_y = attn_weights_217_transpose_y_0, x = var_5261_cast_fp16_1, y = var_5274_1)[name = string("attn_weights_217_cast_fp16")]; fp16 var_5289_to_fp16 = const()[name = string("op_5289_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_219_cast_fp16 = mul(x = attn_weights_217_cast_fp16, y = var_5289_to_fp16)[name = string("attn_weights_219_cast_fp16")]; tensor attn_weights_221_cast_fp16 = add(x = attn_weights_219_cast_fp16, y = attn_mask_1)[name = string("attn_weights_221_cast_fp16")]; int32 var_5293 = const()[name = string("op_5293"), val = int32(-2)]; tensor attn_weights_223_cast_fp16 = softmax(axis = var_5293, x = attn_weights_221_cast_fp16)[name = string("attn_weights_223_cast_fp16")]; bool attn_output_105_transpose_x_1 = const()[name = string("attn_output_105_transpose_x_1"), val = bool(true)]; bool attn_output_105_transpose_y_1 = const()[name = string("attn_output_105_transpose_y_1"), val = bool(false)]; tensor attn_output_105_cast_fp16 = matmul(transpose_x = attn_output_105_transpose_x_1, transpose_y = attn_output_105_transpose_y_1, x = attn_weights_223_cast_fp16, y = var_5271_cast_fp16_1)[name = string("attn_output_105_cast_fp16")]; int32 var_5301 = const()[name = string("op_5301"), val = int32(1)]; bool attn_output_107_interleave_0 = const()[name = string("attn_output_107_interleave_0"), val = bool(false)]; tensor attn_output_107_cast_fp16 = concat(axis = var_5301, interleave = attn_output_107_interleave_0, values = (var_5287_cast_fp16, attn_output_105_cast_fp16))[name = string("attn_output_107_cast_fp16")]; tensor var_5305_perm_0 = const()[name = string("op_5305_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_167x = const()[name = string("concat_167x"), val = tensor([1, 2048, 1, -1])]; tensor var_5305_cast_fp16 = transpose(perm = var_5305_perm_0, x = attn_output_107_cast_fp16)[name = string("transpose_42")]; tensor attn_output_111_cast_fp16 = reshape(shape = concat_167x, x = var_5305_cast_fp16)[name = string("attn_output_111_cast_fp16")]; tensor hidden_states_133_strides_0 = const()[name = string("hidden_states_133_strides_0"), val = tensor([1, 1])]; string hidden_states_133_pad_type_0 = const()[name = string("hidden_states_133_pad_type_0"), val = string("valid")]; tensor hidden_states_133_pad_0 = const()[name = string("hidden_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_133_dilations_0 = const()[name = string("hidden_states_133_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_133_groups_0 = const()[name = string("hidden_states_133_groups_0"), val = int32(1)]; tensor hidden_states_133_cast_fp16 = conv(dilations = hidden_states_133_dilations_0, groups = hidden_states_133_groups_0, pad = hidden_states_133_pad_0, pad_type = hidden_states_133_pad_type_0, strides = hidden_states_133_strides_0, weight = layers_13_self_attn_o_proj_weight_cast_fp16, x = attn_output_111_cast_fp16)[name = string("hidden_states_133_cast_fp16")]; tensor hidden_states_135_cast_fp16 = add(x = hidden_states_129_cast_fp16, y = hidden_states_133_cast_fp16)[name = string("hidden_states_135_cast_fp16")]; fp16 const_138_promoted_to_fp16 = const()[name = string("const_138_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5338_cast_fp16 = mul(x = hidden_states_135_cast_fp16, y = const_138_promoted_to_fp16)[name = string("op_5338_cast_fp16")]; int32 var_5336 = const()[name = string("op_5336"), val = int32(1)]; bool doubled_109_interleave_0 = const()[name = string("doubled_109_interleave_0"), val = bool(false)]; tensor doubled_109_cast_fp16 = concat(axis = var_5336, interleave = doubled_109_interleave_0, values = (hidden_states_135_cast_fp16, var_5338_cast_fp16))[name = string("doubled_109_cast_fp16")]; tensor out_55_axes_0 = const()[name = string("out_55_axes_0"), val = tensor([1])]; tensor out_55_gamma_0_to_fp16 = const()[name = string("out_55_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415399424)))]; fp16 var_5348_to_fp16 = const()[name = string("op_5348_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_55_cast_fp16 = layer_norm(axes = out_55_axes_0, epsilon = var_5348_to_fp16, gamma = out_55_gamma_0_to_fp16, x = doubled_109_cast_fp16)[name = string("out_55_cast_fp16")]; tensor var_5359_split_sizes_0 = const()[name = string("op_5359_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5359_axis_0 = const()[name = string("op_5359_axis_0"), val = int32(1)]; tensor var_5359_cast_fp16_0, tensor var_5359_cast_fp16_1 = split(axis = var_5359_axis_0, split_sizes = var_5359_split_sizes_0, x = out_55_cast_fp16)[name = string("op_5359_cast_fp16")]; tensor input_27_strides_0 = const()[name = string("input_27_strides_0"), val = tensor([1, 1])]; string input_27_pad_type_0 = const()[name = string("input_27_pad_type_0"), val = string("valid")]; tensor input_27_pad_0 = const()[name = string("input_27_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_27_dilations_0 = const()[name = string("input_27_dilations_0"), val = tensor([1, 1])]; int32 input_27_groups_0 = const()[name = string("input_27_groups_0"), val = int32(1)]; tensor input_27_cast_fp16 = conv(dilations = input_27_dilations_0, groups = input_27_groups_0, pad = input_27_pad_0, pad_type = input_27_pad_type_0, strides = input_27_strides_0, weight = layers_13_mlp_gate_proj_weight_cast_fp16, x = var_5359_cast_fp16_0)[name = string("input_27_cast_fp16")]; tensor var_5376_cast_fp16 = silu(x = input_27_cast_fp16)[name = string("op_5376_cast_fp16")]; tensor layers_13_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_13_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1415407680)))]; tensor var_5382_strides_0 = const()[name = string("op_5382_strides_0"), val = tensor([1, 1])]; string var_5382_pad_type_0 = const()[name = string("op_5382_pad_type_0"), val = string("valid")]; tensor var_5382_pad_0 = const()[name = string("op_5382_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5382_dilations_0 = const()[name = string("op_5382_dilations_0"), val = tensor([1, 1])]; int32 var_5382_groups_0 = const()[name = string("op_5382_groups_0"), val = int32(1)]; tensor var_5382_cast_fp16 = conv(dilations = var_5382_dilations_0, groups = var_5382_groups_0, pad = var_5382_pad_0, pad_type = var_5382_pad_type_0, strides = var_5382_strides_0, weight = layers_13_mlp_up_proj_weight_to_fp16, x = var_5359_cast_fp16_0)[name = string("op_5382_cast_fp16")]; tensor x_139_cast_fp16 = mul(x = var_5376_cast_fp16, y = var_5382_cast_fp16)[name = string("x_139_cast_fp16")]; tensor hidden_states_137_strides_0 = const()[name = string("hidden_states_137_strides_0"), val = tensor([1, 1])]; string hidden_states_137_pad_type_0 = const()[name = string("hidden_states_137_pad_type_0"), val = string("valid")]; tensor hidden_states_137_pad_0 = const()[name = string("hidden_states_137_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_137_dilations_0 = const()[name = string("hidden_states_137_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_137_groups_0 = const()[name = string("hidden_states_137_groups_0"), val = int32(1)]; tensor hidden_states_137_cast_fp16 = conv(dilations = hidden_states_137_dilations_0, groups = hidden_states_137_groups_0, pad = hidden_states_137_pad_0, pad_type = hidden_states_137_pad_type_0, strides = hidden_states_137_strides_0, weight = layers_13_mlp_down_proj_weight_cast_fp16, x = x_139_cast_fp16)[name = string("hidden_states_137_cast_fp16")]; tensor hidden_states_139_cast_fp16 = add(x = hidden_states_135_cast_fp16, y = hidden_states_137_cast_fp16)[name = string("hidden_states_139_cast_fp16")]; fp16 const_140_promoted_to_fp16 = const()[name = string("const_140_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5400_cast_fp16 = mul(x = hidden_states_139_cast_fp16, y = const_140_promoted_to_fp16)[name = string("op_5400_cast_fp16")]; int32 var_5398 = const()[name = string("op_5398"), val = int32(1)]; bool doubled_113_interleave_0 = const()[name = string("doubled_113_interleave_0"), val = bool(false)]; tensor doubled_113_cast_fp16 = concat(axis = var_5398, interleave = doubled_113_interleave_0, values = (hidden_states_139_cast_fp16, var_5400_cast_fp16))[name = string("doubled_113_cast_fp16")]; tensor out_57_axes_0 = const()[name = string("out_57_axes_0"), val = tensor([1])]; tensor out_57_gamma_0_to_fp16 = const()[name = string("out_57_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440573568)))]; fp16 var_5410_to_fp16 = const()[name = string("op_5410_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_57_cast_fp16 = layer_norm(axes = out_57_axes_0, epsilon = var_5410_to_fp16, gamma = out_57_gamma_0_to_fp16, x = doubled_113_cast_fp16)[name = string("out_57_cast_fp16")]; tensor var_5421_split_sizes_0 = const()[name = string("op_5421_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5421_axis_0 = const()[name = string("op_5421_axis_0"), val = int32(1)]; tensor var_5421_cast_fp16_0, tensor var_5421_cast_fp16_1 = split(axis = var_5421_axis_0, split_sizes = var_5421_split_sizes_0, x = out_57_cast_fp16)[name = string("op_5421_cast_fp16")]; tensor query_states_85_strides_0 = const()[name = string("query_states_85_strides_0"), val = tensor([1, 1])]; string query_states_85_pad_type_0 = const()[name = string("query_states_85_pad_type_0"), val = string("valid")]; tensor query_states_85_pad_0 = const()[name = string("query_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_85_dilations_0 = const()[name = string("query_states_85_dilations_0"), val = tensor([1, 1])]; int32 query_states_85_groups_0 = const()[name = string("query_states_85_groups_0"), val = int32(1)]; tensor query_states_85_cast_fp16 = conv(dilations = query_states_85_dilations_0, groups = query_states_85_groups_0, pad = query_states_85_pad_0, pad_type = query_states_85_pad_type_0, strides = query_states_85_strides_0, weight = layers_14_self_attn_q_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("query_states_85_cast_fp16")]; tensor layers_14_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_14_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1440581824)))]; tensor key_states_141_strides_0 = const()[name = string("key_states_141_strides_0"), val = tensor([1, 1])]; string key_states_141_pad_type_0 = const()[name = string("key_states_141_pad_type_0"), val = string("valid")]; tensor key_states_141_pad_0 = const()[name = string("key_states_141_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_141_dilations_0 = const()[name = string("key_states_141_dilations_0"), val = tensor([1, 1])]; int32 key_states_141_groups_0 = const()[name = string("key_states_141_groups_0"), val = int32(1)]; tensor key_states_141_cast_fp16 = conv(dilations = key_states_141_dilations_0, groups = key_states_141_groups_0, pad = key_states_141_pad_0, pad_type = key_states_141_pad_type_0, strides = key_states_141_strides_0, weight = layers_14_self_attn_k_proj_weight_to_fp16, x = var_5421_cast_fp16_0)[name = string("key_states_141_cast_fp16")]; tensor value_states_85_strides_0 = const()[name = string("value_states_85_strides_0"), val = tensor([1, 1])]; string value_states_85_pad_type_0 = const()[name = string("value_states_85_pad_type_0"), val = string("valid")]; tensor value_states_85_pad_0 = const()[name = string("value_states_85_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_85_dilations_0 = const()[name = string("value_states_85_dilations_0"), val = tensor([1, 1])]; int32 value_states_85_groups_0 = const()[name = string("value_states_85_groups_0"), val = int32(1)]; tensor value_states_85_cast_fp16 = conv(dilations = value_states_85_dilations_0, groups = value_states_85_groups_0, pad = value_states_85_pad_0, pad_type = value_states_85_pad_type_0, strides = value_states_85_strides_0, weight = layers_14_self_attn_v_proj_weight_cast_fp16, x = var_5421_cast_fp16_0)[name = string("value_states_85_cast_fp16")]; tensor concat_168x = const()[name = string("concat_168x"), val = tensor([1, 16, 128, -1])]; tensor x_141_cast_fp16 = reshape(shape = concat_168x, x = query_states_85_cast_fp16)[name = string("x_141_cast_fp16")]; tensor concat_169x = const()[name = string("concat_169x"), val = tensor([1, 2, 128, -1])]; tensor var_5478_cast_fp16 = reshape(shape = concat_169x, x = key_states_141_cast_fp16)[name = string("op_5478_cast_fp16")]; tensor concat_170x = const()[name = string("concat_170x"), val = tensor([1, 2, 128, -1])]; tensor var_5485_cast_fp16 = reshape(shape = concat_170x, x = value_states_85_cast_fp16)[name = string("op_5485_cast_fp16")]; tensor var_5489_cast_fp16 = mul(x = x_141_cast_fp16, y = var_869_cast_fp16)[name = string("op_5489_cast_fp16")]; tensor var_5490_split_sizes_0 = const()[name = string("op_5490_split_sizes_0"), val = tensor([64, 64])]; int32 var_5490_axis_0 = const()[name = string("op_5490_axis_0"), val = int32(-2)]; tensor var_5490_cast_fp16_0, tensor var_5490_cast_fp16_1 = split(axis = var_5490_axis_0, split_sizes = var_5490_split_sizes_0, x = x_141_cast_fp16)[name = string("op_5490_cast_fp16")]; fp16 const_142_promoted_to_fp16 = const()[name = string("const_142_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5492_cast_fp16 = mul(x = var_5490_cast_fp16_1, y = const_142_promoted_to_fp16)[name = string("op_5492_cast_fp16")]; int32 var_5494 = const()[name = string("op_5494"), val = int32(-2)]; bool var_5495_interleave_0 = const()[name = string("op_5495_interleave_0"), val = bool(false)]; tensor var_5495_cast_fp16 = concat(axis = var_5494, interleave = var_5495_interleave_0, values = (var_5492_cast_fp16, var_5490_cast_fp16_0))[name = string("op_5495_cast_fp16")]; tensor var_5496_cast_fp16 = mul(x = var_5495_cast_fp16, y = var_878_cast_fp16)[name = string("op_5496_cast_fp16")]; tensor query_states_87_cast_fp16 = add(x = var_5489_cast_fp16, y = var_5496_cast_fp16)[name = string("query_states_87_cast_fp16")]; tensor var_5502_cast_fp16 = mul(x = var_5478_cast_fp16, y = var_869_cast_fp16)[name = string("op_5502_cast_fp16")]; tensor var_5503_split_sizes_0 = const()[name = string("op_5503_split_sizes_0"), val = tensor([64, 64])]; int32 var_5503_axis_0 = const()[name = string("op_5503_axis_0"), val = int32(-2)]; tensor var_5503_cast_fp16_0, tensor var_5503_cast_fp16_1 = split(axis = var_5503_axis_0, split_sizes = var_5503_split_sizes_0, x = var_5478_cast_fp16)[name = string("op_5503_cast_fp16")]; fp16 const_143_promoted_to_fp16 = const()[name = string("const_143_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5505_cast_fp16 = mul(x = var_5503_cast_fp16_1, y = const_143_promoted_to_fp16)[name = string("op_5505_cast_fp16")]; int32 var_5507 = const()[name = string("op_5507"), val = int32(-2)]; bool var_5508_interleave_0 = const()[name = string("op_5508_interleave_0"), val = bool(false)]; tensor var_5508_cast_fp16 = concat(axis = var_5507, interleave = var_5508_interleave_0, values = (var_5505_cast_fp16, var_5503_cast_fp16_0))[name = string("op_5508_cast_fp16")]; tensor var_5509_cast_fp16 = mul(x = var_5508_cast_fp16, y = var_878_cast_fp16)[name = string("op_5509_cast_fp16")]; tensor key_states_145_cast_fp16 = add(x = var_5502_cast_fp16, y = var_5509_cast_fp16)[name = string("key_states_145_cast_fp16")]; tensor expand_dims_168 = const()[name = string("expand_dims_168"), val = tensor([14])]; tensor expand_dims_169 = const()[name = string("expand_dims_169"), val = tensor([0])]; tensor expand_dims_171 = const()[name = string("expand_dims_171"), val = tensor([0])]; int32 concat_173_axis_0 = const()[name = string("concat_173_axis_0"), val = int32(0)]; bool concat_173_interleave_0 = const()[name = string("concat_173_interleave_0"), val = bool(false)]; tensor concat_173 = concat(axis = concat_173_axis_0, interleave = concat_173_interleave_0, values = (expand_dims_168, expand_dims_169, position_id, expand_dims_171))[name = string("concat_173")]; tensor expand_dims_172 = const()[name = string("expand_dims_172"), val = tensor([15])]; tensor concat_174_values1_0 = const()[name = string("concat_174_values1_0"), val = tensor([0])]; tensor concat_174_values3_0 = const()[name = string("concat_174_values3_0"), val = tensor([0])]; int32 concat_174_axis_0 = const()[name = string("concat_174_axis_0"), val = int32(0)]; bool concat_174_interleave_0 = const()[name = string("concat_174_interleave_0"), val = bool(false)]; tensor concat_174 = concat(axis = concat_174_axis_0, interleave = concat_174_interleave_0, values = (expand_dims_172, concat_174_values1_0, cache_position_end, concat_174_values3_0))[name = string("concat_174")]; tensor key_states_147_perm_0 = const()[name = string("key_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_15_stride_0 = const()[name = string("key_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_147_cast_fp16 = transpose(perm = key_states_147_perm_0, x = key_states_145_cast_fp16)[name = string("transpose_41")]; tensor key_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = key_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = key_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_15_squeeze_mask_0, stride = key_cache_internal_tensor_assign_15_stride_0, update = key_states_147_cast_fp16, x = coreml_update_state_26)[name = string("key_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_15_cast_fp16, input = key_cache)[name = string("coreml_update_state_28_write_state")]; tensor coreml_update_state_28 = read_state(input = key_cache)[name = string("coreml_update_state_28")]; tensor value_states_87_perm_0 = const()[name = string("value_states_87_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_15_stride_0 = const()[name = string("value_cache_internal_tensor_assign_15_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_15_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_15_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_15_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_15_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_87_cast_fp16 = transpose(perm = value_states_87_perm_0, x = var_5485_cast_fp16)[name = string("transpose_40")]; tensor value_cache_internal_tensor_assign_15_cast_fp16 = slice_update(begin = concat_173, begin_mask = value_cache_internal_tensor_assign_15_begin_mask_0, end = concat_174, end_mask = value_cache_internal_tensor_assign_15_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_15_squeeze_mask_0, stride = value_cache_internal_tensor_assign_15_stride_0, update = value_states_87_cast_fp16, x = coreml_update_state_27)[name = string("value_cache_internal_tensor_assign_15_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_15_cast_fp16, input = value_cache)[name = string("coreml_update_state_29_write_state")]; tensor coreml_update_state_29 = read_state(input = value_cache)[name = string("coreml_update_state_29")]; tensor var_5579_begin_0 = const()[name = string("op_5579_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5579_end_0 = const()[name = string("op_5579_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5579_end_mask_0 = const()[name = string("op_5579_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5579_cast_fp16 = slice_by_index(begin = var_5579_begin_0, end = var_5579_end_0, end_mask = var_5579_end_mask_0, x = coreml_update_state_28)[name = string("op_5579_cast_fp16")]; tensor tile_28 = const()[name = string("tile_28"), val = tensor([1, 1])]; int32 var_5582_axis_0 = const()[name = string("op_5582_axis_0"), val = int32(1)]; tensor var_5582_cast_fp16_0, tensor var_5582_cast_fp16_1 = split(axis = var_5582_axis_0, split_sizes = tile_28, x = var_5579_cast_fp16)[name = string("op_5582_cast_fp16")]; tensor var_5589_begin_0 = const()[name = string("op_5589_begin_0"), val = tensor([14, 0, 0, 0])]; tensor var_5589_end_0 = const()[name = string("op_5589_end_0"), val = tensor([15, 2, 2048, 128])]; tensor var_5589_end_mask_0 = const()[name = string("op_5589_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5589_cast_fp16 = slice_by_index(begin = var_5589_begin_0, end = var_5589_end_0, end_mask = var_5589_end_mask_0, x = coreml_update_state_29)[name = string("op_5589_cast_fp16")]; tensor tile_29 = const()[name = string("tile_29"), val = tensor([1, 1])]; int32 var_5592_axis_0 = const()[name = string("op_5592_axis_0"), val = int32(1)]; tensor var_5592_cast_fp16_0, tensor var_5592_cast_fp16_1 = split(axis = var_5592_axis_0, split_sizes = tile_29, x = var_5589_cast_fp16)[name = string("op_5592_cast_fp16")]; tensor var_5595_split_sizes_0 = const()[name = string("op_5595_split_sizes_0"), val = tensor([8, 8])]; int32 var_5595_axis_0 = const()[name = string("op_5595_axis_0"), val = int32(1)]; tensor var_5595_0, tensor var_5595_1 = split(axis = var_5595_axis_0, split_sizes = var_5595_split_sizes_0, x = query_states_87_cast_fp16)[name = string("op_5595")]; bool attn_weights_225_transpose_x_0 = const()[name = string("attn_weights_225_transpose_x_0"), val = bool(false)]; bool attn_weights_225_transpose_y_0 = const()[name = string("attn_weights_225_transpose_y_0"), val = bool(false)]; tensor attn_weights_225_cast_fp16 = matmul(transpose_x = attn_weights_225_transpose_x_0, transpose_y = attn_weights_225_transpose_y_0, x = var_5582_cast_fp16_0, y = var_5595_0)[name = string("attn_weights_225_cast_fp16")]; fp16 var_5598_to_fp16 = const()[name = string("op_5598_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_227_cast_fp16 = mul(x = attn_weights_225_cast_fp16, y = var_5598_to_fp16)[name = string("attn_weights_227_cast_fp16")]; tensor attn_weights_229_cast_fp16 = add(x = attn_weights_227_cast_fp16, y = attn_mask_1)[name = string("attn_weights_229_cast_fp16")]; int32 var_5602 = const()[name = string("op_5602"), val = int32(-2)]; tensor attn_weights_231_cast_fp16 = softmax(axis = var_5602, x = attn_weights_229_cast_fp16)[name = string("attn_weights_231_cast_fp16")]; bool var_5608_transpose_x_1 = const()[name = string("op_5608_transpose_x_1"), val = bool(true)]; bool var_5608_transpose_y_1 = const()[name = string("op_5608_transpose_y_1"), val = bool(false)]; tensor var_5608_cast_fp16 = matmul(transpose_x = var_5608_transpose_x_1, transpose_y = var_5608_transpose_y_1, x = attn_weights_231_cast_fp16, y = var_5592_cast_fp16_0)[name = string("op_5608_cast_fp16")]; bool attn_weights_233_transpose_x_0 = const()[name = string("attn_weights_233_transpose_x_0"), val = bool(false)]; bool attn_weights_233_transpose_y_0 = const()[name = string("attn_weights_233_transpose_y_0"), val = bool(false)]; tensor attn_weights_233_cast_fp16 = matmul(transpose_x = attn_weights_233_transpose_x_0, transpose_y = attn_weights_233_transpose_y_0, x = var_5582_cast_fp16_1, y = var_5595_1)[name = string("attn_weights_233_cast_fp16")]; fp16 var_5610_to_fp16 = const()[name = string("op_5610_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_235_cast_fp16 = mul(x = attn_weights_233_cast_fp16, y = var_5610_to_fp16)[name = string("attn_weights_235_cast_fp16")]; tensor attn_weights_237_cast_fp16 = add(x = attn_weights_235_cast_fp16, y = attn_mask_1)[name = string("attn_weights_237_cast_fp16")]; int32 var_5614 = const()[name = string("op_5614"), val = int32(-2)]; tensor attn_weights_239_cast_fp16 = softmax(axis = var_5614, x = attn_weights_237_cast_fp16)[name = string("attn_weights_239_cast_fp16")]; bool attn_output_113_transpose_x_1 = const()[name = string("attn_output_113_transpose_x_1"), val = bool(true)]; bool attn_output_113_transpose_y_1 = const()[name = string("attn_output_113_transpose_y_1"), val = bool(false)]; tensor attn_output_113_cast_fp16 = matmul(transpose_x = attn_output_113_transpose_x_1, transpose_y = attn_output_113_transpose_y_1, x = attn_weights_239_cast_fp16, y = var_5592_cast_fp16_1)[name = string("attn_output_113_cast_fp16")]; int32 var_5622 = const()[name = string("op_5622"), val = int32(1)]; bool attn_output_115_interleave_0 = const()[name = string("attn_output_115_interleave_0"), val = bool(false)]; tensor attn_output_115_cast_fp16 = concat(axis = var_5622, interleave = attn_output_115_interleave_0, values = (var_5608_cast_fp16, attn_output_113_cast_fp16))[name = string("attn_output_115_cast_fp16")]; tensor var_5626_perm_0 = const()[name = string("op_5626_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_179x = const()[name = string("concat_179x"), val = tensor([1, 2048, 1, -1])]; tensor var_5626_cast_fp16 = transpose(perm = var_5626_perm_0, x = attn_output_115_cast_fp16)[name = string("transpose_39")]; tensor attn_output_119_cast_fp16 = reshape(shape = concat_179x, x = var_5626_cast_fp16)[name = string("attn_output_119_cast_fp16")]; tensor hidden_states_143_strides_0 = const()[name = string("hidden_states_143_strides_0"), val = tensor([1, 1])]; string hidden_states_143_pad_type_0 = const()[name = string("hidden_states_143_pad_type_0"), val = string("valid")]; tensor hidden_states_143_pad_0 = const()[name = string("hidden_states_143_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_143_dilations_0 = const()[name = string("hidden_states_143_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_143_groups_0 = const()[name = string("hidden_states_143_groups_0"), val = int32(1)]; tensor hidden_states_143_cast_fp16 = conv(dilations = hidden_states_143_dilations_0, groups = hidden_states_143_groups_0, pad = hidden_states_143_pad_0, pad_type = hidden_states_143_pad_type_0, strides = hidden_states_143_strides_0, weight = layers_14_self_attn_o_proj_weight_cast_fp16, x = attn_output_119_cast_fp16)[name = string("hidden_states_143_cast_fp16")]; tensor hidden_states_145_cast_fp16 = add(x = hidden_states_139_cast_fp16, y = hidden_states_143_cast_fp16)[name = string("hidden_states_145_cast_fp16")]; fp16 const_148_promoted_to_fp16 = const()[name = string("const_148_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5659_cast_fp16 = mul(x = hidden_states_145_cast_fp16, y = const_148_promoted_to_fp16)[name = string("op_5659_cast_fp16")]; int32 var_5657 = const()[name = string("op_5657"), val = int32(1)]; bool doubled_117_interleave_0 = const()[name = string("doubled_117_interleave_0"), val = bool(false)]; tensor doubled_117_cast_fp16 = concat(axis = var_5657, interleave = doubled_117_interleave_0, values = (hidden_states_145_cast_fp16, var_5659_cast_fp16))[name = string("doubled_117_cast_fp16")]; tensor out_59_axes_0 = const()[name = string("out_59_axes_0"), val = tensor([1])]; tensor out_59_gamma_0_to_fp16 = const()[name = string("out_59_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441630464)))]; fp16 var_5669_to_fp16 = const()[name = string("op_5669_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_59_cast_fp16 = layer_norm(axes = out_59_axes_0, epsilon = var_5669_to_fp16, gamma = out_59_gamma_0_to_fp16, x = doubled_117_cast_fp16)[name = string("out_59_cast_fp16")]; tensor var_5680_split_sizes_0 = const()[name = string("op_5680_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5680_axis_0 = const()[name = string("op_5680_axis_0"), val = int32(1)]; tensor var_5680_cast_fp16_0, tensor var_5680_cast_fp16_1 = split(axis = var_5680_axis_0, split_sizes = var_5680_split_sizes_0, x = out_59_cast_fp16)[name = string("op_5680_cast_fp16")]; tensor input_29_strides_0 = const()[name = string("input_29_strides_0"), val = tensor([1, 1])]; string input_29_pad_type_0 = const()[name = string("input_29_pad_type_0"), val = string("valid")]; tensor input_29_pad_0 = const()[name = string("input_29_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_29_dilations_0 = const()[name = string("input_29_dilations_0"), val = tensor([1, 1])]; int32 input_29_groups_0 = const()[name = string("input_29_groups_0"), val = int32(1)]; tensor input_29_cast_fp16 = conv(dilations = input_29_dilations_0, groups = input_29_groups_0, pad = input_29_pad_0, pad_type = input_29_pad_type_0, strides = input_29_strides_0, weight = layers_14_mlp_gate_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("input_29_cast_fp16")]; tensor var_5697_cast_fp16 = silu(x = input_29_cast_fp16)[name = string("op_5697_cast_fp16")]; tensor var_5703_strides_0 = const()[name = string("op_5703_strides_0"), val = tensor([1, 1])]; string var_5703_pad_type_0 = const()[name = string("op_5703_pad_type_0"), val = string("valid")]; tensor var_5703_pad_0 = const()[name = string("op_5703_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_5703_dilations_0 = const()[name = string("op_5703_dilations_0"), val = tensor([1, 1])]; int32 var_5703_groups_0 = const()[name = string("op_5703_groups_0"), val = int32(1)]; tensor var_5703_cast_fp16 = conv(dilations = var_5703_dilations_0, groups = var_5703_groups_0, pad = var_5703_pad_0, pad_type = var_5703_pad_type_0, strides = var_5703_strides_0, weight = layers_14_mlp_up_proj_weight_cast_fp16, x = var_5680_cast_fp16_0)[name = string("op_5703_cast_fp16")]; tensor x_149_cast_fp16 = mul(x = var_5697_cast_fp16, y = var_5703_cast_fp16)[name = string("x_149_cast_fp16")]; tensor hidden_states_147_strides_0 = const()[name = string("hidden_states_147_strides_0"), val = tensor([1, 1])]; string hidden_states_147_pad_type_0 = const()[name = string("hidden_states_147_pad_type_0"), val = string("valid")]; tensor hidden_states_147_pad_0 = const()[name = string("hidden_states_147_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_147_dilations_0 = const()[name = string("hidden_states_147_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_147_groups_0 = const()[name = string("hidden_states_147_groups_0"), val = int32(1)]; tensor hidden_states_147_cast_fp16 = conv(dilations = hidden_states_147_dilations_0, groups = hidden_states_147_groups_0, pad = hidden_states_147_pad_0, pad_type = hidden_states_147_pad_type_0, strides = hidden_states_147_strides_0, weight = layers_14_mlp_down_proj_weight_cast_fp16, x = x_149_cast_fp16)[name = string("hidden_states_147_cast_fp16")]; tensor hidden_states_149_cast_fp16 = add(x = hidden_states_145_cast_fp16, y = hidden_states_147_cast_fp16)[name = string("hidden_states_149_cast_fp16")]; fp16 const_150_promoted_to_fp16 = const()[name = string("const_150_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5721_cast_fp16 = mul(x = hidden_states_149_cast_fp16, y = const_150_promoted_to_fp16)[name = string("op_5721_cast_fp16")]; int32 var_5719 = const()[name = string("op_5719"), val = int32(1)]; bool doubled_121_interleave_0 = const()[name = string("doubled_121_interleave_0"), val = bool(false)]; tensor doubled_121_cast_fp16 = concat(axis = var_5719, interleave = doubled_121_interleave_0, values = (hidden_states_149_cast_fp16, var_5721_cast_fp16))[name = string("doubled_121_cast_fp16")]; tensor out_61_axes_0 = const()[name = string("out_61_axes_0"), val = tensor([1])]; tensor out_61_gamma_0_to_fp16 = const()[name = string("out_61_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441638720)))]; fp16 var_5731_to_fp16 = const()[name = string("op_5731_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_61_cast_fp16 = layer_norm(axes = out_61_axes_0, epsilon = var_5731_to_fp16, gamma = out_61_gamma_0_to_fp16, x = doubled_121_cast_fp16)[name = string("out_61_cast_fp16")]; tensor var_5742_split_sizes_0 = const()[name = string("op_5742_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_5742_axis_0 = const()[name = string("op_5742_axis_0"), val = int32(1)]; tensor var_5742_cast_fp16_0, tensor var_5742_cast_fp16_1 = split(axis = var_5742_axis_0, split_sizes = var_5742_split_sizes_0, x = out_61_cast_fp16)[name = string("op_5742_cast_fp16")]; tensor query_states_91_strides_0 = const()[name = string("query_states_91_strides_0"), val = tensor([1, 1])]; string query_states_91_pad_type_0 = const()[name = string("query_states_91_pad_type_0"), val = string("valid")]; tensor query_states_91_pad_0 = const()[name = string("query_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_91_dilations_0 = const()[name = string("query_states_91_dilations_0"), val = tensor([1, 1])]; int32 query_states_91_groups_0 = const()[name = string("query_states_91_groups_0"), val = int32(1)]; tensor query_states_91_cast_fp16 = conv(dilations = query_states_91_dilations_0, groups = query_states_91_groups_0, pad = query_states_91_pad_0, pad_type = query_states_91_pad_type_0, strides = query_states_91_strides_0, weight = layers_15_self_attn_q_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("query_states_91_cast_fp16")]; tensor key_states_151_strides_0 = const()[name = string("key_states_151_strides_0"), val = tensor([1, 1])]; string key_states_151_pad_type_0 = const()[name = string("key_states_151_pad_type_0"), val = string("valid")]; tensor key_states_151_pad_0 = const()[name = string("key_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_151_dilations_0 = const()[name = string("key_states_151_dilations_0"), val = tensor([1, 1])]; int32 key_states_151_groups_0 = const()[name = string("key_states_151_groups_0"), val = int32(1)]; tensor key_states_151_cast_fp16 = conv(dilations = key_states_151_dilations_0, groups = key_states_151_groups_0, pad = key_states_151_pad_0, pad_type = key_states_151_pad_type_0, strides = key_states_151_strides_0, weight = layers_15_self_attn_k_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("key_states_151_cast_fp16")]; tensor value_states_91_strides_0 = const()[name = string("value_states_91_strides_0"), val = tensor([1, 1])]; string value_states_91_pad_type_0 = const()[name = string("value_states_91_pad_type_0"), val = string("valid")]; tensor value_states_91_pad_0 = const()[name = string("value_states_91_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_91_dilations_0 = const()[name = string("value_states_91_dilations_0"), val = tensor([1, 1])]; int32 value_states_91_groups_0 = const()[name = string("value_states_91_groups_0"), val = int32(1)]; tensor value_states_91_cast_fp16 = conv(dilations = value_states_91_dilations_0, groups = value_states_91_groups_0, pad = value_states_91_pad_0, pad_type = value_states_91_pad_type_0, strides = value_states_91_strides_0, weight = layers_15_self_attn_v_proj_weight_cast_fp16, x = var_5742_cast_fp16_0)[name = string("value_states_91_cast_fp16")]; tensor concat_180x = const()[name = string("concat_180x"), val = tensor([1, 16, 128, -1])]; tensor x_151_cast_fp16 = reshape(shape = concat_180x, x = query_states_91_cast_fp16)[name = string("x_151_cast_fp16")]; tensor concat_181x = const()[name = string("concat_181x"), val = tensor([1, 2, 128, -1])]; tensor var_5799_cast_fp16 = reshape(shape = concat_181x, x = key_states_151_cast_fp16)[name = string("op_5799_cast_fp16")]; tensor concat_182x = const()[name = string("concat_182x"), val = tensor([1, 2, 128, -1])]; tensor var_5806_cast_fp16 = reshape(shape = concat_182x, x = value_states_91_cast_fp16)[name = string("op_5806_cast_fp16")]; tensor var_5810_cast_fp16 = mul(x = x_151_cast_fp16, y = var_869_cast_fp16)[name = string("op_5810_cast_fp16")]; tensor var_5811_split_sizes_0 = const()[name = string("op_5811_split_sizes_0"), val = tensor([64, 64])]; int32 var_5811_axis_0 = const()[name = string("op_5811_axis_0"), val = int32(-2)]; tensor var_5811_cast_fp16_0, tensor var_5811_cast_fp16_1 = split(axis = var_5811_axis_0, split_sizes = var_5811_split_sizes_0, x = x_151_cast_fp16)[name = string("op_5811_cast_fp16")]; fp16 const_152_promoted_to_fp16 = const()[name = string("const_152_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5813_cast_fp16 = mul(x = var_5811_cast_fp16_1, y = const_152_promoted_to_fp16)[name = string("op_5813_cast_fp16")]; int32 var_5815 = const()[name = string("op_5815"), val = int32(-2)]; bool var_5816_interleave_0 = const()[name = string("op_5816_interleave_0"), val = bool(false)]; tensor var_5816_cast_fp16 = concat(axis = var_5815, interleave = var_5816_interleave_0, values = (var_5813_cast_fp16, var_5811_cast_fp16_0))[name = string("op_5816_cast_fp16")]; tensor var_5817_cast_fp16 = mul(x = var_5816_cast_fp16, y = var_878_cast_fp16)[name = string("op_5817_cast_fp16")]; tensor query_states_93_cast_fp16 = add(x = var_5810_cast_fp16, y = var_5817_cast_fp16)[name = string("query_states_93_cast_fp16")]; tensor var_5823_cast_fp16 = mul(x = var_5799_cast_fp16, y = var_869_cast_fp16)[name = string("op_5823_cast_fp16")]; tensor var_5824_split_sizes_0 = const()[name = string("op_5824_split_sizes_0"), val = tensor([64, 64])]; int32 var_5824_axis_0 = const()[name = string("op_5824_axis_0"), val = int32(-2)]; tensor var_5824_cast_fp16_0, tensor var_5824_cast_fp16_1 = split(axis = var_5824_axis_0, split_sizes = var_5824_split_sizes_0, x = var_5799_cast_fp16)[name = string("op_5824_cast_fp16")]; fp16 const_153_promoted_to_fp16 = const()[name = string("const_153_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5826_cast_fp16 = mul(x = var_5824_cast_fp16_1, y = const_153_promoted_to_fp16)[name = string("op_5826_cast_fp16")]; int32 var_5828 = const()[name = string("op_5828"), val = int32(-2)]; bool var_5829_interleave_0 = const()[name = string("op_5829_interleave_0"), val = bool(false)]; tensor var_5829_cast_fp16 = concat(axis = var_5828, interleave = var_5829_interleave_0, values = (var_5826_cast_fp16, var_5824_cast_fp16_0))[name = string("op_5829_cast_fp16")]; tensor var_5830_cast_fp16 = mul(x = var_5829_cast_fp16, y = var_878_cast_fp16)[name = string("op_5830_cast_fp16")]; tensor key_states_155_cast_fp16 = add(x = var_5823_cast_fp16, y = var_5830_cast_fp16)[name = string("key_states_155_cast_fp16")]; tensor expand_dims_180 = const()[name = string("expand_dims_180"), val = tensor([15])]; tensor expand_dims_181 = const()[name = string("expand_dims_181"), val = tensor([0])]; tensor expand_dims_183 = const()[name = string("expand_dims_183"), val = tensor([0])]; int32 concat_185_axis_0 = const()[name = string("concat_185_axis_0"), val = int32(0)]; bool concat_185_interleave_0 = const()[name = string("concat_185_interleave_0"), val = bool(false)]; tensor concat_185 = concat(axis = concat_185_axis_0, interleave = concat_185_interleave_0, values = (expand_dims_180, expand_dims_181, position_id, expand_dims_183))[name = string("concat_185")]; tensor expand_dims_184 = const()[name = string("expand_dims_184"), val = tensor([16])]; tensor concat_186_values1_0 = const()[name = string("concat_186_values1_0"), val = tensor([0])]; tensor concat_186_values3_0 = const()[name = string("concat_186_values3_0"), val = tensor([0])]; int32 concat_186_axis_0 = const()[name = string("concat_186_axis_0"), val = int32(0)]; bool concat_186_interleave_0 = const()[name = string("concat_186_interleave_0"), val = bool(false)]; tensor concat_186 = concat(axis = concat_186_axis_0, interleave = concat_186_interleave_0, values = (expand_dims_184, concat_186_values1_0, cache_position_end, concat_186_values3_0))[name = string("concat_186")]; tensor key_states_157_perm_0 = const()[name = string("key_states_157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_16_stride_0 = const()[name = string("key_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_157_cast_fp16 = transpose(perm = key_states_157_perm_0, x = key_states_155_cast_fp16)[name = string("transpose_38")]; tensor key_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = key_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = key_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_16_squeeze_mask_0, stride = key_cache_internal_tensor_assign_16_stride_0, update = key_states_157_cast_fp16, x = coreml_update_state_28)[name = string("key_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_16_cast_fp16, input = key_cache)[name = string("coreml_update_state_30_write_state")]; tensor coreml_update_state_30 = read_state(input = key_cache)[name = string("coreml_update_state_30")]; tensor value_states_93_perm_0 = const()[name = string("value_states_93_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_16_stride_0 = const()[name = string("value_cache_internal_tensor_assign_16_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_16_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_16_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_16_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_16_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_93_cast_fp16 = transpose(perm = value_states_93_perm_0, x = var_5806_cast_fp16)[name = string("transpose_37")]; tensor value_cache_internal_tensor_assign_16_cast_fp16 = slice_update(begin = concat_185, begin_mask = value_cache_internal_tensor_assign_16_begin_mask_0, end = concat_186, end_mask = value_cache_internal_tensor_assign_16_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_16_squeeze_mask_0, stride = value_cache_internal_tensor_assign_16_stride_0, update = value_states_93_cast_fp16, x = coreml_update_state_29)[name = string("value_cache_internal_tensor_assign_16_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_16_cast_fp16, input = value_cache)[name = string("coreml_update_state_31_write_state")]; tensor coreml_update_state_31 = read_state(input = value_cache)[name = string("coreml_update_state_31")]; tensor var_5900_begin_0 = const()[name = string("op_5900_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5900_end_0 = const()[name = string("op_5900_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5900_end_mask_0 = const()[name = string("op_5900_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5900_cast_fp16 = slice_by_index(begin = var_5900_begin_0, end = var_5900_end_0, end_mask = var_5900_end_mask_0, x = coreml_update_state_30)[name = string("op_5900_cast_fp16")]; tensor tile_30 = const()[name = string("tile_30"), val = tensor([1, 1])]; int32 var_5903_axis_0 = const()[name = string("op_5903_axis_0"), val = int32(1)]; tensor var_5903_cast_fp16_0, tensor var_5903_cast_fp16_1 = split(axis = var_5903_axis_0, split_sizes = tile_30, x = var_5900_cast_fp16)[name = string("op_5903_cast_fp16")]; tensor var_5910_begin_0 = const()[name = string("op_5910_begin_0"), val = tensor([15, 0, 0, 0])]; tensor var_5910_end_0 = const()[name = string("op_5910_end_0"), val = tensor([16, 2, 2048, 128])]; tensor var_5910_end_mask_0 = const()[name = string("op_5910_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_5910_cast_fp16 = slice_by_index(begin = var_5910_begin_0, end = var_5910_end_0, end_mask = var_5910_end_mask_0, x = coreml_update_state_31)[name = string("op_5910_cast_fp16")]; tensor tile_31 = const()[name = string("tile_31"), val = tensor([1, 1])]; int32 var_5913_axis_0 = const()[name = string("op_5913_axis_0"), val = int32(1)]; tensor var_5913_cast_fp16_0, tensor var_5913_cast_fp16_1 = split(axis = var_5913_axis_0, split_sizes = tile_31, x = var_5910_cast_fp16)[name = string("op_5913_cast_fp16")]; tensor var_5916_split_sizes_0 = const()[name = string("op_5916_split_sizes_0"), val = tensor([8, 8])]; int32 var_5916_axis_0 = const()[name = string("op_5916_axis_0"), val = int32(1)]; tensor var_5916_0, tensor var_5916_1 = split(axis = var_5916_axis_0, split_sizes = var_5916_split_sizes_0, x = query_states_93_cast_fp16)[name = string("op_5916")]; bool attn_weights_241_transpose_x_0 = const()[name = string("attn_weights_241_transpose_x_0"), val = bool(false)]; bool attn_weights_241_transpose_y_0 = const()[name = string("attn_weights_241_transpose_y_0"), val = bool(false)]; tensor attn_weights_241_cast_fp16 = matmul(transpose_x = attn_weights_241_transpose_x_0, transpose_y = attn_weights_241_transpose_y_0, x = var_5903_cast_fp16_0, y = var_5916_0)[name = string("attn_weights_241_cast_fp16")]; fp16 var_5919_to_fp16 = const()[name = string("op_5919_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_243_cast_fp16 = mul(x = attn_weights_241_cast_fp16, y = var_5919_to_fp16)[name = string("attn_weights_243_cast_fp16")]; tensor attn_weights_245_cast_fp16 = add(x = attn_weights_243_cast_fp16, y = attn_mask_1)[name = string("attn_weights_245_cast_fp16")]; int32 var_5923 = const()[name = string("op_5923"), val = int32(-2)]; tensor attn_weights_247_cast_fp16 = softmax(axis = var_5923, x = attn_weights_245_cast_fp16)[name = string("attn_weights_247_cast_fp16")]; bool var_5929_transpose_x_1 = const()[name = string("op_5929_transpose_x_1"), val = bool(true)]; bool var_5929_transpose_y_1 = const()[name = string("op_5929_transpose_y_1"), val = bool(false)]; tensor var_5929_cast_fp16 = matmul(transpose_x = var_5929_transpose_x_1, transpose_y = var_5929_transpose_y_1, x = attn_weights_247_cast_fp16, y = var_5913_cast_fp16_0)[name = string("op_5929_cast_fp16")]; bool attn_weights_249_transpose_x_0 = const()[name = string("attn_weights_249_transpose_x_0"), val = bool(false)]; bool attn_weights_249_transpose_y_0 = const()[name = string("attn_weights_249_transpose_y_0"), val = bool(false)]; tensor attn_weights_249_cast_fp16 = matmul(transpose_x = attn_weights_249_transpose_x_0, transpose_y = attn_weights_249_transpose_y_0, x = var_5903_cast_fp16_1, y = var_5916_1)[name = string("attn_weights_249_cast_fp16")]; fp16 var_5931_to_fp16 = const()[name = string("op_5931_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_251_cast_fp16 = mul(x = attn_weights_249_cast_fp16, y = var_5931_to_fp16)[name = string("attn_weights_251_cast_fp16")]; tensor attn_weights_253_cast_fp16 = add(x = attn_weights_251_cast_fp16, y = attn_mask_1)[name = string("attn_weights_253_cast_fp16")]; int32 var_5935 = const()[name = string("op_5935"), val = int32(-2)]; tensor attn_weights_255_cast_fp16 = softmax(axis = var_5935, x = attn_weights_253_cast_fp16)[name = string("attn_weights_255_cast_fp16")]; bool attn_output_121_transpose_x_1 = const()[name = string("attn_output_121_transpose_x_1"), val = bool(true)]; bool attn_output_121_transpose_y_1 = const()[name = string("attn_output_121_transpose_y_1"), val = bool(false)]; tensor attn_output_121_cast_fp16 = matmul(transpose_x = attn_output_121_transpose_x_1, transpose_y = attn_output_121_transpose_y_1, x = attn_weights_255_cast_fp16, y = var_5913_cast_fp16_1)[name = string("attn_output_121_cast_fp16")]; int32 var_5943 = const()[name = string("op_5943"), val = int32(1)]; bool attn_output_123_interleave_0 = const()[name = string("attn_output_123_interleave_0"), val = bool(false)]; tensor attn_output_123_cast_fp16 = concat(axis = var_5943, interleave = attn_output_123_interleave_0, values = (var_5929_cast_fp16, attn_output_121_cast_fp16))[name = string("attn_output_123_cast_fp16")]; tensor var_5947_perm_0 = const()[name = string("op_5947_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_191x = const()[name = string("concat_191x"), val = tensor([1, 2048, 1, -1])]; tensor var_5947_cast_fp16 = transpose(perm = var_5947_perm_0, x = attn_output_123_cast_fp16)[name = string("transpose_36")]; tensor attn_output_127_cast_fp16 = reshape(shape = concat_191x, x = var_5947_cast_fp16)[name = string("attn_output_127_cast_fp16")]; tensor hidden_states_153_strides_0 = const()[name = string("hidden_states_153_strides_0"), val = tensor([1, 1])]; string hidden_states_153_pad_type_0 = const()[name = string("hidden_states_153_pad_type_0"), val = string("valid")]; tensor hidden_states_153_pad_0 = const()[name = string("hidden_states_153_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_153_dilations_0 = const()[name = string("hidden_states_153_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_153_groups_0 = const()[name = string("hidden_states_153_groups_0"), val = int32(1)]; tensor hidden_states_153_cast_fp16 = conv(dilations = hidden_states_153_dilations_0, groups = hidden_states_153_groups_0, pad = hidden_states_153_pad_0, pad_type = hidden_states_153_pad_type_0, strides = hidden_states_153_strides_0, weight = layers_15_self_attn_o_proj_weight_cast_fp16, x = attn_output_127_cast_fp16)[name = string("hidden_states_153_cast_fp16")]; tensor hidden_states_155_cast_fp16 = add(x = hidden_states_149_cast_fp16, y = hidden_states_153_cast_fp16)[name = string("hidden_states_155_cast_fp16")]; fp16 const_158_promoted_to_fp16 = const()[name = string("const_158_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_5980_cast_fp16 = mul(x = hidden_states_155_cast_fp16, y = const_158_promoted_to_fp16)[name = string("op_5980_cast_fp16")]; int32 var_5978 = const()[name = string("op_5978"), val = int32(1)]; bool doubled_125_interleave_0 = const()[name = string("doubled_125_interleave_0"), val = bool(false)]; tensor doubled_125_cast_fp16 = concat(axis = var_5978, interleave = doubled_125_interleave_0, values = (hidden_states_155_cast_fp16, var_5980_cast_fp16))[name = string("doubled_125_cast_fp16")]; tensor out_63_axes_0 = const()[name = string("out_63_axes_0"), val = tensor([1])]; tensor out_63_gamma_0_to_fp16 = const()[name = string("out_63_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441646976)))]; fp16 var_5990_to_fp16 = const()[name = string("op_5990_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_63_cast_fp16 = layer_norm(axes = out_63_axes_0, epsilon = var_5990_to_fp16, gamma = out_63_gamma_0_to_fp16, x = doubled_125_cast_fp16)[name = string("out_63_cast_fp16")]; tensor var_6001_split_sizes_0 = const()[name = string("op_6001_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6001_axis_0 = const()[name = string("op_6001_axis_0"), val = int32(1)]; tensor var_6001_cast_fp16_0, tensor var_6001_cast_fp16_1 = split(axis = var_6001_axis_0, split_sizes = var_6001_split_sizes_0, x = out_63_cast_fp16)[name = string("op_6001_cast_fp16")]; tensor input_31_strides_0 = const()[name = string("input_31_strides_0"), val = tensor([1, 1])]; string input_31_pad_type_0 = const()[name = string("input_31_pad_type_0"), val = string("valid")]; tensor input_31_pad_0 = const()[name = string("input_31_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_31_dilations_0 = const()[name = string("input_31_dilations_0"), val = tensor([1, 1])]; int32 input_31_groups_0 = const()[name = string("input_31_groups_0"), val = int32(1)]; tensor input_31_cast_fp16 = conv(dilations = input_31_dilations_0, groups = input_31_groups_0, pad = input_31_pad_0, pad_type = input_31_pad_type_0, strides = input_31_strides_0, weight = layers_15_mlp_gate_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("input_31_cast_fp16")]; tensor var_6018_cast_fp16 = silu(x = input_31_cast_fp16)[name = string("op_6018_cast_fp16")]; tensor var_6024_strides_0 = const()[name = string("op_6024_strides_0"), val = tensor([1, 1])]; string var_6024_pad_type_0 = const()[name = string("op_6024_pad_type_0"), val = string("valid")]; tensor var_6024_pad_0 = const()[name = string("op_6024_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6024_dilations_0 = const()[name = string("op_6024_dilations_0"), val = tensor([1, 1])]; int32 var_6024_groups_0 = const()[name = string("op_6024_groups_0"), val = int32(1)]; tensor var_6024_cast_fp16 = conv(dilations = var_6024_dilations_0, groups = var_6024_groups_0, pad = var_6024_pad_0, pad_type = var_6024_pad_type_0, strides = var_6024_strides_0, weight = layers_15_mlp_up_proj_weight_cast_fp16, x = var_6001_cast_fp16_0)[name = string("op_6024_cast_fp16")]; tensor x_159_cast_fp16 = mul(x = var_6018_cast_fp16, y = var_6024_cast_fp16)[name = string("x_159_cast_fp16")]; tensor hidden_states_157_strides_0 = const()[name = string("hidden_states_157_strides_0"), val = tensor([1, 1])]; string hidden_states_157_pad_type_0 = const()[name = string("hidden_states_157_pad_type_0"), val = string("valid")]; tensor hidden_states_157_pad_0 = const()[name = string("hidden_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_157_dilations_0 = const()[name = string("hidden_states_157_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_157_groups_0 = const()[name = string("hidden_states_157_groups_0"), val = int32(1)]; tensor hidden_states_157_cast_fp16 = conv(dilations = hidden_states_157_dilations_0, groups = hidden_states_157_groups_0, pad = hidden_states_157_pad_0, pad_type = hidden_states_157_pad_type_0, strides = hidden_states_157_strides_0, weight = layers_15_mlp_down_proj_weight_cast_fp16, x = x_159_cast_fp16)[name = string("hidden_states_157_cast_fp16")]; tensor hidden_states_159_cast_fp16 = add(x = hidden_states_155_cast_fp16, y = hidden_states_157_cast_fp16)[name = string("hidden_states_159_cast_fp16")]; fp16 const_160_promoted_to_fp16 = const()[name = string("const_160_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6042_cast_fp16 = mul(x = hidden_states_159_cast_fp16, y = const_160_promoted_to_fp16)[name = string("op_6042_cast_fp16")]; int32 var_6040 = const()[name = string("op_6040"), val = int32(1)]; bool doubled_129_interleave_0 = const()[name = string("doubled_129_interleave_0"), val = bool(false)]; tensor doubled_129_cast_fp16 = concat(axis = var_6040, interleave = doubled_129_interleave_0, values = (hidden_states_159_cast_fp16, var_6042_cast_fp16))[name = string("doubled_129_cast_fp16")]; tensor out_65_axes_0 = const()[name = string("out_65_axes_0"), val = tensor([1])]; tensor out_65_gamma_0_to_fp16 = const()[name = string("out_65_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441655232)))]; fp16 var_6052_to_fp16 = const()[name = string("op_6052_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_65_cast_fp16 = layer_norm(axes = out_65_axes_0, epsilon = var_6052_to_fp16, gamma = out_65_gamma_0_to_fp16, x = doubled_129_cast_fp16)[name = string("out_65_cast_fp16")]; tensor var_6063_split_sizes_0 = const()[name = string("op_6063_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6063_axis_0 = const()[name = string("op_6063_axis_0"), val = int32(1)]; tensor var_6063_cast_fp16_0, tensor var_6063_cast_fp16_1 = split(axis = var_6063_axis_0, split_sizes = var_6063_split_sizes_0, x = out_65_cast_fp16)[name = string("op_6063_cast_fp16")]; tensor query_states_97_strides_0 = const()[name = string("query_states_97_strides_0"), val = tensor([1, 1])]; string query_states_97_pad_type_0 = const()[name = string("query_states_97_pad_type_0"), val = string("valid")]; tensor query_states_97_pad_0 = const()[name = string("query_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_97_dilations_0 = const()[name = string("query_states_97_dilations_0"), val = tensor([1, 1])]; int32 query_states_97_groups_0 = const()[name = string("query_states_97_groups_0"), val = int32(1)]; tensor query_states_97_cast_fp16 = conv(dilations = query_states_97_dilations_0, groups = query_states_97_groups_0, pad = query_states_97_pad_0, pad_type = query_states_97_pad_type_0, strides = query_states_97_strides_0, weight = layers_16_self_attn_q_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("query_states_97_cast_fp16")]; tensor key_states_161_strides_0 = const()[name = string("key_states_161_strides_0"), val = tensor([1, 1])]; string key_states_161_pad_type_0 = const()[name = string("key_states_161_pad_type_0"), val = string("valid")]; tensor key_states_161_pad_0 = const()[name = string("key_states_161_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_161_dilations_0 = const()[name = string("key_states_161_dilations_0"), val = tensor([1, 1])]; int32 key_states_161_groups_0 = const()[name = string("key_states_161_groups_0"), val = int32(1)]; tensor key_states_161_cast_fp16 = conv(dilations = key_states_161_dilations_0, groups = key_states_161_groups_0, pad = key_states_161_pad_0, pad_type = key_states_161_pad_type_0, strides = key_states_161_strides_0, weight = layers_16_self_attn_k_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("key_states_161_cast_fp16")]; tensor value_states_97_strides_0 = const()[name = string("value_states_97_strides_0"), val = tensor([1, 1])]; string value_states_97_pad_type_0 = const()[name = string("value_states_97_pad_type_0"), val = string("valid")]; tensor value_states_97_pad_0 = const()[name = string("value_states_97_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_97_dilations_0 = const()[name = string("value_states_97_dilations_0"), val = tensor([1, 1])]; int32 value_states_97_groups_0 = const()[name = string("value_states_97_groups_0"), val = int32(1)]; tensor value_states_97_cast_fp16 = conv(dilations = value_states_97_dilations_0, groups = value_states_97_groups_0, pad = value_states_97_pad_0, pad_type = value_states_97_pad_type_0, strides = value_states_97_strides_0, weight = layers_16_self_attn_v_proj_weight_cast_fp16, x = var_6063_cast_fp16_0)[name = string("value_states_97_cast_fp16")]; tensor concat_192x = const()[name = string("concat_192x"), val = tensor([1, 16, 128, -1])]; tensor x_161_cast_fp16 = reshape(shape = concat_192x, x = query_states_97_cast_fp16)[name = string("x_161_cast_fp16")]; tensor concat_193x = const()[name = string("concat_193x"), val = tensor([1, 2, 128, -1])]; tensor var_6120_cast_fp16 = reshape(shape = concat_193x, x = key_states_161_cast_fp16)[name = string("op_6120_cast_fp16")]; tensor concat_194x = const()[name = string("concat_194x"), val = tensor([1, 2, 128, -1])]; tensor var_6127_cast_fp16 = reshape(shape = concat_194x, x = value_states_97_cast_fp16)[name = string("op_6127_cast_fp16")]; tensor var_6131_cast_fp16 = mul(x = x_161_cast_fp16, y = var_869_cast_fp16)[name = string("op_6131_cast_fp16")]; tensor var_6132_split_sizes_0 = const()[name = string("op_6132_split_sizes_0"), val = tensor([64, 64])]; int32 var_6132_axis_0 = const()[name = string("op_6132_axis_0"), val = int32(-2)]; tensor var_6132_cast_fp16_0, tensor var_6132_cast_fp16_1 = split(axis = var_6132_axis_0, split_sizes = var_6132_split_sizes_0, x = x_161_cast_fp16)[name = string("op_6132_cast_fp16")]; fp16 const_162_promoted_to_fp16 = const()[name = string("const_162_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6134_cast_fp16 = mul(x = var_6132_cast_fp16_1, y = const_162_promoted_to_fp16)[name = string("op_6134_cast_fp16")]; int32 var_6136 = const()[name = string("op_6136"), val = int32(-2)]; bool var_6137_interleave_0 = const()[name = string("op_6137_interleave_0"), val = bool(false)]; tensor var_6137_cast_fp16 = concat(axis = var_6136, interleave = var_6137_interleave_0, values = (var_6134_cast_fp16, var_6132_cast_fp16_0))[name = string("op_6137_cast_fp16")]; tensor var_6138_cast_fp16 = mul(x = var_6137_cast_fp16, y = var_878_cast_fp16)[name = string("op_6138_cast_fp16")]; tensor query_states_99_cast_fp16 = add(x = var_6131_cast_fp16, y = var_6138_cast_fp16)[name = string("query_states_99_cast_fp16")]; tensor var_6144_cast_fp16 = mul(x = var_6120_cast_fp16, y = var_869_cast_fp16)[name = string("op_6144_cast_fp16")]; tensor var_6145_split_sizes_0 = const()[name = string("op_6145_split_sizes_0"), val = tensor([64, 64])]; int32 var_6145_axis_0 = const()[name = string("op_6145_axis_0"), val = int32(-2)]; tensor var_6145_cast_fp16_0, tensor var_6145_cast_fp16_1 = split(axis = var_6145_axis_0, split_sizes = var_6145_split_sizes_0, x = var_6120_cast_fp16)[name = string("op_6145_cast_fp16")]; fp16 const_163_promoted_to_fp16 = const()[name = string("const_163_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6147_cast_fp16 = mul(x = var_6145_cast_fp16_1, y = const_163_promoted_to_fp16)[name = string("op_6147_cast_fp16")]; int32 var_6149 = const()[name = string("op_6149"), val = int32(-2)]; bool var_6150_interleave_0 = const()[name = string("op_6150_interleave_0"), val = bool(false)]; tensor var_6150_cast_fp16 = concat(axis = var_6149, interleave = var_6150_interleave_0, values = (var_6147_cast_fp16, var_6145_cast_fp16_0))[name = string("op_6150_cast_fp16")]; tensor var_6151_cast_fp16 = mul(x = var_6150_cast_fp16, y = var_878_cast_fp16)[name = string("op_6151_cast_fp16")]; tensor key_states_165_cast_fp16 = add(x = var_6144_cast_fp16, y = var_6151_cast_fp16)[name = string("key_states_165_cast_fp16")]; tensor expand_dims_192 = const()[name = string("expand_dims_192"), val = tensor([16])]; tensor expand_dims_193 = const()[name = string("expand_dims_193"), val = tensor([0])]; tensor expand_dims_195 = const()[name = string("expand_dims_195"), val = tensor([0])]; int32 concat_197_axis_0 = const()[name = string("concat_197_axis_0"), val = int32(0)]; bool concat_197_interleave_0 = const()[name = string("concat_197_interleave_0"), val = bool(false)]; tensor concat_197 = concat(axis = concat_197_axis_0, interleave = concat_197_interleave_0, values = (expand_dims_192, expand_dims_193, position_id, expand_dims_195))[name = string("concat_197")]; tensor expand_dims_196 = const()[name = string("expand_dims_196"), val = tensor([17])]; tensor concat_198_values1_0 = const()[name = string("concat_198_values1_0"), val = tensor([0])]; tensor concat_198_values3_0 = const()[name = string("concat_198_values3_0"), val = tensor([0])]; int32 concat_198_axis_0 = const()[name = string("concat_198_axis_0"), val = int32(0)]; bool concat_198_interleave_0 = const()[name = string("concat_198_interleave_0"), val = bool(false)]; tensor concat_198 = concat(axis = concat_198_axis_0, interleave = concat_198_interleave_0, values = (expand_dims_196, concat_198_values1_0, cache_position_end, concat_198_values3_0))[name = string("concat_198")]; tensor key_states_167_perm_0 = const()[name = string("key_states_167_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_17_stride_0 = const()[name = string("key_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_167_cast_fp16 = transpose(perm = key_states_167_perm_0, x = key_states_165_cast_fp16)[name = string("transpose_35")]; tensor key_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = key_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = key_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_17_squeeze_mask_0, stride = key_cache_internal_tensor_assign_17_stride_0, update = key_states_167_cast_fp16, x = coreml_update_state_30)[name = string("key_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_17_cast_fp16, input = key_cache)[name = string("coreml_update_state_32_write_state")]; tensor coreml_update_state_32 = read_state(input = key_cache)[name = string("coreml_update_state_32")]; tensor value_states_99_perm_0 = const()[name = string("value_states_99_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_17_stride_0 = const()[name = string("value_cache_internal_tensor_assign_17_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_17_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_17_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_17_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_17_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_99_cast_fp16 = transpose(perm = value_states_99_perm_0, x = var_6127_cast_fp16)[name = string("transpose_34")]; tensor value_cache_internal_tensor_assign_17_cast_fp16 = slice_update(begin = concat_197, begin_mask = value_cache_internal_tensor_assign_17_begin_mask_0, end = concat_198, end_mask = value_cache_internal_tensor_assign_17_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_17_squeeze_mask_0, stride = value_cache_internal_tensor_assign_17_stride_0, update = value_states_99_cast_fp16, x = coreml_update_state_31)[name = string("value_cache_internal_tensor_assign_17_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_17_cast_fp16, input = value_cache)[name = string("coreml_update_state_33_write_state")]; tensor coreml_update_state_33 = read_state(input = value_cache)[name = string("coreml_update_state_33")]; tensor var_6221_begin_0 = const()[name = string("op_6221_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6221_end_0 = const()[name = string("op_6221_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6221_end_mask_0 = const()[name = string("op_6221_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6221_cast_fp16 = slice_by_index(begin = var_6221_begin_0, end = var_6221_end_0, end_mask = var_6221_end_mask_0, x = coreml_update_state_32)[name = string("op_6221_cast_fp16")]; tensor tile_32 = const()[name = string("tile_32"), val = tensor([1, 1])]; int32 var_6224_axis_0 = const()[name = string("op_6224_axis_0"), val = int32(1)]; tensor var_6224_cast_fp16_0, tensor var_6224_cast_fp16_1 = split(axis = var_6224_axis_0, split_sizes = tile_32, x = var_6221_cast_fp16)[name = string("op_6224_cast_fp16")]; tensor var_6231_begin_0 = const()[name = string("op_6231_begin_0"), val = tensor([16, 0, 0, 0])]; tensor var_6231_end_0 = const()[name = string("op_6231_end_0"), val = tensor([17, 2, 2048, 128])]; tensor var_6231_end_mask_0 = const()[name = string("op_6231_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6231_cast_fp16 = slice_by_index(begin = var_6231_begin_0, end = var_6231_end_0, end_mask = var_6231_end_mask_0, x = coreml_update_state_33)[name = string("op_6231_cast_fp16")]; tensor tile_33 = const()[name = string("tile_33"), val = tensor([1, 1])]; int32 var_6234_axis_0 = const()[name = string("op_6234_axis_0"), val = int32(1)]; tensor var_6234_cast_fp16_0, tensor var_6234_cast_fp16_1 = split(axis = var_6234_axis_0, split_sizes = tile_33, x = var_6231_cast_fp16)[name = string("op_6234_cast_fp16")]; tensor var_6237_split_sizes_0 = const()[name = string("op_6237_split_sizes_0"), val = tensor([8, 8])]; int32 var_6237_axis_0 = const()[name = string("op_6237_axis_0"), val = int32(1)]; tensor var_6237_0, tensor var_6237_1 = split(axis = var_6237_axis_0, split_sizes = var_6237_split_sizes_0, x = query_states_99_cast_fp16)[name = string("op_6237")]; bool attn_weights_257_transpose_x_0 = const()[name = string("attn_weights_257_transpose_x_0"), val = bool(false)]; bool attn_weights_257_transpose_y_0 = const()[name = string("attn_weights_257_transpose_y_0"), val = bool(false)]; tensor attn_weights_257_cast_fp16 = matmul(transpose_x = attn_weights_257_transpose_x_0, transpose_y = attn_weights_257_transpose_y_0, x = var_6224_cast_fp16_0, y = var_6237_0)[name = string("attn_weights_257_cast_fp16")]; fp16 var_6240_to_fp16 = const()[name = string("op_6240_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_259_cast_fp16 = mul(x = attn_weights_257_cast_fp16, y = var_6240_to_fp16)[name = string("attn_weights_259_cast_fp16")]; tensor attn_weights_261_cast_fp16 = add(x = attn_weights_259_cast_fp16, y = attn_mask_1)[name = string("attn_weights_261_cast_fp16")]; int32 var_6244 = const()[name = string("op_6244"), val = int32(-2)]; tensor attn_weights_263_cast_fp16 = softmax(axis = var_6244, x = attn_weights_261_cast_fp16)[name = string("attn_weights_263_cast_fp16")]; bool var_6250_transpose_x_1 = const()[name = string("op_6250_transpose_x_1"), val = bool(true)]; bool var_6250_transpose_y_1 = const()[name = string("op_6250_transpose_y_1"), val = bool(false)]; tensor var_6250_cast_fp16 = matmul(transpose_x = var_6250_transpose_x_1, transpose_y = var_6250_transpose_y_1, x = attn_weights_263_cast_fp16, y = var_6234_cast_fp16_0)[name = string("op_6250_cast_fp16")]; bool attn_weights_265_transpose_x_0 = const()[name = string("attn_weights_265_transpose_x_0"), val = bool(false)]; bool attn_weights_265_transpose_y_0 = const()[name = string("attn_weights_265_transpose_y_0"), val = bool(false)]; tensor attn_weights_265_cast_fp16 = matmul(transpose_x = attn_weights_265_transpose_x_0, transpose_y = attn_weights_265_transpose_y_0, x = var_6224_cast_fp16_1, y = var_6237_1)[name = string("attn_weights_265_cast_fp16")]; fp16 var_6252_to_fp16 = const()[name = string("op_6252_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_267_cast_fp16 = mul(x = attn_weights_265_cast_fp16, y = var_6252_to_fp16)[name = string("attn_weights_267_cast_fp16")]; tensor attn_weights_269_cast_fp16 = add(x = attn_weights_267_cast_fp16, y = attn_mask_1)[name = string("attn_weights_269_cast_fp16")]; int32 var_6256 = const()[name = string("op_6256"), val = int32(-2)]; tensor attn_weights_271_cast_fp16 = softmax(axis = var_6256, x = attn_weights_269_cast_fp16)[name = string("attn_weights_271_cast_fp16")]; bool attn_output_129_transpose_x_1 = const()[name = string("attn_output_129_transpose_x_1"), val = bool(true)]; bool attn_output_129_transpose_y_1 = const()[name = string("attn_output_129_transpose_y_1"), val = bool(false)]; tensor attn_output_129_cast_fp16 = matmul(transpose_x = attn_output_129_transpose_x_1, transpose_y = attn_output_129_transpose_y_1, x = attn_weights_271_cast_fp16, y = var_6234_cast_fp16_1)[name = string("attn_output_129_cast_fp16")]; int32 var_6264 = const()[name = string("op_6264"), val = int32(1)]; bool attn_output_131_interleave_0 = const()[name = string("attn_output_131_interleave_0"), val = bool(false)]; tensor attn_output_131_cast_fp16 = concat(axis = var_6264, interleave = attn_output_131_interleave_0, values = (var_6250_cast_fp16, attn_output_129_cast_fp16))[name = string("attn_output_131_cast_fp16")]; tensor var_6268_perm_0 = const()[name = string("op_6268_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_203x = const()[name = string("concat_203x"), val = tensor([1, 2048, 1, -1])]; tensor var_6268_cast_fp16 = transpose(perm = var_6268_perm_0, x = attn_output_131_cast_fp16)[name = string("transpose_33")]; tensor attn_output_135_cast_fp16 = reshape(shape = concat_203x, x = var_6268_cast_fp16)[name = string("attn_output_135_cast_fp16")]; tensor hidden_states_163_strides_0 = const()[name = string("hidden_states_163_strides_0"), val = tensor([1, 1])]; string hidden_states_163_pad_type_0 = const()[name = string("hidden_states_163_pad_type_0"), val = string("valid")]; tensor hidden_states_163_pad_0 = const()[name = string("hidden_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_163_dilations_0 = const()[name = string("hidden_states_163_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_163_groups_0 = const()[name = string("hidden_states_163_groups_0"), val = int32(1)]; tensor hidden_states_163_cast_fp16 = conv(dilations = hidden_states_163_dilations_0, groups = hidden_states_163_groups_0, pad = hidden_states_163_pad_0, pad_type = hidden_states_163_pad_type_0, strides = hidden_states_163_strides_0, weight = layers_16_self_attn_o_proj_weight_cast_fp16, x = attn_output_135_cast_fp16)[name = string("hidden_states_163_cast_fp16")]; tensor hidden_states_165_cast_fp16 = add(x = hidden_states_159_cast_fp16, y = hidden_states_163_cast_fp16)[name = string("hidden_states_165_cast_fp16")]; fp16 const_168_promoted_to_fp16 = const()[name = string("const_168_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6301_cast_fp16 = mul(x = hidden_states_165_cast_fp16, y = const_168_promoted_to_fp16)[name = string("op_6301_cast_fp16")]; int32 var_6299 = const()[name = string("op_6299"), val = int32(1)]; bool doubled_133_interleave_0 = const()[name = string("doubled_133_interleave_0"), val = bool(false)]; tensor doubled_133_cast_fp16 = concat(axis = var_6299, interleave = doubled_133_interleave_0, values = (hidden_states_165_cast_fp16, var_6301_cast_fp16))[name = string("doubled_133_cast_fp16")]; tensor out_67_axes_0 = const()[name = string("out_67_axes_0"), val = tensor([1])]; tensor out_67_gamma_0_to_fp16 = const()[name = string("out_67_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441663488)))]; fp16 var_6311_to_fp16 = const()[name = string("op_6311_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_67_cast_fp16 = layer_norm(axes = out_67_axes_0, epsilon = var_6311_to_fp16, gamma = out_67_gamma_0_to_fp16, x = doubled_133_cast_fp16)[name = string("out_67_cast_fp16")]; tensor var_6322_split_sizes_0 = const()[name = string("op_6322_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6322_axis_0 = const()[name = string("op_6322_axis_0"), val = int32(1)]; tensor var_6322_cast_fp16_0, tensor var_6322_cast_fp16_1 = split(axis = var_6322_axis_0, split_sizes = var_6322_split_sizes_0, x = out_67_cast_fp16)[name = string("op_6322_cast_fp16")]; tensor layers_16_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1441671744)))]; tensor input_33_strides_0 = const()[name = string("input_33_strides_0"), val = tensor([1, 1])]; string input_33_pad_type_0 = const()[name = string("input_33_pad_type_0"), val = string("valid")]; tensor input_33_pad_0 = const()[name = string("input_33_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_33_dilations_0 = const()[name = string("input_33_dilations_0"), val = tensor([1, 1])]; int32 input_33_groups_0 = const()[name = string("input_33_groups_0"), val = int32(1)]; tensor input_33_cast_fp16 = conv(dilations = input_33_dilations_0, groups = input_33_groups_0, pad = input_33_pad_0, pad_type = input_33_pad_type_0, strides = input_33_strides_0, weight = layers_16_mlp_gate_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("input_33_cast_fp16")]; tensor var_6339_cast_fp16 = silu(x = input_33_cast_fp16)[name = string("op_6339_cast_fp16")]; tensor layers_16_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_16_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1466837632)))]; tensor var_6345_strides_0 = const()[name = string("op_6345_strides_0"), val = tensor([1, 1])]; string var_6345_pad_type_0 = const()[name = string("op_6345_pad_type_0"), val = string("valid")]; tensor var_6345_pad_0 = const()[name = string("op_6345_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6345_dilations_0 = const()[name = string("op_6345_dilations_0"), val = tensor([1, 1])]; int32 var_6345_groups_0 = const()[name = string("op_6345_groups_0"), val = int32(1)]; tensor var_6345_cast_fp16 = conv(dilations = var_6345_dilations_0, groups = var_6345_groups_0, pad = var_6345_pad_0, pad_type = var_6345_pad_type_0, strides = var_6345_strides_0, weight = layers_16_mlp_up_proj_weight_to_fp16, x = var_6322_cast_fp16_0)[name = string("op_6345_cast_fp16")]; tensor x_169_cast_fp16 = mul(x = var_6339_cast_fp16, y = var_6345_cast_fp16)[name = string("x_169_cast_fp16")]; tensor hidden_states_167_strides_0 = const()[name = string("hidden_states_167_strides_0"), val = tensor([1, 1])]; string hidden_states_167_pad_type_0 = const()[name = string("hidden_states_167_pad_type_0"), val = string("valid")]; tensor hidden_states_167_pad_0 = const()[name = string("hidden_states_167_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_167_dilations_0 = const()[name = string("hidden_states_167_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_167_groups_0 = const()[name = string("hidden_states_167_groups_0"), val = int32(1)]; tensor hidden_states_167_cast_fp16 = conv(dilations = hidden_states_167_dilations_0, groups = hidden_states_167_groups_0, pad = hidden_states_167_pad_0, pad_type = hidden_states_167_pad_type_0, strides = hidden_states_167_strides_0, weight = layers_16_mlp_down_proj_weight_cast_fp16, x = x_169_cast_fp16)[name = string("hidden_states_167_cast_fp16")]; tensor hidden_states_169_cast_fp16 = add(x = hidden_states_165_cast_fp16, y = hidden_states_167_cast_fp16)[name = string("hidden_states_169_cast_fp16")]; fp16 const_170_promoted_to_fp16 = const()[name = string("const_170_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6363_cast_fp16 = mul(x = hidden_states_169_cast_fp16, y = const_170_promoted_to_fp16)[name = string("op_6363_cast_fp16")]; int32 var_6361 = const()[name = string("op_6361"), val = int32(1)]; bool doubled_137_interleave_0 = const()[name = string("doubled_137_interleave_0"), val = bool(false)]; tensor doubled_137_cast_fp16 = concat(axis = var_6361, interleave = doubled_137_interleave_0, values = (hidden_states_169_cast_fp16, var_6363_cast_fp16))[name = string("doubled_137_cast_fp16")]; tensor out_69_axes_0 = const()[name = string("out_69_axes_0"), val = tensor([1])]; tensor out_69_gamma_0_to_fp16 = const()[name = string("out_69_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492003520)))]; fp16 var_6373_to_fp16 = const()[name = string("op_6373_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_69_cast_fp16 = layer_norm(axes = out_69_axes_0, epsilon = var_6373_to_fp16, gamma = out_69_gamma_0_to_fp16, x = doubled_137_cast_fp16)[name = string("out_69_cast_fp16")]; tensor var_6384_split_sizes_0 = const()[name = string("op_6384_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6384_axis_0 = const()[name = string("op_6384_axis_0"), val = int32(1)]; tensor var_6384_cast_fp16_0, tensor var_6384_cast_fp16_1 = split(axis = var_6384_axis_0, split_sizes = var_6384_split_sizes_0, x = out_69_cast_fp16)[name = string("op_6384_cast_fp16")]; tensor query_states_103_strides_0 = const()[name = string("query_states_103_strides_0"), val = tensor([1, 1])]; string query_states_103_pad_type_0 = const()[name = string("query_states_103_pad_type_0"), val = string("valid")]; tensor query_states_103_pad_0 = const()[name = string("query_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_103_dilations_0 = const()[name = string("query_states_103_dilations_0"), val = tensor([1, 1])]; int32 query_states_103_groups_0 = const()[name = string("query_states_103_groups_0"), val = int32(1)]; tensor query_states_103_cast_fp16 = conv(dilations = query_states_103_dilations_0, groups = query_states_103_groups_0, pad = query_states_103_pad_0, pad_type = query_states_103_pad_type_0, strides = query_states_103_strides_0, weight = layers_17_self_attn_q_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("query_states_103_cast_fp16")]; tensor key_states_171_strides_0 = const()[name = string("key_states_171_strides_0"), val = tensor([1, 1])]; string key_states_171_pad_type_0 = const()[name = string("key_states_171_pad_type_0"), val = string("valid")]; tensor key_states_171_pad_0 = const()[name = string("key_states_171_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_171_dilations_0 = const()[name = string("key_states_171_dilations_0"), val = tensor([1, 1])]; int32 key_states_171_groups_0 = const()[name = string("key_states_171_groups_0"), val = int32(1)]; tensor key_states_171_cast_fp16 = conv(dilations = key_states_171_dilations_0, groups = key_states_171_groups_0, pad = key_states_171_pad_0, pad_type = key_states_171_pad_type_0, strides = key_states_171_strides_0, weight = layers_17_self_attn_k_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("key_states_171_cast_fp16")]; tensor value_states_103_strides_0 = const()[name = string("value_states_103_strides_0"), val = tensor([1, 1])]; string value_states_103_pad_type_0 = const()[name = string("value_states_103_pad_type_0"), val = string("valid")]; tensor value_states_103_pad_0 = const()[name = string("value_states_103_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_103_dilations_0 = const()[name = string("value_states_103_dilations_0"), val = tensor([1, 1])]; int32 value_states_103_groups_0 = const()[name = string("value_states_103_groups_0"), val = int32(1)]; tensor value_states_103_cast_fp16 = conv(dilations = value_states_103_dilations_0, groups = value_states_103_groups_0, pad = value_states_103_pad_0, pad_type = value_states_103_pad_type_0, strides = value_states_103_strides_0, weight = layers_17_self_attn_v_proj_weight_cast_fp16, x = var_6384_cast_fp16_0)[name = string("value_states_103_cast_fp16")]; tensor concat_204x = const()[name = string("concat_204x"), val = tensor([1, 16, 128, -1])]; tensor x_171_cast_fp16 = reshape(shape = concat_204x, x = query_states_103_cast_fp16)[name = string("x_171_cast_fp16")]; tensor concat_205x = const()[name = string("concat_205x"), val = tensor([1, 2, 128, -1])]; tensor var_6441_cast_fp16 = reshape(shape = concat_205x, x = key_states_171_cast_fp16)[name = string("op_6441_cast_fp16")]; tensor concat_206x = const()[name = string("concat_206x"), val = tensor([1, 2, 128, -1])]; tensor var_6448_cast_fp16 = reshape(shape = concat_206x, x = value_states_103_cast_fp16)[name = string("op_6448_cast_fp16")]; tensor var_6452_cast_fp16 = mul(x = x_171_cast_fp16, y = var_869_cast_fp16)[name = string("op_6452_cast_fp16")]; tensor var_6453_split_sizes_0 = const()[name = string("op_6453_split_sizes_0"), val = tensor([64, 64])]; int32 var_6453_axis_0 = const()[name = string("op_6453_axis_0"), val = int32(-2)]; tensor var_6453_cast_fp16_0, tensor var_6453_cast_fp16_1 = split(axis = var_6453_axis_0, split_sizes = var_6453_split_sizes_0, x = x_171_cast_fp16)[name = string("op_6453_cast_fp16")]; fp16 const_172_promoted_to_fp16 = const()[name = string("const_172_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6455_cast_fp16 = mul(x = var_6453_cast_fp16_1, y = const_172_promoted_to_fp16)[name = string("op_6455_cast_fp16")]; int32 var_6457 = const()[name = string("op_6457"), val = int32(-2)]; bool var_6458_interleave_0 = const()[name = string("op_6458_interleave_0"), val = bool(false)]; tensor var_6458_cast_fp16 = concat(axis = var_6457, interleave = var_6458_interleave_0, values = (var_6455_cast_fp16, var_6453_cast_fp16_0))[name = string("op_6458_cast_fp16")]; tensor var_6459_cast_fp16 = mul(x = var_6458_cast_fp16, y = var_878_cast_fp16)[name = string("op_6459_cast_fp16")]; tensor query_states_105_cast_fp16 = add(x = var_6452_cast_fp16, y = var_6459_cast_fp16)[name = string("query_states_105_cast_fp16")]; tensor var_6465_cast_fp16 = mul(x = var_6441_cast_fp16, y = var_869_cast_fp16)[name = string("op_6465_cast_fp16")]; tensor var_6466_split_sizes_0 = const()[name = string("op_6466_split_sizes_0"), val = tensor([64, 64])]; int32 var_6466_axis_0 = const()[name = string("op_6466_axis_0"), val = int32(-2)]; tensor var_6466_cast_fp16_0, tensor var_6466_cast_fp16_1 = split(axis = var_6466_axis_0, split_sizes = var_6466_split_sizes_0, x = var_6441_cast_fp16)[name = string("op_6466_cast_fp16")]; fp16 const_173_promoted_to_fp16 = const()[name = string("const_173_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6468_cast_fp16 = mul(x = var_6466_cast_fp16_1, y = const_173_promoted_to_fp16)[name = string("op_6468_cast_fp16")]; int32 var_6470 = const()[name = string("op_6470"), val = int32(-2)]; bool var_6471_interleave_0 = const()[name = string("op_6471_interleave_0"), val = bool(false)]; tensor var_6471_cast_fp16 = concat(axis = var_6470, interleave = var_6471_interleave_0, values = (var_6468_cast_fp16, var_6466_cast_fp16_0))[name = string("op_6471_cast_fp16")]; tensor var_6472_cast_fp16 = mul(x = var_6471_cast_fp16, y = var_878_cast_fp16)[name = string("op_6472_cast_fp16")]; tensor key_states_175_cast_fp16 = add(x = var_6465_cast_fp16, y = var_6472_cast_fp16)[name = string("key_states_175_cast_fp16")]; tensor expand_dims_204 = const()[name = string("expand_dims_204"), val = tensor([17])]; tensor expand_dims_205 = const()[name = string("expand_dims_205"), val = tensor([0])]; tensor expand_dims_207 = const()[name = string("expand_dims_207"), val = tensor([0])]; int32 concat_209_axis_0 = const()[name = string("concat_209_axis_0"), val = int32(0)]; bool concat_209_interleave_0 = const()[name = string("concat_209_interleave_0"), val = bool(false)]; tensor concat_209 = concat(axis = concat_209_axis_0, interleave = concat_209_interleave_0, values = (expand_dims_204, expand_dims_205, position_id, expand_dims_207))[name = string("concat_209")]; tensor expand_dims_208 = const()[name = string("expand_dims_208"), val = tensor([18])]; tensor concat_210_values1_0 = const()[name = string("concat_210_values1_0"), val = tensor([0])]; tensor concat_210_values3_0 = const()[name = string("concat_210_values3_0"), val = tensor([0])]; int32 concat_210_axis_0 = const()[name = string("concat_210_axis_0"), val = int32(0)]; bool concat_210_interleave_0 = const()[name = string("concat_210_interleave_0"), val = bool(false)]; tensor concat_210 = concat(axis = concat_210_axis_0, interleave = concat_210_interleave_0, values = (expand_dims_208, concat_210_values1_0, cache_position_end, concat_210_values3_0))[name = string("concat_210")]; tensor key_states_177_perm_0 = const()[name = string("key_states_177_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_18_stride_0 = const()[name = string("key_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_177_cast_fp16 = transpose(perm = key_states_177_perm_0, x = key_states_175_cast_fp16)[name = string("transpose_32")]; tensor key_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = key_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = key_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_18_squeeze_mask_0, stride = key_cache_internal_tensor_assign_18_stride_0, update = key_states_177_cast_fp16, x = coreml_update_state_32)[name = string("key_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_18_cast_fp16, input = key_cache)[name = string("coreml_update_state_34_write_state")]; tensor coreml_update_state_34 = read_state(input = key_cache)[name = string("coreml_update_state_34")]; tensor value_states_105_perm_0 = const()[name = string("value_states_105_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_18_stride_0 = const()[name = string("value_cache_internal_tensor_assign_18_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_18_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_18_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_18_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_18_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_105_cast_fp16 = transpose(perm = value_states_105_perm_0, x = var_6448_cast_fp16)[name = string("transpose_31")]; tensor value_cache_internal_tensor_assign_18_cast_fp16 = slice_update(begin = concat_209, begin_mask = value_cache_internal_tensor_assign_18_begin_mask_0, end = concat_210, end_mask = value_cache_internal_tensor_assign_18_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_18_squeeze_mask_0, stride = value_cache_internal_tensor_assign_18_stride_0, update = value_states_105_cast_fp16, x = coreml_update_state_33)[name = string("value_cache_internal_tensor_assign_18_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_18_cast_fp16, input = value_cache)[name = string("coreml_update_state_35_write_state")]; tensor coreml_update_state_35 = read_state(input = value_cache)[name = string("coreml_update_state_35")]; tensor var_6542_begin_0 = const()[name = string("op_6542_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6542_end_0 = const()[name = string("op_6542_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6542_end_mask_0 = const()[name = string("op_6542_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6542_cast_fp16 = slice_by_index(begin = var_6542_begin_0, end = var_6542_end_0, end_mask = var_6542_end_mask_0, x = coreml_update_state_34)[name = string("op_6542_cast_fp16")]; tensor tile_34 = const()[name = string("tile_34"), val = tensor([1, 1])]; int32 var_6545_axis_0 = const()[name = string("op_6545_axis_0"), val = int32(1)]; tensor var_6545_cast_fp16_0, tensor var_6545_cast_fp16_1 = split(axis = var_6545_axis_0, split_sizes = tile_34, x = var_6542_cast_fp16)[name = string("op_6545_cast_fp16")]; tensor var_6552_begin_0 = const()[name = string("op_6552_begin_0"), val = tensor([17, 0, 0, 0])]; tensor var_6552_end_0 = const()[name = string("op_6552_end_0"), val = tensor([18, 2, 2048, 128])]; tensor var_6552_end_mask_0 = const()[name = string("op_6552_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6552_cast_fp16 = slice_by_index(begin = var_6552_begin_0, end = var_6552_end_0, end_mask = var_6552_end_mask_0, x = coreml_update_state_35)[name = string("op_6552_cast_fp16")]; tensor tile_35 = const()[name = string("tile_35"), val = tensor([1, 1])]; int32 var_6555_axis_0 = const()[name = string("op_6555_axis_0"), val = int32(1)]; tensor var_6555_cast_fp16_0, tensor var_6555_cast_fp16_1 = split(axis = var_6555_axis_0, split_sizes = tile_35, x = var_6552_cast_fp16)[name = string("op_6555_cast_fp16")]; tensor var_6558_split_sizes_0 = const()[name = string("op_6558_split_sizes_0"), val = tensor([8, 8])]; int32 var_6558_axis_0 = const()[name = string("op_6558_axis_0"), val = int32(1)]; tensor var_6558_0, tensor var_6558_1 = split(axis = var_6558_axis_0, split_sizes = var_6558_split_sizes_0, x = query_states_105_cast_fp16)[name = string("op_6558")]; bool attn_weights_273_transpose_x_0 = const()[name = string("attn_weights_273_transpose_x_0"), val = bool(false)]; bool attn_weights_273_transpose_y_0 = const()[name = string("attn_weights_273_transpose_y_0"), val = bool(false)]; tensor attn_weights_273_cast_fp16 = matmul(transpose_x = attn_weights_273_transpose_x_0, transpose_y = attn_weights_273_transpose_y_0, x = var_6545_cast_fp16_0, y = var_6558_0)[name = string("attn_weights_273_cast_fp16")]; fp16 var_6561_to_fp16 = const()[name = string("op_6561_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_275_cast_fp16 = mul(x = attn_weights_273_cast_fp16, y = var_6561_to_fp16)[name = string("attn_weights_275_cast_fp16")]; tensor attn_weights_277_cast_fp16 = add(x = attn_weights_275_cast_fp16, y = attn_mask_1)[name = string("attn_weights_277_cast_fp16")]; int32 var_6565 = const()[name = string("op_6565"), val = int32(-2)]; tensor attn_weights_279_cast_fp16 = softmax(axis = var_6565, x = attn_weights_277_cast_fp16)[name = string("attn_weights_279_cast_fp16")]; bool var_6571_transpose_x_1 = const()[name = string("op_6571_transpose_x_1"), val = bool(true)]; bool var_6571_transpose_y_1 = const()[name = string("op_6571_transpose_y_1"), val = bool(false)]; tensor var_6571_cast_fp16 = matmul(transpose_x = var_6571_transpose_x_1, transpose_y = var_6571_transpose_y_1, x = attn_weights_279_cast_fp16, y = var_6555_cast_fp16_0)[name = string("op_6571_cast_fp16")]; bool attn_weights_281_transpose_x_0 = const()[name = string("attn_weights_281_transpose_x_0"), val = bool(false)]; bool attn_weights_281_transpose_y_0 = const()[name = string("attn_weights_281_transpose_y_0"), val = bool(false)]; tensor attn_weights_281_cast_fp16 = matmul(transpose_x = attn_weights_281_transpose_x_0, transpose_y = attn_weights_281_transpose_y_0, x = var_6545_cast_fp16_1, y = var_6558_1)[name = string("attn_weights_281_cast_fp16")]; fp16 var_6573_to_fp16 = const()[name = string("op_6573_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_283_cast_fp16 = mul(x = attn_weights_281_cast_fp16, y = var_6573_to_fp16)[name = string("attn_weights_283_cast_fp16")]; tensor attn_weights_285_cast_fp16 = add(x = attn_weights_283_cast_fp16, y = attn_mask_1)[name = string("attn_weights_285_cast_fp16")]; int32 var_6577 = const()[name = string("op_6577"), val = int32(-2)]; tensor attn_weights_287_cast_fp16 = softmax(axis = var_6577, x = attn_weights_285_cast_fp16)[name = string("attn_weights_287_cast_fp16")]; bool attn_output_137_transpose_x_1 = const()[name = string("attn_output_137_transpose_x_1"), val = bool(true)]; bool attn_output_137_transpose_y_1 = const()[name = string("attn_output_137_transpose_y_1"), val = bool(false)]; tensor attn_output_137_cast_fp16 = matmul(transpose_x = attn_output_137_transpose_x_1, transpose_y = attn_output_137_transpose_y_1, x = attn_weights_287_cast_fp16, y = var_6555_cast_fp16_1)[name = string("attn_output_137_cast_fp16")]; int32 var_6585 = const()[name = string("op_6585"), val = int32(1)]; bool attn_output_139_interleave_0 = const()[name = string("attn_output_139_interleave_0"), val = bool(false)]; tensor attn_output_139_cast_fp16 = concat(axis = var_6585, interleave = attn_output_139_interleave_0, values = (var_6571_cast_fp16, attn_output_137_cast_fp16))[name = string("attn_output_139_cast_fp16")]; tensor var_6589_perm_0 = const()[name = string("op_6589_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_215x = const()[name = string("concat_215x"), val = tensor([1, 2048, 1, -1])]; tensor var_6589_cast_fp16 = transpose(perm = var_6589_perm_0, x = attn_output_139_cast_fp16)[name = string("transpose_30")]; tensor attn_output_143_cast_fp16 = reshape(shape = concat_215x, x = var_6589_cast_fp16)[name = string("attn_output_143_cast_fp16")]; tensor hidden_states_173_strides_0 = const()[name = string("hidden_states_173_strides_0"), val = tensor([1, 1])]; string hidden_states_173_pad_type_0 = const()[name = string("hidden_states_173_pad_type_0"), val = string("valid")]; tensor hidden_states_173_pad_0 = const()[name = string("hidden_states_173_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_173_dilations_0 = const()[name = string("hidden_states_173_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_173_groups_0 = const()[name = string("hidden_states_173_groups_0"), val = int32(1)]; tensor hidden_states_173_cast_fp16 = conv(dilations = hidden_states_173_dilations_0, groups = hidden_states_173_groups_0, pad = hidden_states_173_pad_0, pad_type = hidden_states_173_pad_type_0, strides = hidden_states_173_strides_0, weight = layers_17_self_attn_o_proj_weight_cast_fp16, x = attn_output_143_cast_fp16)[name = string("hidden_states_173_cast_fp16")]; tensor hidden_states_175_cast_fp16 = add(x = hidden_states_169_cast_fp16, y = hidden_states_173_cast_fp16)[name = string("hidden_states_175_cast_fp16")]; fp16 const_178_promoted_to_fp16 = const()[name = string("const_178_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6622_cast_fp16 = mul(x = hidden_states_175_cast_fp16, y = const_178_promoted_to_fp16)[name = string("op_6622_cast_fp16")]; int32 var_6620 = const()[name = string("op_6620"), val = int32(1)]; bool doubled_141_interleave_0 = const()[name = string("doubled_141_interleave_0"), val = bool(false)]; tensor doubled_141_cast_fp16 = concat(axis = var_6620, interleave = doubled_141_interleave_0, values = (hidden_states_175_cast_fp16, var_6622_cast_fp16))[name = string("doubled_141_cast_fp16")]; tensor out_71_axes_0 = const()[name = string("out_71_axes_0"), val = tensor([1])]; tensor out_71_gamma_0_to_fp16 = const()[name = string("out_71_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492011776)))]; fp16 var_6632_to_fp16 = const()[name = string("op_6632_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_71_cast_fp16 = layer_norm(axes = out_71_axes_0, epsilon = var_6632_to_fp16, gamma = out_71_gamma_0_to_fp16, x = doubled_141_cast_fp16)[name = string("out_71_cast_fp16")]; tensor var_6643_split_sizes_0 = const()[name = string("op_6643_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6643_axis_0 = const()[name = string("op_6643_axis_0"), val = int32(1)]; tensor var_6643_cast_fp16_0, tensor var_6643_cast_fp16_1 = split(axis = var_6643_axis_0, split_sizes = var_6643_split_sizes_0, x = out_71_cast_fp16)[name = string("op_6643_cast_fp16")]; tensor input_35_strides_0 = const()[name = string("input_35_strides_0"), val = tensor([1, 1])]; string input_35_pad_type_0 = const()[name = string("input_35_pad_type_0"), val = string("valid")]; tensor input_35_pad_0 = const()[name = string("input_35_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_35_dilations_0 = const()[name = string("input_35_dilations_0"), val = tensor([1, 1])]; int32 input_35_groups_0 = const()[name = string("input_35_groups_0"), val = int32(1)]; tensor input_35_cast_fp16 = conv(dilations = input_35_dilations_0, groups = input_35_groups_0, pad = input_35_pad_0, pad_type = input_35_pad_type_0, strides = input_35_strides_0, weight = layers_17_mlp_gate_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("input_35_cast_fp16")]; tensor var_6660_cast_fp16 = silu(x = input_35_cast_fp16)[name = string("op_6660_cast_fp16")]; tensor var_6666_strides_0 = const()[name = string("op_6666_strides_0"), val = tensor([1, 1])]; string var_6666_pad_type_0 = const()[name = string("op_6666_pad_type_0"), val = string("valid")]; tensor var_6666_pad_0 = const()[name = string("op_6666_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6666_dilations_0 = const()[name = string("op_6666_dilations_0"), val = tensor([1, 1])]; int32 var_6666_groups_0 = const()[name = string("op_6666_groups_0"), val = int32(1)]; tensor var_6666_cast_fp16 = conv(dilations = var_6666_dilations_0, groups = var_6666_groups_0, pad = var_6666_pad_0, pad_type = var_6666_pad_type_0, strides = var_6666_strides_0, weight = layers_17_mlp_up_proj_weight_cast_fp16, x = var_6643_cast_fp16_0)[name = string("op_6666_cast_fp16")]; tensor x_179_cast_fp16 = mul(x = var_6660_cast_fp16, y = var_6666_cast_fp16)[name = string("x_179_cast_fp16")]; tensor hidden_states_177_strides_0 = const()[name = string("hidden_states_177_strides_0"), val = tensor([1, 1])]; string hidden_states_177_pad_type_0 = const()[name = string("hidden_states_177_pad_type_0"), val = string("valid")]; tensor hidden_states_177_pad_0 = const()[name = string("hidden_states_177_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_177_dilations_0 = const()[name = string("hidden_states_177_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_177_groups_0 = const()[name = string("hidden_states_177_groups_0"), val = int32(1)]; tensor hidden_states_177_cast_fp16 = conv(dilations = hidden_states_177_dilations_0, groups = hidden_states_177_groups_0, pad = hidden_states_177_pad_0, pad_type = hidden_states_177_pad_type_0, strides = hidden_states_177_strides_0, weight = layers_17_mlp_down_proj_weight_cast_fp16, x = x_179_cast_fp16)[name = string("hidden_states_177_cast_fp16")]; tensor hidden_states_179_cast_fp16 = add(x = hidden_states_175_cast_fp16, y = hidden_states_177_cast_fp16)[name = string("hidden_states_179_cast_fp16")]; fp16 const_180_promoted_to_fp16 = const()[name = string("const_180_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6684_cast_fp16 = mul(x = hidden_states_179_cast_fp16, y = const_180_promoted_to_fp16)[name = string("op_6684_cast_fp16")]; int32 var_6682 = const()[name = string("op_6682"), val = int32(1)]; bool doubled_145_interleave_0 = const()[name = string("doubled_145_interleave_0"), val = bool(false)]; tensor doubled_145_cast_fp16 = concat(axis = var_6682, interleave = doubled_145_interleave_0, values = (hidden_states_179_cast_fp16, var_6684_cast_fp16))[name = string("doubled_145_cast_fp16")]; tensor out_73_axes_0 = const()[name = string("out_73_axes_0"), val = tensor([1])]; tensor out_73_gamma_0_to_fp16 = const()[name = string("out_73_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492020032)))]; fp16 var_6694_to_fp16 = const()[name = string("op_6694_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_73_cast_fp16 = layer_norm(axes = out_73_axes_0, epsilon = var_6694_to_fp16, gamma = out_73_gamma_0_to_fp16, x = doubled_145_cast_fp16)[name = string("out_73_cast_fp16")]; tensor var_6705_split_sizes_0 = const()[name = string("op_6705_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6705_axis_0 = const()[name = string("op_6705_axis_0"), val = int32(1)]; tensor var_6705_cast_fp16_0, tensor var_6705_cast_fp16_1 = split(axis = var_6705_axis_0, split_sizes = var_6705_split_sizes_0, x = out_73_cast_fp16)[name = string("op_6705_cast_fp16")]; tensor query_states_109_strides_0 = const()[name = string("query_states_109_strides_0"), val = tensor([1, 1])]; string query_states_109_pad_type_0 = const()[name = string("query_states_109_pad_type_0"), val = string("valid")]; tensor query_states_109_pad_0 = const()[name = string("query_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_109_dilations_0 = const()[name = string("query_states_109_dilations_0"), val = tensor([1, 1])]; int32 query_states_109_groups_0 = const()[name = string("query_states_109_groups_0"), val = int32(1)]; tensor query_states_109_cast_fp16 = conv(dilations = query_states_109_dilations_0, groups = query_states_109_groups_0, pad = query_states_109_pad_0, pad_type = query_states_109_pad_type_0, strides = query_states_109_strides_0, weight = layers_18_self_attn_q_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("query_states_109_cast_fp16")]; tensor key_states_181_strides_0 = const()[name = string("key_states_181_strides_0"), val = tensor([1, 1])]; string key_states_181_pad_type_0 = const()[name = string("key_states_181_pad_type_0"), val = string("valid")]; tensor key_states_181_pad_0 = const()[name = string("key_states_181_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_181_dilations_0 = const()[name = string("key_states_181_dilations_0"), val = tensor([1, 1])]; int32 key_states_181_groups_0 = const()[name = string("key_states_181_groups_0"), val = int32(1)]; tensor key_states_181_cast_fp16 = conv(dilations = key_states_181_dilations_0, groups = key_states_181_groups_0, pad = key_states_181_pad_0, pad_type = key_states_181_pad_type_0, strides = key_states_181_strides_0, weight = layers_18_self_attn_k_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("key_states_181_cast_fp16")]; tensor value_states_109_strides_0 = const()[name = string("value_states_109_strides_0"), val = tensor([1, 1])]; string value_states_109_pad_type_0 = const()[name = string("value_states_109_pad_type_0"), val = string("valid")]; tensor value_states_109_pad_0 = const()[name = string("value_states_109_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_109_dilations_0 = const()[name = string("value_states_109_dilations_0"), val = tensor([1, 1])]; int32 value_states_109_groups_0 = const()[name = string("value_states_109_groups_0"), val = int32(1)]; tensor value_states_109_cast_fp16 = conv(dilations = value_states_109_dilations_0, groups = value_states_109_groups_0, pad = value_states_109_pad_0, pad_type = value_states_109_pad_type_0, strides = value_states_109_strides_0, weight = layers_18_self_attn_v_proj_weight_cast_fp16, x = var_6705_cast_fp16_0)[name = string("value_states_109_cast_fp16")]; tensor concat_216x = const()[name = string("concat_216x"), val = tensor([1, 16, 128, -1])]; tensor x_181_cast_fp16 = reshape(shape = concat_216x, x = query_states_109_cast_fp16)[name = string("x_181_cast_fp16")]; tensor concat_217x = const()[name = string("concat_217x"), val = tensor([1, 2, 128, -1])]; tensor var_6762_cast_fp16 = reshape(shape = concat_217x, x = key_states_181_cast_fp16)[name = string("op_6762_cast_fp16")]; tensor concat_218x = const()[name = string("concat_218x"), val = tensor([1, 2, 128, -1])]; tensor var_6769_cast_fp16 = reshape(shape = concat_218x, x = value_states_109_cast_fp16)[name = string("op_6769_cast_fp16")]; tensor var_6773_cast_fp16 = mul(x = x_181_cast_fp16, y = var_869_cast_fp16)[name = string("op_6773_cast_fp16")]; tensor var_6774_split_sizes_0 = const()[name = string("op_6774_split_sizes_0"), val = tensor([64, 64])]; int32 var_6774_axis_0 = const()[name = string("op_6774_axis_0"), val = int32(-2)]; tensor var_6774_cast_fp16_0, tensor var_6774_cast_fp16_1 = split(axis = var_6774_axis_0, split_sizes = var_6774_split_sizes_0, x = x_181_cast_fp16)[name = string("op_6774_cast_fp16")]; fp16 const_182_promoted_to_fp16 = const()[name = string("const_182_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6776_cast_fp16 = mul(x = var_6774_cast_fp16_1, y = const_182_promoted_to_fp16)[name = string("op_6776_cast_fp16")]; int32 var_6778 = const()[name = string("op_6778"), val = int32(-2)]; bool var_6779_interleave_0 = const()[name = string("op_6779_interleave_0"), val = bool(false)]; tensor var_6779_cast_fp16 = concat(axis = var_6778, interleave = var_6779_interleave_0, values = (var_6776_cast_fp16, var_6774_cast_fp16_0))[name = string("op_6779_cast_fp16")]; tensor var_6780_cast_fp16 = mul(x = var_6779_cast_fp16, y = var_878_cast_fp16)[name = string("op_6780_cast_fp16")]; tensor query_states_111_cast_fp16 = add(x = var_6773_cast_fp16, y = var_6780_cast_fp16)[name = string("query_states_111_cast_fp16")]; tensor var_6786_cast_fp16 = mul(x = var_6762_cast_fp16, y = var_869_cast_fp16)[name = string("op_6786_cast_fp16")]; tensor var_6787_split_sizes_0 = const()[name = string("op_6787_split_sizes_0"), val = tensor([64, 64])]; int32 var_6787_axis_0 = const()[name = string("op_6787_axis_0"), val = int32(-2)]; tensor var_6787_cast_fp16_0, tensor var_6787_cast_fp16_1 = split(axis = var_6787_axis_0, split_sizes = var_6787_split_sizes_0, x = var_6762_cast_fp16)[name = string("op_6787_cast_fp16")]; fp16 const_183_promoted_to_fp16 = const()[name = string("const_183_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6789_cast_fp16 = mul(x = var_6787_cast_fp16_1, y = const_183_promoted_to_fp16)[name = string("op_6789_cast_fp16")]; int32 var_6791 = const()[name = string("op_6791"), val = int32(-2)]; bool var_6792_interleave_0 = const()[name = string("op_6792_interleave_0"), val = bool(false)]; tensor var_6792_cast_fp16 = concat(axis = var_6791, interleave = var_6792_interleave_0, values = (var_6789_cast_fp16, var_6787_cast_fp16_0))[name = string("op_6792_cast_fp16")]; tensor var_6793_cast_fp16 = mul(x = var_6792_cast_fp16, y = var_878_cast_fp16)[name = string("op_6793_cast_fp16")]; tensor key_states_185_cast_fp16 = add(x = var_6786_cast_fp16, y = var_6793_cast_fp16)[name = string("key_states_185_cast_fp16")]; tensor expand_dims_216 = const()[name = string("expand_dims_216"), val = tensor([18])]; tensor expand_dims_217 = const()[name = string("expand_dims_217"), val = tensor([0])]; tensor expand_dims_219 = const()[name = string("expand_dims_219"), val = tensor([0])]; int32 concat_221_axis_0 = const()[name = string("concat_221_axis_0"), val = int32(0)]; bool concat_221_interleave_0 = const()[name = string("concat_221_interleave_0"), val = bool(false)]; tensor concat_221 = concat(axis = concat_221_axis_0, interleave = concat_221_interleave_0, values = (expand_dims_216, expand_dims_217, position_id, expand_dims_219))[name = string("concat_221")]; tensor expand_dims_220 = const()[name = string("expand_dims_220"), val = tensor([19])]; tensor concat_222_values1_0 = const()[name = string("concat_222_values1_0"), val = tensor([0])]; tensor concat_222_values3_0 = const()[name = string("concat_222_values3_0"), val = tensor([0])]; int32 concat_222_axis_0 = const()[name = string("concat_222_axis_0"), val = int32(0)]; bool concat_222_interleave_0 = const()[name = string("concat_222_interleave_0"), val = bool(false)]; tensor concat_222 = concat(axis = concat_222_axis_0, interleave = concat_222_interleave_0, values = (expand_dims_220, concat_222_values1_0, cache_position_end, concat_222_values3_0))[name = string("concat_222")]; tensor key_states_187_perm_0 = const()[name = string("key_states_187_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_19_stride_0 = const()[name = string("key_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_187_cast_fp16 = transpose(perm = key_states_187_perm_0, x = key_states_185_cast_fp16)[name = string("transpose_29")]; tensor key_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = key_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = key_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_19_squeeze_mask_0, stride = key_cache_internal_tensor_assign_19_stride_0, update = key_states_187_cast_fp16, x = coreml_update_state_34)[name = string("key_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_19_cast_fp16, input = key_cache)[name = string("coreml_update_state_36_write_state")]; tensor coreml_update_state_36 = read_state(input = key_cache)[name = string("coreml_update_state_36")]; tensor value_states_111_perm_0 = const()[name = string("value_states_111_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_19_stride_0 = const()[name = string("value_cache_internal_tensor_assign_19_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_19_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_19_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_19_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_19_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_111_cast_fp16 = transpose(perm = value_states_111_perm_0, x = var_6769_cast_fp16)[name = string("transpose_28")]; tensor value_cache_internal_tensor_assign_19_cast_fp16 = slice_update(begin = concat_221, begin_mask = value_cache_internal_tensor_assign_19_begin_mask_0, end = concat_222, end_mask = value_cache_internal_tensor_assign_19_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_19_squeeze_mask_0, stride = value_cache_internal_tensor_assign_19_stride_0, update = value_states_111_cast_fp16, x = coreml_update_state_35)[name = string("value_cache_internal_tensor_assign_19_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_19_cast_fp16, input = value_cache)[name = string("coreml_update_state_37_write_state")]; tensor coreml_update_state_37 = read_state(input = value_cache)[name = string("coreml_update_state_37")]; tensor var_6863_begin_0 = const()[name = string("op_6863_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6863_end_0 = const()[name = string("op_6863_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6863_end_mask_0 = const()[name = string("op_6863_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6863_cast_fp16 = slice_by_index(begin = var_6863_begin_0, end = var_6863_end_0, end_mask = var_6863_end_mask_0, x = coreml_update_state_36)[name = string("op_6863_cast_fp16")]; tensor tile_36 = const()[name = string("tile_36"), val = tensor([1, 1])]; int32 var_6866_axis_0 = const()[name = string("op_6866_axis_0"), val = int32(1)]; tensor var_6866_cast_fp16_0, tensor var_6866_cast_fp16_1 = split(axis = var_6866_axis_0, split_sizes = tile_36, x = var_6863_cast_fp16)[name = string("op_6866_cast_fp16")]; tensor var_6873_begin_0 = const()[name = string("op_6873_begin_0"), val = tensor([18, 0, 0, 0])]; tensor var_6873_end_0 = const()[name = string("op_6873_end_0"), val = tensor([19, 2, 2048, 128])]; tensor var_6873_end_mask_0 = const()[name = string("op_6873_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_6873_cast_fp16 = slice_by_index(begin = var_6873_begin_0, end = var_6873_end_0, end_mask = var_6873_end_mask_0, x = coreml_update_state_37)[name = string("op_6873_cast_fp16")]; tensor tile_37 = const()[name = string("tile_37"), val = tensor([1, 1])]; int32 var_6876_axis_0 = const()[name = string("op_6876_axis_0"), val = int32(1)]; tensor var_6876_cast_fp16_0, tensor var_6876_cast_fp16_1 = split(axis = var_6876_axis_0, split_sizes = tile_37, x = var_6873_cast_fp16)[name = string("op_6876_cast_fp16")]; tensor var_6879_split_sizes_0 = const()[name = string("op_6879_split_sizes_0"), val = tensor([8, 8])]; int32 var_6879_axis_0 = const()[name = string("op_6879_axis_0"), val = int32(1)]; tensor var_6879_0, tensor var_6879_1 = split(axis = var_6879_axis_0, split_sizes = var_6879_split_sizes_0, x = query_states_111_cast_fp16)[name = string("op_6879")]; bool attn_weights_289_transpose_x_0 = const()[name = string("attn_weights_289_transpose_x_0"), val = bool(false)]; bool attn_weights_289_transpose_y_0 = const()[name = string("attn_weights_289_transpose_y_0"), val = bool(false)]; tensor attn_weights_289_cast_fp16 = matmul(transpose_x = attn_weights_289_transpose_x_0, transpose_y = attn_weights_289_transpose_y_0, x = var_6866_cast_fp16_0, y = var_6879_0)[name = string("attn_weights_289_cast_fp16")]; fp16 var_6882_to_fp16 = const()[name = string("op_6882_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_291_cast_fp16 = mul(x = attn_weights_289_cast_fp16, y = var_6882_to_fp16)[name = string("attn_weights_291_cast_fp16")]; tensor attn_weights_293_cast_fp16 = add(x = attn_weights_291_cast_fp16, y = attn_mask_1)[name = string("attn_weights_293_cast_fp16")]; int32 var_6886 = const()[name = string("op_6886"), val = int32(-2)]; tensor attn_weights_295_cast_fp16 = softmax(axis = var_6886, x = attn_weights_293_cast_fp16)[name = string("attn_weights_295_cast_fp16")]; bool var_6892_transpose_x_1 = const()[name = string("op_6892_transpose_x_1"), val = bool(true)]; bool var_6892_transpose_y_1 = const()[name = string("op_6892_transpose_y_1"), val = bool(false)]; tensor var_6892_cast_fp16 = matmul(transpose_x = var_6892_transpose_x_1, transpose_y = var_6892_transpose_y_1, x = attn_weights_295_cast_fp16, y = var_6876_cast_fp16_0)[name = string("op_6892_cast_fp16")]; bool attn_weights_297_transpose_x_0 = const()[name = string("attn_weights_297_transpose_x_0"), val = bool(false)]; bool attn_weights_297_transpose_y_0 = const()[name = string("attn_weights_297_transpose_y_0"), val = bool(false)]; tensor attn_weights_297_cast_fp16 = matmul(transpose_x = attn_weights_297_transpose_x_0, transpose_y = attn_weights_297_transpose_y_0, x = var_6866_cast_fp16_1, y = var_6879_1)[name = string("attn_weights_297_cast_fp16")]; fp16 var_6894_to_fp16 = const()[name = string("op_6894_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_299_cast_fp16 = mul(x = attn_weights_297_cast_fp16, y = var_6894_to_fp16)[name = string("attn_weights_299_cast_fp16")]; tensor attn_weights_301_cast_fp16 = add(x = attn_weights_299_cast_fp16, y = attn_mask_1)[name = string("attn_weights_301_cast_fp16")]; int32 var_6898 = const()[name = string("op_6898"), val = int32(-2)]; tensor attn_weights_303_cast_fp16 = softmax(axis = var_6898, x = attn_weights_301_cast_fp16)[name = string("attn_weights_303_cast_fp16")]; bool attn_output_145_transpose_x_1 = const()[name = string("attn_output_145_transpose_x_1"), val = bool(true)]; bool attn_output_145_transpose_y_1 = const()[name = string("attn_output_145_transpose_y_1"), val = bool(false)]; tensor attn_output_145_cast_fp16 = matmul(transpose_x = attn_output_145_transpose_x_1, transpose_y = attn_output_145_transpose_y_1, x = attn_weights_303_cast_fp16, y = var_6876_cast_fp16_1)[name = string("attn_output_145_cast_fp16")]; int32 var_6906 = const()[name = string("op_6906"), val = int32(1)]; bool attn_output_147_interleave_0 = const()[name = string("attn_output_147_interleave_0"), val = bool(false)]; tensor attn_output_147_cast_fp16 = concat(axis = var_6906, interleave = attn_output_147_interleave_0, values = (var_6892_cast_fp16, attn_output_145_cast_fp16))[name = string("attn_output_147_cast_fp16")]; tensor var_6910_perm_0 = const()[name = string("op_6910_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_227x = const()[name = string("concat_227x"), val = tensor([1, 2048, 1, -1])]; tensor var_6910_cast_fp16 = transpose(perm = var_6910_perm_0, x = attn_output_147_cast_fp16)[name = string("transpose_27")]; tensor attn_output_151_cast_fp16 = reshape(shape = concat_227x, x = var_6910_cast_fp16)[name = string("attn_output_151_cast_fp16")]; tensor hidden_states_183_strides_0 = const()[name = string("hidden_states_183_strides_0"), val = tensor([1, 1])]; string hidden_states_183_pad_type_0 = const()[name = string("hidden_states_183_pad_type_0"), val = string("valid")]; tensor hidden_states_183_pad_0 = const()[name = string("hidden_states_183_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_183_dilations_0 = const()[name = string("hidden_states_183_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_183_groups_0 = const()[name = string("hidden_states_183_groups_0"), val = int32(1)]; tensor hidden_states_183_cast_fp16 = conv(dilations = hidden_states_183_dilations_0, groups = hidden_states_183_groups_0, pad = hidden_states_183_pad_0, pad_type = hidden_states_183_pad_type_0, strides = hidden_states_183_strides_0, weight = layers_18_self_attn_o_proj_weight_cast_fp16, x = attn_output_151_cast_fp16)[name = string("hidden_states_183_cast_fp16")]; tensor hidden_states_185_cast_fp16 = add(x = hidden_states_179_cast_fp16, y = hidden_states_183_cast_fp16)[name = string("hidden_states_185_cast_fp16")]; fp16 const_188_promoted_to_fp16 = const()[name = string("const_188_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_6943_cast_fp16 = mul(x = hidden_states_185_cast_fp16, y = const_188_promoted_to_fp16)[name = string("op_6943_cast_fp16")]; int32 var_6941 = const()[name = string("op_6941"), val = int32(1)]; bool doubled_149_interleave_0 = const()[name = string("doubled_149_interleave_0"), val = bool(false)]; tensor doubled_149_cast_fp16 = concat(axis = var_6941, interleave = doubled_149_interleave_0, values = (hidden_states_185_cast_fp16, var_6943_cast_fp16))[name = string("doubled_149_cast_fp16")]; tensor out_75_axes_0 = const()[name = string("out_75_axes_0"), val = tensor([1])]; tensor out_75_gamma_0_to_fp16 = const()[name = string("out_75_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492028288)))]; fp16 var_6953_to_fp16 = const()[name = string("op_6953_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_75_cast_fp16 = layer_norm(axes = out_75_axes_0, epsilon = var_6953_to_fp16, gamma = out_75_gamma_0_to_fp16, x = doubled_149_cast_fp16)[name = string("out_75_cast_fp16")]; tensor var_6964_split_sizes_0 = const()[name = string("op_6964_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_6964_axis_0 = const()[name = string("op_6964_axis_0"), val = int32(1)]; tensor var_6964_cast_fp16_0, tensor var_6964_cast_fp16_1 = split(axis = var_6964_axis_0, split_sizes = var_6964_split_sizes_0, x = out_75_cast_fp16)[name = string("op_6964_cast_fp16")]; tensor input_37_strides_0 = const()[name = string("input_37_strides_0"), val = tensor([1, 1])]; string input_37_pad_type_0 = const()[name = string("input_37_pad_type_0"), val = string("valid")]; tensor input_37_pad_0 = const()[name = string("input_37_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_37_dilations_0 = const()[name = string("input_37_dilations_0"), val = tensor([1, 1])]; int32 input_37_groups_0 = const()[name = string("input_37_groups_0"), val = int32(1)]; tensor input_37_cast_fp16 = conv(dilations = input_37_dilations_0, groups = input_37_groups_0, pad = input_37_pad_0, pad_type = input_37_pad_type_0, strides = input_37_strides_0, weight = layers_18_mlp_gate_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("input_37_cast_fp16")]; tensor var_6981_cast_fp16 = silu(x = input_37_cast_fp16)[name = string("op_6981_cast_fp16")]; tensor var_6987_strides_0 = const()[name = string("op_6987_strides_0"), val = tensor([1, 1])]; string var_6987_pad_type_0 = const()[name = string("op_6987_pad_type_0"), val = string("valid")]; tensor var_6987_pad_0 = const()[name = string("op_6987_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_6987_dilations_0 = const()[name = string("op_6987_dilations_0"), val = tensor([1, 1])]; int32 var_6987_groups_0 = const()[name = string("op_6987_groups_0"), val = int32(1)]; tensor var_6987_cast_fp16 = conv(dilations = var_6987_dilations_0, groups = var_6987_groups_0, pad = var_6987_pad_0, pad_type = var_6987_pad_type_0, strides = var_6987_strides_0, weight = layers_18_mlp_up_proj_weight_cast_fp16, x = var_6964_cast_fp16_0)[name = string("op_6987_cast_fp16")]; tensor x_189_cast_fp16 = mul(x = var_6981_cast_fp16, y = var_6987_cast_fp16)[name = string("x_189_cast_fp16")]; tensor hidden_states_187_strides_0 = const()[name = string("hidden_states_187_strides_0"), val = tensor([1, 1])]; string hidden_states_187_pad_type_0 = const()[name = string("hidden_states_187_pad_type_0"), val = string("valid")]; tensor hidden_states_187_pad_0 = const()[name = string("hidden_states_187_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_187_dilations_0 = const()[name = string("hidden_states_187_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_187_groups_0 = const()[name = string("hidden_states_187_groups_0"), val = int32(1)]; tensor hidden_states_187_cast_fp16 = conv(dilations = hidden_states_187_dilations_0, groups = hidden_states_187_groups_0, pad = hidden_states_187_pad_0, pad_type = hidden_states_187_pad_type_0, strides = hidden_states_187_strides_0, weight = layers_18_mlp_down_proj_weight_cast_fp16, x = x_189_cast_fp16)[name = string("hidden_states_187_cast_fp16")]; tensor hidden_states_189_cast_fp16 = add(x = hidden_states_185_cast_fp16, y = hidden_states_187_cast_fp16)[name = string("hidden_states_189_cast_fp16")]; fp16 const_190_promoted_to_fp16 = const()[name = string("const_190_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7005_cast_fp16 = mul(x = hidden_states_189_cast_fp16, y = const_190_promoted_to_fp16)[name = string("op_7005_cast_fp16")]; int32 var_7003 = const()[name = string("op_7003"), val = int32(1)]; bool doubled_153_interleave_0 = const()[name = string("doubled_153_interleave_0"), val = bool(false)]; tensor doubled_153_cast_fp16 = concat(axis = var_7003, interleave = doubled_153_interleave_0, values = (hidden_states_189_cast_fp16, var_7005_cast_fp16))[name = string("doubled_153_cast_fp16")]; tensor out_77_axes_0 = const()[name = string("out_77_axes_0"), val = tensor([1])]; tensor out_77_gamma_0_to_fp16 = const()[name = string("out_77_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492036544)))]; fp16 var_7015_to_fp16 = const()[name = string("op_7015_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_77_cast_fp16 = layer_norm(axes = out_77_axes_0, epsilon = var_7015_to_fp16, gamma = out_77_gamma_0_to_fp16, x = doubled_153_cast_fp16)[name = string("out_77_cast_fp16")]; tensor var_7026_split_sizes_0 = const()[name = string("op_7026_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7026_axis_0 = const()[name = string("op_7026_axis_0"), val = int32(1)]; tensor var_7026_cast_fp16_0, tensor var_7026_cast_fp16_1 = split(axis = var_7026_axis_0, split_sizes = var_7026_split_sizes_0, x = out_77_cast_fp16)[name = string("op_7026_cast_fp16")]; tensor query_states_115_strides_0 = const()[name = string("query_states_115_strides_0"), val = tensor([1, 1])]; string query_states_115_pad_type_0 = const()[name = string("query_states_115_pad_type_0"), val = string("valid")]; tensor query_states_115_pad_0 = const()[name = string("query_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_115_dilations_0 = const()[name = string("query_states_115_dilations_0"), val = tensor([1, 1])]; int32 query_states_115_groups_0 = const()[name = string("query_states_115_groups_0"), val = int32(1)]; tensor query_states_115_cast_fp16 = conv(dilations = query_states_115_dilations_0, groups = query_states_115_groups_0, pad = query_states_115_pad_0, pad_type = query_states_115_pad_type_0, strides = query_states_115_strides_0, weight = layers_19_self_attn_q_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("query_states_115_cast_fp16")]; tensor key_states_191_strides_0 = const()[name = string("key_states_191_strides_0"), val = tensor([1, 1])]; string key_states_191_pad_type_0 = const()[name = string("key_states_191_pad_type_0"), val = string("valid")]; tensor key_states_191_pad_0 = const()[name = string("key_states_191_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_191_dilations_0 = const()[name = string("key_states_191_dilations_0"), val = tensor([1, 1])]; int32 key_states_191_groups_0 = const()[name = string("key_states_191_groups_0"), val = int32(1)]; tensor key_states_191_cast_fp16 = conv(dilations = key_states_191_dilations_0, groups = key_states_191_groups_0, pad = key_states_191_pad_0, pad_type = key_states_191_pad_type_0, strides = key_states_191_strides_0, weight = layers_19_self_attn_k_proj_weight_cast_fp16, x = var_7026_cast_fp16_0)[name = string("key_states_191_cast_fp16")]; tensor layers_19_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1492044800)))]; tensor value_states_115_strides_0 = const()[name = string("value_states_115_strides_0"), val = tensor([1, 1])]; string value_states_115_pad_type_0 = const()[name = string("value_states_115_pad_type_0"), val = string("valid")]; tensor value_states_115_pad_0 = const()[name = string("value_states_115_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_115_dilations_0 = const()[name = string("value_states_115_dilations_0"), val = tensor([1, 1])]; int32 value_states_115_groups_0 = const()[name = string("value_states_115_groups_0"), val = int32(1)]; tensor value_states_115_cast_fp16 = conv(dilations = value_states_115_dilations_0, groups = value_states_115_groups_0, pad = value_states_115_pad_0, pad_type = value_states_115_pad_type_0, strides = value_states_115_strides_0, weight = layers_19_self_attn_v_proj_weight_to_fp16, x = var_7026_cast_fp16_0)[name = string("value_states_115_cast_fp16")]; tensor concat_228x = const()[name = string("concat_228x"), val = tensor([1, 16, 128, -1])]; tensor x_191_cast_fp16 = reshape(shape = concat_228x, x = query_states_115_cast_fp16)[name = string("x_191_cast_fp16")]; tensor concat_229x = const()[name = string("concat_229x"), val = tensor([1, 2, 128, -1])]; tensor var_7083_cast_fp16 = reshape(shape = concat_229x, x = key_states_191_cast_fp16)[name = string("op_7083_cast_fp16")]; tensor concat_230x = const()[name = string("concat_230x"), val = tensor([1, 2, 128, -1])]; tensor var_7090_cast_fp16 = reshape(shape = concat_230x, x = value_states_115_cast_fp16)[name = string("op_7090_cast_fp16")]; tensor var_7094_cast_fp16 = mul(x = x_191_cast_fp16, y = var_869_cast_fp16)[name = string("op_7094_cast_fp16")]; tensor var_7095_split_sizes_0 = const()[name = string("op_7095_split_sizes_0"), val = tensor([64, 64])]; int32 var_7095_axis_0 = const()[name = string("op_7095_axis_0"), val = int32(-2)]; tensor var_7095_cast_fp16_0, tensor var_7095_cast_fp16_1 = split(axis = var_7095_axis_0, split_sizes = var_7095_split_sizes_0, x = x_191_cast_fp16)[name = string("op_7095_cast_fp16")]; fp16 const_192_promoted_to_fp16 = const()[name = string("const_192_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7097_cast_fp16 = mul(x = var_7095_cast_fp16_1, y = const_192_promoted_to_fp16)[name = string("op_7097_cast_fp16")]; int32 var_7099 = const()[name = string("op_7099"), val = int32(-2)]; bool var_7100_interleave_0 = const()[name = string("op_7100_interleave_0"), val = bool(false)]; tensor var_7100_cast_fp16 = concat(axis = var_7099, interleave = var_7100_interleave_0, values = (var_7097_cast_fp16, var_7095_cast_fp16_0))[name = string("op_7100_cast_fp16")]; tensor var_7101_cast_fp16 = mul(x = var_7100_cast_fp16, y = var_878_cast_fp16)[name = string("op_7101_cast_fp16")]; tensor query_states_117_cast_fp16 = add(x = var_7094_cast_fp16, y = var_7101_cast_fp16)[name = string("query_states_117_cast_fp16")]; tensor var_7107_cast_fp16 = mul(x = var_7083_cast_fp16, y = var_869_cast_fp16)[name = string("op_7107_cast_fp16")]; tensor var_7108_split_sizes_0 = const()[name = string("op_7108_split_sizes_0"), val = tensor([64, 64])]; int32 var_7108_axis_0 = const()[name = string("op_7108_axis_0"), val = int32(-2)]; tensor var_7108_cast_fp16_0, tensor var_7108_cast_fp16_1 = split(axis = var_7108_axis_0, split_sizes = var_7108_split_sizes_0, x = var_7083_cast_fp16)[name = string("op_7108_cast_fp16")]; fp16 const_193_promoted_to_fp16 = const()[name = string("const_193_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7110_cast_fp16 = mul(x = var_7108_cast_fp16_1, y = const_193_promoted_to_fp16)[name = string("op_7110_cast_fp16")]; int32 var_7112 = const()[name = string("op_7112"), val = int32(-2)]; bool var_7113_interleave_0 = const()[name = string("op_7113_interleave_0"), val = bool(false)]; tensor var_7113_cast_fp16 = concat(axis = var_7112, interleave = var_7113_interleave_0, values = (var_7110_cast_fp16, var_7108_cast_fp16_0))[name = string("op_7113_cast_fp16")]; tensor var_7114_cast_fp16 = mul(x = var_7113_cast_fp16, y = var_878_cast_fp16)[name = string("op_7114_cast_fp16")]; tensor key_states_195_cast_fp16 = add(x = var_7107_cast_fp16, y = var_7114_cast_fp16)[name = string("key_states_195_cast_fp16")]; tensor expand_dims_228 = const()[name = string("expand_dims_228"), val = tensor([19])]; tensor expand_dims_229 = const()[name = string("expand_dims_229"), val = tensor([0])]; tensor expand_dims_231 = const()[name = string("expand_dims_231"), val = tensor([0])]; int32 concat_233_axis_0 = const()[name = string("concat_233_axis_0"), val = int32(0)]; bool concat_233_interleave_0 = const()[name = string("concat_233_interleave_0"), val = bool(false)]; tensor concat_233 = concat(axis = concat_233_axis_0, interleave = concat_233_interleave_0, values = (expand_dims_228, expand_dims_229, position_id, expand_dims_231))[name = string("concat_233")]; tensor expand_dims_232 = const()[name = string("expand_dims_232"), val = tensor([20])]; tensor concat_234_values1_0 = const()[name = string("concat_234_values1_0"), val = tensor([0])]; tensor concat_234_values3_0 = const()[name = string("concat_234_values3_0"), val = tensor([0])]; int32 concat_234_axis_0 = const()[name = string("concat_234_axis_0"), val = int32(0)]; bool concat_234_interleave_0 = const()[name = string("concat_234_interleave_0"), val = bool(false)]; tensor concat_234 = concat(axis = concat_234_axis_0, interleave = concat_234_interleave_0, values = (expand_dims_232, concat_234_values1_0, cache_position_end, concat_234_values3_0))[name = string("concat_234")]; tensor key_states_197_perm_0 = const()[name = string("key_states_197_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_20_stride_0 = const()[name = string("key_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_197_cast_fp16 = transpose(perm = key_states_197_perm_0, x = key_states_195_cast_fp16)[name = string("transpose_26")]; tensor key_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = key_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = key_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_20_squeeze_mask_0, stride = key_cache_internal_tensor_assign_20_stride_0, update = key_states_197_cast_fp16, x = coreml_update_state_36)[name = string("key_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_20_cast_fp16, input = key_cache)[name = string("coreml_update_state_38_write_state")]; tensor coreml_update_state_38 = read_state(input = key_cache)[name = string("coreml_update_state_38")]; tensor value_states_117_perm_0 = const()[name = string("value_states_117_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_20_stride_0 = const()[name = string("value_cache_internal_tensor_assign_20_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_20_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_20_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_20_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_20_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_117_cast_fp16 = transpose(perm = value_states_117_perm_0, x = var_7090_cast_fp16)[name = string("transpose_25")]; tensor value_cache_internal_tensor_assign_20_cast_fp16 = slice_update(begin = concat_233, begin_mask = value_cache_internal_tensor_assign_20_begin_mask_0, end = concat_234, end_mask = value_cache_internal_tensor_assign_20_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_20_squeeze_mask_0, stride = value_cache_internal_tensor_assign_20_stride_0, update = value_states_117_cast_fp16, x = coreml_update_state_37)[name = string("value_cache_internal_tensor_assign_20_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_20_cast_fp16, input = value_cache)[name = string("coreml_update_state_39_write_state")]; tensor coreml_update_state_39 = read_state(input = value_cache)[name = string("coreml_update_state_39")]; tensor var_7184_begin_0 = const()[name = string("op_7184_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7184_end_0 = const()[name = string("op_7184_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7184_end_mask_0 = const()[name = string("op_7184_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7184_cast_fp16 = slice_by_index(begin = var_7184_begin_0, end = var_7184_end_0, end_mask = var_7184_end_mask_0, x = coreml_update_state_38)[name = string("op_7184_cast_fp16")]; tensor tile_38 = const()[name = string("tile_38"), val = tensor([1, 1])]; int32 var_7187_axis_0 = const()[name = string("op_7187_axis_0"), val = int32(1)]; tensor var_7187_cast_fp16_0, tensor var_7187_cast_fp16_1 = split(axis = var_7187_axis_0, split_sizes = tile_38, x = var_7184_cast_fp16)[name = string("op_7187_cast_fp16")]; tensor var_7194_begin_0 = const()[name = string("op_7194_begin_0"), val = tensor([19, 0, 0, 0])]; tensor var_7194_end_0 = const()[name = string("op_7194_end_0"), val = tensor([20, 2, 2048, 128])]; tensor var_7194_end_mask_0 = const()[name = string("op_7194_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7194_cast_fp16 = slice_by_index(begin = var_7194_begin_0, end = var_7194_end_0, end_mask = var_7194_end_mask_0, x = coreml_update_state_39)[name = string("op_7194_cast_fp16")]; tensor tile_39 = const()[name = string("tile_39"), val = tensor([1, 1])]; int32 var_7197_axis_0 = const()[name = string("op_7197_axis_0"), val = int32(1)]; tensor var_7197_cast_fp16_0, tensor var_7197_cast_fp16_1 = split(axis = var_7197_axis_0, split_sizes = tile_39, x = var_7194_cast_fp16)[name = string("op_7197_cast_fp16")]; tensor var_7200_split_sizes_0 = const()[name = string("op_7200_split_sizes_0"), val = tensor([8, 8])]; int32 var_7200_axis_0 = const()[name = string("op_7200_axis_0"), val = int32(1)]; tensor var_7200_0, tensor var_7200_1 = split(axis = var_7200_axis_0, split_sizes = var_7200_split_sizes_0, x = query_states_117_cast_fp16)[name = string("op_7200")]; bool attn_weights_305_transpose_x_0 = const()[name = string("attn_weights_305_transpose_x_0"), val = bool(false)]; bool attn_weights_305_transpose_y_0 = const()[name = string("attn_weights_305_transpose_y_0"), val = bool(false)]; tensor attn_weights_305_cast_fp16 = matmul(transpose_x = attn_weights_305_transpose_x_0, transpose_y = attn_weights_305_transpose_y_0, x = var_7187_cast_fp16_0, y = var_7200_0)[name = string("attn_weights_305_cast_fp16")]; fp16 var_7203_to_fp16 = const()[name = string("op_7203_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_307_cast_fp16 = mul(x = attn_weights_305_cast_fp16, y = var_7203_to_fp16)[name = string("attn_weights_307_cast_fp16")]; tensor attn_weights_309_cast_fp16 = add(x = attn_weights_307_cast_fp16, y = attn_mask_1)[name = string("attn_weights_309_cast_fp16")]; int32 var_7207 = const()[name = string("op_7207"), val = int32(-2)]; tensor attn_weights_311_cast_fp16 = softmax(axis = var_7207, x = attn_weights_309_cast_fp16)[name = string("attn_weights_311_cast_fp16")]; bool var_7213_transpose_x_1 = const()[name = string("op_7213_transpose_x_1"), val = bool(true)]; bool var_7213_transpose_y_1 = const()[name = string("op_7213_transpose_y_1"), val = bool(false)]; tensor var_7213_cast_fp16 = matmul(transpose_x = var_7213_transpose_x_1, transpose_y = var_7213_transpose_y_1, x = attn_weights_311_cast_fp16, y = var_7197_cast_fp16_0)[name = string("op_7213_cast_fp16")]; bool attn_weights_313_transpose_x_0 = const()[name = string("attn_weights_313_transpose_x_0"), val = bool(false)]; bool attn_weights_313_transpose_y_0 = const()[name = string("attn_weights_313_transpose_y_0"), val = bool(false)]; tensor attn_weights_313_cast_fp16 = matmul(transpose_x = attn_weights_313_transpose_x_0, transpose_y = attn_weights_313_transpose_y_0, x = var_7187_cast_fp16_1, y = var_7200_1)[name = string("attn_weights_313_cast_fp16")]; fp16 var_7215_to_fp16 = const()[name = string("op_7215_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_315_cast_fp16 = mul(x = attn_weights_313_cast_fp16, y = var_7215_to_fp16)[name = string("attn_weights_315_cast_fp16")]; tensor attn_weights_317_cast_fp16 = add(x = attn_weights_315_cast_fp16, y = attn_mask_1)[name = string("attn_weights_317_cast_fp16")]; int32 var_7219 = const()[name = string("op_7219"), val = int32(-2)]; tensor attn_weights_319_cast_fp16 = softmax(axis = var_7219, x = attn_weights_317_cast_fp16)[name = string("attn_weights_319_cast_fp16")]; bool attn_output_153_transpose_x_1 = const()[name = string("attn_output_153_transpose_x_1"), val = bool(true)]; bool attn_output_153_transpose_y_1 = const()[name = string("attn_output_153_transpose_y_1"), val = bool(false)]; tensor attn_output_153_cast_fp16 = matmul(transpose_x = attn_output_153_transpose_x_1, transpose_y = attn_output_153_transpose_y_1, x = attn_weights_319_cast_fp16, y = var_7197_cast_fp16_1)[name = string("attn_output_153_cast_fp16")]; int32 var_7227 = const()[name = string("op_7227"), val = int32(1)]; bool attn_output_155_interleave_0 = const()[name = string("attn_output_155_interleave_0"), val = bool(false)]; tensor attn_output_155_cast_fp16 = concat(axis = var_7227, interleave = attn_output_155_interleave_0, values = (var_7213_cast_fp16, attn_output_153_cast_fp16))[name = string("attn_output_155_cast_fp16")]; tensor var_7231_perm_0 = const()[name = string("op_7231_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_239x = const()[name = string("concat_239x"), val = tensor([1, 2048, 1, -1])]; tensor var_7231_cast_fp16 = transpose(perm = var_7231_perm_0, x = attn_output_155_cast_fp16)[name = string("transpose_24")]; tensor attn_output_159_cast_fp16 = reshape(shape = concat_239x, x = var_7231_cast_fp16)[name = string("attn_output_159_cast_fp16")]; tensor layers_19_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_19_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1493093440)))]; tensor hidden_states_193_strides_0 = const()[name = string("hidden_states_193_strides_0"), val = tensor([1, 1])]; string hidden_states_193_pad_type_0 = const()[name = string("hidden_states_193_pad_type_0"), val = string("valid")]; tensor hidden_states_193_pad_0 = const()[name = string("hidden_states_193_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_193_dilations_0 = const()[name = string("hidden_states_193_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_193_groups_0 = const()[name = string("hidden_states_193_groups_0"), val = int32(1)]; tensor hidden_states_193_cast_fp16 = conv(dilations = hidden_states_193_dilations_0, groups = hidden_states_193_groups_0, pad = hidden_states_193_pad_0, pad_type = hidden_states_193_pad_type_0, strides = hidden_states_193_strides_0, weight = layers_19_self_attn_o_proj_weight_to_fp16, x = attn_output_159_cast_fp16)[name = string("hidden_states_193_cast_fp16")]; tensor hidden_states_195_cast_fp16 = add(x = hidden_states_189_cast_fp16, y = hidden_states_193_cast_fp16)[name = string("hidden_states_195_cast_fp16")]; fp16 const_198_promoted_to_fp16 = const()[name = string("const_198_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7264_cast_fp16 = mul(x = hidden_states_195_cast_fp16, y = const_198_promoted_to_fp16)[name = string("op_7264_cast_fp16")]; int32 var_7262 = const()[name = string("op_7262"), val = int32(1)]; bool doubled_157_interleave_0 = const()[name = string("doubled_157_interleave_0"), val = bool(false)]; tensor doubled_157_cast_fp16 = concat(axis = var_7262, interleave = doubled_157_interleave_0, values = (hidden_states_195_cast_fp16, var_7264_cast_fp16))[name = string("doubled_157_cast_fp16")]; tensor out_79_axes_0 = const()[name = string("out_79_axes_0"), val = tensor([1])]; tensor out_79_gamma_0_to_fp16 = const()[name = string("out_79_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501482112)))]; fp16 var_7274_to_fp16 = const()[name = string("op_7274_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_79_cast_fp16 = layer_norm(axes = out_79_axes_0, epsilon = var_7274_to_fp16, gamma = out_79_gamma_0_to_fp16, x = doubled_157_cast_fp16)[name = string("out_79_cast_fp16")]; tensor var_7285_split_sizes_0 = const()[name = string("op_7285_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7285_axis_0 = const()[name = string("op_7285_axis_0"), val = int32(1)]; tensor var_7285_cast_fp16_0, tensor var_7285_cast_fp16_1 = split(axis = var_7285_axis_0, split_sizes = var_7285_split_sizes_0, x = out_79_cast_fp16)[name = string("op_7285_cast_fp16")]; tensor input_39_strides_0 = const()[name = string("input_39_strides_0"), val = tensor([1, 1])]; string input_39_pad_type_0 = const()[name = string("input_39_pad_type_0"), val = string("valid")]; tensor input_39_pad_0 = const()[name = string("input_39_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_39_dilations_0 = const()[name = string("input_39_dilations_0"), val = tensor([1, 1])]; int32 input_39_groups_0 = const()[name = string("input_39_groups_0"), val = int32(1)]; tensor input_39_cast_fp16 = conv(dilations = input_39_dilations_0, groups = input_39_groups_0, pad = input_39_pad_0, pad_type = input_39_pad_type_0, strides = input_39_strides_0, weight = layers_19_mlp_gate_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("input_39_cast_fp16")]; tensor var_7302_cast_fp16 = silu(x = input_39_cast_fp16)[name = string("op_7302_cast_fp16")]; tensor var_7308_strides_0 = const()[name = string("op_7308_strides_0"), val = tensor([1, 1])]; string var_7308_pad_type_0 = const()[name = string("op_7308_pad_type_0"), val = string("valid")]; tensor var_7308_pad_0 = const()[name = string("op_7308_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7308_dilations_0 = const()[name = string("op_7308_dilations_0"), val = tensor([1, 1])]; int32 var_7308_groups_0 = const()[name = string("op_7308_groups_0"), val = int32(1)]; tensor var_7308_cast_fp16 = conv(dilations = var_7308_dilations_0, groups = var_7308_groups_0, pad = var_7308_pad_0, pad_type = var_7308_pad_type_0, strides = var_7308_strides_0, weight = layers_19_mlp_up_proj_weight_cast_fp16, x = var_7285_cast_fp16_0)[name = string("op_7308_cast_fp16")]; tensor x_199_cast_fp16 = mul(x = var_7302_cast_fp16, y = var_7308_cast_fp16)[name = string("x_199_cast_fp16")]; tensor hidden_states_197_strides_0 = const()[name = string("hidden_states_197_strides_0"), val = tensor([1, 1])]; string hidden_states_197_pad_type_0 = const()[name = string("hidden_states_197_pad_type_0"), val = string("valid")]; tensor hidden_states_197_pad_0 = const()[name = string("hidden_states_197_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_197_dilations_0 = const()[name = string("hidden_states_197_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_197_groups_0 = const()[name = string("hidden_states_197_groups_0"), val = int32(1)]; tensor hidden_states_197_cast_fp16 = conv(dilations = hidden_states_197_dilations_0, groups = hidden_states_197_groups_0, pad = hidden_states_197_pad_0, pad_type = hidden_states_197_pad_type_0, strides = hidden_states_197_strides_0, weight = layers_19_mlp_down_proj_weight_cast_fp16, x = x_199_cast_fp16)[name = string("hidden_states_197_cast_fp16")]; tensor hidden_states_199_cast_fp16 = add(x = hidden_states_195_cast_fp16, y = hidden_states_197_cast_fp16)[name = string("hidden_states_199_cast_fp16")]; fp16 const_200_promoted_to_fp16 = const()[name = string("const_200_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7326_cast_fp16 = mul(x = hidden_states_199_cast_fp16, y = const_200_promoted_to_fp16)[name = string("op_7326_cast_fp16")]; int32 var_7324 = const()[name = string("op_7324"), val = int32(1)]; bool doubled_161_interleave_0 = const()[name = string("doubled_161_interleave_0"), val = bool(false)]; tensor doubled_161_cast_fp16 = concat(axis = var_7324, interleave = doubled_161_interleave_0, values = (hidden_states_199_cast_fp16, var_7326_cast_fp16))[name = string("doubled_161_cast_fp16")]; tensor out_81_axes_0 = const()[name = string("out_81_axes_0"), val = tensor([1])]; tensor out_81_gamma_0_to_fp16 = const()[name = string("out_81_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501490368)))]; fp16 var_7336_to_fp16 = const()[name = string("op_7336_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_81_cast_fp16 = layer_norm(axes = out_81_axes_0, epsilon = var_7336_to_fp16, gamma = out_81_gamma_0_to_fp16, x = doubled_161_cast_fp16)[name = string("out_81_cast_fp16")]; tensor var_7347_split_sizes_0 = const()[name = string("op_7347_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7347_axis_0 = const()[name = string("op_7347_axis_0"), val = int32(1)]; tensor var_7347_cast_fp16_0, tensor var_7347_cast_fp16_1 = split(axis = var_7347_axis_0, split_sizes = var_7347_split_sizes_0, x = out_81_cast_fp16)[name = string("op_7347_cast_fp16")]; tensor query_states_121_strides_0 = const()[name = string("query_states_121_strides_0"), val = tensor([1, 1])]; string query_states_121_pad_type_0 = const()[name = string("query_states_121_pad_type_0"), val = string("valid")]; tensor query_states_121_pad_0 = const()[name = string("query_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_121_dilations_0 = const()[name = string("query_states_121_dilations_0"), val = tensor([1, 1])]; int32 query_states_121_groups_0 = const()[name = string("query_states_121_groups_0"), val = int32(1)]; tensor query_states_121_cast_fp16 = conv(dilations = query_states_121_dilations_0, groups = query_states_121_groups_0, pad = query_states_121_pad_0, pad_type = query_states_121_pad_type_0, strides = query_states_121_strides_0, weight = layers_20_self_attn_q_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("query_states_121_cast_fp16")]; tensor key_states_201_strides_0 = const()[name = string("key_states_201_strides_0"), val = tensor([1, 1])]; string key_states_201_pad_type_0 = const()[name = string("key_states_201_pad_type_0"), val = string("valid")]; tensor key_states_201_pad_0 = const()[name = string("key_states_201_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_201_dilations_0 = const()[name = string("key_states_201_dilations_0"), val = tensor([1, 1])]; int32 key_states_201_groups_0 = const()[name = string("key_states_201_groups_0"), val = int32(1)]; tensor key_states_201_cast_fp16 = conv(dilations = key_states_201_dilations_0, groups = key_states_201_groups_0, pad = key_states_201_pad_0, pad_type = key_states_201_pad_type_0, strides = key_states_201_strides_0, weight = layers_20_self_attn_k_proj_weight_cast_fp16, x = var_7347_cast_fp16_0)[name = string("key_states_201_cast_fp16")]; tensor layers_20_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_20_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1501498624)))]; tensor value_states_121_strides_0 = const()[name = string("value_states_121_strides_0"), val = tensor([1, 1])]; string value_states_121_pad_type_0 = const()[name = string("value_states_121_pad_type_0"), val = string("valid")]; tensor value_states_121_pad_0 = const()[name = string("value_states_121_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_121_dilations_0 = const()[name = string("value_states_121_dilations_0"), val = tensor([1, 1])]; int32 value_states_121_groups_0 = const()[name = string("value_states_121_groups_0"), val = int32(1)]; tensor value_states_121_cast_fp16 = conv(dilations = value_states_121_dilations_0, groups = value_states_121_groups_0, pad = value_states_121_pad_0, pad_type = value_states_121_pad_type_0, strides = value_states_121_strides_0, weight = layers_20_self_attn_v_proj_weight_to_fp16, x = var_7347_cast_fp16_0)[name = string("value_states_121_cast_fp16")]; tensor concat_240x = const()[name = string("concat_240x"), val = tensor([1, 16, 128, -1])]; tensor x_201_cast_fp16 = reshape(shape = concat_240x, x = query_states_121_cast_fp16)[name = string("x_201_cast_fp16")]; tensor concat_241x = const()[name = string("concat_241x"), val = tensor([1, 2, 128, -1])]; tensor var_7404_cast_fp16 = reshape(shape = concat_241x, x = key_states_201_cast_fp16)[name = string("op_7404_cast_fp16")]; tensor concat_242x = const()[name = string("concat_242x"), val = tensor([1, 2, 128, -1])]; tensor var_7411_cast_fp16 = reshape(shape = concat_242x, x = value_states_121_cast_fp16)[name = string("op_7411_cast_fp16")]; tensor var_7415_cast_fp16 = mul(x = x_201_cast_fp16, y = var_869_cast_fp16)[name = string("op_7415_cast_fp16")]; tensor var_7416_split_sizes_0 = const()[name = string("op_7416_split_sizes_0"), val = tensor([64, 64])]; int32 var_7416_axis_0 = const()[name = string("op_7416_axis_0"), val = int32(-2)]; tensor var_7416_cast_fp16_0, tensor var_7416_cast_fp16_1 = split(axis = var_7416_axis_0, split_sizes = var_7416_split_sizes_0, x = x_201_cast_fp16)[name = string("op_7416_cast_fp16")]; fp16 const_202_promoted_to_fp16 = const()[name = string("const_202_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7418_cast_fp16 = mul(x = var_7416_cast_fp16_1, y = const_202_promoted_to_fp16)[name = string("op_7418_cast_fp16")]; int32 var_7420 = const()[name = string("op_7420"), val = int32(-2)]; bool var_7421_interleave_0 = const()[name = string("op_7421_interleave_0"), val = bool(false)]; tensor var_7421_cast_fp16 = concat(axis = var_7420, interleave = var_7421_interleave_0, values = (var_7418_cast_fp16, var_7416_cast_fp16_0))[name = string("op_7421_cast_fp16")]; tensor var_7422_cast_fp16 = mul(x = var_7421_cast_fp16, y = var_878_cast_fp16)[name = string("op_7422_cast_fp16")]; tensor query_states_123_cast_fp16 = add(x = var_7415_cast_fp16, y = var_7422_cast_fp16)[name = string("query_states_123_cast_fp16")]; tensor var_7428_cast_fp16 = mul(x = var_7404_cast_fp16, y = var_869_cast_fp16)[name = string("op_7428_cast_fp16")]; tensor var_7429_split_sizes_0 = const()[name = string("op_7429_split_sizes_0"), val = tensor([64, 64])]; int32 var_7429_axis_0 = const()[name = string("op_7429_axis_0"), val = int32(-2)]; tensor var_7429_cast_fp16_0, tensor var_7429_cast_fp16_1 = split(axis = var_7429_axis_0, split_sizes = var_7429_split_sizes_0, x = var_7404_cast_fp16)[name = string("op_7429_cast_fp16")]; fp16 const_203_promoted_to_fp16 = const()[name = string("const_203_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7431_cast_fp16 = mul(x = var_7429_cast_fp16_1, y = const_203_promoted_to_fp16)[name = string("op_7431_cast_fp16")]; int32 var_7433 = const()[name = string("op_7433"), val = int32(-2)]; bool var_7434_interleave_0 = const()[name = string("op_7434_interleave_0"), val = bool(false)]; tensor var_7434_cast_fp16 = concat(axis = var_7433, interleave = var_7434_interleave_0, values = (var_7431_cast_fp16, var_7429_cast_fp16_0))[name = string("op_7434_cast_fp16")]; tensor var_7435_cast_fp16 = mul(x = var_7434_cast_fp16, y = var_878_cast_fp16)[name = string("op_7435_cast_fp16")]; tensor key_states_205_cast_fp16 = add(x = var_7428_cast_fp16, y = var_7435_cast_fp16)[name = string("key_states_205_cast_fp16")]; tensor expand_dims_240 = const()[name = string("expand_dims_240"), val = tensor([20])]; tensor expand_dims_241 = const()[name = string("expand_dims_241"), val = tensor([0])]; tensor expand_dims_243 = const()[name = string("expand_dims_243"), val = tensor([0])]; int32 concat_245_axis_0 = const()[name = string("concat_245_axis_0"), val = int32(0)]; bool concat_245_interleave_0 = const()[name = string("concat_245_interleave_0"), val = bool(false)]; tensor concat_245 = concat(axis = concat_245_axis_0, interleave = concat_245_interleave_0, values = (expand_dims_240, expand_dims_241, position_id, expand_dims_243))[name = string("concat_245")]; tensor expand_dims_244 = const()[name = string("expand_dims_244"), val = tensor([21])]; tensor concat_246_values1_0 = const()[name = string("concat_246_values1_0"), val = tensor([0])]; tensor concat_246_values3_0 = const()[name = string("concat_246_values3_0"), val = tensor([0])]; int32 concat_246_axis_0 = const()[name = string("concat_246_axis_0"), val = int32(0)]; bool concat_246_interleave_0 = const()[name = string("concat_246_interleave_0"), val = bool(false)]; tensor concat_246 = concat(axis = concat_246_axis_0, interleave = concat_246_interleave_0, values = (expand_dims_244, concat_246_values1_0, cache_position_end, concat_246_values3_0))[name = string("concat_246")]; tensor key_states_207_perm_0 = const()[name = string("key_states_207_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_21_stride_0 = const()[name = string("key_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_207_cast_fp16 = transpose(perm = key_states_207_perm_0, x = key_states_205_cast_fp16)[name = string("transpose_23")]; tensor key_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = key_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = key_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_21_squeeze_mask_0, stride = key_cache_internal_tensor_assign_21_stride_0, update = key_states_207_cast_fp16, x = coreml_update_state_38)[name = string("key_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_21_cast_fp16, input = key_cache)[name = string("coreml_update_state_40_write_state")]; tensor coreml_update_state_40 = read_state(input = key_cache)[name = string("coreml_update_state_40")]; tensor value_states_123_perm_0 = const()[name = string("value_states_123_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_21_stride_0 = const()[name = string("value_cache_internal_tensor_assign_21_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_21_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_21_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_21_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_21_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_123_cast_fp16 = transpose(perm = value_states_123_perm_0, x = var_7411_cast_fp16)[name = string("transpose_22")]; tensor value_cache_internal_tensor_assign_21_cast_fp16 = slice_update(begin = concat_245, begin_mask = value_cache_internal_tensor_assign_21_begin_mask_0, end = concat_246, end_mask = value_cache_internal_tensor_assign_21_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_21_squeeze_mask_0, stride = value_cache_internal_tensor_assign_21_stride_0, update = value_states_123_cast_fp16, x = coreml_update_state_39)[name = string("value_cache_internal_tensor_assign_21_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_21_cast_fp16, input = value_cache)[name = string("coreml_update_state_41_write_state")]; tensor coreml_update_state_41 = read_state(input = value_cache)[name = string("coreml_update_state_41")]; tensor var_7505_begin_0 = const()[name = string("op_7505_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7505_end_0 = const()[name = string("op_7505_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7505_end_mask_0 = const()[name = string("op_7505_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7505_cast_fp16 = slice_by_index(begin = var_7505_begin_0, end = var_7505_end_0, end_mask = var_7505_end_mask_0, x = coreml_update_state_40)[name = string("op_7505_cast_fp16")]; tensor tile_40 = const()[name = string("tile_40"), val = tensor([1, 1])]; int32 var_7508_axis_0 = const()[name = string("op_7508_axis_0"), val = int32(1)]; tensor var_7508_cast_fp16_0, tensor var_7508_cast_fp16_1 = split(axis = var_7508_axis_0, split_sizes = tile_40, x = var_7505_cast_fp16)[name = string("op_7508_cast_fp16")]; tensor var_7515_begin_0 = const()[name = string("op_7515_begin_0"), val = tensor([20, 0, 0, 0])]; tensor var_7515_end_0 = const()[name = string("op_7515_end_0"), val = tensor([21, 2, 2048, 128])]; tensor var_7515_end_mask_0 = const()[name = string("op_7515_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7515_cast_fp16 = slice_by_index(begin = var_7515_begin_0, end = var_7515_end_0, end_mask = var_7515_end_mask_0, x = coreml_update_state_41)[name = string("op_7515_cast_fp16")]; tensor tile_41 = const()[name = string("tile_41"), val = tensor([1, 1])]; int32 var_7518_axis_0 = const()[name = string("op_7518_axis_0"), val = int32(1)]; tensor var_7518_cast_fp16_0, tensor var_7518_cast_fp16_1 = split(axis = var_7518_axis_0, split_sizes = tile_41, x = var_7515_cast_fp16)[name = string("op_7518_cast_fp16")]; tensor var_7521_split_sizes_0 = const()[name = string("op_7521_split_sizes_0"), val = tensor([8, 8])]; int32 var_7521_axis_0 = const()[name = string("op_7521_axis_0"), val = int32(1)]; tensor var_7521_0, tensor var_7521_1 = split(axis = var_7521_axis_0, split_sizes = var_7521_split_sizes_0, x = query_states_123_cast_fp16)[name = string("op_7521")]; bool attn_weights_321_transpose_x_0 = const()[name = string("attn_weights_321_transpose_x_0"), val = bool(false)]; bool attn_weights_321_transpose_y_0 = const()[name = string("attn_weights_321_transpose_y_0"), val = bool(false)]; tensor attn_weights_321_cast_fp16 = matmul(transpose_x = attn_weights_321_transpose_x_0, transpose_y = attn_weights_321_transpose_y_0, x = var_7508_cast_fp16_0, y = var_7521_0)[name = string("attn_weights_321_cast_fp16")]; fp16 var_7524_to_fp16 = const()[name = string("op_7524_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_323_cast_fp16 = mul(x = attn_weights_321_cast_fp16, y = var_7524_to_fp16)[name = string("attn_weights_323_cast_fp16")]; tensor attn_weights_325_cast_fp16 = add(x = attn_weights_323_cast_fp16, y = attn_mask_1)[name = string("attn_weights_325_cast_fp16")]; int32 var_7528 = const()[name = string("op_7528"), val = int32(-2)]; tensor attn_weights_327_cast_fp16 = softmax(axis = var_7528, x = attn_weights_325_cast_fp16)[name = string("attn_weights_327_cast_fp16")]; bool var_7534_transpose_x_1 = const()[name = string("op_7534_transpose_x_1"), val = bool(true)]; bool var_7534_transpose_y_1 = const()[name = string("op_7534_transpose_y_1"), val = bool(false)]; tensor var_7534_cast_fp16 = matmul(transpose_x = var_7534_transpose_x_1, transpose_y = var_7534_transpose_y_1, x = attn_weights_327_cast_fp16, y = var_7518_cast_fp16_0)[name = string("op_7534_cast_fp16")]; bool attn_weights_329_transpose_x_0 = const()[name = string("attn_weights_329_transpose_x_0"), val = bool(false)]; bool attn_weights_329_transpose_y_0 = const()[name = string("attn_weights_329_transpose_y_0"), val = bool(false)]; tensor attn_weights_329_cast_fp16 = matmul(transpose_x = attn_weights_329_transpose_x_0, transpose_y = attn_weights_329_transpose_y_0, x = var_7508_cast_fp16_1, y = var_7521_1)[name = string("attn_weights_329_cast_fp16")]; fp16 var_7536_to_fp16 = const()[name = string("op_7536_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_331_cast_fp16 = mul(x = attn_weights_329_cast_fp16, y = var_7536_to_fp16)[name = string("attn_weights_331_cast_fp16")]; tensor attn_weights_333_cast_fp16 = add(x = attn_weights_331_cast_fp16, y = attn_mask_1)[name = string("attn_weights_333_cast_fp16")]; int32 var_7540 = const()[name = string("op_7540"), val = int32(-2)]; tensor attn_weights_335_cast_fp16 = softmax(axis = var_7540, x = attn_weights_333_cast_fp16)[name = string("attn_weights_335_cast_fp16")]; bool attn_output_161_transpose_x_1 = const()[name = string("attn_output_161_transpose_x_1"), val = bool(true)]; bool attn_output_161_transpose_y_1 = const()[name = string("attn_output_161_transpose_y_1"), val = bool(false)]; tensor attn_output_161_cast_fp16 = matmul(transpose_x = attn_output_161_transpose_x_1, transpose_y = attn_output_161_transpose_y_1, x = attn_weights_335_cast_fp16, y = var_7518_cast_fp16_1)[name = string("attn_output_161_cast_fp16")]; int32 var_7548 = const()[name = string("op_7548"), val = int32(1)]; bool attn_output_163_interleave_0 = const()[name = string("attn_output_163_interleave_0"), val = bool(false)]; tensor attn_output_163_cast_fp16 = concat(axis = var_7548, interleave = attn_output_163_interleave_0, values = (var_7534_cast_fp16, attn_output_161_cast_fp16))[name = string("attn_output_163_cast_fp16")]; tensor var_7552_perm_0 = const()[name = string("op_7552_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_251x = const()[name = string("concat_251x"), val = tensor([1, 2048, 1, -1])]; tensor var_7552_cast_fp16 = transpose(perm = var_7552_perm_0, x = attn_output_163_cast_fp16)[name = string("transpose_21")]; tensor attn_output_167_cast_fp16 = reshape(shape = concat_251x, x = var_7552_cast_fp16)[name = string("attn_output_167_cast_fp16")]; tensor hidden_states_203_strides_0 = const()[name = string("hidden_states_203_strides_0"), val = tensor([1, 1])]; string hidden_states_203_pad_type_0 = const()[name = string("hidden_states_203_pad_type_0"), val = string("valid")]; tensor hidden_states_203_pad_0 = const()[name = string("hidden_states_203_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_203_dilations_0 = const()[name = string("hidden_states_203_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_203_groups_0 = const()[name = string("hidden_states_203_groups_0"), val = int32(1)]; tensor hidden_states_203_cast_fp16 = conv(dilations = hidden_states_203_dilations_0, groups = hidden_states_203_groups_0, pad = hidden_states_203_pad_0, pad_type = hidden_states_203_pad_type_0, strides = hidden_states_203_strides_0, weight = layers_20_self_attn_o_proj_weight_cast_fp16, x = attn_output_167_cast_fp16)[name = string("hidden_states_203_cast_fp16")]; tensor hidden_states_205_cast_fp16 = add(x = hidden_states_199_cast_fp16, y = hidden_states_203_cast_fp16)[name = string("hidden_states_205_cast_fp16")]; fp16 const_208_promoted_to_fp16 = const()[name = string("const_208_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7585_cast_fp16 = mul(x = hidden_states_205_cast_fp16, y = const_208_promoted_to_fp16)[name = string("op_7585_cast_fp16")]; int32 var_7583 = const()[name = string("op_7583"), val = int32(1)]; bool doubled_165_interleave_0 = const()[name = string("doubled_165_interleave_0"), val = bool(false)]; tensor doubled_165_cast_fp16 = concat(axis = var_7583, interleave = doubled_165_interleave_0, values = (hidden_states_205_cast_fp16, var_7585_cast_fp16))[name = string("doubled_165_cast_fp16")]; tensor out_83_axes_0 = const()[name = string("out_83_axes_0"), val = tensor([1])]; tensor out_83_gamma_0_to_fp16 = const()[name = string("out_83_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502547264)))]; fp16 var_7595_to_fp16 = const()[name = string("op_7595_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_83_cast_fp16 = layer_norm(axes = out_83_axes_0, epsilon = var_7595_to_fp16, gamma = out_83_gamma_0_to_fp16, x = doubled_165_cast_fp16)[name = string("out_83_cast_fp16")]; tensor var_7606_split_sizes_0 = const()[name = string("op_7606_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7606_axis_0 = const()[name = string("op_7606_axis_0"), val = int32(1)]; tensor var_7606_cast_fp16_0, tensor var_7606_cast_fp16_1 = split(axis = var_7606_axis_0, split_sizes = var_7606_split_sizes_0, x = out_83_cast_fp16)[name = string("op_7606_cast_fp16")]; tensor input_41_strides_0 = const()[name = string("input_41_strides_0"), val = tensor([1, 1])]; string input_41_pad_type_0 = const()[name = string("input_41_pad_type_0"), val = string("valid")]; tensor input_41_pad_0 = const()[name = string("input_41_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_41_dilations_0 = const()[name = string("input_41_dilations_0"), val = tensor([1, 1])]; int32 input_41_groups_0 = const()[name = string("input_41_groups_0"), val = int32(1)]; tensor input_41_cast_fp16 = conv(dilations = input_41_dilations_0, groups = input_41_groups_0, pad = input_41_pad_0, pad_type = input_41_pad_type_0, strides = input_41_strides_0, weight = layers_20_mlp_gate_proj_weight_cast_fp16, x = var_7606_cast_fp16_0)[name = string("input_41_cast_fp16")]; tensor var_7623_cast_fp16 = silu(x = input_41_cast_fp16)[name = string("op_7623_cast_fp16")]; tensor layers_20_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_20_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1502555520)))]; tensor var_7629_strides_0 = const()[name = string("op_7629_strides_0"), val = tensor([1, 1])]; string var_7629_pad_type_0 = const()[name = string("op_7629_pad_type_0"), val = string("valid")]; tensor var_7629_pad_0 = const()[name = string("op_7629_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7629_dilations_0 = const()[name = string("op_7629_dilations_0"), val = tensor([1, 1])]; int32 var_7629_groups_0 = const()[name = string("op_7629_groups_0"), val = int32(1)]; tensor var_7629_cast_fp16 = conv(dilations = var_7629_dilations_0, groups = var_7629_groups_0, pad = var_7629_pad_0, pad_type = var_7629_pad_type_0, strides = var_7629_strides_0, weight = layers_20_mlp_up_proj_weight_to_fp16, x = var_7606_cast_fp16_0)[name = string("op_7629_cast_fp16")]; tensor x_209_cast_fp16 = mul(x = var_7623_cast_fp16, y = var_7629_cast_fp16)[name = string("x_209_cast_fp16")]; tensor hidden_states_207_strides_0 = const()[name = string("hidden_states_207_strides_0"), val = tensor([1, 1])]; string hidden_states_207_pad_type_0 = const()[name = string("hidden_states_207_pad_type_0"), val = string("valid")]; tensor hidden_states_207_pad_0 = const()[name = string("hidden_states_207_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_207_dilations_0 = const()[name = string("hidden_states_207_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_207_groups_0 = const()[name = string("hidden_states_207_groups_0"), val = int32(1)]; tensor hidden_states_207_cast_fp16 = conv(dilations = hidden_states_207_dilations_0, groups = hidden_states_207_groups_0, pad = hidden_states_207_pad_0, pad_type = hidden_states_207_pad_type_0, strides = hidden_states_207_strides_0, weight = layers_20_mlp_down_proj_weight_cast_fp16, x = x_209_cast_fp16)[name = string("hidden_states_207_cast_fp16")]; tensor hidden_states_209_cast_fp16 = add(x = hidden_states_205_cast_fp16, y = hidden_states_207_cast_fp16)[name = string("hidden_states_209_cast_fp16")]; fp16 const_210_promoted_to_fp16 = const()[name = string("const_210_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7647_cast_fp16 = mul(x = hidden_states_209_cast_fp16, y = const_210_promoted_to_fp16)[name = string("op_7647_cast_fp16")]; int32 var_7645 = const()[name = string("op_7645"), val = int32(1)]; bool doubled_169_interleave_0 = const()[name = string("doubled_169_interleave_0"), val = bool(false)]; tensor doubled_169_cast_fp16 = concat(axis = var_7645, interleave = doubled_169_interleave_0, values = (hidden_states_209_cast_fp16, var_7647_cast_fp16))[name = string("doubled_169_cast_fp16")]; tensor out_85_axes_0 = const()[name = string("out_85_axes_0"), val = tensor([1])]; tensor out_85_gamma_0_to_fp16 = const()[name = string("out_85_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527721408)))]; fp16 var_7657_to_fp16 = const()[name = string("op_7657_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_85_cast_fp16 = layer_norm(axes = out_85_axes_0, epsilon = var_7657_to_fp16, gamma = out_85_gamma_0_to_fp16, x = doubled_169_cast_fp16)[name = string("out_85_cast_fp16")]; tensor var_7668_split_sizes_0 = const()[name = string("op_7668_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7668_axis_0 = const()[name = string("op_7668_axis_0"), val = int32(1)]; tensor var_7668_cast_fp16_0, tensor var_7668_cast_fp16_1 = split(axis = var_7668_axis_0, split_sizes = var_7668_split_sizes_0, x = out_85_cast_fp16)[name = string("op_7668_cast_fp16")]; tensor query_states_127_strides_0 = const()[name = string("query_states_127_strides_0"), val = tensor([1, 1])]; string query_states_127_pad_type_0 = const()[name = string("query_states_127_pad_type_0"), val = string("valid")]; tensor query_states_127_pad_0 = const()[name = string("query_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_127_dilations_0 = const()[name = string("query_states_127_dilations_0"), val = tensor([1, 1])]; int32 query_states_127_groups_0 = const()[name = string("query_states_127_groups_0"), val = int32(1)]; tensor query_states_127_cast_fp16 = conv(dilations = query_states_127_dilations_0, groups = query_states_127_groups_0, pad = query_states_127_pad_0, pad_type = query_states_127_pad_type_0, strides = query_states_127_strides_0, weight = layers_21_self_attn_q_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("query_states_127_cast_fp16")]; tensor key_states_211_strides_0 = const()[name = string("key_states_211_strides_0"), val = tensor([1, 1])]; string key_states_211_pad_type_0 = const()[name = string("key_states_211_pad_type_0"), val = string("valid")]; tensor key_states_211_pad_0 = const()[name = string("key_states_211_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_211_dilations_0 = const()[name = string("key_states_211_dilations_0"), val = tensor([1, 1])]; int32 key_states_211_groups_0 = const()[name = string("key_states_211_groups_0"), val = int32(1)]; tensor key_states_211_cast_fp16 = conv(dilations = key_states_211_dilations_0, groups = key_states_211_groups_0, pad = key_states_211_pad_0, pad_type = key_states_211_pad_type_0, strides = key_states_211_strides_0, weight = layers_21_self_attn_k_proj_weight_cast_fp16, x = var_7668_cast_fp16_0)[name = string("key_states_211_cast_fp16")]; tensor layers_21_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_21_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1527729664)))]; tensor value_states_127_strides_0 = const()[name = string("value_states_127_strides_0"), val = tensor([1, 1])]; string value_states_127_pad_type_0 = const()[name = string("value_states_127_pad_type_0"), val = string("valid")]; tensor value_states_127_pad_0 = const()[name = string("value_states_127_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_127_dilations_0 = const()[name = string("value_states_127_dilations_0"), val = tensor([1, 1])]; int32 value_states_127_groups_0 = const()[name = string("value_states_127_groups_0"), val = int32(1)]; tensor value_states_127_cast_fp16 = conv(dilations = value_states_127_dilations_0, groups = value_states_127_groups_0, pad = value_states_127_pad_0, pad_type = value_states_127_pad_type_0, strides = value_states_127_strides_0, weight = layers_21_self_attn_v_proj_weight_to_fp16, x = var_7668_cast_fp16_0)[name = string("value_states_127_cast_fp16")]; tensor concat_252x = const()[name = string("concat_252x"), val = tensor([1, 16, 128, -1])]; tensor x_211_cast_fp16 = reshape(shape = concat_252x, x = query_states_127_cast_fp16)[name = string("x_211_cast_fp16")]; tensor concat_253x = const()[name = string("concat_253x"), val = tensor([1, 2, 128, -1])]; tensor var_7725_cast_fp16 = reshape(shape = concat_253x, x = key_states_211_cast_fp16)[name = string("op_7725_cast_fp16")]; tensor concat_254x = const()[name = string("concat_254x"), val = tensor([1, 2, 128, -1])]; tensor var_7732_cast_fp16 = reshape(shape = concat_254x, x = value_states_127_cast_fp16)[name = string("op_7732_cast_fp16")]; tensor var_7736_cast_fp16 = mul(x = x_211_cast_fp16, y = var_869_cast_fp16)[name = string("op_7736_cast_fp16")]; tensor var_7737_split_sizes_0 = const()[name = string("op_7737_split_sizes_0"), val = tensor([64, 64])]; int32 var_7737_axis_0 = const()[name = string("op_7737_axis_0"), val = int32(-2)]; tensor var_7737_cast_fp16_0, tensor var_7737_cast_fp16_1 = split(axis = var_7737_axis_0, split_sizes = var_7737_split_sizes_0, x = x_211_cast_fp16)[name = string("op_7737_cast_fp16")]; fp16 const_212_promoted_to_fp16 = const()[name = string("const_212_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7739_cast_fp16 = mul(x = var_7737_cast_fp16_1, y = const_212_promoted_to_fp16)[name = string("op_7739_cast_fp16")]; int32 var_7741 = const()[name = string("op_7741"), val = int32(-2)]; bool var_7742_interleave_0 = const()[name = string("op_7742_interleave_0"), val = bool(false)]; tensor var_7742_cast_fp16 = concat(axis = var_7741, interleave = var_7742_interleave_0, values = (var_7739_cast_fp16, var_7737_cast_fp16_0))[name = string("op_7742_cast_fp16")]; tensor var_7743_cast_fp16 = mul(x = var_7742_cast_fp16, y = var_878_cast_fp16)[name = string("op_7743_cast_fp16")]; tensor query_states_129_cast_fp16 = add(x = var_7736_cast_fp16, y = var_7743_cast_fp16)[name = string("query_states_129_cast_fp16")]; tensor var_7749_cast_fp16 = mul(x = var_7725_cast_fp16, y = var_869_cast_fp16)[name = string("op_7749_cast_fp16")]; tensor var_7750_split_sizes_0 = const()[name = string("op_7750_split_sizes_0"), val = tensor([64, 64])]; int32 var_7750_axis_0 = const()[name = string("op_7750_axis_0"), val = int32(-2)]; tensor var_7750_cast_fp16_0, tensor var_7750_cast_fp16_1 = split(axis = var_7750_axis_0, split_sizes = var_7750_split_sizes_0, x = var_7725_cast_fp16)[name = string("op_7750_cast_fp16")]; fp16 const_213_promoted_to_fp16 = const()[name = string("const_213_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7752_cast_fp16 = mul(x = var_7750_cast_fp16_1, y = const_213_promoted_to_fp16)[name = string("op_7752_cast_fp16")]; int32 var_7754 = const()[name = string("op_7754"), val = int32(-2)]; bool var_7755_interleave_0 = const()[name = string("op_7755_interleave_0"), val = bool(false)]; tensor var_7755_cast_fp16 = concat(axis = var_7754, interleave = var_7755_interleave_0, values = (var_7752_cast_fp16, var_7750_cast_fp16_0))[name = string("op_7755_cast_fp16")]; tensor var_7756_cast_fp16 = mul(x = var_7755_cast_fp16, y = var_878_cast_fp16)[name = string("op_7756_cast_fp16")]; tensor key_states_215_cast_fp16 = add(x = var_7749_cast_fp16, y = var_7756_cast_fp16)[name = string("key_states_215_cast_fp16")]; tensor expand_dims_252 = const()[name = string("expand_dims_252"), val = tensor([21])]; tensor expand_dims_253 = const()[name = string("expand_dims_253"), val = tensor([0])]; tensor expand_dims_255 = const()[name = string("expand_dims_255"), val = tensor([0])]; int32 concat_257_axis_0 = const()[name = string("concat_257_axis_0"), val = int32(0)]; bool concat_257_interleave_0 = const()[name = string("concat_257_interleave_0"), val = bool(false)]; tensor concat_257 = concat(axis = concat_257_axis_0, interleave = concat_257_interleave_0, values = (expand_dims_252, expand_dims_253, position_id, expand_dims_255))[name = string("concat_257")]; tensor expand_dims_256 = const()[name = string("expand_dims_256"), val = tensor([22])]; tensor concat_258_values1_0 = const()[name = string("concat_258_values1_0"), val = tensor([0])]; tensor concat_258_values3_0 = const()[name = string("concat_258_values3_0"), val = tensor([0])]; int32 concat_258_axis_0 = const()[name = string("concat_258_axis_0"), val = int32(0)]; bool concat_258_interleave_0 = const()[name = string("concat_258_interleave_0"), val = bool(false)]; tensor concat_258 = concat(axis = concat_258_axis_0, interleave = concat_258_interleave_0, values = (expand_dims_256, concat_258_values1_0, cache_position_end, concat_258_values3_0))[name = string("concat_258")]; tensor key_states_217_perm_0 = const()[name = string("key_states_217_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_22_stride_0 = const()[name = string("key_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_217_cast_fp16 = transpose(perm = key_states_217_perm_0, x = key_states_215_cast_fp16)[name = string("transpose_20")]; tensor key_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = key_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = key_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_22_squeeze_mask_0, stride = key_cache_internal_tensor_assign_22_stride_0, update = key_states_217_cast_fp16, x = coreml_update_state_40)[name = string("key_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_22_cast_fp16, input = key_cache)[name = string("coreml_update_state_42_write_state")]; tensor coreml_update_state_42 = read_state(input = key_cache)[name = string("coreml_update_state_42")]; tensor value_states_129_perm_0 = const()[name = string("value_states_129_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_22_stride_0 = const()[name = string("value_cache_internal_tensor_assign_22_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_22_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_22_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_22_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_22_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_129_cast_fp16 = transpose(perm = value_states_129_perm_0, x = var_7732_cast_fp16)[name = string("transpose_19")]; tensor value_cache_internal_tensor_assign_22_cast_fp16 = slice_update(begin = concat_257, begin_mask = value_cache_internal_tensor_assign_22_begin_mask_0, end = concat_258, end_mask = value_cache_internal_tensor_assign_22_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_22_squeeze_mask_0, stride = value_cache_internal_tensor_assign_22_stride_0, update = value_states_129_cast_fp16, x = coreml_update_state_41)[name = string("value_cache_internal_tensor_assign_22_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_22_cast_fp16, input = value_cache)[name = string("coreml_update_state_43_write_state")]; tensor coreml_update_state_43 = read_state(input = value_cache)[name = string("coreml_update_state_43")]; tensor var_7826_begin_0 = const()[name = string("op_7826_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7826_end_0 = const()[name = string("op_7826_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7826_end_mask_0 = const()[name = string("op_7826_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7826_cast_fp16 = slice_by_index(begin = var_7826_begin_0, end = var_7826_end_0, end_mask = var_7826_end_mask_0, x = coreml_update_state_42)[name = string("op_7826_cast_fp16")]; tensor tile_42 = const()[name = string("tile_42"), val = tensor([1, 1])]; int32 var_7829_axis_0 = const()[name = string("op_7829_axis_0"), val = int32(1)]; tensor var_7829_cast_fp16_0, tensor var_7829_cast_fp16_1 = split(axis = var_7829_axis_0, split_sizes = tile_42, x = var_7826_cast_fp16)[name = string("op_7829_cast_fp16")]; tensor var_7836_begin_0 = const()[name = string("op_7836_begin_0"), val = tensor([21, 0, 0, 0])]; tensor var_7836_end_0 = const()[name = string("op_7836_end_0"), val = tensor([22, 2, 2048, 128])]; tensor var_7836_end_mask_0 = const()[name = string("op_7836_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_7836_cast_fp16 = slice_by_index(begin = var_7836_begin_0, end = var_7836_end_0, end_mask = var_7836_end_mask_0, x = coreml_update_state_43)[name = string("op_7836_cast_fp16")]; tensor tile_43 = const()[name = string("tile_43"), val = tensor([1, 1])]; int32 var_7839_axis_0 = const()[name = string("op_7839_axis_0"), val = int32(1)]; tensor var_7839_cast_fp16_0, tensor var_7839_cast_fp16_1 = split(axis = var_7839_axis_0, split_sizes = tile_43, x = var_7836_cast_fp16)[name = string("op_7839_cast_fp16")]; tensor var_7842_split_sizes_0 = const()[name = string("op_7842_split_sizes_0"), val = tensor([8, 8])]; int32 var_7842_axis_0 = const()[name = string("op_7842_axis_0"), val = int32(1)]; tensor var_7842_0, tensor var_7842_1 = split(axis = var_7842_axis_0, split_sizes = var_7842_split_sizes_0, x = query_states_129_cast_fp16)[name = string("op_7842")]; bool attn_weights_337_transpose_x_0 = const()[name = string("attn_weights_337_transpose_x_0"), val = bool(false)]; bool attn_weights_337_transpose_y_0 = const()[name = string("attn_weights_337_transpose_y_0"), val = bool(false)]; tensor attn_weights_337_cast_fp16 = matmul(transpose_x = attn_weights_337_transpose_x_0, transpose_y = attn_weights_337_transpose_y_0, x = var_7829_cast_fp16_0, y = var_7842_0)[name = string("attn_weights_337_cast_fp16")]; fp16 var_7845_to_fp16 = const()[name = string("op_7845_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_339_cast_fp16 = mul(x = attn_weights_337_cast_fp16, y = var_7845_to_fp16)[name = string("attn_weights_339_cast_fp16")]; tensor attn_weights_341_cast_fp16 = add(x = attn_weights_339_cast_fp16, y = attn_mask_1)[name = string("attn_weights_341_cast_fp16")]; int32 var_7849 = const()[name = string("op_7849"), val = int32(-2)]; tensor attn_weights_343_cast_fp16 = softmax(axis = var_7849, x = attn_weights_341_cast_fp16)[name = string("attn_weights_343_cast_fp16")]; bool var_7855_transpose_x_1 = const()[name = string("op_7855_transpose_x_1"), val = bool(true)]; bool var_7855_transpose_y_1 = const()[name = string("op_7855_transpose_y_1"), val = bool(false)]; tensor var_7855_cast_fp16 = matmul(transpose_x = var_7855_transpose_x_1, transpose_y = var_7855_transpose_y_1, x = attn_weights_343_cast_fp16, y = var_7839_cast_fp16_0)[name = string("op_7855_cast_fp16")]; bool attn_weights_345_transpose_x_0 = const()[name = string("attn_weights_345_transpose_x_0"), val = bool(false)]; bool attn_weights_345_transpose_y_0 = const()[name = string("attn_weights_345_transpose_y_0"), val = bool(false)]; tensor attn_weights_345_cast_fp16 = matmul(transpose_x = attn_weights_345_transpose_x_0, transpose_y = attn_weights_345_transpose_y_0, x = var_7829_cast_fp16_1, y = var_7842_1)[name = string("attn_weights_345_cast_fp16")]; fp16 var_7857_to_fp16 = const()[name = string("op_7857_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_347_cast_fp16 = mul(x = attn_weights_345_cast_fp16, y = var_7857_to_fp16)[name = string("attn_weights_347_cast_fp16")]; tensor attn_weights_349_cast_fp16 = add(x = attn_weights_347_cast_fp16, y = attn_mask_1)[name = string("attn_weights_349_cast_fp16")]; int32 var_7861 = const()[name = string("op_7861"), val = int32(-2)]; tensor attn_weights_351_cast_fp16 = softmax(axis = var_7861, x = attn_weights_349_cast_fp16)[name = string("attn_weights_351_cast_fp16")]; bool attn_output_169_transpose_x_1 = const()[name = string("attn_output_169_transpose_x_1"), val = bool(true)]; bool attn_output_169_transpose_y_1 = const()[name = string("attn_output_169_transpose_y_1"), val = bool(false)]; tensor attn_output_169_cast_fp16 = matmul(transpose_x = attn_output_169_transpose_x_1, transpose_y = attn_output_169_transpose_y_1, x = attn_weights_351_cast_fp16, y = var_7839_cast_fp16_1)[name = string("attn_output_169_cast_fp16")]; int32 var_7869 = const()[name = string("op_7869"), val = int32(1)]; bool attn_output_171_interleave_0 = const()[name = string("attn_output_171_interleave_0"), val = bool(false)]; tensor attn_output_171_cast_fp16 = concat(axis = var_7869, interleave = attn_output_171_interleave_0, values = (var_7855_cast_fp16, attn_output_169_cast_fp16))[name = string("attn_output_171_cast_fp16")]; tensor var_7873_perm_0 = const()[name = string("op_7873_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_263x = const()[name = string("concat_263x"), val = tensor([1, 2048, 1, -1])]; tensor var_7873_cast_fp16 = transpose(perm = var_7873_perm_0, x = attn_output_171_cast_fp16)[name = string("transpose_18")]; tensor attn_output_175_cast_fp16 = reshape(shape = concat_263x, x = var_7873_cast_fp16)[name = string("attn_output_175_cast_fp16")]; tensor hidden_states_213_strides_0 = const()[name = string("hidden_states_213_strides_0"), val = tensor([1, 1])]; string hidden_states_213_pad_type_0 = const()[name = string("hidden_states_213_pad_type_0"), val = string("valid")]; tensor hidden_states_213_pad_0 = const()[name = string("hidden_states_213_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_213_dilations_0 = const()[name = string("hidden_states_213_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_213_groups_0 = const()[name = string("hidden_states_213_groups_0"), val = int32(1)]; tensor hidden_states_213_cast_fp16 = conv(dilations = hidden_states_213_dilations_0, groups = hidden_states_213_groups_0, pad = hidden_states_213_pad_0, pad_type = hidden_states_213_pad_type_0, strides = hidden_states_213_strides_0, weight = layers_21_self_attn_o_proj_weight_cast_fp16, x = attn_output_175_cast_fp16)[name = string("hidden_states_213_cast_fp16")]; tensor hidden_states_215_cast_fp16 = add(x = hidden_states_209_cast_fp16, y = hidden_states_213_cast_fp16)[name = string("hidden_states_215_cast_fp16")]; fp16 const_218_promoted_to_fp16 = const()[name = string("const_218_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7906_cast_fp16 = mul(x = hidden_states_215_cast_fp16, y = const_218_promoted_to_fp16)[name = string("op_7906_cast_fp16")]; int32 var_7904 = const()[name = string("op_7904"), val = int32(1)]; bool doubled_173_interleave_0 = const()[name = string("doubled_173_interleave_0"), val = bool(false)]; tensor doubled_173_cast_fp16 = concat(axis = var_7904, interleave = doubled_173_interleave_0, values = (hidden_states_215_cast_fp16, var_7906_cast_fp16))[name = string("doubled_173_cast_fp16")]; tensor out_87_axes_0 = const()[name = string("out_87_axes_0"), val = tensor([1])]; tensor out_87_gamma_0_to_fp16 = const()[name = string("out_87_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528778304)))]; fp16 var_7916_to_fp16 = const()[name = string("op_7916_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_87_cast_fp16 = layer_norm(axes = out_87_axes_0, epsilon = var_7916_to_fp16, gamma = out_87_gamma_0_to_fp16, x = doubled_173_cast_fp16)[name = string("out_87_cast_fp16")]; tensor var_7927_split_sizes_0 = const()[name = string("op_7927_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7927_axis_0 = const()[name = string("op_7927_axis_0"), val = int32(1)]; tensor var_7927_cast_fp16_0, tensor var_7927_cast_fp16_1 = split(axis = var_7927_axis_0, split_sizes = var_7927_split_sizes_0, x = out_87_cast_fp16)[name = string("op_7927_cast_fp16")]; tensor input_43_strides_0 = const()[name = string("input_43_strides_0"), val = tensor([1, 1])]; string input_43_pad_type_0 = const()[name = string("input_43_pad_type_0"), val = string("valid")]; tensor input_43_pad_0 = const()[name = string("input_43_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_43_dilations_0 = const()[name = string("input_43_dilations_0"), val = tensor([1, 1])]; int32 input_43_groups_0 = const()[name = string("input_43_groups_0"), val = int32(1)]; tensor input_43_cast_fp16 = conv(dilations = input_43_dilations_0, groups = input_43_groups_0, pad = input_43_pad_0, pad_type = input_43_pad_type_0, strides = input_43_strides_0, weight = layers_21_mlp_gate_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("input_43_cast_fp16")]; tensor var_7944_cast_fp16 = silu(x = input_43_cast_fp16)[name = string("op_7944_cast_fp16")]; tensor var_7950_strides_0 = const()[name = string("op_7950_strides_0"), val = tensor([1, 1])]; string var_7950_pad_type_0 = const()[name = string("op_7950_pad_type_0"), val = string("valid")]; tensor var_7950_pad_0 = const()[name = string("op_7950_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_7950_dilations_0 = const()[name = string("op_7950_dilations_0"), val = tensor([1, 1])]; int32 var_7950_groups_0 = const()[name = string("op_7950_groups_0"), val = int32(1)]; tensor var_7950_cast_fp16 = conv(dilations = var_7950_dilations_0, groups = var_7950_groups_0, pad = var_7950_pad_0, pad_type = var_7950_pad_type_0, strides = var_7950_strides_0, weight = layers_21_mlp_up_proj_weight_cast_fp16, x = var_7927_cast_fp16_0)[name = string("op_7950_cast_fp16")]; tensor x_219_cast_fp16 = mul(x = var_7944_cast_fp16, y = var_7950_cast_fp16)[name = string("x_219_cast_fp16")]; tensor hidden_states_217_strides_0 = const()[name = string("hidden_states_217_strides_0"), val = tensor([1, 1])]; string hidden_states_217_pad_type_0 = const()[name = string("hidden_states_217_pad_type_0"), val = string("valid")]; tensor hidden_states_217_pad_0 = const()[name = string("hidden_states_217_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_217_dilations_0 = const()[name = string("hidden_states_217_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_217_groups_0 = const()[name = string("hidden_states_217_groups_0"), val = int32(1)]; tensor hidden_states_217_cast_fp16 = conv(dilations = hidden_states_217_dilations_0, groups = hidden_states_217_groups_0, pad = hidden_states_217_pad_0, pad_type = hidden_states_217_pad_type_0, strides = hidden_states_217_strides_0, weight = layers_21_mlp_down_proj_weight_cast_fp16, x = x_219_cast_fp16)[name = string("hidden_states_217_cast_fp16")]; tensor hidden_states_219_cast_fp16 = add(x = hidden_states_215_cast_fp16, y = hidden_states_217_cast_fp16)[name = string("hidden_states_219_cast_fp16")]; fp16 const_220_promoted_to_fp16 = const()[name = string("const_220_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_7968_cast_fp16 = mul(x = hidden_states_219_cast_fp16, y = const_220_promoted_to_fp16)[name = string("op_7968_cast_fp16")]; int32 var_7966 = const()[name = string("op_7966"), val = int32(1)]; bool doubled_177_interleave_0 = const()[name = string("doubled_177_interleave_0"), val = bool(false)]; tensor doubled_177_cast_fp16 = concat(axis = var_7966, interleave = doubled_177_interleave_0, values = (hidden_states_219_cast_fp16, var_7968_cast_fp16))[name = string("doubled_177_cast_fp16")]; tensor out_89_axes_0 = const()[name = string("out_89_axes_0"), val = tensor([1])]; tensor out_89_gamma_0_to_fp16 = const()[name = string("out_89_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528786560)))]; fp16 var_7978_to_fp16 = const()[name = string("op_7978_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_89_cast_fp16 = layer_norm(axes = out_89_axes_0, epsilon = var_7978_to_fp16, gamma = out_89_gamma_0_to_fp16, x = doubled_177_cast_fp16)[name = string("out_89_cast_fp16")]; tensor var_7989_split_sizes_0 = const()[name = string("op_7989_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_7989_axis_0 = const()[name = string("op_7989_axis_0"), val = int32(1)]; tensor var_7989_cast_fp16_0, tensor var_7989_cast_fp16_1 = split(axis = var_7989_axis_0, split_sizes = var_7989_split_sizes_0, x = out_89_cast_fp16)[name = string("op_7989_cast_fp16")]; tensor query_states_133_strides_0 = const()[name = string("query_states_133_strides_0"), val = tensor([1, 1])]; string query_states_133_pad_type_0 = const()[name = string("query_states_133_pad_type_0"), val = string("valid")]; tensor query_states_133_pad_0 = const()[name = string("query_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_133_dilations_0 = const()[name = string("query_states_133_dilations_0"), val = tensor([1, 1])]; int32 query_states_133_groups_0 = const()[name = string("query_states_133_groups_0"), val = int32(1)]; tensor query_states_133_cast_fp16 = conv(dilations = query_states_133_dilations_0, groups = query_states_133_groups_0, pad = query_states_133_pad_0, pad_type = query_states_133_pad_type_0, strides = query_states_133_strides_0, weight = layers_22_self_attn_q_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("query_states_133_cast_fp16")]; tensor key_states_221_strides_0 = const()[name = string("key_states_221_strides_0"), val = tensor([1, 1])]; string key_states_221_pad_type_0 = const()[name = string("key_states_221_pad_type_0"), val = string("valid")]; tensor key_states_221_pad_0 = const()[name = string("key_states_221_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_221_dilations_0 = const()[name = string("key_states_221_dilations_0"), val = tensor([1, 1])]; int32 key_states_221_groups_0 = const()[name = string("key_states_221_groups_0"), val = int32(1)]; tensor key_states_221_cast_fp16 = conv(dilations = key_states_221_dilations_0, groups = key_states_221_groups_0, pad = key_states_221_pad_0, pad_type = key_states_221_pad_type_0, strides = key_states_221_strides_0, weight = layers_22_self_attn_k_proj_weight_cast_fp16, x = var_7989_cast_fp16_0)[name = string("key_states_221_cast_fp16")]; tensor layers_22_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1528794816)))]; tensor value_states_133_strides_0 = const()[name = string("value_states_133_strides_0"), val = tensor([1, 1])]; string value_states_133_pad_type_0 = const()[name = string("value_states_133_pad_type_0"), val = string("valid")]; tensor value_states_133_pad_0 = const()[name = string("value_states_133_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_133_dilations_0 = const()[name = string("value_states_133_dilations_0"), val = tensor([1, 1])]; int32 value_states_133_groups_0 = const()[name = string("value_states_133_groups_0"), val = int32(1)]; tensor value_states_133_cast_fp16 = conv(dilations = value_states_133_dilations_0, groups = value_states_133_groups_0, pad = value_states_133_pad_0, pad_type = value_states_133_pad_type_0, strides = value_states_133_strides_0, weight = layers_22_self_attn_v_proj_weight_to_fp16, x = var_7989_cast_fp16_0)[name = string("value_states_133_cast_fp16")]; tensor concat_264x = const()[name = string("concat_264x"), val = tensor([1, 16, 128, -1])]; tensor x_221_cast_fp16 = reshape(shape = concat_264x, x = query_states_133_cast_fp16)[name = string("x_221_cast_fp16")]; tensor concat_265x = const()[name = string("concat_265x"), val = tensor([1, 2, 128, -1])]; tensor var_8046_cast_fp16 = reshape(shape = concat_265x, x = key_states_221_cast_fp16)[name = string("op_8046_cast_fp16")]; tensor concat_266x = const()[name = string("concat_266x"), val = tensor([1, 2, 128, -1])]; tensor var_8053_cast_fp16 = reshape(shape = concat_266x, x = value_states_133_cast_fp16)[name = string("op_8053_cast_fp16")]; tensor var_8057_cast_fp16 = mul(x = x_221_cast_fp16, y = var_869_cast_fp16)[name = string("op_8057_cast_fp16")]; tensor var_8058_split_sizes_0 = const()[name = string("op_8058_split_sizes_0"), val = tensor([64, 64])]; int32 var_8058_axis_0 = const()[name = string("op_8058_axis_0"), val = int32(-2)]; tensor var_8058_cast_fp16_0, tensor var_8058_cast_fp16_1 = split(axis = var_8058_axis_0, split_sizes = var_8058_split_sizes_0, x = x_221_cast_fp16)[name = string("op_8058_cast_fp16")]; fp16 const_222_promoted_to_fp16 = const()[name = string("const_222_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8060_cast_fp16 = mul(x = var_8058_cast_fp16_1, y = const_222_promoted_to_fp16)[name = string("op_8060_cast_fp16")]; int32 var_8062 = const()[name = string("op_8062"), val = int32(-2)]; bool var_8063_interleave_0 = const()[name = string("op_8063_interleave_0"), val = bool(false)]; tensor var_8063_cast_fp16 = concat(axis = var_8062, interleave = var_8063_interleave_0, values = (var_8060_cast_fp16, var_8058_cast_fp16_0))[name = string("op_8063_cast_fp16")]; tensor var_8064_cast_fp16 = mul(x = var_8063_cast_fp16, y = var_878_cast_fp16)[name = string("op_8064_cast_fp16")]; tensor query_states_135_cast_fp16 = add(x = var_8057_cast_fp16, y = var_8064_cast_fp16)[name = string("query_states_135_cast_fp16")]; tensor var_8070_cast_fp16 = mul(x = var_8046_cast_fp16, y = var_869_cast_fp16)[name = string("op_8070_cast_fp16")]; tensor var_8071_split_sizes_0 = const()[name = string("op_8071_split_sizes_0"), val = tensor([64, 64])]; int32 var_8071_axis_0 = const()[name = string("op_8071_axis_0"), val = int32(-2)]; tensor var_8071_cast_fp16_0, tensor var_8071_cast_fp16_1 = split(axis = var_8071_axis_0, split_sizes = var_8071_split_sizes_0, x = var_8046_cast_fp16)[name = string("op_8071_cast_fp16")]; fp16 const_223_promoted_to_fp16 = const()[name = string("const_223_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8073_cast_fp16 = mul(x = var_8071_cast_fp16_1, y = const_223_promoted_to_fp16)[name = string("op_8073_cast_fp16")]; int32 var_8075 = const()[name = string("op_8075"), val = int32(-2)]; bool var_8076_interleave_0 = const()[name = string("op_8076_interleave_0"), val = bool(false)]; tensor var_8076_cast_fp16 = concat(axis = var_8075, interleave = var_8076_interleave_0, values = (var_8073_cast_fp16, var_8071_cast_fp16_0))[name = string("op_8076_cast_fp16")]; tensor var_8077_cast_fp16 = mul(x = var_8076_cast_fp16, y = var_878_cast_fp16)[name = string("op_8077_cast_fp16")]; tensor key_states_225_cast_fp16 = add(x = var_8070_cast_fp16, y = var_8077_cast_fp16)[name = string("key_states_225_cast_fp16")]; tensor expand_dims_264 = const()[name = string("expand_dims_264"), val = tensor([22])]; tensor expand_dims_265 = const()[name = string("expand_dims_265"), val = tensor([0])]; tensor expand_dims_267 = const()[name = string("expand_dims_267"), val = tensor([0])]; int32 concat_269_axis_0 = const()[name = string("concat_269_axis_0"), val = int32(0)]; bool concat_269_interleave_0 = const()[name = string("concat_269_interleave_0"), val = bool(false)]; tensor concat_269 = concat(axis = concat_269_axis_0, interleave = concat_269_interleave_0, values = (expand_dims_264, expand_dims_265, position_id, expand_dims_267))[name = string("concat_269")]; tensor expand_dims_268 = const()[name = string("expand_dims_268"), val = tensor([23])]; tensor concat_270_values1_0 = const()[name = string("concat_270_values1_0"), val = tensor([0])]; tensor concat_270_values3_0 = const()[name = string("concat_270_values3_0"), val = tensor([0])]; int32 concat_270_axis_0 = const()[name = string("concat_270_axis_0"), val = int32(0)]; bool concat_270_interleave_0 = const()[name = string("concat_270_interleave_0"), val = bool(false)]; tensor concat_270 = concat(axis = concat_270_axis_0, interleave = concat_270_interleave_0, values = (expand_dims_268, concat_270_values1_0, cache_position_end, concat_270_values3_0))[name = string("concat_270")]; tensor key_states_227_perm_0 = const()[name = string("key_states_227_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_23_stride_0 = const()[name = string("key_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_227_cast_fp16 = transpose(perm = key_states_227_perm_0, x = key_states_225_cast_fp16)[name = string("transpose_17")]; tensor key_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = key_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = key_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_23_squeeze_mask_0, stride = key_cache_internal_tensor_assign_23_stride_0, update = key_states_227_cast_fp16, x = coreml_update_state_42)[name = string("key_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_23_cast_fp16, input = key_cache)[name = string("coreml_update_state_44_write_state")]; tensor coreml_update_state_44 = read_state(input = key_cache)[name = string("coreml_update_state_44")]; tensor value_states_135_perm_0 = const()[name = string("value_states_135_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_23_stride_0 = const()[name = string("value_cache_internal_tensor_assign_23_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_23_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_23_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_23_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_23_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_135_cast_fp16 = transpose(perm = value_states_135_perm_0, x = var_8053_cast_fp16)[name = string("transpose_16")]; tensor value_cache_internal_tensor_assign_23_cast_fp16 = slice_update(begin = concat_269, begin_mask = value_cache_internal_tensor_assign_23_begin_mask_0, end = concat_270, end_mask = value_cache_internal_tensor_assign_23_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_23_squeeze_mask_0, stride = value_cache_internal_tensor_assign_23_stride_0, update = value_states_135_cast_fp16, x = coreml_update_state_43)[name = string("value_cache_internal_tensor_assign_23_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_23_cast_fp16, input = value_cache)[name = string("coreml_update_state_45_write_state")]; tensor coreml_update_state_45 = read_state(input = value_cache)[name = string("coreml_update_state_45")]; tensor var_8147_begin_0 = const()[name = string("op_8147_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8147_end_0 = const()[name = string("op_8147_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8147_end_mask_0 = const()[name = string("op_8147_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8147_cast_fp16 = slice_by_index(begin = var_8147_begin_0, end = var_8147_end_0, end_mask = var_8147_end_mask_0, x = coreml_update_state_44)[name = string("op_8147_cast_fp16")]; tensor tile_44 = const()[name = string("tile_44"), val = tensor([1, 1])]; int32 var_8150_axis_0 = const()[name = string("op_8150_axis_0"), val = int32(1)]; tensor var_8150_cast_fp16_0, tensor var_8150_cast_fp16_1 = split(axis = var_8150_axis_0, split_sizes = tile_44, x = var_8147_cast_fp16)[name = string("op_8150_cast_fp16")]; tensor var_8157_begin_0 = const()[name = string("op_8157_begin_0"), val = tensor([22, 0, 0, 0])]; tensor var_8157_end_0 = const()[name = string("op_8157_end_0"), val = tensor([23, 2, 2048, 128])]; tensor var_8157_end_mask_0 = const()[name = string("op_8157_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8157_cast_fp16 = slice_by_index(begin = var_8157_begin_0, end = var_8157_end_0, end_mask = var_8157_end_mask_0, x = coreml_update_state_45)[name = string("op_8157_cast_fp16")]; tensor tile_45 = const()[name = string("tile_45"), val = tensor([1, 1])]; int32 var_8160_axis_0 = const()[name = string("op_8160_axis_0"), val = int32(1)]; tensor var_8160_cast_fp16_0, tensor var_8160_cast_fp16_1 = split(axis = var_8160_axis_0, split_sizes = tile_45, x = var_8157_cast_fp16)[name = string("op_8160_cast_fp16")]; tensor var_8163_split_sizes_0 = const()[name = string("op_8163_split_sizes_0"), val = tensor([8, 8])]; int32 var_8163_axis_0 = const()[name = string("op_8163_axis_0"), val = int32(1)]; tensor var_8163_0, tensor var_8163_1 = split(axis = var_8163_axis_0, split_sizes = var_8163_split_sizes_0, x = query_states_135_cast_fp16)[name = string("op_8163")]; bool attn_weights_353_transpose_x_0 = const()[name = string("attn_weights_353_transpose_x_0"), val = bool(false)]; bool attn_weights_353_transpose_y_0 = const()[name = string("attn_weights_353_transpose_y_0"), val = bool(false)]; tensor attn_weights_353_cast_fp16 = matmul(transpose_x = attn_weights_353_transpose_x_0, transpose_y = attn_weights_353_transpose_y_0, x = var_8150_cast_fp16_0, y = var_8163_0)[name = string("attn_weights_353_cast_fp16")]; fp16 var_8166_to_fp16 = const()[name = string("op_8166_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_355_cast_fp16 = mul(x = attn_weights_353_cast_fp16, y = var_8166_to_fp16)[name = string("attn_weights_355_cast_fp16")]; tensor attn_weights_357_cast_fp16 = add(x = attn_weights_355_cast_fp16, y = attn_mask_1)[name = string("attn_weights_357_cast_fp16")]; int32 var_8170 = const()[name = string("op_8170"), val = int32(-2)]; tensor attn_weights_359_cast_fp16 = softmax(axis = var_8170, x = attn_weights_357_cast_fp16)[name = string("attn_weights_359_cast_fp16")]; bool var_8176_transpose_x_1 = const()[name = string("op_8176_transpose_x_1"), val = bool(true)]; bool var_8176_transpose_y_1 = const()[name = string("op_8176_transpose_y_1"), val = bool(false)]; tensor var_8176_cast_fp16 = matmul(transpose_x = var_8176_transpose_x_1, transpose_y = var_8176_transpose_y_1, x = attn_weights_359_cast_fp16, y = var_8160_cast_fp16_0)[name = string("op_8176_cast_fp16")]; bool attn_weights_361_transpose_x_0 = const()[name = string("attn_weights_361_transpose_x_0"), val = bool(false)]; bool attn_weights_361_transpose_y_0 = const()[name = string("attn_weights_361_transpose_y_0"), val = bool(false)]; tensor attn_weights_361_cast_fp16 = matmul(transpose_x = attn_weights_361_transpose_x_0, transpose_y = attn_weights_361_transpose_y_0, x = var_8150_cast_fp16_1, y = var_8163_1)[name = string("attn_weights_361_cast_fp16")]; fp16 var_8178_to_fp16 = const()[name = string("op_8178_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_363_cast_fp16 = mul(x = attn_weights_361_cast_fp16, y = var_8178_to_fp16)[name = string("attn_weights_363_cast_fp16")]; tensor attn_weights_365_cast_fp16 = add(x = attn_weights_363_cast_fp16, y = attn_mask_1)[name = string("attn_weights_365_cast_fp16")]; int32 var_8182 = const()[name = string("op_8182"), val = int32(-2)]; tensor attn_weights_367_cast_fp16 = softmax(axis = var_8182, x = attn_weights_365_cast_fp16)[name = string("attn_weights_367_cast_fp16")]; bool attn_output_177_transpose_x_1 = const()[name = string("attn_output_177_transpose_x_1"), val = bool(true)]; bool attn_output_177_transpose_y_1 = const()[name = string("attn_output_177_transpose_y_1"), val = bool(false)]; tensor attn_output_177_cast_fp16 = matmul(transpose_x = attn_output_177_transpose_x_1, transpose_y = attn_output_177_transpose_y_1, x = attn_weights_367_cast_fp16, y = var_8160_cast_fp16_1)[name = string("attn_output_177_cast_fp16")]; int32 var_8190 = const()[name = string("op_8190"), val = int32(1)]; bool attn_output_179_interleave_0 = const()[name = string("attn_output_179_interleave_0"), val = bool(false)]; tensor attn_output_179_cast_fp16 = concat(axis = var_8190, interleave = attn_output_179_interleave_0, values = (var_8176_cast_fp16, attn_output_177_cast_fp16))[name = string("attn_output_179_cast_fp16")]; tensor var_8194_perm_0 = const()[name = string("op_8194_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_275x = const()[name = string("concat_275x"), val = tensor([1, 2048, 1, -1])]; tensor var_8194_cast_fp16 = transpose(perm = var_8194_perm_0, x = attn_output_179_cast_fp16)[name = string("transpose_15")]; tensor attn_output_183_cast_fp16 = reshape(shape = concat_275x, x = var_8194_cast_fp16)[name = string("attn_output_183_cast_fp16")]; tensor layers_22_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_22_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1529843456)))]; tensor hidden_states_223_strides_0 = const()[name = string("hidden_states_223_strides_0"), val = tensor([1, 1])]; string hidden_states_223_pad_type_0 = const()[name = string("hidden_states_223_pad_type_0"), val = string("valid")]; tensor hidden_states_223_pad_0 = const()[name = string("hidden_states_223_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_223_dilations_0 = const()[name = string("hidden_states_223_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_223_groups_0 = const()[name = string("hidden_states_223_groups_0"), val = int32(1)]; tensor hidden_states_223_cast_fp16 = conv(dilations = hidden_states_223_dilations_0, groups = hidden_states_223_groups_0, pad = hidden_states_223_pad_0, pad_type = hidden_states_223_pad_type_0, strides = hidden_states_223_strides_0, weight = layers_22_self_attn_o_proj_weight_to_fp16, x = attn_output_183_cast_fp16)[name = string("hidden_states_223_cast_fp16")]; tensor hidden_states_225_cast_fp16 = add(x = hidden_states_219_cast_fp16, y = hidden_states_223_cast_fp16)[name = string("hidden_states_225_cast_fp16")]; fp16 const_228_promoted_to_fp16 = const()[name = string("const_228_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8227_cast_fp16 = mul(x = hidden_states_225_cast_fp16, y = const_228_promoted_to_fp16)[name = string("op_8227_cast_fp16")]; int32 var_8225 = const()[name = string("op_8225"), val = int32(1)]; bool doubled_181_interleave_0 = const()[name = string("doubled_181_interleave_0"), val = bool(false)]; tensor doubled_181_cast_fp16 = concat(axis = var_8225, interleave = doubled_181_interleave_0, values = (hidden_states_225_cast_fp16, var_8227_cast_fp16))[name = string("doubled_181_cast_fp16")]; tensor out_91_axes_0 = const()[name = string("out_91_axes_0"), val = tensor([1])]; tensor out_91_gamma_0_to_fp16 = const()[name = string("out_91_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538232128)))]; fp16 var_8237_to_fp16 = const()[name = string("op_8237_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_91_cast_fp16 = layer_norm(axes = out_91_axes_0, epsilon = var_8237_to_fp16, gamma = out_91_gamma_0_to_fp16, x = doubled_181_cast_fp16)[name = string("out_91_cast_fp16")]; tensor var_8248_split_sizes_0 = const()[name = string("op_8248_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8248_axis_0 = const()[name = string("op_8248_axis_0"), val = int32(1)]; tensor var_8248_cast_fp16_0, tensor var_8248_cast_fp16_1 = split(axis = var_8248_axis_0, split_sizes = var_8248_split_sizes_0, x = out_91_cast_fp16)[name = string("op_8248_cast_fp16")]; tensor input_45_strides_0 = const()[name = string("input_45_strides_0"), val = tensor([1, 1])]; string input_45_pad_type_0 = const()[name = string("input_45_pad_type_0"), val = string("valid")]; tensor input_45_pad_0 = const()[name = string("input_45_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_45_dilations_0 = const()[name = string("input_45_dilations_0"), val = tensor([1, 1])]; int32 input_45_groups_0 = const()[name = string("input_45_groups_0"), val = int32(1)]; tensor input_45_cast_fp16 = conv(dilations = input_45_dilations_0, groups = input_45_groups_0, pad = input_45_pad_0, pad_type = input_45_pad_type_0, strides = input_45_strides_0, weight = layers_22_mlp_gate_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("input_45_cast_fp16")]; tensor var_8265_cast_fp16 = silu(x = input_45_cast_fp16)[name = string("op_8265_cast_fp16")]; tensor var_8271_strides_0 = const()[name = string("op_8271_strides_0"), val = tensor([1, 1])]; string var_8271_pad_type_0 = const()[name = string("op_8271_pad_type_0"), val = string("valid")]; tensor var_8271_pad_0 = const()[name = string("op_8271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8271_dilations_0 = const()[name = string("op_8271_dilations_0"), val = tensor([1, 1])]; int32 var_8271_groups_0 = const()[name = string("op_8271_groups_0"), val = int32(1)]; tensor var_8271_cast_fp16 = conv(dilations = var_8271_dilations_0, groups = var_8271_groups_0, pad = var_8271_pad_0, pad_type = var_8271_pad_type_0, strides = var_8271_strides_0, weight = layers_22_mlp_up_proj_weight_cast_fp16, x = var_8248_cast_fp16_0)[name = string("op_8271_cast_fp16")]; tensor x_229_cast_fp16 = mul(x = var_8265_cast_fp16, y = var_8271_cast_fp16)[name = string("x_229_cast_fp16")]; tensor hidden_states_227_strides_0 = const()[name = string("hidden_states_227_strides_0"), val = tensor([1, 1])]; string hidden_states_227_pad_type_0 = const()[name = string("hidden_states_227_pad_type_0"), val = string("valid")]; tensor hidden_states_227_pad_0 = const()[name = string("hidden_states_227_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_227_dilations_0 = const()[name = string("hidden_states_227_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_227_groups_0 = const()[name = string("hidden_states_227_groups_0"), val = int32(1)]; tensor hidden_states_227_cast_fp16 = conv(dilations = hidden_states_227_dilations_0, groups = hidden_states_227_groups_0, pad = hidden_states_227_pad_0, pad_type = hidden_states_227_pad_type_0, strides = hidden_states_227_strides_0, weight = layers_22_mlp_down_proj_weight_cast_fp16, x = x_229_cast_fp16)[name = string("hidden_states_227_cast_fp16")]; tensor hidden_states_229_cast_fp16 = add(x = hidden_states_225_cast_fp16, y = hidden_states_227_cast_fp16)[name = string("hidden_states_229_cast_fp16")]; fp16 const_230_promoted_to_fp16 = const()[name = string("const_230_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8289_cast_fp16 = mul(x = hidden_states_229_cast_fp16, y = const_230_promoted_to_fp16)[name = string("op_8289_cast_fp16")]; int32 var_8287 = const()[name = string("op_8287"), val = int32(1)]; bool doubled_185_interleave_0 = const()[name = string("doubled_185_interleave_0"), val = bool(false)]; tensor doubled_185_cast_fp16 = concat(axis = var_8287, interleave = doubled_185_interleave_0, values = (hidden_states_229_cast_fp16, var_8289_cast_fp16))[name = string("doubled_185_cast_fp16")]; tensor out_93_axes_0 = const()[name = string("out_93_axes_0"), val = tensor([1])]; tensor out_93_gamma_0_to_fp16 = const()[name = string("out_93_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538240384)))]; fp16 var_8299_to_fp16 = const()[name = string("op_8299_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_93_cast_fp16 = layer_norm(axes = out_93_axes_0, epsilon = var_8299_to_fp16, gamma = out_93_gamma_0_to_fp16, x = doubled_185_cast_fp16)[name = string("out_93_cast_fp16")]; tensor var_8310_split_sizes_0 = const()[name = string("op_8310_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8310_axis_0 = const()[name = string("op_8310_axis_0"), val = int32(1)]; tensor var_8310_cast_fp16_0, tensor var_8310_cast_fp16_1 = split(axis = var_8310_axis_0, split_sizes = var_8310_split_sizes_0, x = out_93_cast_fp16)[name = string("op_8310_cast_fp16")]; tensor query_states_139_strides_0 = const()[name = string("query_states_139_strides_0"), val = tensor([1, 1])]; string query_states_139_pad_type_0 = const()[name = string("query_states_139_pad_type_0"), val = string("valid")]; tensor query_states_139_pad_0 = const()[name = string("query_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_139_dilations_0 = const()[name = string("query_states_139_dilations_0"), val = tensor([1, 1])]; int32 query_states_139_groups_0 = const()[name = string("query_states_139_groups_0"), val = int32(1)]; tensor query_states_139_cast_fp16 = conv(dilations = query_states_139_dilations_0, groups = query_states_139_groups_0, pad = query_states_139_pad_0, pad_type = query_states_139_pad_type_0, strides = query_states_139_strides_0, weight = layers_23_self_attn_q_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("query_states_139_cast_fp16")]; tensor key_states_231_strides_0 = const()[name = string("key_states_231_strides_0"), val = tensor([1, 1])]; string key_states_231_pad_type_0 = const()[name = string("key_states_231_pad_type_0"), val = string("valid")]; tensor key_states_231_pad_0 = const()[name = string("key_states_231_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_231_dilations_0 = const()[name = string("key_states_231_dilations_0"), val = tensor([1, 1])]; int32 key_states_231_groups_0 = const()[name = string("key_states_231_groups_0"), val = int32(1)]; tensor key_states_231_cast_fp16 = conv(dilations = key_states_231_dilations_0, groups = key_states_231_groups_0, pad = key_states_231_pad_0, pad_type = key_states_231_pad_type_0, strides = key_states_231_strides_0, weight = layers_23_self_attn_k_proj_weight_cast_fp16, x = var_8310_cast_fp16_0)[name = string("key_states_231_cast_fp16")]; tensor layers_23_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_23_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1538248640)))]; tensor value_states_139_strides_0 = const()[name = string("value_states_139_strides_0"), val = tensor([1, 1])]; string value_states_139_pad_type_0 = const()[name = string("value_states_139_pad_type_0"), val = string("valid")]; tensor value_states_139_pad_0 = const()[name = string("value_states_139_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_139_dilations_0 = const()[name = string("value_states_139_dilations_0"), val = tensor([1, 1])]; int32 value_states_139_groups_0 = const()[name = string("value_states_139_groups_0"), val = int32(1)]; tensor value_states_139_cast_fp16 = conv(dilations = value_states_139_dilations_0, groups = value_states_139_groups_0, pad = value_states_139_pad_0, pad_type = value_states_139_pad_type_0, strides = value_states_139_strides_0, weight = layers_23_self_attn_v_proj_weight_to_fp16, x = var_8310_cast_fp16_0)[name = string("value_states_139_cast_fp16")]; tensor concat_276x = const()[name = string("concat_276x"), val = tensor([1, 16, 128, -1])]; tensor x_231_cast_fp16 = reshape(shape = concat_276x, x = query_states_139_cast_fp16)[name = string("x_231_cast_fp16")]; tensor concat_277x = const()[name = string("concat_277x"), val = tensor([1, 2, 128, -1])]; tensor var_8367_cast_fp16 = reshape(shape = concat_277x, x = key_states_231_cast_fp16)[name = string("op_8367_cast_fp16")]; tensor concat_278x = const()[name = string("concat_278x"), val = tensor([1, 2, 128, -1])]; tensor var_8374_cast_fp16 = reshape(shape = concat_278x, x = value_states_139_cast_fp16)[name = string("op_8374_cast_fp16")]; tensor var_8378_cast_fp16 = mul(x = x_231_cast_fp16, y = var_869_cast_fp16)[name = string("op_8378_cast_fp16")]; tensor var_8379_split_sizes_0 = const()[name = string("op_8379_split_sizes_0"), val = tensor([64, 64])]; int32 var_8379_axis_0 = const()[name = string("op_8379_axis_0"), val = int32(-2)]; tensor var_8379_cast_fp16_0, tensor var_8379_cast_fp16_1 = split(axis = var_8379_axis_0, split_sizes = var_8379_split_sizes_0, x = x_231_cast_fp16)[name = string("op_8379_cast_fp16")]; fp16 const_232_promoted_to_fp16 = const()[name = string("const_232_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8381_cast_fp16 = mul(x = var_8379_cast_fp16_1, y = const_232_promoted_to_fp16)[name = string("op_8381_cast_fp16")]; int32 var_8383 = const()[name = string("op_8383"), val = int32(-2)]; bool var_8384_interleave_0 = const()[name = string("op_8384_interleave_0"), val = bool(false)]; tensor var_8384_cast_fp16 = concat(axis = var_8383, interleave = var_8384_interleave_0, values = (var_8381_cast_fp16, var_8379_cast_fp16_0))[name = string("op_8384_cast_fp16")]; tensor var_8385_cast_fp16 = mul(x = var_8384_cast_fp16, y = var_878_cast_fp16)[name = string("op_8385_cast_fp16")]; tensor query_states_141_cast_fp16 = add(x = var_8378_cast_fp16, y = var_8385_cast_fp16)[name = string("query_states_141_cast_fp16")]; tensor var_8391_cast_fp16 = mul(x = var_8367_cast_fp16, y = var_869_cast_fp16)[name = string("op_8391_cast_fp16")]; tensor var_8392_split_sizes_0 = const()[name = string("op_8392_split_sizes_0"), val = tensor([64, 64])]; int32 var_8392_axis_0 = const()[name = string("op_8392_axis_0"), val = int32(-2)]; tensor var_8392_cast_fp16_0, tensor var_8392_cast_fp16_1 = split(axis = var_8392_axis_0, split_sizes = var_8392_split_sizes_0, x = var_8367_cast_fp16)[name = string("op_8392_cast_fp16")]; fp16 const_233_promoted_to_fp16 = const()[name = string("const_233_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8394_cast_fp16 = mul(x = var_8392_cast_fp16_1, y = const_233_promoted_to_fp16)[name = string("op_8394_cast_fp16")]; int32 var_8396 = const()[name = string("op_8396"), val = int32(-2)]; bool var_8397_interleave_0 = const()[name = string("op_8397_interleave_0"), val = bool(false)]; tensor var_8397_cast_fp16 = concat(axis = var_8396, interleave = var_8397_interleave_0, values = (var_8394_cast_fp16, var_8392_cast_fp16_0))[name = string("op_8397_cast_fp16")]; tensor var_8398_cast_fp16 = mul(x = var_8397_cast_fp16, y = var_878_cast_fp16)[name = string("op_8398_cast_fp16")]; tensor key_states_235_cast_fp16 = add(x = var_8391_cast_fp16, y = var_8398_cast_fp16)[name = string("key_states_235_cast_fp16")]; tensor expand_dims_276 = const()[name = string("expand_dims_276"), val = tensor([23])]; tensor expand_dims_277 = const()[name = string("expand_dims_277"), val = tensor([0])]; tensor expand_dims_279 = const()[name = string("expand_dims_279"), val = tensor([0])]; int32 concat_281_axis_0 = const()[name = string("concat_281_axis_0"), val = int32(0)]; bool concat_281_interleave_0 = const()[name = string("concat_281_interleave_0"), val = bool(false)]; tensor concat_281 = concat(axis = concat_281_axis_0, interleave = concat_281_interleave_0, values = (expand_dims_276, expand_dims_277, position_id, expand_dims_279))[name = string("concat_281")]; tensor expand_dims_280 = const()[name = string("expand_dims_280"), val = tensor([24])]; tensor concat_282_values1_0 = const()[name = string("concat_282_values1_0"), val = tensor([0])]; tensor concat_282_values3_0 = const()[name = string("concat_282_values3_0"), val = tensor([0])]; int32 concat_282_axis_0 = const()[name = string("concat_282_axis_0"), val = int32(0)]; bool concat_282_interleave_0 = const()[name = string("concat_282_interleave_0"), val = bool(false)]; tensor concat_282 = concat(axis = concat_282_axis_0, interleave = concat_282_interleave_0, values = (expand_dims_280, concat_282_values1_0, cache_position_end, concat_282_values3_0))[name = string("concat_282")]; tensor key_states_237_perm_0 = const()[name = string("key_states_237_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_24_stride_0 = const()[name = string("key_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_237_cast_fp16 = transpose(perm = key_states_237_perm_0, x = key_states_235_cast_fp16)[name = string("transpose_14")]; tensor key_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = key_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = key_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_24_squeeze_mask_0, stride = key_cache_internal_tensor_assign_24_stride_0, update = key_states_237_cast_fp16, x = coreml_update_state_44)[name = string("key_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_24_cast_fp16, input = key_cache)[name = string("coreml_update_state_46_write_state")]; tensor coreml_update_state_46 = read_state(input = key_cache)[name = string("coreml_update_state_46")]; tensor value_states_141_perm_0 = const()[name = string("value_states_141_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_24_stride_0 = const()[name = string("value_cache_internal_tensor_assign_24_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_24_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_24_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_24_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_24_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_141_cast_fp16 = transpose(perm = value_states_141_perm_0, x = var_8374_cast_fp16)[name = string("transpose_13")]; tensor value_cache_internal_tensor_assign_24_cast_fp16 = slice_update(begin = concat_281, begin_mask = value_cache_internal_tensor_assign_24_begin_mask_0, end = concat_282, end_mask = value_cache_internal_tensor_assign_24_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_24_squeeze_mask_0, stride = value_cache_internal_tensor_assign_24_stride_0, update = value_states_141_cast_fp16, x = coreml_update_state_45)[name = string("value_cache_internal_tensor_assign_24_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_24_cast_fp16, input = value_cache)[name = string("coreml_update_state_47_write_state")]; tensor coreml_update_state_47 = read_state(input = value_cache)[name = string("coreml_update_state_47")]; tensor var_8468_begin_0 = const()[name = string("op_8468_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8468_end_0 = const()[name = string("op_8468_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8468_end_mask_0 = const()[name = string("op_8468_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8468_cast_fp16 = slice_by_index(begin = var_8468_begin_0, end = var_8468_end_0, end_mask = var_8468_end_mask_0, x = coreml_update_state_46)[name = string("op_8468_cast_fp16")]; tensor tile_46 = const()[name = string("tile_46"), val = tensor([1, 1])]; int32 var_8471_axis_0 = const()[name = string("op_8471_axis_0"), val = int32(1)]; tensor var_8471_cast_fp16_0, tensor var_8471_cast_fp16_1 = split(axis = var_8471_axis_0, split_sizes = tile_46, x = var_8468_cast_fp16)[name = string("op_8471_cast_fp16")]; tensor var_8478_begin_0 = const()[name = string("op_8478_begin_0"), val = tensor([23, 0, 0, 0])]; tensor var_8478_end_0 = const()[name = string("op_8478_end_0"), val = tensor([24, 2, 2048, 128])]; tensor var_8478_end_mask_0 = const()[name = string("op_8478_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8478_cast_fp16 = slice_by_index(begin = var_8478_begin_0, end = var_8478_end_0, end_mask = var_8478_end_mask_0, x = coreml_update_state_47)[name = string("op_8478_cast_fp16")]; tensor tile_47 = const()[name = string("tile_47"), val = tensor([1, 1])]; int32 var_8481_axis_0 = const()[name = string("op_8481_axis_0"), val = int32(1)]; tensor var_8481_cast_fp16_0, tensor var_8481_cast_fp16_1 = split(axis = var_8481_axis_0, split_sizes = tile_47, x = var_8478_cast_fp16)[name = string("op_8481_cast_fp16")]; tensor var_8484_split_sizes_0 = const()[name = string("op_8484_split_sizes_0"), val = tensor([8, 8])]; int32 var_8484_axis_0 = const()[name = string("op_8484_axis_0"), val = int32(1)]; tensor var_8484_0, tensor var_8484_1 = split(axis = var_8484_axis_0, split_sizes = var_8484_split_sizes_0, x = query_states_141_cast_fp16)[name = string("op_8484")]; bool attn_weights_369_transpose_x_0 = const()[name = string("attn_weights_369_transpose_x_0"), val = bool(false)]; bool attn_weights_369_transpose_y_0 = const()[name = string("attn_weights_369_transpose_y_0"), val = bool(false)]; tensor attn_weights_369_cast_fp16 = matmul(transpose_x = attn_weights_369_transpose_x_0, transpose_y = attn_weights_369_transpose_y_0, x = var_8471_cast_fp16_0, y = var_8484_0)[name = string("attn_weights_369_cast_fp16")]; fp16 var_8487_to_fp16 = const()[name = string("op_8487_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_371_cast_fp16 = mul(x = attn_weights_369_cast_fp16, y = var_8487_to_fp16)[name = string("attn_weights_371_cast_fp16")]; tensor attn_weights_373_cast_fp16 = add(x = attn_weights_371_cast_fp16, y = attn_mask_1)[name = string("attn_weights_373_cast_fp16")]; int32 var_8491 = const()[name = string("op_8491"), val = int32(-2)]; tensor attn_weights_375_cast_fp16 = softmax(axis = var_8491, x = attn_weights_373_cast_fp16)[name = string("attn_weights_375_cast_fp16")]; bool var_8497_transpose_x_1 = const()[name = string("op_8497_transpose_x_1"), val = bool(true)]; bool var_8497_transpose_y_1 = const()[name = string("op_8497_transpose_y_1"), val = bool(false)]; tensor var_8497_cast_fp16 = matmul(transpose_x = var_8497_transpose_x_1, transpose_y = var_8497_transpose_y_1, x = attn_weights_375_cast_fp16, y = var_8481_cast_fp16_0)[name = string("op_8497_cast_fp16")]; bool attn_weights_377_transpose_x_0 = const()[name = string("attn_weights_377_transpose_x_0"), val = bool(false)]; bool attn_weights_377_transpose_y_0 = const()[name = string("attn_weights_377_transpose_y_0"), val = bool(false)]; tensor attn_weights_377_cast_fp16 = matmul(transpose_x = attn_weights_377_transpose_x_0, transpose_y = attn_weights_377_transpose_y_0, x = var_8471_cast_fp16_1, y = var_8484_1)[name = string("attn_weights_377_cast_fp16")]; fp16 var_8499_to_fp16 = const()[name = string("op_8499_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_379_cast_fp16 = mul(x = attn_weights_377_cast_fp16, y = var_8499_to_fp16)[name = string("attn_weights_379_cast_fp16")]; tensor attn_weights_381_cast_fp16 = add(x = attn_weights_379_cast_fp16, y = attn_mask_1)[name = string("attn_weights_381_cast_fp16")]; int32 var_8503 = const()[name = string("op_8503"), val = int32(-2)]; tensor attn_weights_383_cast_fp16 = softmax(axis = var_8503, x = attn_weights_381_cast_fp16)[name = string("attn_weights_383_cast_fp16")]; bool attn_output_185_transpose_x_1 = const()[name = string("attn_output_185_transpose_x_1"), val = bool(true)]; bool attn_output_185_transpose_y_1 = const()[name = string("attn_output_185_transpose_y_1"), val = bool(false)]; tensor attn_output_185_cast_fp16 = matmul(transpose_x = attn_output_185_transpose_x_1, transpose_y = attn_output_185_transpose_y_1, x = attn_weights_383_cast_fp16, y = var_8481_cast_fp16_1)[name = string("attn_output_185_cast_fp16")]; int32 var_8511 = const()[name = string("op_8511"), val = int32(1)]; bool attn_output_187_interleave_0 = const()[name = string("attn_output_187_interleave_0"), val = bool(false)]; tensor attn_output_187_cast_fp16 = concat(axis = var_8511, interleave = attn_output_187_interleave_0, values = (var_8497_cast_fp16, attn_output_185_cast_fp16))[name = string("attn_output_187_cast_fp16")]; tensor var_8515_perm_0 = const()[name = string("op_8515_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_287x = const()[name = string("concat_287x"), val = tensor([1, 2048, 1, -1])]; tensor var_8515_cast_fp16 = transpose(perm = var_8515_perm_0, x = attn_output_187_cast_fp16)[name = string("transpose_12")]; tensor attn_output_191_cast_fp16 = reshape(shape = concat_287x, x = var_8515_cast_fp16)[name = string("attn_output_191_cast_fp16")]; tensor hidden_states_233_strides_0 = const()[name = string("hidden_states_233_strides_0"), val = tensor([1, 1])]; string hidden_states_233_pad_type_0 = const()[name = string("hidden_states_233_pad_type_0"), val = string("valid")]; tensor hidden_states_233_pad_0 = const()[name = string("hidden_states_233_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_233_dilations_0 = const()[name = string("hidden_states_233_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_233_groups_0 = const()[name = string("hidden_states_233_groups_0"), val = int32(1)]; tensor hidden_states_233_cast_fp16 = conv(dilations = hidden_states_233_dilations_0, groups = hidden_states_233_groups_0, pad = hidden_states_233_pad_0, pad_type = hidden_states_233_pad_type_0, strides = hidden_states_233_strides_0, weight = layers_23_self_attn_o_proj_weight_cast_fp16, x = attn_output_191_cast_fp16)[name = string("hidden_states_233_cast_fp16")]; tensor hidden_states_235_cast_fp16 = add(x = hidden_states_229_cast_fp16, y = hidden_states_233_cast_fp16)[name = string("hidden_states_235_cast_fp16")]; fp16 const_238_promoted_to_fp16 = const()[name = string("const_238_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8548_cast_fp16 = mul(x = hidden_states_235_cast_fp16, y = const_238_promoted_to_fp16)[name = string("op_8548_cast_fp16")]; int32 var_8546 = const()[name = string("op_8546"), val = int32(1)]; bool doubled_189_interleave_0 = const()[name = string("doubled_189_interleave_0"), val = bool(false)]; tensor doubled_189_cast_fp16 = concat(axis = var_8546, interleave = doubled_189_interleave_0, values = (hidden_states_235_cast_fp16, var_8548_cast_fp16))[name = string("doubled_189_cast_fp16")]; tensor out_95_axes_0 = const()[name = string("out_95_axes_0"), val = tensor([1])]; tensor out_95_gamma_0_to_fp16 = const()[name = string("out_95_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539297280)))]; fp16 var_8558_to_fp16 = const()[name = string("op_8558_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_95_cast_fp16 = layer_norm(axes = out_95_axes_0, epsilon = var_8558_to_fp16, gamma = out_95_gamma_0_to_fp16, x = doubled_189_cast_fp16)[name = string("out_95_cast_fp16")]; tensor var_8569_split_sizes_0 = const()[name = string("op_8569_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8569_axis_0 = const()[name = string("op_8569_axis_0"), val = int32(1)]; tensor var_8569_cast_fp16_0, tensor var_8569_cast_fp16_1 = split(axis = var_8569_axis_0, split_sizes = var_8569_split_sizes_0, x = out_95_cast_fp16)[name = string("op_8569_cast_fp16")]; tensor input_47_strides_0 = const()[name = string("input_47_strides_0"), val = tensor([1, 1])]; string input_47_pad_type_0 = const()[name = string("input_47_pad_type_0"), val = string("valid")]; tensor input_47_pad_0 = const()[name = string("input_47_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_47_dilations_0 = const()[name = string("input_47_dilations_0"), val = tensor([1, 1])]; int32 input_47_groups_0 = const()[name = string("input_47_groups_0"), val = int32(1)]; tensor input_47_cast_fp16 = conv(dilations = input_47_dilations_0, groups = input_47_groups_0, pad = input_47_pad_0, pad_type = input_47_pad_type_0, strides = input_47_strides_0, weight = layers_23_mlp_gate_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("input_47_cast_fp16")]; tensor var_8586_cast_fp16 = silu(x = input_47_cast_fp16)[name = string("op_8586_cast_fp16")]; tensor var_8592_strides_0 = const()[name = string("op_8592_strides_0"), val = tensor([1, 1])]; string var_8592_pad_type_0 = const()[name = string("op_8592_pad_type_0"), val = string("valid")]; tensor var_8592_pad_0 = const()[name = string("op_8592_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8592_dilations_0 = const()[name = string("op_8592_dilations_0"), val = tensor([1, 1])]; int32 var_8592_groups_0 = const()[name = string("op_8592_groups_0"), val = int32(1)]; tensor var_8592_cast_fp16 = conv(dilations = var_8592_dilations_0, groups = var_8592_groups_0, pad = var_8592_pad_0, pad_type = var_8592_pad_type_0, strides = var_8592_strides_0, weight = layers_23_mlp_up_proj_weight_cast_fp16, x = var_8569_cast_fp16_0)[name = string("op_8592_cast_fp16")]; tensor x_239_cast_fp16 = mul(x = var_8586_cast_fp16, y = var_8592_cast_fp16)[name = string("x_239_cast_fp16")]; tensor hidden_states_237_strides_0 = const()[name = string("hidden_states_237_strides_0"), val = tensor([1, 1])]; string hidden_states_237_pad_type_0 = const()[name = string("hidden_states_237_pad_type_0"), val = string("valid")]; tensor hidden_states_237_pad_0 = const()[name = string("hidden_states_237_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_237_dilations_0 = const()[name = string("hidden_states_237_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_237_groups_0 = const()[name = string("hidden_states_237_groups_0"), val = int32(1)]; tensor hidden_states_237_cast_fp16 = conv(dilations = hidden_states_237_dilations_0, groups = hidden_states_237_groups_0, pad = hidden_states_237_pad_0, pad_type = hidden_states_237_pad_type_0, strides = hidden_states_237_strides_0, weight = layers_23_mlp_down_proj_weight_cast_fp16, x = x_239_cast_fp16)[name = string("hidden_states_237_cast_fp16")]; tensor hidden_states_239_cast_fp16 = add(x = hidden_states_235_cast_fp16, y = hidden_states_237_cast_fp16)[name = string("hidden_states_239_cast_fp16")]; fp16 const_240_promoted_to_fp16 = const()[name = string("const_240_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8610_cast_fp16 = mul(x = hidden_states_239_cast_fp16, y = const_240_promoted_to_fp16)[name = string("op_8610_cast_fp16")]; int32 var_8608 = const()[name = string("op_8608"), val = int32(1)]; bool doubled_193_interleave_0 = const()[name = string("doubled_193_interleave_0"), val = bool(false)]; tensor doubled_193_cast_fp16 = concat(axis = var_8608, interleave = doubled_193_interleave_0, values = (hidden_states_239_cast_fp16, var_8610_cast_fp16))[name = string("doubled_193_cast_fp16")]; tensor out_97_axes_0 = const()[name = string("out_97_axes_0"), val = tensor([1])]; tensor out_97_gamma_0_to_fp16 = const()[name = string("out_97_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539305536)))]; fp16 var_8620_to_fp16 = const()[name = string("op_8620_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_97_cast_fp16 = layer_norm(axes = out_97_axes_0, epsilon = var_8620_to_fp16, gamma = out_97_gamma_0_to_fp16, x = doubled_193_cast_fp16)[name = string("out_97_cast_fp16")]; tensor var_8631_split_sizes_0 = const()[name = string("op_8631_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8631_axis_0 = const()[name = string("op_8631_axis_0"), val = int32(1)]; tensor var_8631_cast_fp16_0, tensor var_8631_cast_fp16_1 = split(axis = var_8631_axis_0, split_sizes = var_8631_split_sizes_0, x = out_97_cast_fp16)[name = string("op_8631_cast_fp16")]; tensor query_states_145_strides_0 = const()[name = string("query_states_145_strides_0"), val = tensor([1, 1])]; string query_states_145_pad_type_0 = const()[name = string("query_states_145_pad_type_0"), val = string("valid")]; tensor query_states_145_pad_0 = const()[name = string("query_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_145_dilations_0 = const()[name = string("query_states_145_dilations_0"), val = tensor([1, 1])]; int32 query_states_145_groups_0 = const()[name = string("query_states_145_groups_0"), val = int32(1)]; tensor query_states_145_cast_fp16 = conv(dilations = query_states_145_dilations_0, groups = query_states_145_groups_0, pad = query_states_145_pad_0, pad_type = query_states_145_pad_type_0, strides = query_states_145_strides_0, weight = layers_24_self_attn_q_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("query_states_145_cast_fp16")]; tensor key_states_241_strides_0 = const()[name = string("key_states_241_strides_0"), val = tensor([1, 1])]; string key_states_241_pad_type_0 = const()[name = string("key_states_241_pad_type_0"), val = string("valid")]; tensor key_states_241_pad_0 = const()[name = string("key_states_241_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_241_dilations_0 = const()[name = string("key_states_241_dilations_0"), val = tensor([1, 1])]; int32 key_states_241_groups_0 = const()[name = string("key_states_241_groups_0"), val = int32(1)]; tensor key_states_241_cast_fp16 = conv(dilations = key_states_241_dilations_0, groups = key_states_241_groups_0, pad = key_states_241_pad_0, pad_type = key_states_241_pad_type_0, strides = key_states_241_strides_0, weight = layers_24_self_attn_k_proj_weight_cast_fp16, x = var_8631_cast_fp16_0)[name = string("key_states_241_cast_fp16")]; tensor layers_24_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_24_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1539313792)))]; tensor value_states_145_strides_0 = const()[name = string("value_states_145_strides_0"), val = tensor([1, 1])]; string value_states_145_pad_type_0 = const()[name = string("value_states_145_pad_type_0"), val = string("valid")]; tensor value_states_145_pad_0 = const()[name = string("value_states_145_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_145_dilations_0 = const()[name = string("value_states_145_dilations_0"), val = tensor([1, 1])]; int32 value_states_145_groups_0 = const()[name = string("value_states_145_groups_0"), val = int32(1)]; tensor value_states_145_cast_fp16 = conv(dilations = value_states_145_dilations_0, groups = value_states_145_groups_0, pad = value_states_145_pad_0, pad_type = value_states_145_pad_type_0, strides = value_states_145_strides_0, weight = layers_24_self_attn_v_proj_weight_to_fp16, x = var_8631_cast_fp16_0)[name = string("value_states_145_cast_fp16")]; tensor concat_288x = const()[name = string("concat_288x"), val = tensor([1, 16, 128, -1])]; tensor x_241_cast_fp16 = reshape(shape = concat_288x, x = query_states_145_cast_fp16)[name = string("x_241_cast_fp16")]; tensor concat_289x = const()[name = string("concat_289x"), val = tensor([1, 2, 128, -1])]; tensor var_8688_cast_fp16 = reshape(shape = concat_289x, x = key_states_241_cast_fp16)[name = string("op_8688_cast_fp16")]; tensor concat_290x = const()[name = string("concat_290x"), val = tensor([1, 2, 128, -1])]; tensor var_8695_cast_fp16 = reshape(shape = concat_290x, x = value_states_145_cast_fp16)[name = string("op_8695_cast_fp16")]; tensor var_8699_cast_fp16 = mul(x = x_241_cast_fp16, y = var_869_cast_fp16)[name = string("op_8699_cast_fp16")]; tensor var_8700_split_sizes_0 = const()[name = string("op_8700_split_sizes_0"), val = tensor([64, 64])]; int32 var_8700_axis_0 = const()[name = string("op_8700_axis_0"), val = int32(-2)]; tensor var_8700_cast_fp16_0, tensor var_8700_cast_fp16_1 = split(axis = var_8700_axis_0, split_sizes = var_8700_split_sizes_0, x = x_241_cast_fp16)[name = string("op_8700_cast_fp16")]; fp16 const_242_promoted_to_fp16 = const()[name = string("const_242_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8702_cast_fp16 = mul(x = var_8700_cast_fp16_1, y = const_242_promoted_to_fp16)[name = string("op_8702_cast_fp16")]; int32 var_8704 = const()[name = string("op_8704"), val = int32(-2)]; bool var_8705_interleave_0 = const()[name = string("op_8705_interleave_0"), val = bool(false)]; tensor var_8705_cast_fp16 = concat(axis = var_8704, interleave = var_8705_interleave_0, values = (var_8702_cast_fp16, var_8700_cast_fp16_0))[name = string("op_8705_cast_fp16")]; tensor var_8706_cast_fp16 = mul(x = var_8705_cast_fp16, y = var_878_cast_fp16)[name = string("op_8706_cast_fp16")]; tensor query_states_147_cast_fp16 = add(x = var_8699_cast_fp16, y = var_8706_cast_fp16)[name = string("query_states_147_cast_fp16")]; tensor var_8712_cast_fp16 = mul(x = var_8688_cast_fp16, y = var_869_cast_fp16)[name = string("op_8712_cast_fp16")]; tensor var_8713_split_sizes_0 = const()[name = string("op_8713_split_sizes_0"), val = tensor([64, 64])]; int32 var_8713_axis_0 = const()[name = string("op_8713_axis_0"), val = int32(-2)]; tensor var_8713_cast_fp16_0, tensor var_8713_cast_fp16_1 = split(axis = var_8713_axis_0, split_sizes = var_8713_split_sizes_0, x = var_8688_cast_fp16)[name = string("op_8713_cast_fp16")]; fp16 const_243_promoted_to_fp16 = const()[name = string("const_243_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8715_cast_fp16 = mul(x = var_8713_cast_fp16_1, y = const_243_promoted_to_fp16)[name = string("op_8715_cast_fp16")]; int32 var_8717 = const()[name = string("op_8717"), val = int32(-2)]; bool var_8718_interleave_0 = const()[name = string("op_8718_interleave_0"), val = bool(false)]; tensor var_8718_cast_fp16 = concat(axis = var_8717, interleave = var_8718_interleave_0, values = (var_8715_cast_fp16, var_8713_cast_fp16_0))[name = string("op_8718_cast_fp16")]; tensor var_8719_cast_fp16 = mul(x = var_8718_cast_fp16, y = var_878_cast_fp16)[name = string("op_8719_cast_fp16")]; tensor key_states_245_cast_fp16 = add(x = var_8712_cast_fp16, y = var_8719_cast_fp16)[name = string("key_states_245_cast_fp16")]; tensor expand_dims_288 = const()[name = string("expand_dims_288"), val = tensor([24])]; tensor expand_dims_289 = const()[name = string("expand_dims_289"), val = tensor([0])]; tensor expand_dims_291 = const()[name = string("expand_dims_291"), val = tensor([0])]; int32 concat_293_axis_0 = const()[name = string("concat_293_axis_0"), val = int32(0)]; bool concat_293_interleave_0 = const()[name = string("concat_293_interleave_0"), val = bool(false)]; tensor concat_293 = concat(axis = concat_293_axis_0, interleave = concat_293_interleave_0, values = (expand_dims_288, expand_dims_289, position_id, expand_dims_291))[name = string("concat_293")]; tensor expand_dims_292 = const()[name = string("expand_dims_292"), val = tensor([25])]; tensor concat_294_values1_0 = const()[name = string("concat_294_values1_0"), val = tensor([0])]; tensor concat_294_values3_0 = const()[name = string("concat_294_values3_0"), val = tensor([0])]; int32 concat_294_axis_0 = const()[name = string("concat_294_axis_0"), val = int32(0)]; bool concat_294_interleave_0 = const()[name = string("concat_294_interleave_0"), val = bool(false)]; tensor concat_294 = concat(axis = concat_294_axis_0, interleave = concat_294_interleave_0, values = (expand_dims_292, concat_294_values1_0, cache_position_end, concat_294_values3_0))[name = string("concat_294")]; tensor key_states_247_perm_0 = const()[name = string("key_states_247_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_25_stride_0 = const()[name = string("key_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_247_cast_fp16 = transpose(perm = key_states_247_perm_0, x = key_states_245_cast_fp16)[name = string("transpose_11")]; tensor key_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = key_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = key_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_25_squeeze_mask_0, stride = key_cache_internal_tensor_assign_25_stride_0, update = key_states_247_cast_fp16, x = coreml_update_state_46)[name = string("key_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_25_cast_fp16, input = key_cache)[name = string("coreml_update_state_48_write_state")]; tensor coreml_update_state_48 = read_state(input = key_cache)[name = string("coreml_update_state_48")]; tensor value_states_147_perm_0 = const()[name = string("value_states_147_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_25_stride_0 = const()[name = string("value_cache_internal_tensor_assign_25_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_25_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_25_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_25_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_25_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_147_cast_fp16 = transpose(perm = value_states_147_perm_0, x = var_8695_cast_fp16)[name = string("transpose_10")]; tensor value_cache_internal_tensor_assign_25_cast_fp16 = slice_update(begin = concat_293, begin_mask = value_cache_internal_tensor_assign_25_begin_mask_0, end = concat_294, end_mask = value_cache_internal_tensor_assign_25_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_25_squeeze_mask_0, stride = value_cache_internal_tensor_assign_25_stride_0, update = value_states_147_cast_fp16, x = coreml_update_state_47)[name = string("value_cache_internal_tensor_assign_25_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_25_cast_fp16, input = value_cache)[name = string("coreml_update_state_49_write_state")]; tensor coreml_update_state_49 = read_state(input = value_cache)[name = string("coreml_update_state_49")]; tensor var_8789_begin_0 = const()[name = string("op_8789_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8789_end_0 = const()[name = string("op_8789_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8789_end_mask_0 = const()[name = string("op_8789_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8789_cast_fp16 = slice_by_index(begin = var_8789_begin_0, end = var_8789_end_0, end_mask = var_8789_end_mask_0, x = coreml_update_state_48)[name = string("op_8789_cast_fp16")]; tensor tile_48 = const()[name = string("tile_48"), val = tensor([1, 1])]; int32 var_8792_axis_0 = const()[name = string("op_8792_axis_0"), val = int32(1)]; tensor var_8792_cast_fp16_0, tensor var_8792_cast_fp16_1 = split(axis = var_8792_axis_0, split_sizes = tile_48, x = var_8789_cast_fp16)[name = string("op_8792_cast_fp16")]; tensor var_8799_begin_0 = const()[name = string("op_8799_begin_0"), val = tensor([24, 0, 0, 0])]; tensor var_8799_end_0 = const()[name = string("op_8799_end_0"), val = tensor([25, 2, 2048, 128])]; tensor var_8799_end_mask_0 = const()[name = string("op_8799_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_8799_cast_fp16 = slice_by_index(begin = var_8799_begin_0, end = var_8799_end_0, end_mask = var_8799_end_mask_0, x = coreml_update_state_49)[name = string("op_8799_cast_fp16")]; tensor tile_49 = const()[name = string("tile_49"), val = tensor([1, 1])]; int32 var_8802_axis_0 = const()[name = string("op_8802_axis_0"), val = int32(1)]; tensor var_8802_cast_fp16_0, tensor var_8802_cast_fp16_1 = split(axis = var_8802_axis_0, split_sizes = tile_49, x = var_8799_cast_fp16)[name = string("op_8802_cast_fp16")]; tensor var_8805_split_sizes_0 = const()[name = string("op_8805_split_sizes_0"), val = tensor([8, 8])]; int32 var_8805_axis_0 = const()[name = string("op_8805_axis_0"), val = int32(1)]; tensor var_8805_0, tensor var_8805_1 = split(axis = var_8805_axis_0, split_sizes = var_8805_split_sizes_0, x = query_states_147_cast_fp16)[name = string("op_8805")]; bool attn_weights_385_transpose_x_0 = const()[name = string("attn_weights_385_transpose_x_0"), val = bool(false)]; bool attn_weights_385_transpose_y_0 = const()[name = string("attn_weights_385_transpose_y_0"), val = bool(false)]; tensor attn_weights_385_cast_fp16 = matmul(transpose_x = attn_weights_385_transpose_x_0, transpose_y = attn_weights_385_transpose_y_0, x = var_8792_cast_fp16_0, y = var_8805_0)[name = string("attn_weights_385_cast_fp16")]; fp16 var_8808_to_fp16 = const()[name = string("op_8808_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_387_cast_fp16 = mul(x = attn_weights_385_cast_fp16, y = var_8808_to_fp16)[name = string("attn_weights_387_cast_fp16")]; tensor attn_weights_389_cast_fp16 = add(x = attn_weights_387_cast_fp16, y = attn_mask_1)[name = string("attn_weights_389_cast_fp16")]; int32 var_8812 = const()[name = string("op_8812"), val = int32(-2)]; tensor attn_weights_391_cast_fp16 = softmax(axis = var_8812, x = attn_weights_389_cast_fp16)[name = string("attn_weights_391_cast_fp16")]; bool var_8818_transpose_x_1 = const()[name = string("op_8818_transpose_x_1"), val = bool(true)]; bool var_8818_transpose_y_1 = const()[name = string("op_8818_transpose_y_1"), val = bool(false)]; tensor var_8818_cast_fp16 = matmul(transpose_x = var_8818_transpose_x_1, transpose_y = var_8818_transpose_y_1, x = attn_weights_391_cast_fp16, y = var_8802_cast_fp16_0)[name = string("op_8818_cast_fp16")]; bool attn_weights_393_transpose_x_0 = const()[name = string("attn_weights_393_transpose_x_0"), val = bool(false)]; bool attn_weights_393_transpose_y_0 = const()[name = string("attn_weights_393_transpose_y_0"), val = bool(false)]; tensor attn_weights_393_cast_fp16 = matmul(transpose_x = attn_weights_393_transpose_x_0, transpose_y = attn_weights_393_transpose_y_0, x = var_8792_cast_fp16_1, y = var_8805_1)[name = string("attn_weights_393_cast_fp16")]; fp16 var_8820_to_fp16 = const()[name = string("op_8820_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_395_cast_fp16 = mul(x = attn_weights_393_cast_fp16, y = var_8820_to_fp16)[name = string("attn_weights_395_cast_fp16")]; tensor attn_weights_397_cast_fp16 = add(x = attn_weights_395_cast_fp16, y = attn_mask_1)[name = string("attn_weights_397_cast_fp16")]; int32 var_8824 = const()[name = string("op_8824"), val = int32(-2)]; tensor attn_weights_399_cast_fp16 = softmax(axis = var_8824, x = attn_weights_397_cast_fp16)[name = string("attn_weights_399_cast_fp16")]; bool attn_output_193_transpose_x_1 = const()[name = string("attn_output_193_transpose_x_1"), val = bool(true)]; bool attn_output_193_transpose_y_1 = const()[name = string("attn_output_193_transpose_y_1"), val = bool(false)]; tensor attn_output_193_cast_fp16 = matmul(transpose_x = attn_output_193_transpose_x_1, transpose_y = attn_output_193_transpose_y_1, x = attn_weights_399_cast_fp16, y = var_8802_cast_fp16_1)[name = string("attn_output_193_cast_fp16")]; int32 var_8832 = const()[name = string("op_8832"), val = int32(1)]; bool attn_output_195_interleave_0 = const()[name = string("attn_output_195_interleave_0"), val = bool(false)]; tensor attn_output_195_cast_fp16 = concat(axis = var_8832, interleave = attn_output_195_interleave_0, values = (var_8818_cast_fp16, attn_output_193_cast_fp16))[name = string("attn_output_195_cast_fp16")]; tensor var_8836_perm_0 = const()[name = string("op_8836_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_299x = const()[name = string("concat_299x"), val = tensor([1, 2048, 1, -1])]; tensor var_8836_cast_fp16 = transpose(perm = var_8836_perm_0, x = attn_output_195_cast_fp16)[name = string("transpose_9")]; tensor attn_output_199_cast_fp16 = reshape(shape = concat_299x, x = var_8836_cast_fp16)[name = string("attn_output_199_cast_fp16")]; tensor hidden_states_243_strides_0 = const()[name = string("hidden_states_243_strides_0"), val = tensor([1, 1])]; string hidden_states_243_pad_type_0 = const()[name = string("hidden_states_243_pad_type_0"), val = string("valid")]; tensor hidden_states_243_pad_0 = const()[name = string("hidden_states_243_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_243_dilations_0 = const()[name = string("hidden_states_243_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_243_groups_0 = const()[name = string("hidden_states_243_groups_0"), val = int32(1)]; tensor hidden_states_243_cast_fp16 = conv(dilations = hidden_states_243_dilations_0, groups = hidden_states_243_groups_0, pad = hidden_states_243_pad_0, pad_type = hidden_states_243_pad_type_0, strides = hidden_states_243_strides_0, weight = layers_24_self_attn_o_proj_weight_cast_fp16, x = attn_output_199_cast_fp16)[name = string("hidden_states_243_cast_fp16")]; tensor hidden_states_245_cast_fp16 = add(x = hidden_states_239_cast_fp16, y = hidden_states_243_cast_fp16)[name = string("hidden_states_245_cast_fp16")]; fp16 const_248_promoted_to_fp16 = const()[name = string("const_248_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8869_cast_fp16 = mul(x = hidden_states_245_cast_fp16, y = const_248_promoted_to_fp16)[name = string("op_8869_cast_fp16")]; int32 var_8867 = const()[name = string("op_8867"), val = int32(1)]; bool doubled_197_interleave_0 = const()[name = string("doubled_197_interleave_0"), val = bool(false)]; tensor doubled_197_cast_fp16 = concat(axis = var_8867, interleave = doubled_197_interleave_0, values = (hidden_states_245_cast_fp16, var_8869_cast_fp16))[name = string("doubled_197_cast_fp16")]; tensor out_99_axes_0 = const()[name = string("out_99_axes_0"), val = tensor([1])]; tensor out_99_gamma_0_to_fp16 = const()[name = string("out_99_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540362432)))]; fp16 var_8879_to_fp16 = const()[name = string("op_8879_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_99_cast_fp16 = layer_norm(axes = out_99_axes_0, epsilon = var_8879_to_fp16, gamma = out_99_gamma_0_to_fp16, x = doubled_197_cast_fp16)[name = string("out_99_cast_fp16")]; tensor var_8890_split_sizes_0 = const()[name = string("op_8890_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8890_axis_0 = const()[name = string("op_8890_axis_0"), val = int32(1)]; tensor var_8890_cast_fp16_0, tensor var_8890_cast_fp16_1 = split(axis = var_8890_axis_0, split_sizes = var_8890_split_sizes_0, x = out_99_cast_fp16)[name = string("op_8890_cast_fp16")]; tensor input_49_strides_0 = const()[name = string("input_49_strides_0"), val = tensor([1, 1])]; string input_49_pad_type_0 = const()[name = string("input_49_pad_type_0"), val = string("valid")]; tensor input_49_pad_0 = const()[name = string("input_49_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_49_dilations_0 = const()[name = string("input_49_dilations_0"), val = tensor([1, 1])]; int32 input_49_groups_0 = const()[name = string("input_49_groups_0"), val = int32(1)]; tensor input_49_cast_fp16 = conv(dilations = input_49_dilations_0, groups = input_49_groups_0, pad = input_49_pad_0, pad_type = input_49_pad_type_0, strides = input_49_strides_0, weight = layers_24_mlp_gate_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("input_49_cast_fp16")]; tensor var_8907_cast_fp16 = silu(x = input_49_cast_fp16)[name = string("op_8907_cast_fp16")]; tensor var_8913_strides_0 = const()[name = string("op_8913_strides_0"), val = tensor([1, 1])]; string var_8913_pad_type_0 = const()[name = string("op_8913_pad_type_0"), val = string("valid")]; tensor var_8913_pad_0 = const()[name = string("op_8913_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_8913_dilations_0 = const()[name = string("op_8913_dilations_0"), val = tensor([1, 1])]; int32 var_8913_groups_0 = const()[name = string("op_8913_groups_0"), val = int32(1)]; tensor var_8913_cast_fp16 = conv(dilations = var_8913_dilations_0, groups = var_8913_groups_0, pad = var_8913_pad_0, pad_type = var_8913_pad_type_0, strides = var_8913_strides_0, weight = layers_24_mlp_up_proj_weight_cast_fp16, x = var_8890_cast_fp16_0)[name = string("op_8913_cast_fp16")]; tensor x_249_cast_fp16 = mul(x = var_8907_cast_fp16, y = var_8913_cast_fp16)[name = string("x_249_cast_fp16")]; tensor hidden_states_247_strides_0 = const()[name = string("hidden_states_247_strides_0"), val = tensor([1, 1])]; string hidden_states_247_pad_type_0 = const()[name = string("hidden_states_247_pad_type_0"), val = string("valid")]; tensor hidden_states_247_pad_0 = const()[name = string("hidden_states_247_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_247_dilations_0 = const()[name = string("hidden_states_247_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_247_groups_0 = const()[name = string("hidden_states_247_groups_0"), val = int32(1)]; tensor hidden_states_247_cast_fp16 = conv(dilations = hidden_states_247_dilations_0, groups = hidden_states_247_groups_0, pad = hidden_states_247_pad_0, pad_type = hidden_states_247_pad_type_0, strides = hidden_states_247_strides_0, weight = layers_24_mlp_down_proj_weight_cast_fp16, x = x_249_cast_fp16)[name = string("hidden_states_247_cast_fp16")]; tensor hidden_states_249_cast_fp16 = add(x = hidden_states_245_cast_fp16, y = hidden_states_247_cast_fp16)[name = string("hidden_states_249_cast_fp16")]; fp16 const_250_promoted_to_fp16 = const()[name = string("const_250_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_8931_cast_fp16 = mul(x = hidden_states_249_cast_fp16, y = const_250_promoted_to_fp16)[name = string("op_8931_cast_fp16")]; int32 var_8929 = const()[name = string("op_8929"), val = int32(1)]; bool doubled_201_interleave_0 = const()[name = string("doubled_201_interleave_0"), val = bool(false)]; tensor doubled_201_cast_fp16 = concat(axis = var_8929, interleave = doubled_201_interleave_0, values = (hidden_states_249_cast_fp16, var_8931_cast_fp16))[name = string("doubled_201_cast_fp16")]; tensor out_101_axes_0 = const()[name = string("out_101_axes_0"), val = tensor([1])]; tensor out_101_gamma_0_to_fp16 = const()[name = string("out_101_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540370688)))]; fp16 var_8941_to_fp16 = const()[name = string("op_8941_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_101_cast_fp16 = layer_norm(axes = out_101_axes_0, epsilon = var_8941_to_fp16, gamma = out_101_gamma_0_to_fp16, x = doubled_201_cast_fp16)[name = string("out_101_cast_fp16")]; tensor var_8952_split_sizes_0 = const()[name = string("op_8952_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_8952_axis_0 = const()[name = string("op_8952_axis_0"), val = int32(1)]; tensor var_8952_cast_fp16_0, tensor var_8952_cast_fp16_1 = split(axis = var_8952_axis_0, split_sizes = var_8952_split_sizes_0, x = out_101_cast_fp16)[name = string("op_8952_cast_fp16")]; tensor query_states_151_strides_0 = const()[name = string("query_states_151_strides_0"), val = tensor([1, 1])]; string query_states_151_pad_type_0 = const()[name = string("query_states_151_pad_type_0"), val = string("valid")]; tensor query_states_151_pad_0 = const()[name = string("query_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_151_dilations_0 = const()[name = string("query_states_151_dilations_0"), val = tensor([1, 1])]; int32 query_states_151_groups_0 = const()[name = string("query_states_151_groups_0"), val = int32(1)]; tensor query_states_151_cast_fp16 = conv(dilations = query_states_151_dilations_0, groups = query_states_151_groups_0, pad = query_states_151_pad_0, pad_type = query_states_151_pad_type_0, strides = query_states_151_strides_0, weight = layers_25_self_attn_q_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("query_states_151_cast_fp16")]; tensor key_states_251_strides_0 = const()[name = string("key_states_251_strides_0"), val = tensor([1, 1])]; string key_states_251_pad_type_0 = const()[name = string("key_states_251_pad_type_0"), val = string("valid")]; tensor key_states_251_pad_0 = const()[name = string("key_states_251_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_251_dilations_0 = const()[name = string("key_states_251_dilations_0"), val = tensor([1, 1])]; int32 key_states_251_groups_0 = const()[name = string("key_states_251_groups_0"), val = int32(1)]; tensor key_states_251_cast_fp16 = conv(dilations = key_states_251_dilations_0, groups = key_states_251_groups_0, pad = key_states_251_pad_0, pad_type = key_states_251_pad_type_0, strides = key_states_251_strides_0, weight = layers_25_self_attn_k_proj_weight_cast_fp16, x = var_8952_cast_fp16_0)[name = string("key_states_251_cast_fp16")]; tensor layers_25_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_25_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1540378944)))]; tensor value_states_151_strides_0 = const()[name = string("value_states_151_strides_0"), val = tensor([1, 1])]; string value_states_151_pad_type_0 = const()[name = string("value_states_151_pad_type_0"), val = string("valid")]; tensor value_states_151_pad_0 = const()[name = string("value_states_151_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_151_dilations_0 = const()[name = string("value_states_151_dilations_0"), val = tensor([1, 1])]; int32 value_states_151_groups_0 = const()[name = string("value_states_151_groups_0"), val = int32(1)]; tensor value_states_151_cast_fp16 = conv(dilations = value_states_151_dilations_0, groups = value_states_151_groups_0, pad = value_states_151_pad_0, pad_type = value_states_151_pad_type_0, strides = value_states_151_strides_0, weight = layers_25_self_attn_v_proj_weight_to_fp16, x = var_8952_cast_fp16_0)[name = string("value_states_151_cast_fp16")]; tensor concat_300x = const()[name = string("concat_300x"), val = tensor([1, 16, 128, -1])]; tensor x_251_cast_fp16 = reshape(shape = concat_300x, x = query_states_151_cast_fp16)[name = string("x_251_cast_fp16")]; tensor concat_301x = const()[name = string("concat_301x"), val = tensor([1, 2, 128, -1])]; tensor var_9009_cast_fp16 = reshape(shape = concat_301x, x = key_states_251_cast_fp16)[name = string("op_9009_cast_fp16")]; tensor concat_302x = const()[name = string("concat_302x"), val = tensor([1, 2, 128, -1])]; tensor var_9016_cast_fp16 = reshape(shape = concat_302x, x = value_states_151_cast_fp16)[name = string("op_9016_cast_fp16")]; tensor var_9020_cast_fp16 = mul(x = x_251_cast_fp16, y = var_869_cast_fp16)[name = string("op_9020_cast_fp16")]; tensor var_9021_split_sizes_0 = const()[name = string("op_9021_split_sizes_0"), val = tensor([64, 64])]; int32 var_9021_axis_0 = const()[name = string("op_9021_axis_0"), val = int32(-2)]; tensor var_9021_cast_fp16_0, tensor var_9021_cast_fp16_1 = split(axis = var_9021_axis_0, split_sizes = var_9021_split_sizes_0, x = x_251_cast_fp16)[name = string("op_9021_cast_fp16")]; fp16 const_252_promoted_to_fp16 = const()[name = string("const_252_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9023_cast_fp16 = mul(x = var_9021_cast_fp16_1, y = const_252_promoted_to_fp16)[name = string("op_9023_cast_fp16")]; int32 var_9025 = const()[name = string("op_9025"), val = int32(-2)]; bool var_9026_interleave_0 = const()[name = string("op_9026_interleave_0"), val = bool(false)]; tensor var_9026_cast_fp16 = concat(axis = var_9025, interleave = var_9026_interleave_0, values = (var_9023_cast_fp16, var_9021_cast_fp16_0))[name = string("op_9026_cast_fp16")]; tensor var_9027_cast_fp16 = mul(x = var_9026_cast_fp16, y = var_878_cast_fp16)[name = string("op_9027_cast_fp16")]; tensor query_states_153_cast_fp16 = add(x = var_9020_cast_fp16, y = var_9027_cast_fp16)[name = string("query_states_153_cast_fp16")]; tensor var_9033_cast_fp16 = mul(x = var_9009_cast_fp16, y = var_869_cast_fp16)[name = string("op_9033_cast_fp16")]; tensor var_9034_split_sizes_0 = const()[name = string("op_9034_split_sizes_0"), val = tensor([64, 64])]; int32 var_9034_axis_0 = const()[name = string("op_9034_axis_0"), val = int32(-2)]; tensor var_9034_cast_fp16_0, tensor var_9034_cast_fp16_1 = split(axis = var_9034_axis_0, split_sizes = var_9034_split_sizes_0, x = var_9009_cast_fp16)[name = string("op_9034_cast_fp16")]; fp16 const_253_promoted_to_fp16 = const()[name = string("const_253_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9036_cast_fp16 = mul(x = var_9034_cast_fp16_1, y = const_253_promoted_to_fp16)[name = string("op_9036_cast_fp16")]; int32 var_9038 = const()[name = string("op_9038"), val = int32(-2)]; bool var_9039_interleave_0 = const()[name = string("op_9039_interleave_0"), val = bool(false)]; tensor var_9039_cast_fp16 = concat(axis = var_9038, interleave = var_9039_interleave_0, values = (var_9036_cast_fp16, var_9034_cast_fp16_0))[name = string("op_9039_cast_fp16")]; tensor var_9040_cast_fp16 = mul(x = var_9039_cast_fp16, y = var_878_cast_fp16)[name = string("op_9040_cast_fp16")]; tensor key_states_255_cast_fp16 = add(x = var_9033_cast_fp16, y = var_9040_cast_fp16)[name = string("key_states_255_cast_fp16")]; tensor expand_dims_300 = const()[name = string("expand_dims_300"), val = tensor([25])]; tensor expand_dims_301 = const()[name = string("expand_dims_301"), val = tensor([0])]; tensor expand_dims_303 = const()[name = string("expand_dims_303"), val = tensor([0])]; int32 concat_305_axis_0 = const()[name = string("concat_305_axis_0"), val = int32(0)]; bool concat_305_interleave_0 = const()[name = string("concat_305_interleave_0"), val = bool(false)]; tensor concat_305 = concat(axis = concat_305_axis_0, interleave = concat_305_interleave_0, values = (expand_dims_300, expand_dims_301, position_id, expand_dims_303))[name = string("concat_305")]; tensor expand_dims_304 = const()[name = string("expand_dims_304"), val = tensor([26])]; tensor concat_306_values1_0 = const()[name = string("concat_306_values1_0"), val = tensor([0])]; tensor concat_306_values3_0 = const()[name = string("concat_306_values3_0"), val = tensor([0])]; int32 concat_306_axis_0 = const()[name = string("concat_306_axis_0"), val = int32(0)]; bool concat_306_interleave_0 = const()[name = string("concat_306_interleave_0"), val = bool(false)]; tensor concat_306 = concat(axis = concat_306_axis_0, interleave = concat_306_interleave_0, values = (expand_dims_304, concat_306_values1_0, cache_position_end, concat_306_values3_0))[name = string("concat_306")]; tensor key_states_257_perm_0 = const()[name = string("key_states_257_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_26_stride_0 = const()[name = string("key_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_257_cast_fp16 = transpose(perm = key_states_257_perm_0, x = key_states_255_cast_fp16)[name = string("transpose_8")]; tensor key_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = key_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = key_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_26_squeeze_mask_0, stride = key_cache_internal_tensor_assign_26_stride_0, update = key_states_257_cast_fp16, x = coreml_update_state_48)[name = string("key_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_26_cast_fp16, input = key_cache)[name = string("coreml_update_state_50_write_state")]; tensor coreml_update_state_50 = read_state(input = key_cache)[name = string("coreml_update_state_50")]; tensor value_states_153_perm_0 = const()[name = string("value_states_153_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_26_stride_0 = const()[name = string("value_cache_internal_tensor_assign_26_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_26_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_26_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_26_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_26_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_153_cast_fp16 = transpose(perm = value_states_153_perm_0, x = var_9016_cast_fp16)[name = string("transpose_7")]; tensor value_cache_internal_tensor_assign_26_cast_fp16 = slice_update(begin = concat_305, begin_mask = value_cache_internal_tensor_assign_26_begin_mask_0, end = concat_306, end_mask = value_cache_internal_tensor_assign_26_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_26_squeeze_mask_0, stride = value_cache_internal_tensor_assign_26_stride_0, update = value_states_153_cast_fp16, x = coreml_update_state_49)[name = string("value_cache_internal_tensor_assign_26_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_26_cast_fp16, input = value_cache)[name = string("coreml_update_state_51_write_state")]; tensor coreml_update_state_51 = read_state(input = value_cache)[name = string("coreml_update_state_51")]; tensor var_9110_begin_0 = const()[name = string("op_9110_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9110_end_0 = const()[name = string("op_9110_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9110_end_mask_0 = const()[name = string("op_9110_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9110_cast_fp16 = slice_by_index(begin = var_9110_begin_0, end = var_9110_end_0, end_mask = var_9110_end_mask_0, x = coreml_update_state_50)[name = string("op_9110_cast_fp16")]; tensor tile_50 = const()[name = string("tile_50"), val = tensor([1, 1])]; int32 var_9113_axis_0 = const()[name = string("op_9113_axis_0"), val = int32(1)]; tensor var_9113_cast_fp16_0, tensor var_9113_cast_fp16_1 = split(axis = var_9113_axis_0, split_sizes = tile_50, x = var_9110_cast_fp16)[name = string("op_9113_cast_fp16")]; tensor var_9120_begin_0 = const()[name = string("op_9120_begin_0"), val = tensor([25, 0, 0, 0])]; tensor var_9120_end_0 = const()[name = string("op_9120_end_0"), val = tensor([26, 2, 2048, 128])]; tensor var_9120_end_mask_0 = const()[name = string("op_9120_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9120_cast_fp16 = slice_by_index(begin = var_9120_begin_0, end = var_9120_end_0, end_mask = var_9120_end_mask_0, x = coreml_update_state_51)[name = string("op_9120_cast_fp16")]; tensor tile_51 = const()[name = string("tile_51"), val = tensor([1, 1])]; int32 var_9123_axis_0 = const()[name = string("op_9123_axis_0"), val = int32(1)]; tensor var_9123_cast_fp16_0, tensor var_9123_cast_fp16_1 = split(axis = var_9123_axis_0, split_sizes = tile_51, x = var_9120_cast_fp16)[name = string("op_9123_cast_fp16")]; tensor var_9126_split_sizes_0 = const()[name = string("op_9126_split_sizes_0"), val = tensor([8, 8])]; int32 var_9126_axis_0 = const()[name = string("op_9126_axis_0"), val = int32(1)]; tensor var_9126_0, tensor var_9126_1 = split(axis = var_9126_axis_0, split_sizes = var_9126_split_sizes_0, x = query_states_153_cast_fp16)[name = string("op_9126")]; bool attn_weights_401_transpose_x_0 = const()[name = string("attn_weights_401_transpose_x_0"), val = bool(false)]; bool attn_weights_401_transpose_y_0 = const()[name = string("attn_weights_401_transpose_y_0"), val = bool(false)]; tensor attn_weights_401_cast_fp16 = matmul(transpose_x = attn_weights_401_transpose_x_0, transpose_y = attn_weights_401_transpose_y_0, x = var_9113_cast_fp16_0, y = var_9126_0)[name = string("attn_weights_401_cast_fp16")]; fp16 var_9129_to_fp16 = const()[name = string("op_9129_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_403_cast_fp16 = mul(x = attn_weights_401_cast_fp16, y = var_9129_to_fp16)[name = string("attn_weights_403_cast_fp16")]; tensor attn_weights_405_cast_fp16 = add(x = attn_weights_403_cast_fp16, y = attn_mask_1)[name = string("attn_weights_405_cast_fp16")]; int32 var_9133 = const()[name = string("op_9133"), val = int32(-2)]; tensor attn_weights_407_cast_fp16 = softmax(axis = var_9133, x = attn_weights_405_cast_fp16)[name = string("attn_weights_407_cast_fp16")]; bool var_9139_transpose_x_1 = const()[name = string("op_9139_transpose_x_1"), val = bool(true)]; bool var_9139_transpose_y_1 = const()[name = string("op_9139_transpose_y_1"), val = bool(false)]; tensor var_9139_cast_fp16 = matmul(transpose_x = var_9139_transpose_x_1, transpose_y = var_9139_transpose_y_1, x = attn_weights_407_cast_fp16, y = var_9123_cast_fp16_0)[name = string("op_9139_cast_fp16")]; bool attn_weights_409_transpose_x_0 = const()[name = string("attn_weights_409_transpose_x_0"), val = bool(false)]; bool attn_weights_409_transpose_y_0 = const()[name = string("attn_weights_409_transpose_y_0"), val = bool(false)]; tensor attn_weights_409_cast_fp16 = matmul(transpose_x = attn_weights_409_transpose_x_0, transpose_y = attn_weights_409_transpose_y_0, x = var_9113_cast_fp16_1, y = var_9126_1)[name = string("attn_weights_409_cast_fp16")]; fp16 var_9141_to_fp16 = const()[name = string("op_9141_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_411_cast_fp16 = mul(x = attn_weights_409_cast_fp16, y = var_9141_to_fp16)[name = string("attn_weights_411_cast_fp16")]; tensor attn_weights_413_cast_fp16 = add(x = attn_weights_411_cast_fp16, y = attn_mask_1)[name = string("attn_weights_413_cast_fp16")]; int32 var_9145 = const()[name = string("op_9145"), val = int32(-2)]; tensor attn_weights_415_cast_fp16 = softmax(axis = var_9145, x = attn_weights_413_cast_fp16)[name = string("attn_weights_415_cast_fp16")]; bool attn_output_201_transpose_x_1 = const()[name = string("attn_output_201_transpose_x_1"), val = bool(true)]; bool attn_output_201_transpose_y_1 = const()[name = string("attn_output_201_transpose_y_1"), val = bool(false)]; tensor attn_output_201_cast_fp16 = matmul(transpose_x = attn_output_201_transpose_x_1, transpose_y = attn_output_201_transpose_y_1, x = attn_weights_415_cast_fp16, y = var_9123_cast_fp16_1)[name = string("attn_output_201_cast_fp16")]; int32 var_9153 = const()[name = string("op_9153"), val = int32(1)]; bool attn_output_203_interleave_0 = const()[name = string("attn_output_203_interleave_0"), val = bool(false)]; tensor attn_output_203_cast_fp16 = concat(axis = var_9153, interleave = attn_output_203_interleave_0, values = (var_9139_cast_fp16, attn_output_201_cast_fp16))[name = string("attn_output_203_cast_fp16")]; tensor var_9157_perm_0 = const()[name = string("op_9157_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_311x = const()[name = string("concat_311x"), val = tensor([1, 2048, 1, -1])]; tensor var_9157_cast_fp16 = transpose(perm = var_9157_perm_0, x = attn_output_203_cast_fp16)[name = string("transpose_6")]; tensor attn_output_207_cast_fp16 = reshape(shape = concat_311x, x = var_9157_cast_fp16)[name = string("attn_output_207_cast_fp16")]; tensor hidden_states_253_strides_0 = const()[name = string("hidden_states_253_strides_0"), val = tensor([1, 1])]; string hidden_states_253_pad_type_0 = const()[name = string("hidden_states_253_pad_type_0"), val = string("valid")]; tensor hidden_states_253_pad_0 = const()[name = string("hidden_states_253_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_253_dilations_0 = const()[name = string("hidden_states_253_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_253_groups_0 = const()[name = string("hidden_states_253_groups_0"), val = int32(1)]; tensor hidden_states_253_cast_fp16 = conv(dilations = hidden_states_253_dilations_0, groups = hidden_states_253_groups_0, pad = hidden_states_253_pad_0, pad_type = hidden_states_253_pad_type_0, strides = hidden_states_253_strides_0, weight = layers_25_self_attn_o_proj_weight_cast_fp16, x = attn_output_207_cast_fp16)[name = string("hidden_states_253_cast_fp16")]; tensor hidden_states_255_cast_fp16 = add(x = hidden_states_249_cast_fp16, y = hidden_states_253_cast_fp16)[name = string("hidden_states_255_cast_fp16")]; fp16 const_258_promoted_to_fp16 = const()[name = string("const_258_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9190_cast_fp16 = mul(x = hidden_states_255_cast_fp16, y = const_258_promoted_to_fp16)[name = string("op_9190_cast_fp16")]; int32 var_9188 = const()[name = string("op_9188"), val = int32(1)]; bool doubled_205_interleave_0 = const()[name = string("doubled_205_interleave_0"), val = bool(false)]; tensor doubled_205_cast_fp16 = concat(axis = var_9188, interleave = doubled_205_interleave_0, values = (hidden_states_255_cast_fp16, var_9190_cast_fp16))[name = string("doubled_205_cast_fp16")]; tensor out_103_axes_0 = const()[name = string("out_103_axes_0"), val = tensor([1])]; tensor out_103_gamma_0_to_fp16 = const()[name = string("out_103_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541427584)))]; fp16 var_9200_to_fp16 = const()[name = string("op_9200_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_103_cast_fp16 = layer_norm(axes = out_103_axes_0, epsilon = var_9200_to_fp16, gamma = out_103_gamma_0_to_fp16, x = doubled_205_cast_fp16)[name = string("out_103_cast_fp16")]; tensor var_9211_split_sizes_0 = const()[name = string("op_9211_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9211_axis_0 = const()[name = string("op_9211_axis_0"), val = int32(1)]; tensor var_9211_cast_fp16_0, tensor var_9211_cast_fp16_1 = split(axis = var_9211_axis_0, split_sizes = var_9211_split_sizes_0, x = out_103_cast_fp16)[name = string("op_9211_cast_fp16")]; tensor input_51_strides_0 = const()[name = string("input_51_strides_0"), val = tensor([1, 1])]; string input_51_pad_type_0 = const()[name = string("input_51_pad_type_0"), val = string("valid")]; tensor input_51_pad_0 = const()[name = string("input_51_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_51_dilations_0 = const()[name = string("input_51_dilations_0"), val = tensor([1, 1])]; int32 input_51_groups_0 = const()[name = string("input_51_groups_0"), val = int32(1)]; tensor input_51_cast_fp16 = conv(dilations = input_51_dilations_0, groups = input_51_groups_0, pad = input_51_pad_0, pad_type = input_51_pad_type_0, strides = input_51_strides_0, weight = layers_25_mlp_gate_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("input_51_cast_fp16")]; tensor var_9228_cast_fp16 = silu(x = input_51_cast_fp16)[name = string("op_9228_cast_fp16")]; tensor var_9234_strides_0 = const()[name = string("op_9234_strides_0"), val = tensor([1, 1])]; string var_9234_pad_type_0 = const()[name = string("op_9234_pad_type_0"), val = string("valid")]; tensor var_9234_pad_0 = const()[name = string("op_9234_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9234_dilations_0 = const()[name = string("op_9234_dilations_0"), val = tensor([1, 1])]; int32 var_9234_groups_0 = const()[name = string("op_9234_groups_0"), val = int32(1)]; tensor var_9234_cast_fp16 = conv(dilations = var_9234_dilations_0, groups = var_9234_groups_0, pad = var_9234_pad_0, pad_type = var_9234_pad_type_0, strides = var_9234_strides_0, weight = layers_25_mlp_up_proj_weight_cast_fp16, x = var_9211_cast_fp16_0)[name = string("op_9234_cast_fp16")]; tensor x_259_cast_fp16 = mul(x = var_9228_cast_fp16, y = var_9234_cast_fp16)[name = string("x_259_cast_fp16")]; tensor layers_25_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_25_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1541435840)))]; tensor hidden_states_257_strides_0 = const()[name = string("hidden_states_257_strides_0"), val = tensor([1, 1])]; string hidden_states_257_pad_type_0 = const()[name = string("hidden_states_257_pad_type_0"), val = string("valid")]; tensor hidden_states_257_pad_0 = const()[name = string("hidden_states_257_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_257_dilations_0 = const()[name = string("hidden_states_257_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_257_groups_0 = const()[name = string("hidden_states_257_groups_0"), val = int32(1)]; tensor hidden_states_257_cast_fp16 = conv(dilations = hidden_states_257_dilations_0, groups = hidden_states_257_groups_0, pad = hidden_states_257_pad_0, pad_type = hidden_states_257_pad_type_0, strides = hidden_states_257_strides_0, weight = layers_25_mlp_down_proj_weight_to_fp16, x = x_259_cast_fp16)[name = string("hidden_states_257_cast_fp16")]; tensor hidden_states_259_cast_fp16 = add(x = hidden_states_255_cast_fp16, y = hidden_states_257_cast_fp16)[name = string("hidden_states_259_cast_fp16")]; fp16 const_260_promoted_to_fp16 = const()[name = string("const_260_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9252_cast_fp16 = mul(x = hidden_states_259_cast_fp16, y = const_260_promoted_to_fp16)[name = string("op_9252_cast_fp16")]; int32 var_9250 = const()[name = string("op_9250"), val = int32(1)]; bool doubled_209_interleave_0 = const()[name = string("doubled_209_interleave_0"), val = bool(false)]; tensor doubled_209_cast_fp16 = concat(axis = var_9250, interleave = doubled_209_interleave_0, values = (hidden_states_259_cast_fp16, var_9252_cast_fp16))[name = string("doubled_209_cast_fp16")]; tensor out_105_axes_0 = const()[name = string("out_105_axes_0"), val = tensor([1])]; tensor out_105_gamma_0_to_fp16 = const()[name = string("out_105_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566601728)))]; fp16 var_9262_to_fp16 = const()[name = string("op_9262_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_105_cast_fp16 = layer_norm(axes = out_105_axes_0, epsilon = var_9262_to_fp16, gamma = out_105_gamma_0_to_fp16, x = doubled_209_cast_fp16)[name = string("out_105_cast_fp16")]; tensor var_9273_split_sizes_0 = const()[name = string("op_9273_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9273_axis_0 = const()[name = string("op_9273_axis_0"), val = int32(1)]; tensor var_9273_cast_fp16_0, tensor var_9273_cast_fp16_1 = split(axis = var_9273_axis_0, split_sizes = var_9273_split_sizes_0, x = out_105_cast_fp16)[name = string("op_9273_cast_fp16")]; tensor query_states_157_strides_0 = const()[name = string("query_states_157_strides_0"), val = tensor([1, 1])]; string query_states_157_pad_type_0 = const()[name = string("query_states_157_pad_type_0"), val = string("valid")]; tensor query_states_157_pad_0 = const()[name = string("query_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_157_dilations_0 = const()[name = string("query_states_157_dilations_0"), val = tensor([1, 1])]; int32 query_states_157_groups_0 = const()[name = string("query_states_157_groups_0"), val = int32(1)]; tensor query_states_157_cast_fp16 = conv(dilations = query_states_157_dilations_0, groups = query_states_157_groups_0, pad = query_states_157_pad_0, pad_type = query_states_157_pad_type_0, strides = query_states_157_strides_0, weight = layers_26_self_attn_q_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("query_states_157_cast_fp16")]; tensor key_states_261_strides_0 = const()[name = string("key_states_261_strides_0"), val = tensor([1, 1])]; string key_states_261_pad_type_0 = const()[name = string("key_states_261_pad_type_0"), val = string("valid")]; tensor key_states_261_pad_0 = const()[name = string("key_states_261_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_261_dilations_0 = const()[name = string("key_states_261_dilations_0"), val = tensor([1, 1])]; int32 key_states_261_groups_0 = const()[name = string("key_states_261_groups_0"), val = int32(1)]; tensor key_states_261_cast_fp16 = conv(dilations = key_states_261_dilations_0, groups = key_states_261_groups_0, pad = key_states_261_pad_0, pad_type = key_states_261_pad_type_0, strides = key_states_261_strides_0, weight = layers_26_self_attn_k_proj_weight_cast_fp16, x = var_9273_cast_fp16_0)[name = string("key_states_261_cast_fp16")]; tensor layers_26_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_26_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1566609984)))]; tensor value_states_157_strides_0 = const()[name = string("value_states_157_strides_0"), val = tensor([1, 1])]; string value_states_157_pad_type_0 = const()[name = string("value_states_157_pad_type_0"), val = string("valid")]; tensor value_states_157_pad_0 = const()[name = string("value_states_157_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_157_dilations_0 = const()[name = string("value_states_157_dilations_0"), val = tensor([1, 1])]; int32 value_states_157_groups_0 = const()[name = string("value_states_157_groups_0"), val = int32(1)]; tensor value_states_157_cast_fp16 = conv(dilations = value_states_157_dilations_0, groups = value_states_157_groups_0, pad = value_states_157_pad_0, pad_type = value_states_157_pad_type_0, strides = value_states_157_strides_0, weight = layers_26_self_attn_v_proj_weight_to_fp16, x = var_9273_cast_fp16_0)[name = string("value_states_157_cast_fp16")]; tensor concat_312x = const()[name = string("concat_312x"), val = tensor([1, 16, 128, -1])]; tensor x_261_cast_fp16 = reshape(shape = concat_312x, x = query_states_157_cast_fp16)[name = string("x_261_cast_fp16")]; tensor concat_313x = const()[name = string("concat_313x"), val = tensor([1, 2, 128, -1])]; tensor var_9330_cast_fp16 = reshape(shape = concat_313x, x = key_states_261_cast_fp16)[name = string("op_9330_cast_fp16")]; tensor concat_314x = const()[name = string("concat_314x"), val = tensor([1, 2, 128, -1])]; tensor var_9337_cast_fp16 = reshape(shape = concat_314x, x = value_states_157_cast_fp16)[name = string("op_9337_cast_fp16")]; tensor var_9341_cast_fp16 = mul(x = x_261_cast_fp16, y = var_869_cast_fp16)[name = string("op_9341_cast_fp16")]; tensor var_9342_split_sizes_0 = const()[name = string("op_9342_split_sizes_0"), val = tensor([64, 64])]; int32 var_9342_axis_0 = const()[name = string("op_9342_axis_0"), val = int32(-2)]; tensor var_9342_cast_fp16_0, tensor var_9342_cast_fp16_1 = split(axis = var_9342_axis_0, split_sizes = var_9342_split_sizes_0, x = x_261_cast_fp16)[name = string("op_9342_cast_fp16")]; fp16 const_262_promoted_to_fp16 = const()[name = string("const_262_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9344_cast_fp16 = mul(x = var_9342_cast_fp16_1, y = const_262_promoted_to_fp16)[name = string("op_9344_cast_fp16")]; int32 var_9346 = const()[name = string("op_9346"), val = int32(-2)]; bool var_9347_interleave_0 = const()[name = string("op_9347_interleave_0"), val = bool(false)]; tensor var_9347_cast_fp16 = concat(axis = var_9346, interleave = var_9347_interleave_0, values = (var_9344_cast_fp16, var_9342_cast_fp16_0))[name = string("op_9347_cast_fp16")]; tensor var_9348_cast_fp16 = mul(x = var_9347_cast_fp16, y = var_878_cast_fp16)[name = string("op_9348_cast_fp16")]; tensor query_states_159_cast_fp16 = add(x = var_9341_cast_fp16, y = var_9348_cast_fp16)[name = string("query_states_159_cast_fp16")]; tensor var_9354_cast_fp16 = mul(x = var_9330_cast_fp16, y = var_869_cast_fp16)[name = string("op_9354_cast_fp16")]; tensor var_9355_split_sizes_0 = const()[name = string("op_9355_split_sizes_0"), val = tensor([64, 64])]; int32 var_9355_axis_0 = const()[name = string("op_9355_axis_0"), val = int32(-2)]; tensor var_9355_cast_fp16_0, tensor var_9355_cast_fp16_1 = split(axis = var_9355_axis_0, split_sizes = var_9355_split_sizes_0, x = var_9330_cast_fp16)[name = string("op_9355_cast_fp16")]; fp16 const_263_promoted_to_fp16 = const()[name = string("const_263_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9357_cast_fp16 = mul(x = var_9355_cast_fp16_1, y = const_263_promoted_to_fp16)[name = string("op_9357_cast_fp16")]; int32 var_9359 = const()[name = string("op_9359"), val = int32(-2)]; bool var_9360_interleave_0 = const()[name = string("op_9360_interleave_0"), val = bool(false)]; tensor var_9360_cast_fp16 = concat(axis = var_9359, interleave = var_9360_interleave_0, values = (var_9357_cast_fp16, var_9355_cast_fp16_0))[name = string("op_9360_cast_fp16")]; tensor var_9361_cast_fp16 = mul(x = var_9360_cast_fp16, y = var_878_cast_fp16)[name = string("op_9361_cast_fp16")]; tensor key_states_265_cast_fp16 = add(x = var_9354_cast_fp16, y = var_9361_cast_fp16)[name = string("key_states_265_cast_fp16")]; tensor expand_dims_312 = const()[name = string("expand_dims_312"), val = tensor([26])]; tensor expand_dims_313 = const()[name = string("expand_dims_313"), val = tensor([0])]; tensor expand_dims_315 = const()[name = string("expand_dims_315"), val = tensor([0])]; int32 concat_317_axis_0 = const()[name = string("concat_317_axis_0"), val = int32(0)]; bool concat_317_interleave_0 = const()[name = string("concat_317_interleave_0"), val = bool(false)]; tensor concat_317 = concat(axis = concat_317_axis_0, interleave = concat_317_interleave_0, values = (expand_dims_312, expand_dims_313, position_id, expand_dims_315))[name = string("concat_317")]; tensor expand_dims_316 = const()[name = string("expand_dims_316"), val = tensor([27])]; tensor concat_318_values1_0 = const()[name = string("concat_318_values1_0"), val = tensor([0])]; tensor concat_318_values3_0 = const()[name = string("concat_318_values3_0"), val = tensor([0])]; int32 concat_318_axis_0 = const()[name = string("concat_318_axis_0"), val = int32(0)]; bool concat_318_interleave_0 = const()[name = string("concat_318_interleave_0"), val = bool(false)]; tensor concat_318 = concat(axis = concat_318_axis_0, interleave = concat_318_interleave_0, values = (expand_dims_316, concat_318_values1_0, cache_position_end, concat_318_values3_0))[name = string("concat_318")]; tensor key_states_267_perm_0 = const()[name = string("key_states_267_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_27_stride_0 = const()[name = string("key_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_267_cast_fp16 = transpose(perm = key_states_267_perm_0, x = key_states_265_cast_fp16)[name = string("transpose_5")]; tensor key_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = key_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = key_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_27_squeeze_mask_0, stride = key_cache_internal_tensor_assign_27_stride_0, update = key_states_267_cast_fp16, x = coreml_update_state_50)[name = string("key_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_27_cast_fp16, input = key_cache)[name = string("coreml_update_state_52_write_state")]; tensor coreml_update_state_52 = read_state(input = key_cache)[name = string("coreml_update_state_52")]; tensor value_states_159_perm_0 = const()[name = string("value_states_159_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_27_stride_0 = const()[name = string("value_cache_internal_tensor_assign_27_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_27_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_27_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_27_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_27_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_159_cast_fp16 = transpose(perm = value_states_159_perm_0, x = var_9337_cast_fp16)[name = string("transpose_4")]; tensor value_cache_internal_tensor_assign_27_cast_fp16 = slice_update(begin = concat_317, begin_mask = value_cache_internal_tensor_assign_27_begin_mask_0, end = concat_318, end_mask = value_cache_internal_tensor_assign_27_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_27_squeeze_mask_0, stride = value_cache_internal_tensor_assign_27_stride_0, update = value_states_159_cast_fp16, x = coreml_update_state_51)[name = string("value_cache_internal_tensor_assign_27_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_27_cast_fp16, input = value_cache)[name = string("coreml_update_state_53_write_state")]; tensor coreml_update_state_53 = read_state(input = value_cache)[name = string("coreml_update_state_53")]; tensor var_9431_begin_0 = const()[name = string("op_9431_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9431_end_0 = const()[name = string("op_9431_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9431_end_mask_0 = const()[name = string("op_9431_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9431_cast_fp16 = slice_by_index(begin = var_9431_begin_0, end = var_9431_end_0, end_mask = var_9431_end_mask_0, x = coreml_update_state_52)[name = string("op_9431_cast_fp16")]; tensor tile_52 = const()[name = string("tile_52"), val = tensor([1, 1])]; int32 var_9434_axis_0 = const()[name = string("op_9434_axis_0"), val = int32(1)]; tensor var_9434_cast_fp16_0, tensor var_9434_cast_fp16_1 = split(axis = var_9434_axis_0, split_sizes = tile_52, x = var_9431_cast_fp16)[name = string("op_9434_cast_fp16")]; tensor var_9441_begin_0 = const()[name = string("op_9441_begin_0"), val = tensor([26, 0, 0, 0])]; tensor var_9441_end_0 = const()[name = string("op_9441_end_0"), val = tensor([27, 2, 2048, 128])]; tensor var_9441_end_mask_0 = const()[name = string("op_9441_end_mask_0"), val = tensor([false, true, true, true])]; tensor var_9441_cast_fp16 = slice_by_index(begin = var_9441_begin_0, end = var_9441_end_0, end_mask = var_9441_end_mask_0, x = coreml_update_state_53)[name = string("op_9441_cast_fp16")]; tensor tile_53 = const()[name = string("tile_53"), val = tensor([1, 1])]; int32 var_9444_axis_0 = const()[name = string("op_9444_axis_0"), val = int32(1)]; tensor var_9444_cast_fp16_0, tensor var_9444_cast_fp16_1 = split(axis = var_9444_axis_0, split_sizes = tile_53, x = var_9441_cast_fp16)[name = string("op_9444_cast_fp16")]; tensor var_9447_split_sizes_0 = const()[name = string("op_9447_split_sizes_0"), val = tensor([8, 8])]; int32 var_9447_axis_0 = const()[name = string("op_9447_axis_0"), val = int32(1)]; tensor var_9447_0, tensor var_9447_1 = split(axis = var_9447_axis_0, split_sizes = var_9447_split_sizes_0, x = query_states_159_cast_fp16)[name = string("op_9447")]; bool attn_weights_417_transpose_x_0 = const()[name = string("attn_weights_417_transpose_x_0"), val = bool(false)]; bool attn_weights_417_transpose_y_0 = const()[name = string("attn_weights_417_transpose_y_0"), val = bool(false)]; tensor attn_weights_417_cast_fp16 = matmul(transpose_x = attn_weights_417_transpose_x_0, transpose_y = attn_weights_417_transpose_y_0, x = var_9434_cast_fp16_0, y = var_9447_0)[name = string("attn_weights_417_cast_fp16")]; fp16 var_9450_to_fp16 = const()[name = string("op_9450_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_419_cast_fp16 = mul(x = attn_weights_417_cast_fp16, y = var_9450_to_fp16)[name = string("attn_weights_419_cast_fp16")]; tensor attn_weights_421_cast_fp16 = add(x = attn_weights_419_cast_fp16, y = attn_mask_1)[name = string("attn_weights_421_cast_fp16")]; int32 var_9454 = const()[name = string("op_9454"), val = int32(-2)]; tensor attn_weights_423_cast_fp16 = softmax(axis = var_9454, x = attn_weights_421_cast_fp16)[name = string("attn_weights_423_cast_fp16")]; bool var_9460_transpose_x_1 = const()[name = string("op_9460_transpose_x_1"), val = bool(true)]; bool var_9460_transpose_y_1 = const()[name = string("op_9460_transpose_y_1"), val = bool(false)]; tensor var_9460_cast_fp16 = matmul(transpose_x = var_9460_transpose_x_1, transpose_y = var_9460_transpose_y_1, x = attn_weights_423_cast_fp16, y = var_9444_cast_fp16_0)[name = string("op_9460_cast_fp16")]; bool attn_weights_425_transpose_x_0 = const()[name = string("attn_weights_425_transpose_x_0"), val = bool(false)]; bool attn_weights_425_transpose_y_0 = const()[name = string("attn_weights_425_transpose_y_0"), val = bool(false)]; tensor attn_weights_425_cast_fp16 = matmul(transpose_x = attn_weights_425_transpose_x_0, transpose_y = attn_weights_425_transpose_y_0, x = var_9434_cast_fp16_1, y = var_9447_1)[name = string("attn_weights_425_cast_fp16")]; fp16 var_9462_to_fp16 = const()[name = string("op_9462_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_427_cast_fp16 = mul(x = attn_weights_425_cast_fp16, y = var_9462_to_fp16)[name = string("attn_weights_427_cast_fp16")]; tensor attn_weights_429_cast_fp16 = add(x = attn_weights_427_cast_fp16, y = attn_mask_1)[name = string("attn_weights_429_cast_fp16")]; int32 var_9466 = const()[name = string("op_9466"), val = int32(-2)]; tensor attn_weights_431_cast_fp16 = softmax(axis = var_9466, x = attn_weights_429_cast_fp16)[name = string("attn_weights_431_cast_fp16")]; bool attn_output_209_transpose_x_1 = const()[name = string("attn_output_209_transpose_x_1"), val = bool(true)]; bool attn_output_209_transpose_y_1 = const()[name = string("attn_output_209_transpose_y_1"), val = bool(false)]; tensor attn_output_209_cast_fp16 = matmul(transpose_x = attn_output_209_transpose_x_1, transpose_y = attn_output_209_transpose_y_1, x = attn_weights_431_cast_fp16, y = var_9444_cast_fp16_1)[name = string("attn_output_209_cast_fp16")]; int32 var_9474 = const()[name = string("op_9474"), val = int32(1)]; bool attn_output_211_interleave_0 = const()[name = string("attn_output_211_interleave_0"), val = bool(false)]; tensor attn_output_211_cast_fp16 = concat(axis = var_9474, interleave = attn_output_211_interleave_0, values = (var_9460_cast_fp16, attn_output_209_cast_fp16))[name = string("attn_output_211_cast_fp16")]; tensor var_9478_perm_0 = const()[name = string("op_9478_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_323x = const()[name = string("concat_323x"), val = tensor([1, 2048, 1, -1])]; tensor var_9478_cast_fp16 = transpose(perm = var_9478_perm_0, x = attn_output_211_cast_fp16)[name = string("transpose_3")]; tensor attn_output_215_cast_fp16 = reshape(shape = concat_323x, x = var_9478_cast_fp16)[name = string("attn_output_215_cast_fp16")]; tensor hidden_states_263_strides_0 = const()[name = string("hidden_states_263_strides_0"), val = tensor([1, 1])]; string hidden_states_263_pad_type_0 = const()[name = string("hidden_states_263_pad_type_0"), val = string("valid")]; tensor hidden_states_263_pad_0 = const()[name = string("hidden_states_263_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_263_dilations_0 = const()[name = string("hidden_states_263_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_263_groups_0 = const()[name = string("hidden_states_263_groups_0"), val = int32(1)]; tensor hidden_states_263_cast_fp16 = conv(dilations = hidden_states_263_dilations_0, groups = hidden_states_263_groups_0, pad = hidden_states_263_pad_0, pad_type = hidden_states_263_pad_type_0, strides = hidden_states_263_strides_0, weight = layers_26_self_attn_o_proj_weight_cast_fp16, x = attn_output_215_cast_fp16)[name = string("hidden_states_263_cast_fp16")]; tensor hidden_states_265_cast_fp16 = add(x = hidden_states_259_cast_fp16, y = hidden_states_263_cast_fp16)[name = string("hidden_states_265_cast_fp16")]; fp16 const_268_promoted_to_fp16 = const()[name = string("const_268_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9511_cast_fp16 = mul(x = hidden_states_265_cast_fp16, y = const_268_promoted_to_fp16)[name = string("op_9511_cast_fp16")]; int32 var_9509 = const()[name = string("op_9509"), val = int32(1)]; bool doubled_213_interleave_0 = const()[name = string("doubled_213_interleave_0"), val = bool(false)]; tensor doubled_213_cast_fp16 = concat(axis = var_9509, interleave = doubled_213_interleave_0, values = (hidden_states_265_cast_fp16, var_9511_cast_fp16))[name = string("doubled_213_cast_fp16")]; tensor out_107_axes_0 = const()[name = string("out_107_axes_0"), val = tensor([1])]; tensor out_107_gamma_0_to_fp16 = const()[name = string("out_107_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567658624)))]; fp16 var_9521_to_fp16 = const()[name = string("op_9521_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_107_cast_fp16 = layer_norm(axes = out_107_axes_0, epsilon = var_9521_to_fp16, gamma = out_107_gamma_0_to_fp16, x = doubled_213_cast_fp16)[name = string("out_107_cast_fp16")]; tensor var_9532_split_sizes_0 = const()[name = string("op_9532_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9532_axis_0 = const()[name = string("op_9532_axis_0"), val = int32(1)]; tensor var_9532_cast_fp16_0, tensor var_9532_cast_fp16_1 = split(axis = var_9532_axis_0, split_sizes = var_9532_split_sizes_0, x = out_107_cast_fp16)[name = string("op_9532_cast_fp16")]; tensor input_53_strides_0 = const()[name = string("input_53_strides_0"), val = tensor([1, 1])]; string input_53_pad_type_0 = const()[name = string("input_53_pad_type_0"), val = string("valid")]; tensor input_53_pad_0 = const()[name = string("input_53_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_53_dilations_0 = const()[name = string("input_53_dilations_0"), val = tensor([1, 1])]; int32 input_53_groups_0 = const()[name = string("input_53_groups_0"), val = int32(1)]; tensor input_53_cast_fp16 = conv(dilations = input_53_dilations_0, groups = input_53_groups_0, pad = input_53_pad_0, pad_type = input_53_pad_type_0, strides = input_53_strides_0, weight = layers_26_mlp_gate_proj_weight_cast_fp16, x = var_9532_cast_fp16_0)[name = string("input_53_cast_fp16")]; tensor var_9549_cast_fp16 = silu(x = input_53_cast_fp16)[name = string("op_9549_cast_fp16")]; tensor layers_26_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1567666880)))]; tensor var_9555_strides_0 = const()[name = string("op_9555_strides_0"), val = tensor([1, 1])]; string var_9555_pad_type_0 = const()[name = string("op_9555_pad_type_0"), val = string("valid")]; tensor var_9555_pad_0 = const()[name = string("op_9555_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9555_dilations_0 = const()[name = string("op_9555_dilations_0"), val = tensor([1, 1])]; int32 var_9555_groups_0 = const()[name = string("op_9555_groups_0"), val = int32(1)]; tensor var_9555_cast_fp16 = conv(dilations = var_9555_dilations_0, groups = var_9555_groups_0, pad = var_9555_pad_0, pad_type = var_9555_pad_type_0, strides = var_9555_strides_0, weight = layers_26_mlp_up_proj_weight_to_fp16, x = var_9532_cast_fp16_0)[name = string("op_9555_cast_fp16")]; tensor x_269_cast_fp16 = mul(x = var_9549_cast_fp16, y = var_9555_cast_fp16)[name = string("x_269_cast_fp16")]; tensor layers_26_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_26_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1592832768)))]; tensor hidden_states_267_strides_0 = const()[name = string("hidden_states_267_strides_0"), val = tensor([1, 1])]; string hidden_states_267_pad_type_0 = const()[name = string("hidden_states_267_pad_type_0"), val = string("valid")]; tensor hidden_states_267_pad_0 = const()[name = string("hidden_states_267_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_267_dilations_0 = const()[name = string("hidden_states_267_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_267_groups_0 = const()[name = string("hidden_states_267_groups_0"), val = int32(1)]; tensor hidden_states_267_cast_fp16 = conv(dilations = hidden_states_267_dilations_0, groups = hidden_states_267_groups_0, pad = hidden_states_267_pad_0, pad_type = hidden_states_267_pad_type_0, strides = hidden_states_267_strides_0, weight = layers_26_mlp_down_proj_weight_to_fp16, x = x_269_cast_fp16)[name = string("hidden_states_267_cast_fp16")]; tensor hidden_states_269_cast_fp16 = add(x = hidden_states_265_cast_fp16, y = hidden_states_267_cast_fp16)[name = string("hidden_states_269_cast_fp16")]; fp16 const_270_promoted_to_fp16 = const()[name = string("const_270_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9573_cast_fp16 = mul(x = hidden_states_269_cast_fp16, y = const_270_promoted_to_fp16)[name = string("op_9573_cast_fp16")]; int32 var_9571 = const()[name = string("op_9571"), val = int32(1)]; bool doubled_217_interleave_0 = const()[name = string("doubled_217_interleave_0"), val = bool(false)]; tensor doubled_217_cast_fp16 = concat(axis = var_9571, interleave = doubled_217_interleave_0, values = (hidden_states_269_cast_fp16, var_9573_cast_fp16))[name = string("doubled_217_cast_fp16")]; tensor out_109_axes_0 = const()[name = string("out_109_axes_0"), val = tensor([1])]; tensor out_109_gamma_0_to_fp16 = const()[name = string("out_109_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1617998656)))]; fp16 var_9583_to_fp16 = const()[name = string("op_9583_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_109_cast_fp16 = layer_norm(axes = out_109_axes_0, epsilon = var_9583_to_fp16, gamma = out_109_gamma_0_to_fp16, x = doubled_217_cast_fp16)[name = string("out_109_cast_fp16")]; tensor var_9594_split_sizes_0 = const()[name = string("op_9594_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9594_axis_0 = const()[name = string("op_9594_axis_0"), val = int32(1)]; tensor var_9594_cast_fp16_0, tensor var_9594_cast_fp16_1 = split(axis = var_9594_axis_0, split_sizes = var_9594_split_sizes_0, x = out_109_cast_fp16)[name = string("op_9594_cast_fp16")]; tensor layers_27_self_attn_q_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_q_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1618006912)))]; tensor query_states_163_strides_0 = const()[name = string("query_states_163_strides_0"), val = tensor([1, 1])]; string query_states_163_pad_type_0 = const()[name = string("query_states_163_pad_type_0"), val = string("valid")]; tensor query_states_163_pad_0 = const()[name = string("query_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor query_states_163_dilations_0 = const()[name = string("query_states_163_dilations_0"), val = tensor([1, 1])]; int32 query_states_163_groups_0 = const()[name = string("query_states_163_groups_0"), val = int32(1)]; tensor query_states_163_cast_fp16 = conv(dilations = query_states_163_dilations_0, groups = query_states_163_groups_0, pad = query_states_163_pad_0, pad_type = query_states_163_pad_type_0, strides = query_states_163_strides_0, weight = layers_27_self_attn_q_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("query_states_163_cast_fp16")]; tensor layers_27_self_attn_k_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_k_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1626395584)))]; tensor key_states_271_strides_0 = const()[name = string("key_states_271_strides_0"), val = tensor([1, 1])]; string key_states_271_pad_type_0 = const()[name = string("key_states_271_pad_type_0"), val = string("valid")]; tensor key_states_271_pad_0 = const()[name = string("key_states_271_pad_0"), val = tensor([0, 0, 0, 0])]; tensor key_states_271_dilations_0 = const()[name = string("key_states_271_dilations_0"), val = tensor([1, 1])]; int32 key_states_271_groups_0 = const()[name = string("key_states_271_groups_0"), val = int32(1)]; tensor key_states_271_cast_fp16 = conv(dilations = key_states_271_dilations_0, groups = key_states_271_groups_0, pad = key_states_271_pad_0, pad_type = key_states_271_pad_type_0, strides = key_states_271_strides_0, weight = layers_27_self_attn_k_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("key_states_271_cast_fp16")]; tensor layers_27_self_attn_v_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_v_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1627444224)))]; tensor value_states_163_strides_0 = const()[name = string("value_states_163_strides_0"), val = tensor([1, 1])]; string value_states_163_pad_type_0 = const()[name = string("value_states_163_pad_type_0"), val = string("valid")]; tensor value_states_163_pad_0 = const()[name = string("value_states_163_pad_0"), val = tensor([0, 0, 0, 0])]; tensor value_states_163_dilations_0 = const()[name = string("value_states_163_dilations_0"), val = tensor([1, 1])]; int32 value_states_163_groups_0 = const()[name = string("value_states_163_groups_0"), val = int32(1)]; tensor value_states_163_cast_fp16 = conv(dilations = value_states_163_dilations_0, groups = value_states_163_groups_0, pad = value_states_163_pad_0, pad_type = value_states_163_pad_type_0, strides = value_states_163_strides_0, weight = layers_27_self_attn_v_proj_weight_to_fp16, x = var_9594_cast_fp16_0)[name = string("value_states_163_cast_fp16")]; tensor concat_324x = const()[name = string("concat_324x"), val = tensor([1, 16, 128, -1])]; tensor x_271_cast_fp16 = reshape(shape = concat_324x, x = query_states_163_cast_fp16)[name = string("x_271_cast_fp16")]; tensor concat_325x = const()[name = string("concat_325x"), val = tensor([1, 2, 128, -1])]; tensor var_9651_cast_fp16 = reshape(shape = concat_325x, x = key_states_271_cast_fp16)[name = string("op_9651_cast_fp16")]; tensor concat_326x = const()[name = string("concat_326x"), val = tensor([1, 2, 128, -1])]; tensor var_9658_cast_fp16 = reshape(shape = concat_326x, x = value_states_163_cast_fp16)[name = string("op_9658_cast_fp16")]; tensor var_9662_cast_fp16 = mul(x = x_271_cast_fp16, y = var_869_cast_fp16)[name = string("op_9662_cast_fp16")]; tensor var_9663_split_sizes_0 = const()[name = string("op_9663_split_sizes_0"), val = tensor([64, 64])]; int32 var_9663_axis_0 = const()[name = string("op_9663_axis_0"), val = int32(-2)]; tensor var_9663_cast_fp16_0, tensor var_9663_cast_fp16_1 = split(axis = var_9663_axis_0, split_sizes = var_9663_split_sizes_0, x = x_271_cast_fp16)[name = string("op_9663_cast_fp16")]; fp16 const_272_promoted_to_fp16 = const()[name = string("const_272_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9665_cast_fp16 = mul(x = var_9663_cast_fp16_1, y = const_272_promoted_to_fp16)[name = string("op_9665_cast_fp16")]; int32 var_9667 = const()[name = string("op_9667"), val = int32(-2)]; bool var_9668_interleave_0 = const()[name = string("op_9668_interleave_0"), val = bool(false)]; tensor var_9668_cast_fp16 = concat(axis = var_9667, interleave = var_9668_interleave_0, values = (var_9665_cast_fp16, var_9663_cast_fp16_0))[name = string("op_9668_cast_fp16")]; tensor var_9669_cast_fp16 = mul(x = var_9668_cast_fp16, y = var_878_cast_fp16)[name = string("op_9669_cast_fp16")]; tensor query_states_165_cast_fp16 = add(x = var_9662_cast_fp16, y = var_9669_cast_fp16)[name = string("query_states_165_cast_fp16")]; tensor var_9675_cast_fp16 = mul(x = var_9651_cast_fp16, y = var_869_cast_fp16)[name = string("op_9675_cast_fp16")]; tensor var_9676_split_sizes_0 = const()[name = string("op_9676_split_sizes_0"), val = tensor([64, 64])]; int32 var_9676_axis_0 = const()[name = string("op_9676_axis_0"), val = int32(-2)]; tensor var_9676_cast_fp16_0, tensor var_9676_cast_fp16_1 = split(axis = var_9676_axis_0, split_sizes = var_9676_split_sizes_0, x = var_9651_cast_fp16)[name = string("op_9676_cast_fp16")]; fp16 const_273_promoted_to_fp16 = const()[name = string("const_273_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9678_cast_fp16 = mul(x = var_9676_cast_fp16_1, y = const_273_promoted_to_fp16)[name = string("op_9678_cast_fp16")]; int32 var_9680 = const()[name = string("op_9680"), val = int32(-2)]; bool var_9681_interleave_0 = const()[name = string("op_9681_interleave_0"), val = bool(false)]; tensor var_9681_cast_fp16 = concat(axis = var_9680, interleave = var_9681_interleave_0, values = (var_9678_cast_fp16, var_9676_cast_fp16_0))[name = string("op_9681_cast_fp16")]; tensor var_9682_cast_fp16 = mul(x = var_9681_cast_fp16, y = var_878_cast_fp16)[name = string("op_9682_cast_fp16")]; tensor key_states_275_cast_fp16 = add(x = var_9675_cast_fp16, y = var_9682_cast_fp16)[name = string("key_states_275_cast_fp16")]; tensor expand_dims_324 = const()[name = string("expand_dims_324"), val = tensor([27])]; tensor expand_dims_325 = const()[name = string("expand_dims_325"), val = tensor([0])]; tensor expand_dims_327 = const()[name = string("expand_dims_327"), val = tensor([0])]; int32 concat_329_axis_0 = const()[name = string("concat_329_axis_0"), val = int32(0)]; bool concat_329_interleave_0 = const()[name = string("concat_329_interleave_0"), val = bool(false)]; tensor concat_329 = concat(axis = concat_329_axis_0, interleave = concat_329_interleave_0, values = (expand_dims_324, expand_dims_325, position_id, expand_dims_327))[name = string("concat_329")]; tensor expand_dims_328 = const()[name = string("expand_dims_328"), val = tensor([28])]; tensor concat_330_values1_0 = const()[name = string("concat_330_values1_0"), val = tensor([0])]; tensor concat_330_values3_0 = const()[name = string("concat_330_values3_0"), val = tensor([0])]; int32 concat_330_axis_0 = const()[name = string("concat_330_axis_0"), val = int32(0)]; bool concat_330_interleave_0 = const()[name = string("concat_330_interleave_0"), val = bool(false)]; tensor concat_330 = concat(axis = concat_330_axis_0, interleave = concat_330_interleave_0, values = (expand_dims_328, concat_330_values1_0, cache_position_end, concat_330_values3_0))[name = string("concat_330")]; tensor key_states_277_perm_0 = const()[name = string("key_states_277_perm_0"), val = tensor([0, 1, 3, 2])]; tensor key_cache_internal_tensor_assign_28_stride_0 = const()[name = string("key_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor key_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor key_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor key_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("key_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor key_states_277_cast_fp16 = transpose(perm = key_states_277_perm_0, x = key_states_275_cast_fp16)[name = string("transpose_2")]; tensor key_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = key_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = key_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = key_cache_internal_tensor_assign_28_squeeze_mask_0, stride = key_cache_internal_tensor_assign_28_stride_0, update = key_states_277_cast_fp16, x = coreml_update_state_52)[name = string("key_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = key_cache_internal_tensor_assign_28_cast_fp16, input = key_cache)[name = string("coreml_update_state_54_write_state")]; tensor coreml_update_state_54 = read_state(input = key_cache)[name = string("coreml_update_state_54")]; tensor value_states_165_perm_0 = const()[name = string("value_states_165_perm_0"), val = tensor([0, 1, 3, 2])]; tensor value_cache_internal_tensor_assign_28_stride_0 = const()[name = string("value_cache_internal_tensor_assign_28_stride_0"), val = tensor([1, 1, 1, 1])]; tensor value_cache_internal_tensor_assign_28_begin_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_begin_mask_0"), val = tensor([false, false, false, false])]; tensor value_cache_internal_tensor_assign_28_end_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_end_mask_0"), val = tensor([false, true, false, true])]; tensor value_cache_internal_tensor_assign_28_squeeze_mask_0 = const()[name = string("value_cache_internal_tensor_assign_28_squeeze_mask_0"), val = tensor([false, false, false, false])]; tensor value_states_165_cast_fp16 = transpose(perm = value_states_165_perm_0, x = var_9658_cast_fp16)[name = string("transpose_1")]; tensor value_cache_internal_tensor_assign_28_cast_fp16 = slice_update(begin = concat_329, begin_mask = value_cache_internal_tensor_assign_28_begin_mask_0, end = concat_330, end_mask = value_cache_internal_tensor_assign_28_end_mask_0, squeeze_mask = value_cache_internal_tensor_assign_28_squeeze_mask_0, stride = value_cache_internal_tensor_assign_28_stride_0, update = value_states_165_cast_fp16, x = coreml_update_state_53)[name = string("value_cache_internal_tensor_assign_28_cast_fp16")]; write_state(data = value_cache_internal_tensor_assign_28_cast_fp16, input = value_cache)[name = string("coreml_update_state_55_write_state")]; tensor coreml_update_state_55 = read_state(input = value_cache)[name = string("coreml_update_state_55")]; tensor var_9752_begin_0 = const()[name = string("op_9752_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9752_end_0 = const()[name = string("op_9752_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9752_end_mask_0 = const()[name = string("op_9752_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9752_cast_fp16 = slice_by_index(begin = var_9752_begin_0, end = var_9752_end_0, end_mask = var_9752_end_mask_0, x = coreml_update_state_54)[name = string("op_9752_cast_fp16")]; tensor tile_54 = const()[name = string("tile_54"), val = tensor([1, 1])]; int32 var_9755_axis_0 = const()[name = string("op_9755_axis_0"), val = int32(1)]; tensor var_9755_cast_fp16_0, tensor var_9755_cast_fp16_1 = split(axis = var_9755_axis_0, split_sizes = tile_54, x = var_9752_cast_fp16)[name = string("op_9755_cast_fp16")]; tensor var_9762_begin_0 = const()[name = string("op_9762_begin_0"), val = tensor([27, 0, 0, 0])]; tensor var_9762_end_0 = const()[name = string("op_9762_end_0"), val = tensor([1, 2, 2048, 128])]; tensor var_9762_end_mask_0 = const()[name = string("op_9762_end_mask_0"), val = tensor([true, true, true, true])]; tensor var_9762_cast_fp16 = slice_by_index(begin = var_9762_begin_0, end = var_9762_end_0, end_mask = var_9762_end_mask_0, x = coreml_update_state_55)[name = string("op_9762_cast_fp16")]; tensor tile_55 = const()[name = string("tile_55"), val = tensor([1, 1])]; int32 var_9765_axis_0 = const()[name = string("op_9765_axis_0"), val = int32(1)]; tensor var_9765_cast_fp16_0, tensor var_9765_cast_fp16_1 = split(axis = var_9765_axis_0, split_sizes = tile_55, x = var_9762_cast_fp16)[name = string("op_9765_cast_fp16")]; tensor var_9768_split_sizes_0 = const()[name = string("op_9768_split_sizes_0"), val = tensor([8, 8])]; int32 var_9768_axis_0 = const()[name = string("op_9768_axis_0"), val = int32(1)]; tensor var_9768_0, tensor var_9768_1 = split(axis = var_9768_axis_0, split_sizes = var_9768_split_sizes_0, x = query_states_165_cast_fp16)[name = string("op_9768")]; bool attn_weights_433_transpose_x_0 = const()[name = string("attn_weights_433_transpose_x_0"), val = bool(false)]; bool attn_weights_433_transpose_y_0 = const()[name = string("attn_weights_433_transpose_y_0"), val = bool(false)]; tensor attn_weights_433_cast_fp16 = matmul(transpose_x = attn_weights_433_transpose_x_0, transpose_y = attn_weights_433_transpose_y_0, x = var_9755_cast_fp16_0, y = var_9768_0)[name = string("attn_weights_433_cast_fp16")]; fp16 var_9771_to_fp16 = const()[name = string("op_9771_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_435_cast_fp16 = mul(x = attn_weights_433_cast_fp16, y = var_9771_to_fp16)[name = string("attn_weights_435_cast_fp16")]; tensor attn_weights_437_cast_fp16 = add(x = attn_weights_435_cast_fp16, y = attn_mask_1)[name = string("attn_weights_437_cast_fp16")]; int32 var_9775 = const()[name = string("op_9775"), val = int32(-2)]; tensor attn_weights_439_cast_fp16 = softmax(axis = var_9775, x = attn_weights_437_cast_fp16)[name = string("attn_weights_439_cast_fp16")]; bool var_9781_transpose_x_1 = const()[name = string("op_9781_transpose_x_1"), val = bool(true)]; bool var_9781_transpose_y_1 = const()[name = string("op_9781_transpose_y_1"), val = bool(false)]; tensor var_9781_cast_fp16 = matmul(transpose_x = var_9781_transpose_x_1, transpose_y = var_9781_transpose_y_1, x = attn_weights_439_cast_fp16, y = var_9765_cast_fp16_0)[name = string("op_9781_cast_fp16")]; bool attn_weights_441_transpose_x_0 = const()[name = string("attn_weights_441_transpose_x_0"), val = bool(false)]; bool attn_weights_441_transpose_y_0 = const()[name = string("attn_weights_441_transpose_y_0"), val = bool(false)]; tensor attn_weights_441_cast_fp16 = matmul(transpose_x = attn_weights_441_transpose_x_0, transpose_y = attn_weights_441_transpose_y_0, x = var_9755_cast_fp16_1, y = var_9768_1)[name = string("attn_weights_441_cast_fp16")]; fp16 var_9783_to_fp16 = const()[name = string("op_9783_to_fp16"), val = fp16(0x1.6ap-4)]; tensor attn_weights_443_cast_fp16 = mul(x = attn_weights_441_cast_fp16, y = var_9783_to_fp16)[name = string("attn_weights_443_cast_fp16")]; tensor attn_weights_445_cast_fp16 = add(x = attn_weights_443_cast_fp16, y = attn_mask_1)[name = string("attn_weights_445_cast_fp16")]; int32 var_9787 = const()[name = string("op_9787"), val = int32(-2)]; tensor attn_weights_cast_fp16 = softmax(axis = var_9787, x = attn_weights_445_cast_fp16)[name = string("attn_weights_cast_fp16")]; bool attn_output_217_transpose_x_1 = const()[name = string("attn_output_217_transpose_x_1"), val = bool(true)]; bool attn_output_217_transpose_y_1 = const()[name = string("attn_output_217_transpose_y_1"), val = bool(false)]; tensor attn_output_217_cast_fp16 = matmul(transpose_x = attn_output_217_transpose_x_1, transpose_y = attn_output_217_transpose_y_1, x = attn_weights_cast_fp16, y = var_9765_cast_fp16_1)[name = string("attn_output_217_cast_fp16")]; int32 var_9795 = const()[name = string("op_9795"), val = int32(1)]; bool attn_output_219_interleave_0 = const()[name = string("attn_output_219_interleave_0"), val = bool(false)]; tensor attn_output_219_cast_fp16 = concat(axis = var_9795, interleave = attn_output_219_interleave_0, values = (var_9781_cast_fp16, attn_output_217_cast_fp16))[name = string("attn_output_219_cast_fp16")]; tensor var_9799_perm_0 = const()[name = string("op_9799_perm_0"), val = tensor([0, 1, 3, 2])]; tensor concat_335x = const()[name = string("concat_335x"), val = tensor([1, 2048, 1, -1])]; tensor var_9799_cast_fp16 = transpose(perm = var_9799_perm_0, x = attn_output_219_cast_fp16)[name = string("transpose_0")]; tensor attn_output_cast_fp16 = reshape(shape = concat_335x, x = var_9799_cast_fp16)[name = string("attn_output_cast_fp16")]; tensor layers_27_self_attn_o_proj_weight_to_fp16 = const()[name = string("layers_27_self_attn_o_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1628492864)))]; tensor hidden_states_273_strides_0 = const()[name = string("hidden_states_273_strides_0"), val = tensor([1, 1])]; string hidden_states_273_pad_type_0 = const()[name = string("hidden_states_273_pad_type_0"), val = string("valid")]; tensor hidden_states_273_pad_0 = const()[name = string("hidden_states_273_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_273_dilations_0 = const()[name = string("hidden_states_273_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_273_groups_0 = const()[name = string("hidden_states_273_groups_0"), val = int32(1)]; tensor hidden_states_273_cast_fp16 = conv(dilations = hidden_states_273_dilations_0, groups = hidden_states_273_groups_0, pad = hidden_states_273_pad_0, pad_type = hidden_states_273_pad_type_0, strides = hidden_states_273_strides_0, weight = layers_27_self_attn_o_proj_weight_to_fp16, x = attn_output_cast_fp16)[name = string("hidden_states_273_cast_fp16")]; tensor hidden_states_275_cast_fp16 = add(x = hidden_states_269_cast_fp16, y = hidden_states_273_cast_fp16)[name = string("hidden_states_275_cast_fp16")]; fp16 const_278_promoted_to_fp16 = const()[name = string("const_278_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9832_cast_fp16 = mul(x = hidden_states_275_cast_fp16, y = const_278_promoted_to_fp16)[name = string("op_9832_cast_fp16")]; int32 var_9830 = const()[name = string("op_9830"), val = int32(1)]; bool doubled_221_interleave_0 = const()[name = string("doubled_221_interleave_0"), val = bool(false)]; tensor doubled_221_cast_fp16 = concat(axis = var_9830, interleave = doubled_221_interleave_0, values = (hidden_states_275_cast_fp16, var_9832_cast_fp16))[name = string("doubled_221_cast_fp16")]; tensor out_111_axes_0 = const()[name = string("out_111_axes_0"), val = tensor([1])]; tensor out_111_gamma_0_to_fp16 = const()[name = string("out_111_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636881536)))]; fp16 var_9842_to_fp16 = const()[name = string("op_9842_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_111_cast_fp16 = layer_norm(axes = out_111_axes_0, epsilon = var_9842_to_fp16, gamma = out_111_gamma_0_to_fp16, x = doubled_221_cast_fp16)[name = string("out_111_cast_fp16")]; tensor var_9853_split_sizes_0 = const()[name = string("op_9853_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9853_axis_0 = const()[name = string("op_9853_axis_0"), val = int32(1)]; tensor var_9853_cast_fp16_0, tensor var_9853_cast_fp16_1 = split(axis = var_9853_axis_0, split_sizes = var_9853_split_sizes_0, x = out_111_cast_fp16)[name = string("op_9853_cast_fp16")]; tensor layers_27_mlp_gate_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_gate_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1636889792)))]; tensor input_strides_0 = const()[name = string("input_strides_0"), val = tensor([1, 1])]; string input_pad_type_0 = const()[name = string("input_pad_type_0"), val = string("valid")]; tensor input_pad_0 = const()[name = string("input_pad_0"), val = tensor([0, 0, 0, 0])]; tensor input_dilations_0 = const()[name = string("input_dilations_0"), val = tensor([1, 1])]; int32 input_groups_0 = const()[name = string("input_groups_0"), val = int32(1)]; tensor input_cast_fp16 = conv(dilations = input_dilations_0, groups = input_groups_0, pad = input_pad_0, pad_type = input_pad_type_0, strides = input_strides_0, weight = layers_27_mlp_gate_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("input_cast_fp16")]; tensor var_9870_cast_fp16 = silu(x = input_cast_fp16)[name = string("op_9870_cast_fp16")]; tensor layers_27_mlp_up_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_up_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1662055680)))]; tensor var_9876_strides_0 = const()[name = string("op_9876_strides_0"), val = tensor([1, 1])]; string var_9876_pad_type_0 = const()[name = string("op_9876_pad_type_0"), val = string("valid")]; tensor var_9876_pad_0 = const()[name = string("op_9876_pad_0"), val = tensor([0, 0, 0, 0])]; tensor var_9876_dilations_0 = const()[name = string("op_9876_dilations_0"), val = tensor([1, 1])]; int32 var_9876_groups_0 = const()[name = string("op_9876_groups_0"), val = int32(1)]; tensor var_9876_cast_fp16 = conv(dilations = var_9876_dilations_0, groups = var_9876_groups_0, pad = var_9876_pad_0, pad_type = var_9876_pad_type_0, strides = var_9876_strides_0, weight = layers_27_mlp_up_proj_weight_to_fp16, x = var_9853_cast_fp16_0)[name = string("op_9876_cast_fp16")]; tensor x_cast_fp16 = mul(x = var_9870_cast_fp16, y = var_9876_cast_fp16)[name = string("x_cast_fp16")]; tensor layers_27_mlp_down_proj_weight_to_fp16 = const()[name = string("layers_27_mlp_down_proj_weight_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1687221568)))]; tensor hidden_states_277_strides_0 = const()[name = string("hidden_states_277_strides_0"), val = tensor([1, 1])]; string hidden_states_277_pad_type_0 = const()[name = string("hidden_states_277_pad_type_0"), val = string("valid")]; tensor hidden_states_277_pad_0 = const()[name = string("hidden_states_277_pad_0"), val = tensor([0, 0, 0, 0])]; tensor hidden_states_277_dilations_0 = const()[name = string("hidden_states_277_dilations_0"), val = tensor([1, 1])]; int32 hidden_states_277_groups_0 = const()[name = string("hidden_states_277_groups_0"), val = int32(1)]; tensor hidden_states_277_cast_fp16 = conv(dilations = hidden_states_277_dilations_0, groups = hidden_states_277_groups_0, pad = hidden_states_277_pad_0, pad_type = hidden_states_277_pad_type_0, strides = hidden_states_277_strides_0, weight = layers_27_mlp_down_proj_weight_to_fp16, x = x_cast_fp16)[name = string("hidden_states_277_cast_fp16")]; tensor hidden_states_cast_fp16 = add(x = hidden_states_275_cast_fp16, y = hidden_states_277_cast_fp16)[name = string("hidden_states_cast_fp16")]; fp16 const_280_promoted_to_fp16 = const()[name = string("const_280_promoted_to_fp16"), val = fp16(-0x1p+0)]; tensor var_9894_cast_fp16 = mul(x = hidden_states_cast_fp16, y = const_280_promoted_to_fp16)[name = string("op_9894_cast_fp16")]; int32 var_9892 = const()[name = string("op_9892"), val = int32(1)]; bool doubled_225_interleave_0 = const()[name = string("doubled_225_interleave_0"), val = bool(false)]; tensor doubled_225_cast_fp16 = concat(axis = var_9892, interleave = doubled_225_interleave_0, values = (hidden_states_cast_fp16, var_9894_cast_fp16))[name = string("doubled_225_cast_fp16")]; tensor out_axes_0 = const()[name = string("out_axes_0"), val = tensor([1])]; tensor out_gamma_0_to_fp16 = const()[name = string("out_gamma_0_to_fp16"), val = tensor(BLOBFILE(path = string("@model_path/weights/weight.bin"), offset = uint64(1712387456)))]; fp16 var_9904_to_fp16 = const()[name = string("op_9904_to_fp16"), val = fp16(0x1.5p-17)]; tensor out_cast_fp16 = layer_norm(axes = out_axes_0, epsilon = var_9904_to_fp16, gamma = out_gamma_0_to_fp16, x = doubled_225_cast_fp16)[name = string("out_cast_fp16")]; tensor var_9915_split_sizes_0 = const()[name = string("op_9915_split_sizes_0"), val = tensor([2048, 2048])]; int32 var_9915_axis_0 = const()[name = string("op_9915_axis_0"), val = int32(1)]; tensor hidden_states, tensor var_9915_cast_fp16_1 = split(axis = var_9915_axis_0, split_sizes = var_9915_split_sizes_0, x = out_cast_fp16)[name = string("op_9915_cast_fp16")]; } -> (hidden_states); }