diff --git "a/cache/Token_modellayers.0input_layernormMLADFRMSNORM_meta.json" "b/cache/Token_modellayers.0input_layernormMLADFRMSNORM_meta.json" new file mode 100755--- /dev/null +++ "b/cache/Token_modellayers.0input_layernormMLADFRMSNORM_meta.json" @@ -0,0 +1,34780 @@ +{ + "dd_meta_major_version": 1, + "dd_meta_minor_version": 4, + "state_table_updates": [ + { + "state_table_idx": 0, + "update_func": 1, + "update_arg": 1 + } + ], + "op_list": [ + { + "name": "/model/layers.0/input_layernorm/MLADFRMSNORM", + "type": "MLADFRMSNORM", + "in_args": [ + "/model/embed_tokens/Gather/output_0.out27_14_0" + ], + "const_args": [ + "model.layers.0.input_layernorm.weight", + "eps_27_14_0" + ], + "out_args": [ + "/model/layers.0/input_layernorm/output_0.out27_14_0" + ], + "attrs": { + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "m_x": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.0/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.0/input_layernorm/output_0.out27_14_0" + ], + "const_args": [ + "model.layers.0.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.0.attn.k_proj.Add.bias.preformat", + "model.layers.0.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.0.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.0/attn/k_proj/Add/output_0.out27_11_1" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.0/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.0/input_layernorm/output_0.out27_14_0" + ], + "const_args": [ + "model.layers.0.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.0.attn.q_proj.Add.bias.preformat", + "model.layers.0.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.0.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.0/attn/q_proj/Add/output_0.out27_11_0" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.0/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.0/input_layernorm/output_0.out27_14_0" + ], + "const_args": [ + "model.layers.0.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.0.attn.v_proj.Add.bias.preformat", + "model.layers.0.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.0.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.0.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "3", + "1" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.0/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.0/attn/q_proj/Add/output_0.out27_11_0", + "/model/layers.0/attn/k_proj/Add/output_0.out27_11_1", + "past_key_values.0.key", + "past_key_values.0.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.0/attn/GroupQueryAttention/output_0.out24_0_0", + "present.0.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "0", + "0", + "3", + "0", + "1", + "1", + "8", + "0", + "2", + "0" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.0/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.0/attn/GroupQueryAttention/output_0.out24_0_0" + ], + "const_args": [ + "model.layers.0.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.0.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.0.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.0.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.0/attn/o_proj/MatMulNBits/output_0.out27_11_2" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.0/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/embed_tokens/Gather/output_0.out27_14_0", + "/model/layers.0/attn/o_proj/MatMulNBits/output_0.out27_11_2" + ], + "const_args": [ + "model.layers.0.post_attention_layernorm.weight.bf", + "eps_26_1_0" + ], + "out_args": [ + "/model/layers.0/post_attention_layernorm/output_3.out26_1_0", + "/model/layers.0/post_attention_layernorm/output_0.out26_1_0" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.0/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.0/post_attention_layernorm/output_0.out26_1_0" + ], + "const_args": [ + "model.layers.0.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.0.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.0.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.0.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.0.mlp.up_proj.MatMulNBits.qweight", + "model.layers.0.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.0.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.0.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.0/mlp/Mul/output_0.out25_0_0" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.0/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.0/mlp/Mul/output_0.out25_0_0" + ], + "const_args": [ + "model.layers.0.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.0.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.0.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.0.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.0/mlp/down_proj/MatMulNBits/output_0.out27_11_3" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.1/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.0/post_attention_layernorm/output_3.out26_1_0", + "/model/layers.0/mlp/down_proj/MatMulNBits/output_0.out27_11_3" + ], + "const_args": [ + "model.layers.1.input_layernorm.weight.bf", + "eps_26_1_1" + ], + "out_args": [ + "/model/layers.1/input_layernorm/output_3.out26_1_1", + "/model/layers.1/input_layernorm/output_0.out26_1_1" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.1/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.1/input_layernorm/output_0.out26_1_1" + ], + "const_args": [ + "model.layers.1.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.1.attn.k_proj.Add.bias.preformat", + "model.layers.1.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.1.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.1/attn/k_proj/Add/output_0.out27_11_5" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.1/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.1/input_layernorm/output_0.out26_1_1" + ], + "const_args": [ + "model.layers.1.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.1.attn.q_proj.Add.bias.preformat", + "model.layers.1.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.1.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.1/attn/q_proj/Add/output_0.out27_11_4" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.1/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.1/input_layernorm/output_0.out26_1_1" + ], + "const_args": [ + "model.layers.1.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.1.attn.v_proj.Add.bias.preformat", + "model.layers.1.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.1.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.1.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "7", + "3" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.1/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.1/attn/q_proj/Add/output_0.out27_11_4", + "/model/layers.1/attn/k_proj/Add/output_0.out27_11_5", + "past_key_values.1.key", + "past_key_values.1.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.1/attn/GroupQueryAttention/output_0.out24_0_1", + "present.1.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "4", + "2", + "3", + "0", + "5", + "3", + "8", + "0", + "6", + "2" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.1/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.1/attn/GroupQueryAttention/output_0.out24_0_1" + ], + "const_args": [ + "model.layers.1.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.1.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.1.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.1.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.1/attn/o_proj/MatMulNBits/output_0.out27_11_6" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.1/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.1/input_layernorm/output_3.out26_1_1", + "/model/layers.1/attn/o_proj/MatMulNBits/output_0.out27_11_6" + ], + "const_args": [ + "model.layers.1.post_attention_layernorm.weight.bf", + "eps_26_1_2" + ], + "out_args": [ + "/model/layers.1/post_attention_layernorm/output_3.out26_1_2", + "/model/layers.1/post_attention_layernorm/output_0.out26_1_2" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.1/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.1/post_attention_layernorm/output_0.out26_1_2" + ], + "const_args": [ + "model.layers.1.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.1.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.1.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.1.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.1.mlp.up_proj.MatMulNBits.qweight", + "model.layers.1.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.1.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.1.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.1/mlp/Mul/output_0.out25_0_1" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.1/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.1/mlp/Mul/output_0.out25_0_1" + ], + "const_args": [ + "model.layers.1.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.1.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.1.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.1.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.1/mlp/down_proj/MatMulNBits/output_0.out27_11_7" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.2/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.1/post_attention_layernorm/output_3.out26_1_2", + "/model/layers.1/mlp/down_proj/MatMulNBits/output_0.out27_11_7" + ], + "const_args": [ + "model.layers.2.input_layernorm.weight.bf", + "eps_26_1_3" + ], + "out_args": [ + "/model/layers.2/input_layernorm/output_3.out26_1_3", + "/model/layers.2/input_layernorm/output_0.out26_1_3" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.2/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.2/input_layernorm/output_0.out26_1_3" + ], + "const_args": [ + "model.layers.2.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.2.attn.k_proj.Add.bias.preformat", + "model.layers.2.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.2.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.2/attn/k_proj/Add/output_0.out27_11_9" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.2/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.2/input_layernorm/output_0.out26_1_3" + ], + "const_args": [ + "model.layers.2.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.2.attn.q_proj.Add.bias.preformat", + "model.layers.2.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.2.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.2/attn/q_proj/Add/output_0.out27_11_8" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.2/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.2/input_layernorm/output_0.out26_1_3" + ], + "const_args": [ + "model.layers.2.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.2.attn.v_proj.Add.bias.preformat", + "model.layers.2.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.2.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.2.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "11", + "5" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.2/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.2/attn/q_proj/Add/output_0.out27_11_8", + "/model/layers.2/attn/k_proj/Add/output_0.out27_11_9", + "past_key_values.2.key", + "past_key_values.2.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.2/attn/GroupQueryAttention/output_0.out24_0_2", + "present.2.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "8", + "4", + "3", + "0", + "9", + "5", + "8", + "0", + "10", + "4" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.2/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.2/attn/GroupQueryAttention/output_0.out24_0_2" + ], + "const_args": [ + "model.layers.2.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.2.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.2.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.2.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.2/attn/o_proj/MatMulNBits/output_0.out27_11_10" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.2/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.2/input_layernorm/output_3.out26_1_3", + "/model/layers.2/attn/o_proj/MatMulNBits/output_0.out27_11_10" + ], + "const_args": [ + "model.layers.2.post_attention_layernorm.weight.bf", + "eps_26_1_4" + ], + "out_args": [ + "/model/layers.2/post_attention_layernorm/output_3.out26_1_4", + "/model/layers.2/post_attention_layernorm/output_0.out26_1_4" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.2/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.2/post_attention_layernorm/output_0.out26_1_4" + ], + "const_args": [ + "model.layers.2.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.2.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.2.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.2.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.2.mlp.up_proj.MatMulNBits.qweight", + "model.layers.2.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.2.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.2.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.2/mlp/Mul/output_0.out25_0_2" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.2/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.2/mlp/Mul/output_0.out25_0_2" + ], + "const_args": [ + "model.layers.2.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.2.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.2.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.2.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.2/mlp/down_proj/MatMulNBits/output_0.out27_11_11" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.3/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.2/post_attention_layernorm/output_3.out26_1_4", + "/model/layers.2/mlp/down_proj/MatMulNBits/output_0.out27_11_11" + ], + "const_args": [ + "model.layers.3.input_layernorm.weight.bf", + "eps_26_1_5" + ], + "out_args": [ + "/model/layers.3/input_layernorm/output_3.out26_1_5", + "/model/layers.3/input_layernorm/output_0.out26_1_5" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.3/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.3/input_layernorm/output_0.out26_1_5" + ], + "const_args": [ + "model.layers.3.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.3.attn.k_proj.Add.bias.preformat", + "model.layers.3.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.3.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.3/attn/k_proj/Add/output_0.out27_11_13" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.3/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.3/input_layernorm/output_0.out26_1_5" + ], + "const_args": [ + "model.layers.3.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.3.attn.q_proj.Add.bias.preformat", + "model.layers.3.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.3.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.3/attn/q_proj/Add/output_0.out27_11_12" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.3/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.3/input_layernorm/output_0.out26_1_5" + ], + "const_args": [ + "model.layers.3.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.3.attn.v_proj.Add.bias.preformat", + "model.layers.3.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.3.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.3.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "15", + "7" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.3/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.3/attn/q_proj/Add/output_0.out27_11_12", + "/model/layers.3/attn/k_proj/Add/output_0.out27_11_13", + "past_key_values.3.key", + "past_key_values.3.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.3/attn/GroupQueryAttention/output_0.out24_0_3", + "present.3.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "12", + "6", + "3", + "0", + "13", + "7", + "8", + "0", + "14", + "6" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.3/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.3/attn/GroupQueryAttention/output_0.out24_0_3" + ], + "const_args": [ + "model.layers.3.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.3.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.3.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.3.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.3/attn/o_proj/MatMulNBits/output_0.out27_11_14" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.3/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.3/input_layernorm/output_3.out26_1_5", + "/model/layers.3/attn/o_proj/MatMulNBits/output_0.out27_11_14" + ], + "const_args": [ + "model.layers.3.post_attention_layernorm.weight.bf", + "eps_26_1_6" + ], + "out_args": [ + "/model/layers.3/post_attention_layernorm/output_3.out26_1_6", + "/model/layers.3/post_attention_layernorm/output_0.out26_1_6" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.3/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.3/post_attention_layernorm/output_0.out26_1_6" + ], + "const_args": [ + "model.layers.3.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.3.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.3.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.3.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.3.mlp.up_proj.MatMulNBits.qweight", + "model.layers.3.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.3.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.3.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.3/mlp/Mul/output_0.out25_0_3" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.3/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.3/mlp/Mul/output_0.out25_0_3" + ], + "const_args": [ + "model.layers.3.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.3.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.3.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.3.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.3/mlp/down_proj/MatMulNBits/output_0.out27_11_15" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.4/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.3/post_attention_layernorm/output_3.out26_1_6", + "/model/layers.3/mlp/down_proj/MatMulNBits/output_0.out27_11_15" + ], + "const_args": [ + "model.layers.4.input_layernorm.weight.bf", + "eps_26_1_7" + ], + "out_args": [ + "/model/layers.4/input_layernorm/output_3.out26_1_7", + "/model/layers.4/input_layernorm/output_0.out26_1_7" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.4/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.4/input_layernorm/output_0.out26_1_7" + ], + "const_args": [ + "model.layers.4.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.4.attn.k_proj.Add.bias.preformat", + "model.layers.4.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.4.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.4/attn/k_proj/Add/output_0.out27_11_17" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.4/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.4/input_layernorm/output_0.out26_1_7" + ], + "const_args": [ + "model.layers.4.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.4.attn.q_proj.Add.bias.preformat", + "model.layers.4.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.4.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.4/attn/q_proj/Add/output_0.out27_11_16" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.4/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.4/input_layernorm/output_0.out26_1_7" + ], + "const_args": [ + "model.layers.4.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.4.attn.v_proj.Add.bias.preformat", + "model.layers.4.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.4.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.4.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "19", + "9" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.4/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.4/attn/q_proj/Add/output_0.out27_11_16", + "/model/layers.4/attn/k_proj/Add/output_0.out27_11_17", + "past_key_values.4.key", + "past_key_values.4.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.4/attn/GroupQueryAttention/output_0.out24_0_4", + "present.4.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "16", + "8", + "3", + "0", + "17", + "9", + "8", + "0", + "18", + "8" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.4/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.4/attn/GroupQueryAttention/output_0.out24_0_4" + ], + "const_args": [ + "model.layers.4.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.4.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.4.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.4.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.4/attn/o_proj/MatMulNBits/output_0.out27_11_18" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.4/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.4/input_layernorm/output_3.out26_1_7", + "/model/layers.4/attn/o_proj/MatMulNBits/output_0.out27_11_18" + ], + "const_args": [ + "model.layers.4.post_attention_layernorm.weight.bf", + "eps_26_1_8" + ], + "out_args": [ + "/model/layers.4/post_attention_layernorm/output_3.out26_1_8", + "/model/layers.4/post_attention_layernorm/output_0.out26_1_8" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.4/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.4/post_attention_layernorm/output_0.out26_1_8" + ], + "const_args": [ + "model.layers.4.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.4.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.4.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.4.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.4.mlp.up_proj.MatMulNBits.qweight", + "model.layers.4.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.4.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.4.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.4/mlp/Mul/output_0.out25_0_4" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.4/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.4/mlp/Mul/output_0.out25_0_4" + ], + "const_args": [ + "model.layers.4.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.4.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.4.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.4.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.4/mlp/down_proj/MatMulNBits/output_0.out27_11_19" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.5/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.4/post_attention_layernorm/output_3.out26_1_8", + "/model/layers.4/mlp/down_proj/MatMulNBits/output_0.out27_11_19" + ], + "const_args": [ + "model.layers.5.input_layernorm.weight.bf", + "eps_26_1_9" + ], + "out_args": [ + "/model/layers.5/input_layernorm/output_3.out26_1_9", + "/model/layers.5/input_layernorm/output_0.out26_1_9" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.5/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.5/input_layernorm/output_0.out26_1_9" + ], + "const_args": [ + "model.layers.5.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.5.attn.k_proj.Add.bias.preformat", + "model.layers.5.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.5.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.5/attn/k_proj/Add/output_0.out27_11_21" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.5/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.5/input_layernorm/output_0.out26_1_9" + ], + "const_args": [ + "model.layers.5.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.5.attn.q_proj.Add.bias.preformat", + "model.layers.5.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.5.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.5/attn/q_proj/Add/output_0.out27_11_20" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.5/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.5/input_layernorm/output_0.out26_1_9" + ], + "const_args": [ + "model.layers.5.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.5.attn.v_proj.Add.bias.preformat", + "model.layers.5.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.5.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.5.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "23", + "11" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.5/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.5/attn/q_proj/Add/output_0.out27_11_20", + "/model/layers.5/attn/k_proj/Add/output_0.out27_11_21", + "past_key_values.5.key", + "past_key_values.5.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.5/attn/GroupQueryAttention/output_0.out24_0_5", + "present.5.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "20", + "10", + "3", + "0", + "21", + "11", + "8", + "0", + "22", + "10" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.5/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.5/attn/GroupQueryAttention/output_0.out24_0_5" + ], + "const_args": [ + "model.layers.5.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.5.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.5.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.5.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.5/attn/o_proj/MatMulNBits/output_0.out27_11_22" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.5/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.5/input_layernorm/output_3.out26_1_9", + "/model/layers.5/attn/o_proj/MatMulNBits/output_0.out27_11_22" + ], + "const_args": [ + "model.layers.5.post_attention_layernorm.weight.bf", + "eps_26_1_10" + ], + "out_args": [ + "/model/layers.5/post_attention_layernorm/output_3.out26_1_10", + "/model/layers.5/post_attention_layernorm/output_0.out26_1_10" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.5/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.5/post_attention_layernorm/output_0.out26_1_10" + ], + "const_args": [ + "model.layers.5.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.5.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.5.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.5.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.5.mlp.up_proj.MatMulNBits.qweight", + "model.layers.5.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.5.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.5.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.5/mlp/Mul/output_0.out25_0_5" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.5/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.5/mlp/Mul/output_0.out25_0_5" + ], + "const_args": [ + "model.layers.5.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.5.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.5.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.5.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.5/mlp/down_proj/MatMulNBits/output_0.out27_11_23" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.6/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.5/post_attention_layernorm/output_3.out26_1_10", + "/model/layers.5/mlp/down_proj/MatMulNBits/output_0.out27_11_23" + ], + "const_args": [ + "model.layers.6.input_layernorm.weight.bf", + "eps_26_1_11" + ], + "out_args": [ + "/model/layers.6/input_layernorm/output_3.out26_1_11", + "/model/layers.6/input_layernorm/output_0.out26_1_11" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.6/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.6/input_layernorm/output_0.out26_1_11" + ], + "const_args": [ + "model.layers.6.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.6.attn.k_proj.Add.bias.preformat", + "model.layers.6.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.6.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.6/attn/k_proj/Add/output_0.out27_11_25" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.6/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.6/input_layernorm/output_0.out26_1_11" + ], + "const_args": [ + "model.layers.6.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.6.attn.q_proj.Add.bias.preformat", + "model.layers.6.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.6.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.6/attn/q_proj/Add/output_0.out27_11_24" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.6/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.6/input_layernorm/output_0.out26_1_11" + ], + "const_args": [ + "model.layers.6.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.6.attn.v_proj.Add.bias.preformat", + "model.layers.6.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.6.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.6.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "27", + "13" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.6/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.6/attn/q_proj/Add/output_0.out27_11_24", + "/model/layers.6/attn/k_proj/Add/output_0.out27_11_25", + "past_key_values.6.key", + "past_key_values.6.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.6/attn/GroupQueryAttention/output_0.out24_0_6", + "present.6.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "24", + "12", + "3", + "0", + "25", + "13", + "8", + "0", + "26", + "12" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.6/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.6/attn/GroupQueryAttention/output_0.out24_0_6" + ], + "const_args": [ + "model.layers.6.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.6.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.6.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.6.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.6/attn/o_proj/MatMulNBits/output_0.out27_11_26" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.6/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.6/input_layernorm/output_3.out26_1_11", + "/model/layers.6/attn/o_proj/MatMulNBits/output_0.out27_11_26" + ], + "const_args": [ + "model.layers.6.post_attention_layernorm.weight.bf", + "eps_26_1_12" + ], + "out_args": [ + "/model/layers.6/post_attention_layernorm/output_3.out26_1_12", + "/model/layers.6/post_attention_layernorm/output_0.out26_1_12" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.6/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.6/post_attention_layernorm/output_0.out26_1_12" + ], + "const_args": [ + "model.layers.6.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.6.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.6.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.6.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.6.mlp.up_proj.MatMulNBits.qweight", + "model.layers.6.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.6.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.6.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.6/mlp/Mul/output_0.out25_0_6" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.6/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.6/mlp/Mul/output_0.out25_0_6" + ], + "const_args": [ + "model.layers.6.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.6.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.6.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.6.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.6/mlp/down_proj/MatMulNBits/output_0.out27_11_27" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.7/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.6/post_attention_layernorm/output_3.out26_1_12", + "/model/layers.6/mlp/down_proj/MatMulNBits/output_0.out27_11_27" + ], + "const_args": [ + "model.layers.7.input_layernorm.weight.bf", + "eps_26_1_13" + ], + "out_args": [ + "/model/layers.7/input_layernorm/output_3.out26_1_13", + "/model/layers.7/input_layernorm/output_0.out26_1_13" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.7/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.7/input_layernorm/output_0.out26_1_13" + ], + "const_args": [ + "model.layers.7.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.7.attn.k_proj.Add.bias.preformat", + "model.layers.7.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.7.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.7/attn/k_proj/Add/output_0.out27_11_29" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.7/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.7/input_layernorm/output_0.out26_1_13" + ], + "const_args": [ + "model.layers.7.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.7.attn.q_proj.Add.bias.preformat", + "model.layers.7.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.7.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.7/attn/q_proj/Add/output_0.out27_11_28" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.7/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.7/input_layernorm/output_0.out26_1_13" + ], + "const_args": [ + "model.layers.7.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.7.attn.v_proj.Add.bias.preformat", + "model.layers.7.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.7.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.7.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "31", + "15" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.7/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.7/attn/q_proj/Add/output_0.out27_11_28", + "/model/layers.7/attn/k_proj/Add/output_0.out27_11_29", + "past_key_values.7.key", + "past_key_values.7.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.7/attn/GroupQueryAttention/output_0.out24_0_7", + "present.7.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "28", + "14", + "3", + "0", + "29", + "15", + "8", + "0", + "30", + "14" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.7/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.7/attn/GroupQueryAttention/output_0.out24_0_7" + ], + "const_args": [ + "model.layers.7.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.7.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.7.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.7.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.7/attn/o_proj/MatMulNBits/output_0.out27_11_30" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.7/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.7/input_layernorm/output_3.out26_1_13", + "/model/layers.7/attn/o_proj/MatMulNBits/output_0.out27_11_30" + ], + "const_args": [ + "model.layers.7.post_attention_layernorm.weight.bf", + "eps_26_1_14" + ], + "out_args": [ + "/model/layers.7/post_attention_layernorm/output_3.out26_1_14", + "/model/layers.7/post_attention_layernorm/output_0.out26_1_14" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.7/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.7/post_attention_layernorm/output_0.out26_1_14" + ], + "const_args": [ + "model.layers.7.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.7.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.7.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.7.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.7.mlp.up_proj.MatMulNBits.qweight", + "model.layers.7.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.7.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.7.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.7/mlp/Mul/output_0.out25_0_7" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.7/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.7/mlp/Mul/output_0.out25_0_7" + ], + "const_args": [ + "model.layers.7.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.7.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.7.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.7.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.7/mlp/down_proj/MatMulNBits/output_0.out27_11_31" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.8/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.7/post_attention_layernorm/output_3.out26_1_14", + "/model/layers.7/mlp/down_proj/MatMulNBits/output_0.out27_11_31" + ], + "const_args": [ + "model.layers.8.input_layernorm.weight.bf", + "eps_26_1_15" + ], + "out_args": [ + "/model/layers.8/input_layernorm/output_3.out26_1_15", + "/model/layers.8/input_layernorm/output_0.out26_1_15" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.8/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.8/input_layernorm/output_0.out26_1_15" + ], + "const_args": [ + "model.layers.8.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.8.attn.k_proj.Add.bias.preformat", + "model.layers.8.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.8.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.8/attn/k_proj/Add/output_0.out27_11_33" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.8/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.8/input_layernorm/output_0.out26_1_15" + ], + "const_args": [ + "model.layers.8.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.8.attn.q_proj.Add.bias.preformat", + "model.layers.8.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.8.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.8/attn/q_proj/Add/output_0.out27_11_32" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.8/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.8/input_layernorm/output_0.out26_1_15" + ], + "const_args": [ + "model.layers.8.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.8.attn.v_proj.Add.bias.preformat", + "model.layers.8.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.8.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.8.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "35", + "17" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.8/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.8/attn/q_proj/Add/output_0.out27_11_32", + "/model/layers.8/attn/k_proj/Add/output_0.out27_11_33", + "past_key_values.8.key", + "past_key_values.8.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.8/attn/GroupQueryAttention/output_0.out24_0_8", + "present.8.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "32", + "16", + "3", + "0", + "33", + "17", + "8", + "0", + "34", + "16" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.8/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.8/attn/GroupQueryAttention/output_0.out24_0_8" + ], + "const_args": [ + "model.layers.8.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.8.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.8.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.8.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.8/attn/o_proj/MatMulNBits/output_0.out27_11_34" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.8/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.8/input_layernorm/output_3.out26_1_15", + "/model/layers.8/attn/o_proj/MatMulNBits/output_0.out27_11_34" + ], + "const_args": [ + "model.layers.8.post_attention_layernorm.weight.bf", + "eps_26_1_16" + ], + "out_args": [ + "/model/layers.8/post_attention_layernorm/output_3.out26_1_16", + "/model/layers.8/post_attention_layernorm/output_0.out26_1_16" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.8/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.8/post_attention_layernorm/output_0.out26_1_16" + ], + "const_args": [ + "model.layers.8.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.8.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.8.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.8.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.8.mlp.up_proj.MatMulNBits.qweight", + "model.layers.8.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.8.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.8.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.8/mlp/Mul/output_0.out25_0_8" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.8/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.8/mlp/Mul/output_0.out25_0_8" + ], + "const_args": [ + "model.layers.8.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.8.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.8.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.8.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.8/mlp/down_proj/MatMulNBits/output_0.out27_11_35" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.9/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.8/post_attention_layernorm/output_3.out26_1_16", + "/model/layers.8/mlp/down_proj/MatMulNBits/output_0.out27_11_35" + ], + "const_args": [ + "model.layers.9.input_layernorm.weight.bf", + "eps_26_1_17" + ], + "out_args": [ + "/model/layers.9/input_layernorm/output_3.out26_1_17", + "/model/layers.9/input_layernorm/output_0.out26_1_17" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.9/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.9/input_layernorm/output_0.out26_1_17" + ], + "const_args": [ + "model.layers.9.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.9.attn.k_proj.Add.bias.preformat", + "model.layers.9.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.9.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.9/attn/k_proj/Add/output_0.out27_11_37" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.9/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.9/input_layernorm/output_0.out26_1_17" + ], + "const_args": [ + "model.layers.9.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.9.attn.q_proj.Add.bias.preformat", + "model.layers.9.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.9.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.9/attn/q_proj/Add/output_0.out27_11_36" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.9/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.9/input_layernorm/output_0.out26_1_17" + ], + "const_args": [ + "model.layers.9.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.9.attn.v_proj.Add.bias.preformat", + "model.layers.9.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.9.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.9.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "39", + "19" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.9/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.9/attn/q_proj/Add/output_0.out27_11_36", + "/model/layers.9/attn/k_proj/Add/output_0.out27_11_37", + "past_key_values.9.key", + "past_key_values.9.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.9/attn/GroupQueryAttention/output_0.out24_0_9", + "present.9.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "36", + "18", + "3", + "0", + "37", + "19", + "8", + "0", + "38", + "18" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.9/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.9/attn/GroupQueryAttention/output_0.out24_0_9" + ], + "const_args": [ + "model.layers.9.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.9.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.9.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.9.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.9/attn/o_proj/MatMulNBits/output_0.out27_11_38" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.9/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.9/input_layernorm/output_3.out26_1_17", + "/model/layers.9/attn/o_proj/MatMulNBits/output_0.out27_11_38" + ], + "const_args": [ + "model.layers.9.post_attention_layernorm.weight.bf", + "eps_26_1_18" + ], + "out_args": [ + "/model/layers.9/post_attention_layernorm/output_3.out26_1_18", + "/model/layers.9/post_attention_layernorm/output_0.out26_1_18" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.9/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.9/post_attention_layernorm/output_0.out26_1_18" + ], + "const_args": [ + "model.layers.9.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.9.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.9.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.9.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.9.mlp.up_proj.MatMulNBits.qweight", + "model.layers.9.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.9.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.9.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.9/mlp/Mul/output_0.out25_0_9" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.9/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.9/mlp/Mul/output_0.out25_0_9" + ], + "const_args": [ + "model.layers.9.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.9.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.9.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.9.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.9/mlp/down_proj/MatMulNBits/output_0.out27_11_39" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.10/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.9/post_attention_layernorm/output_3.out26_1_18", + "/model/layers.9/mlp/down_proj/MatMulNBits/output_0.out27_11_39" + ], + "const_args": [ + "model.layers.10.input_layernorm.weight.bf", + "eps_26_1_19" + ], + "out_args": [ + "/model/layers.10/input_layernorm/output_3.out26_1_19", + "/model/layers.10/input_layernorm/output_0.out26_1_19" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.10/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.10/input_layernorm/output_0.out26_1_19" + ], + "const_args": [ + "model.layers.10.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.10.attn.k_proj.Add.bias.preformat", + "model.layers.10.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.10.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.10/attn/k_proj/Add/output_0.out27_11_41" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.10/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.10/input_layernorm/output_0.out26_1_19" + ], + "const_args": [ + "model.layers.10.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.10.attn.q_proj.Add.bias.preformat", + "model.layers.10.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.10.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.10/attn/q_proj/Add/output_0.out27_11_40" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.10/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.10/input_layernorm/output_0.out26_1_19" + ], + "const_args": [ + "model.layers.10.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.10.attn.v_proj.Add.bias.preformat", + "model.layers.10.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.10.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.10.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "43", + "21" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.10/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.10/attn/q_proj/Add/output_0.out27_11_40", + "/model/layers.10/attn/k_proj/Add/output_0.out27_11_41", + "past_key_values.10.key", + "past_key_values.10.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.10/attn/GroupQueryAttention/output_0.out24_0_10", + "present.10.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "40", + "20", + "3", + "0", + "41", + "21", + "8", + "0", + "42", + "20" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.10/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.10/attn/GroupQueryAttention/output_0.out24_0_10" + ], + "const_args": [ + "model.layers.10.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.10.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.10.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.10.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.10/attn/o_proj/MatMulNBits/output_0.out27_11_42" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.10/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.10/input_layernorm/output_3.out26_1_19", + "/model/layers.10/attn/o_proj/MatMulNBits/output_0.out27_11_42" + ], + "const_args": [ + "model.layers.10.post_attention_layernorm.weight.bf", + "eps_26_1_20" + ], + "out_args": [ + "/model/layers.10/post_attention_layernorm/output_3.out26_1_20", + "/model/layers.10/post_attention_layernorm/output_0.out26_1_20" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.10/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.10/post_attention_layernorm/output_0.out26_1_20" + ], + "const_args": [ + "model.layers.10.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.10.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.10.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.10.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.10.mlp.up_proj.MatMulNBits.qweight", + "model.layers.10.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.10.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.10.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.10/mlp/Mul/output_0.out25_0_10" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.10/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.10/mlp/Mul/output_0.out25_0_10" + ], + "const_args": [ + "model.layers.10.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.10.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.10.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.10.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.10/mlp/down_proj/MatMulNBits/output_0.out27_11_43" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.11/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.10/post_attention_layernorm/output_3.out26_1_20", + "/model/layers.10/mlp/down_proj/MatMulNBits/output_0.out27_11_43" + ], + "const_args": [ + "model.layers.11.input_layernorm.weight.bf", + "eps_26_1_21" + ], + "out_args": [ + "/model/layers.11/input_layernorm/output_3.out26_1_21", + "/model/layers.11/input_layernorm/output_0.out26_1_21" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.11/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.11/input_layernorm/output_0.out26_1_21" + ], + "const_args": [ + "model.layers.11.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.11.attn.k_proj.Add.bias.preformat", + "model.layers.11.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.11.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.11/attn/k_proj/Add/output_0.out27_11_45" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.11/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.11/input_layernorm/output_0.out26_1_21" + ], + "const_args": [ + "model.layers.11.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.11.attn.q_proj.Add.bias.preformat", + "model.layers.11.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.11.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.11/attn/q_proj/Add/output_0.out27_11_44" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.11/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.11/input_layernorm/output_0.out26_1_21" + ], + "const_args": [ + "model.layers.11.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.11.attn.v_proj.Add.bias.preformat", + "model.layers.11.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.11.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.11.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "47", + "23" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.11/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.11/attn/q_proj/Add/output_0.out27_11_44", + "/model/layers.11/attn/k_proj/Add/output_0.out27_11_45", + "past_key_values.11.key", + "past_key_values.11.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.11/attn/GroupQueryAttention/output_0.out24_0_11", + "present.11.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "44", + "22", + "3", + "0", + "45", + "23", + "8", + "0", + "46", + "22" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.11/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.11/attn/GroupQueryAttention/output_0.out24_0_11" + ], + "const_args": [ + "model.layers.11.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.11.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.11.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.11.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.11/attn/o_proj/MatMulNBits/output_0.out27_11_46" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.11/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.11/input_layernorm/output_3.out26_1_21", + "/model/layers.11/attn/o_proj/MatMulNBits/output_0.out27_11_46" + ], + "const_args": [ + "model.layers.11.post_attention_layernorm.weight.bf", + "eps_26_1_22" + ], + "out_args": [ + "/model/layers.11/post_attention_layernorm/output_3.out26_1_22", + "/model/layers.11/post_attention_layernorm/output_0.out26_1_22" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.11/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.11/post_attention_layernorm/output_0.out26_1_22" + ], + "const_args": [ + "model.layers.11.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.11.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.11.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.11.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.11.mlp.up_proj.MatMulNBits.qweight", + "model.layers.11.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.11.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.11.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.11/mlp/Mul/output_0.out25_0_11" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.11/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.11/mlp/Mul/output_0.out25_0_11" + ], + "const_args": [ + "model.layers.11.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.11.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.11.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.11.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.11/mlp/down_proj/MatMulNBits/output_0.out27_11_47" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.12/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.11/post_attention_layernorm/output_3.out26_1_22", + "/model/layers.11/mlp/down_proj/MatMulNBits/output_0.out27_11_47" + ], + "const_args": [ + "model.layers.12.input_layernorm.weight.bf", + "eps_26_1_23" + ], + "out_args": [ + "/model/layers.12/input_layernorm/output_3.out26_1_23", + "/model/layers.12/input_layernorm/output_0.out26_1_23" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.12/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.12/input_layernorm/output_0.out26_1_23" + ], + "const_args": [ + "model.layers.12.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.12.attn.k_proj.Add.bias.preformat", + "model.layers.12.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.12.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.12/attn/k_proj/Add/output_0.out27_11_49" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.12/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.12/input_layernorm/output_0.out26_1_23" + ], + "const_args": [ + "model.layers.12.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.12.attn.q_proj.Add.bias.preformat", + "model.layers.12.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.12.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.12/attn/q_proj/Add/output_0.out27_11_48" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.12/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.12/input_layernorm/output_0.out26_1_23" + ], + "const_args": [ + "model.layers.12.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.12.attn.v_proj.Add.bias.preformat", + "model.layers.12.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.12.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.12.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "51", + "25" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.12/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.12/attn/q_proj/Add/output_0.out27_11_48", + "/model/layers.12/attn/k_proj/Add/output_0.out27_11_49", + "past_key_values.12.key", + "past_key_values.12.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.12/attn/GroupQueryAttention/output_0.out24_0_12", + "present.12.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "48", + "24", + "3", + "0", + "49", + "25", + "8", + "0", + "50", + "24" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.12/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.12/attn/GroupQueryAttention/output_0.out24_0_12" + ], + "const_args": [ + "model.layers.12.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.12.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.12.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.12.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.12/attn/o_proj/MatMulNBits/output_0.out27_11_50" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.12/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.12/input_layernorm/output_3.out26_1_23", + "/model/layers.12/attn/o_proj/MatMulNBits/output_0.out27_11_50" + ], + "const_args": [ + "model.layers.12.post_attention_layernorm.weight.bf", + "eps_26_1_24" + ], + "out_args": [ + "/model/layers.12/post_attention_layernorm/output_3.out26_1_24", + "/model/layers.12/post_attention_layernorm/output_0.out26_1_24" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.12/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.12/post_attention_layernorm/output_0.out26_1_24" + ], + "const_args": [ + "model.layers.12.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.12.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.12.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.12.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.12.mlp.up_proj.MatMulNBits.qweight", + "model.layers.12.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.12.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.12.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.12/mlp/Mul/output_0.out25_0_12" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.12/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.12/mlp/Mul/output_0.out25_0_12" + ], + "const_args": [ + "model.layers.12.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.12.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.12.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.12.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.12/mlp/down_proj/MatMulNBits/output_0.out27_11_51" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.13/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.12/post_attention_layernorm/output_3.out26_1_24", + "/model/layers.12/mlp/down_proj/MatMulNBits/output_0.out27_11_51" + ], + "const_args": [ + "model.layers.13.input_layernorm.weight.bf", + "eps_26_1_25" + ], + "out_args": [ + "/model/layers.13/input_layernorm/output_3.out26_1_25", + "/model/layers.13/input_layernorm/output_0.out26_1_25" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.13/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.13/input_layernorm/output_0.out26_1_25" + ], + "const_args": [ + "model.layers.13.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.13.attn.k_proj.Add.bias.preformat", + "model.layers.13.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.13.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.13/attn/k_proj/Add/output_0.out27_11_53" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.13/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.13/input_layernorm/output_0.out26_1_25" + ], + "const_args": [ + "model.layers.13.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.13.attn.q_proj.Add.bias.preformat", + "model.layers.13.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.13.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.13/attn/q_proj/Add/output_0.out27_11_52" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.13/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.13/input_layernorm/output_0.out26_1_25" + ], + "const_args": [ + "model.layers.13.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.13.attn.v_proj.Add.bias.preformat", + "model.layers.13.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.13.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.13.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "55", + "27" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.13/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.13/attn/q_proj/Add/output_0.out27_11_52", + "/model/layers.13/attn/k_proj/Add/output_0.out27_11_53", + "past_key_values.13.key", + "past_key_values.13.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.13/attn/GroupQueryAttention/output_0.out24_0_13", + "present.13.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "52", + "26", + "3", + "0", + "53", + "27", + "8", + "0", + "54", + "26" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.13/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.13/attn/GroupQueryAttention/output_0.out24_0_13" + ], + "const_args": [ + "model.layers.13.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.13.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.13.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.13.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.13/attn/o_proj/MatMulNBits/output_0.out27_11_54" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.13/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.13/input_layernorm/output_3.out26_1_25", + "/model/layers.13/attn/o_proj/MatMulNBits/output_0.out27_11_54" + ], + "const_args": [ + "model.layers.13.post_attention_layernorm.weight.bf", + "eps_26_1_26" + ], + "out_args": [ + "/model/layers.13/post_attention_layernorm/output_3.out26_1_26", + "/model/layers.13/post_attention_layernorm/output_0.out26_1_26" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.13/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.13/post_attention_layernorm/output_0.out26_1_26" + ], + "const_args": [ + "model.layers.13.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.13.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.13.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.13.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.13.mlp.up_proj.MatMulNBits.qweight", + "model.layers.13.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.13.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.13.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.13/mlp/Mul/output_0.out25_0_13" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.13/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.13/mlp/Mul/output_0.out25_0_13" + ], + "const_args": [ + "model.layers.13.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.13.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.13.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.13.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.13/mlp/down_proj/MatMulNBits/output_0.out27_11_55" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.14/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.13/post_attention_layernorm/output_3.out26_1_26", + "/model/layers.13/mlp/down_proj/MatMulNBits/output_0.out27_11_55" + ], + "const_args": [ + "model.layers.14.input_layernorm.weight.bf", + "eps_26_1_27" + ], + "out_args": [ + "/model/layers.14/input_layernorm/output_3.out26_1_27", + "/model/layers.14/input_layernorm/output_0.out26_1_27" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.14/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.14/input_layernorm/output_0.out26_1_27" + ], + "const_args": [ + "model.layers.14.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.14.attn.k_proj.Add.bias.preformat", + "model.layers.14.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.14.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.14/attn/k_proj/Add/output_0.out27_11_57" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.14/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.14/input_layernorm/output_0.out26_1_27" + ], + "const_args": [ + "model.layers.14.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.14.attn.q_proj.Add.bias.preformat", + "model.layers.14.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.14.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.14/attn/q_proj/Add/output_0.out27_11_56" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.14/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.14/input_layernorm/output_0.out26_1_27" + ], + "const_args": [ + "model.layers.14.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.14.attn.v_proj.Add.bias.preformat", + "model.layers.14.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.14.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.14.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "59", + "29" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.14/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.14/attn/q_proj/Add/output_0.out27_11_56", + "/model/layers.14/attn/k_proj/Add/output_0.out27_11_57", + "past_key_values.14.key", + "past_key_values.14.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.14/attn/GroupQueryAttention/output_0.out24_0_14", + "present.14.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "56", + "28", + "3", + "0", + "57", + "29", + "8", + "0", + "58", + "28" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.14/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.14/attn/GroupQueryAttention/output_0.out24_0_14" + ], + "const_args": [ + "model.layers.14.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.14.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.14.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.14.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.14/attn/o_proj/MatMulNBits/output_0.out27_11_58" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.14/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.14/input_layernorm/output_3.out26_1_27", + "/model/layers.14/attn/o_proj/MatMulNBits/output_0.out27_11_58" + ], + "const_args": [ + "model.layers.14.post_attention_layernorm.weight.bf", + "eps_26_1_28" + ], + "out_args": [ + "/model/layers.14/post_attention_layernorm/output_3.out26_1_28", + "/model/layers.14/post_attention_layernorm/output_0.out26_1_28" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.14/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.14/post_attention_layernorm/output_0.out26_1_28" + ], + "const_args": [ + "model.layers.14.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.14.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.14.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.14.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.14.mlp.up_proj.MatMulNBits.qweight", + "model.layers.14.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.14.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.14.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.14/mlp/Mul/output_0.out25_0_14" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.14/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.14/mlp/Mul/output_0.out25_0_14" + ], + "const_args": [ + "model.layers.14.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.14.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.14.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.14.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.14/mlp/down_proj/MatMulNBits/output_0.out27_11_59" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.15/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.14/post_attention_layernorm/output_3.out26_1_28", + "/model/layers.14/mlp/down_proj/MatMulNBits/output_0.out27_11_59" + ], + "const_args": [ + "model.layers.15.input_layernorm.weight.bf", + "eps_26_1_29" + ], + "out_args": [ + "/model/layers.15/input_layernorm/output_3.out26_1_29", + "/model/layers.15/input_layernorm/output_0.out26_1_29" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.15/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.15/input_layernorm/output_0.out26_1_29" + ], + "const_args": [ + "model.layers.15.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.15.attn.k_proj.Add.bias.preformat", + "model.layers.15.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.15.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.15/attn/k_proj/Add/output_0.out27_11_61" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.15/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.15/input_layernorm/output_0.out26_1_29" + ], + "const_args": [ + "model.layers.15.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.15.attn.q_proj.Add.bias.preformat", + "model.layers.15.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.15.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.15/attn/q_proj/Add/output_0.out27_11_60" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.15/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.15/input_layernorm/output_0.out26_1_29" + ], + "const_args": [ + "model.layers.15.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.15.attn.v_proj.Add.bias.preformat", + "model.layers.15.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.15.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.15.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "63", + "31" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.15/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.15/attn/q_proj/Add/output_0.out27_11_60", + "/model/layers.15/attn/k_proj/Add/output_0.out27_11_61", + "past_key_values.15.key", + "past_key_values.15.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.15/attn/GroupQueryAttention/output_0.out24_0_15", + "present.15.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "60", + "30", + "3", + "0", + "61", + "31", + "8", + "0", + "62", + "30" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.15/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.15/attn/GroupQueryAttention/output_0.out24_0_15" + ], + "const_args": [ + "model.layers.15.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.15.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.15.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.15.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.15/attn/o_proj/MatMulNBits/output_0.out27_11_62" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.15/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.15/input_layernorm/output_3.out26_1_29", + "/model/layers.15/attn/o_proj/MatMulNBits/output_0.out27_11_62" + ], + "const_args": [ + "model.layers.15.post_attention_layernorm.weight.bf", + "eps_26_1_30" + ], + "out_args": [ + "/model/layers.15/post_attention_layernorm/output_3.out26_1_30", + "/model/layers.15/post_attention_layernorm/output_0.out26_1_30" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.15/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.15/post_attention_layernorm/output_0.out26_1_30" + ], + "const_args": [ + "model.layers.15.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.15.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.15.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.15.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.15.mlp.up_proj.MatMulNBits.qweight", + "model.layers.15.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.15.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.15.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.15/mlp/Mul/output_0.out25_0_15" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.15/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.15/mlp/Mul/output_0.out25_0_15" + ], + "const_args": [ + "model.layers.15.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.15.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.15.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.15.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.15/mlp/down_proj/MatMulNBits/output_0.out27_11_63" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.16/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.15/post_attention_layernorm/output_3.out26_1_30", + "/model/layers.15/mlp/down_proj/MatMulNBits/output_0.out27_11_63" + ], + "const_args": [ + "model.layers.16.input_layernorm.weight.bf", + "eps_26_1_31" + ], + "out_args": [ + "/model/layers.16/input_layernorm/output_3.out26_1_31", + "/model/layers.16/input_layernorm/output_0.out26_1_31" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.16/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.16/input_layernorm/output_0.out26_1_31" + ], + "const_args": [ + "model.layers.16.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.16.attn.k_proj.Add.bias.preformat", + "model.layers.16.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.16.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.16/attn/k_proj/Add/output_0.out27_11_65" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.16/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.16/input_layernorm/output_0.out26_1_31" + ], + "const_args": [ + "model.layers.16.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.16.attn.q_proj.Add.bias.preformat", + "model.layers.16.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.16.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.16/attn/q_proj/Add/output_0.out27_11_64" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.16/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.16/input_layernorm/output_0.out26_1_31" + ], + "const_args": [ + "model.layers.16.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.16.attn.v_proj.Add.bias.preformat", + "model.layers.16.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.16.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.16.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "67", + "33" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.16/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.16/attn/q_proj/Add/output_0.out27_11_64", + "/model/layers.16/attn/k_proj/Add/output_0.out27_11_65", + "past_key_values.16.key", + "past_key_values.16.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.16/attn/GroupQueryAttention/output_0.out24_0_16", + "present.16.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "64", + "32", + "3", + "0", + "65", + "33", + "8", + "0", + "66", + "32" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.16/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.16/attn/GroupQueryAttention/output_0.out24_0_16" + ], + "const_args": [ + "model.layers.16.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.16.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.16.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.16.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.16/attn/o_proj/MatMulNBits/output_0.out27_11_66" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.16/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.16/input_layernorm/output_3.out26_1_31", + "/model/layers.16/attn/o_proj/MatMulNBits/output_0.out27_11_66" + ], + "const_args": [ + "model.layers.16.post_attention_layernorm.weight.bf", + "eps_26_1_32" + ], + "out_args": [ + "/model/layers.16/post_attention_layernorm/output_3.out26_1_32", + "/model/layers.16/post_attention_layernorm/output_0.out26_1_32" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.16/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.16/post_attention_layernorm/output_0.out26_1_32" + ], + "const_args": [ + "model.layers.16.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.16.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.16.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.16.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.16.mlp.up_proj.MatMulNBits.qweight", + "model.layers.16.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.16.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.16.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.16/mlp/Mul/output_0.out25_0_16" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.16/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.16/mlp/Mul/output_0.out25_0_16" + ], + "const_args": [ + "model.layers.16.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.16.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.16.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.16.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.16/mlp/down_proj/MatMulNBits/output_0.out27_11_67" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.17/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.16/post_attention_layernorm/output_3.out26_1_32", + "/model/layers.16/mlp/down_proj/MatMulNBits/output_0.out27_11_67" + ], + "const_args": [ + "model.layers.17.input_layernorm.weight.bf", + "eps_26_1_33" + ], + "out_args": [ + "/model/layers.17/input_layernorm/output_3.out26_1_33", + "/model/layers.17/input_layernorm/output_0.out26_1_33" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.17/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.17/input_layernorm/output_0.out26_1_33" + ], + "const_args": [ + "model.layers.17.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.17.attn.k_proj.Add.bias.preformat", + "model.layers.17.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.17.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.17/attn/k_proj/Add/output_0.out27_11_69" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.17/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.17/input_layernorm/output_0.out26_1_33" + ], + "const_args": [ + "model.layers.17.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.17.attn.q_proj.Add.bias.preformat", + "model.layers.17.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.17.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.17/attn/q_proj/Add/output_0.out27_11_68" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.17/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.17/input_layernorm/output_0.out26_1_33" + ], + "const_args": [ + "model.layers.17.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.17.attn.v_proj.Add.bias.preformat", + "model.layers.17.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.17.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.17.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "71", + "35" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.17/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.17/attn/q_proj/Add/output_0.out27_11_68", + "/model/layers.17/attn/k_proj/Add/output_0.out27_11_69", + "past_key_values.17.key", + "past_key_values.17.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.17/attn/GroupQueryAttention/output_0.out24_0_17", + "present.17.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "68", + "34", + "3", + "0", + "69", + "35", + "8", + "0", + "70", + "34" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.17/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.17/attn/GroupQueryAttention/output_0.out24_0_17" + ], + "const_args": [ + "model.layers.17.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.17.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.17.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.17.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.17/attn/o_proj/MatMulNBits/output_0.out27_11_70" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.17/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.17/input_layernorm/output_3.out26_1_33", + "/model/layers.17/attn/o_proj/MatMulNBits/output_0.out27_11_70" + ], + "const_args": [ + "model.layers.17.post_attention_layernorm.weight.bf", + "eps_26_1_34" + ], + "out_args": [ + "/model/layers.17/post_attention_layernorm/output_3.out26_1_34", + "/model/layers.17/post_attention_layernorm/output_0.out26_1_34" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.17/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.17/post_attention_layernorm/output_0.out26_1_34" + ], + "const_args": [ + "model.layers.17.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.17.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.17.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.17.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.17.mlp.up_proj.MatMulNBits.qweight", + "model.layers.17.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.17.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.17.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.17/mlp/Mul/output_0.out25_0_17" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.17/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.17/mlp/Mul/output_0.out25_0_17" + ], + "const_args": [ + "model.layers.17.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.17.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.17.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.17.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.17/mlp/down_proj/MatMulNBits/output_0.out27_11_71" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.18/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.17/post_attention_layernorm/output_3.out26_1_34", + "/model/layers.17/mlp/down_proj/MatMulNBits/output_0.out27_11_71" + ], + "const_args": [ + "model.layers.18.input_layernorm.weight.bf", + "eps_26_1_35" + ], + "out_args": [ + "/model/layers.18/input_layernorm/output_3.out26_1_35", + "/model/layers.18/input_layernorm/output_0.out26_1_35" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.18/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.18/input_layernorm/output_0.out26_1_35" + ], + "const_args": [ + "model.layers.18.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.18.attn.k_proj.Add.bias.preformat", + "model.layers.18.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.18.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.18/attn/k_proj/Add/output_0.out27_11_73" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.18/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.18/input_layernorm/output_0.out26_1_35" + ], + "const_args": [ + "model.layers.18.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.18.attn.q_proj.Add.bias.preformat", + "model.layers.18.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.18.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.18/attn/q_proj/Add/output_0.out27_11_72" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.18/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.18/input_layernorm/output_0.out26_1_35" + ], + "const_args": [ + "model.layers.18.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.18.attn.v_proj.Add.bias.preformat", + "model.layers.18.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.18.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.18.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "75", + "37" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.18/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.18/attn/q_proj/Add/output_0.out27_11_72", + "/model/layers.18/attn/k_proj/Add/output_0.out27_11_73", + "past_key_values.18.key", + "past_key_values.18.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.18/attn/GroupQueryAttention/output_0.out24_0_18", + "present.18.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "72", + "36", + "3", + "0", + "73", + "37", + "8", + "0", + "74", + "36" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.18/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.18/attn/GroupQueryAttention/output_0.out24_0_18" + ], + "const_args": [ + "model.layers.18.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.18.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.18.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.18.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.18/attn/o_proj/MatMulNBits/output_0.out27_11_74" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.18/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.18/input_layernorm/output_3.out26_1_35", + "/model/layers.18/attn/o_proj/MatMulNBits/output_0.out27_11_74" + ], + "const_args": [ + "model.layers.18.post_attention_layernorm.weight.bf", + "eps_26_1_36" + ], + "out_args": [ + "/model/layers.18/post_attention_layernorm/output_3.out26_1_36", + "/model/layers.18/post_attention_layernorm/output_0.out26_1_36" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.18/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.18/post_attention_layernorm/output_0.out26_1_36" + ], + "const_args": [ + "model.layers.18.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.18.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.18.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.18.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.18.mlp.up_proj.MatMulNBits.qweight", + "model.layers.18.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.18.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.18.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.18/mlp/Mul/output_0.out25_0_18" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.18/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.18/mlp/Mul/output_0.out25_0_18" + ], + "const_args": [ + "model.layers.18.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.18.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.18.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.18.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.18/mlp/down_proj/MatMulNBits/output_0.out27_11_75" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.19/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.18/post_attention_layernorm/output_3.out26_1_36", + "/model/layers.18/mlp/down_proj/MatMulNBits/output_0.out27_11_75" + ], + "const_args": [ + "model.layers.19.input_layernorm.weight.bf", + "eps_26_1_37" + ], + "out_args": [ + "/model/layers.19/input_layernorm/output_3.out26_1_37", + "/model/layers.19/input_layernorm/output_0.out26_1_37" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.19/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.19/input_layernorm/output_0.out26_1_37" + ], + "const_args": [ + "model.layers.19.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.19.attn.k_proj.Add.bias.preformat", + "model.layers.19.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.19.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.19/attn/k_proj/Add/output_0.out27_11_77" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.19/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.19/input_layernorm/output_0.out26_1_37" + ], + "const_args": [ + "model.layers.19.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.19.attn.q_proj.Add.bias.preformat", + "model.layers.19.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.19.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.19/attn/q_proj/Add/output_0.out27_11_76" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.19/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.19/input_layernorm/output_0.out26_1_37" + ], + "const_args": [ + "model.layers.19.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.19.attn.v_proj.Add.bias.preformat", + "model.layers.19.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.19.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.19.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "79", + "39" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.19/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.19/attn/q_proj/Add/output_0.out27_11_76", + "/model/layers.19/attn/k_proj/Add/output_0.out27_11_77", + "past_key_values.19.key", + "past_key_values.19.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.19/attn/GroupQueryAttention/output_0.out24_0_19", + "present.19.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "76", + "38", + "3", + "0", + "77", + "39", + "8", + "0", + "78", + "38" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.19/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.19/attn/GroupQueryAttention/output_0.out24_0_19" + ], + "const_args": [ + "model.layers.19.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.19.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.19.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.19.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.19/attn/o_proj/MatMulNBits/output_0.out27_11_78" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.19/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.19/input_layernorm/output_3.out26_1_37", + "/model/layers.19/attn/o_proj/MatMulNBits/output_0.out27_11_78" + ], + "const_args": [ + "model.layers.19.post_attention_layernorm.weight.bf", + "eps_26_1_38" + ], + "out_args": [ + "/model/layers.19/post_attention_layernorm/output_3.out26_1_38", + "/model/layers.19/post_attention_layernorm/output_0.out26_1_38" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.19/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.19/post_attention_layernorm/output_0.out26_1_38" + ], + "const_args": [ + "model.layers.19.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.19.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.19.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.19.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.19.mlp.up_proj.MatMulNBits.qweight", + "model.layers.19.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.19.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.19.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.19/mlp/Mul/output_0.out25_0_19" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.19/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.19/mlp/Mul/output_0.out25_0_19" + ], + "const_args": [ + "model.layers.19.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.19.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.19.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.19.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.19/mlp/down_proj/MatMulNBits/output_0.out27_11_79" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.20/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.19/post_attention_layernorm/output_3.out26_1_38", + "/model/layers.19/mlp/down_proj/MatMulNBits/output_0.out27_11_79" + ], + "const_args": [ + "model.layers.20.input_layernorm.weight.bf", + "eps_26_1_39" + ], + "out_args": [ + "/model/layers.20/input_layernorm/output_3.out26_1_39", + "/model/layers.20/input_layernorm/output_0.out26_1_39" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.20/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.20/input_layernorm/output_0.out26_1_39" + ], + "const_args": [ + "model.layers.20.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.20.attn.k_proj.Add.bias.preformat", + "model.layers.20.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.20.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.20/attn/k_proj/Add/output_0.out27_11_81" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.20/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.20/input_layernorm/output_0.out26_1_39" + ], + "const_args": [ + "model.layers.20.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.20.attn.q_proj.Add.bias.preformat", + "model.layers.20.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.20.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.20/attn/q_proj/Add/output_0.out27_11_80" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.20/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.20/input_layernorm/output_0.out26_1_39" + ], + "const_args": [ + "model.layers.20.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.20.attn.v_proj.Add.bias.preformat", + "model.layers.20.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.20.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.20.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "83", + "41" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.20/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.20/attn/q_proj/Add/output_0.out27_11_80", + "/model/layers.20/attn/k_proj/Add/output_0.out27_11_81", + "past_key_values.20.key", + "past_key_values.20.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.20/attn/GroupQueryAttention/output_0.out24_0_20", + "present.20.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "80", + "40", + "3", + "0", + "81", + "41", + "8", + "0", + "82", + "40" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.20/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.20/attn/GroupQueryAttention/output_0.out24_0_20" + ], + "const_args": [ + "model.layers.20.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.20.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.20.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.20.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.20/attn/o_proj/MatMulNBits/output_0.out27_11_82" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.20/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.20/input_layernorm/output_3.out26_1_39", + "/model/layers.20/attn/o_proj/MatMulNBits/output_0.out27_11_82" + ], + "const_args": [ + "model.layers.20.post_attention_layernorm.weight.bf", + "eps_26_1_40" + ], + "out_args": [ + "/model/layers.20/post_attention_layernorm/output_3.out26_1_40", + "/model/layers.20/post_attention_layernorm/output_0.out26_1_40" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.20/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.20/post_attention_layernorm/output_0.out26_1_40" + ], + "const_args": [ + "model.layers.20.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.20.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.20.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.20.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.20.mlp.up_proj.MatMulNBits.qweight", + "model.layers.20.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.20.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.20.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.20/mlp/Mul/output_0.out25_0_20" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.20/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.20/mlp/Mul/output_0.out25_0_20" + ], + "const_args": [ + "model.layers.20.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.20.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.20.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.20.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.20/mlp/down_proj/MatMulNBits/output_0.out27_11_83" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.21/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.20/post_attention_layernorm/output_3.out26_1_40", + "/model/layers.20/mlp/down_proj/MatMulNBits/output_0.out27_11_83" + ], + "const_args": [ + "model.layers.21.input_layernorm.weight.bf", + "eps_26_1_41" + ], + "out_args": [ + "/model/layers.21/input_layernorm/output_3.out26_1_41", + "/model/layers.21/input_layernorm/output_0.out26_1_41" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.21/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.21/input_layernorm/output_0.out26_1_41" + ], + "const_args": [ + "model.layers.21.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.21.attn.k_proj.Add.bias.preformat", + "model.layers.21.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.21.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.21/attn/k_proj/Add/output_0.out27_11_85" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.21/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.21/input_layernorm/output_0.out26_1_41" + ], + "const_args": [ + "model.layers.21.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.21.attn.q_proj.Add.bias.preformat", + "model.layers.21.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.21.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.21/attn/q_proj/Add/output_0.out27_11_84" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.21/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.21/input_layernorm/output_0.out26_1_41" + ], + "const_args": [ + "model.layers.21.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.21.attn.v_proj.Add.bias.preformat", + "model.layers.21.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.21.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.21.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "87", + "43" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.21/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.21/attn/q_proj/Add/output_0.out27_11_84", + "/model/layers.21/attn/k_proj/Add/output_0.out27_11_85", + "past_key_values.21.key", + "past_key_values.21.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.21/attn/GroupQueryAttention/output_0.out24_0_21", + "present.21.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "84", + "42", + "3", + "0", + "85", + "43", + "8", + "0", + "86", + "42" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.21/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.21/attn/GroupQueryAttention/output_0.out24_0_21" + ], + "const_args": [ + "model.layers.21.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.21.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.21.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.21.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.21/attn/o_proj/MatMulNBits/output_0.out27_11_86" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.21/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.21/input_layernorm/output_3.out26_1_41", + "/model/layers.21/attn/o_proj/MatMulNBits/output_0.out27_11_86" + ], + "const_args": [ + "model.layers.21.post_attention_layernorm.weight.bf", + "eps_26_1_42" + ], + "out_args": [ + "/model/layers.21/post_attention_layernorm/output_3.out26_1_42", + "/model/layers.21/post_attention_layernorm/output_0.out26_1_42" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.21/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.21/post_attention_layernorm/output_0.out26_1_42" + ], + "const_args": [ + "model.layers.21.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.21.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.21.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.21.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.21.mlp.up_proj.MatMulNBits.qweight", + "model.layers.21.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.21.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.21.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.21/mlp/Mul/output_0.out25_0_21" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.21/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.21/mlp/Mul/output_0.out25_0_21" + ], + "const_args": [ + "model.layers.21.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.21.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.21.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.21.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.21/mlp/down_proj/MatMulNBits/output_0.out27_11_87" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.22/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.21/post_attention_layernorm/output_3.out26_1_42", + "/model/layers.21/mlp/down_proj/MatMulNBits/output_0.out27_11_87" + ], + "const_args": [ + "model.layers.22.input_layernorm.weight.bf", + "eps_26_1_43" + ], + "out_args": [ + "/model/layers.22/input_layernorm/output_3.out26_1_43", + "/model/layers.22/input_layernorm/output_0.out26_1_43" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.22/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.22/input_layernorm/output_0.out26_1_43" + ], + "const_args": [ + "model.layers.22.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.22.attn.k_proj.Add.bias.preformat", + "model.layers.22.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.22.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.22/attn/k_proj/Add/output_0.out27_11_89" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.22/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.22/input_layernorm/output_0.out26_1_43" + ], + "const_args": [ + "model.layers.22.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.22.attn.q_proj.Add.bias.preformat", + "model.layers.22.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.22.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.22/attn/q_proj/Add/output_0.out27_11_88" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.22/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.22/input_layernorm/output_0.out26_1_43" + ], + "const_args": [ + "model.layers.22.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.22.attn.v_proj.Add.bias.preformat", + "model.layers.22.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.22.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.22.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "91", + "45" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.22/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.22/attn/q_proj/Add/output_0.out27_11_88", + "/model/layers.22/attn/k_proj/Add/output_0.out27_11_89", + "past_key_values.22.key", + "past_key_values.22.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.22/attn/GroupQueryAttention/output_0.out24_0_22", + "present.22.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "88", + "44", + "3", + "0", + "89", + "45", + "8", + "0", + "90", + "44" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.22/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.22/attn/GroupQueryAttention/output_0.out24_0_22" + ], + "const_args": [ + "model.layers.22.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.22.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.22.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.22.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.22/attn/o_proj/MatMulNBits/output_0.out27_11_90" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.22/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.22/input_layernorm/output_3.out26_1_43", + "/model/layers.22/attn/o_proj/MatMulNBits/output_0.out27_11_90" + ], + "const_args": [ + "model.layers.22.post_attention_layernorm.weight.bf", + "eps_26_1_44" + ], + "out_args": [ + "/model/layers.22/post_attention_layernorm/output_3.out26_1_44", + "/model/layers.22/post_attention_layernorm/output_0.out26_1_44" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.22/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.22/post_attention_layernorm/output_0.out26_1_44" + ], + "const_args": [ + "model.layers.22.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.22.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.22.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.22.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.22.mlp.up_proj.MatMulNBits.qweight", + "model.layers.22.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.22.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.22.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.22/mlp/Mul/output_0.out25_0_22" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.22/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.22/mlp/Mul/output_0.out25_0_22" + ], + "const_args": [ + "model.layers.22.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.22.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.22.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.22.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.22/mlp/down_proj/MatMulNBits/output_0.out27_11_91" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.23/input_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.22/post_attention_layernorm/output_3.out26_1_44", + "/model/layers.22/mlp/down_proj/MatMulNBits/output_0.out27_11_91" + ], + "const_args": [ + "model.layers.23.input_layernorm.weight.bf", + "eps_26_1_45" + ], + "out_args": [ + "/model/layers.23/input_layernorm/output_3.out26_1_45", + "/model/layers.23/input_layernorm/output_0.out26_1_45" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.23/attn/k_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.23/input_layernorm/output_0.out26_1_45" + ], + "const_args": [ + "model.layers.23.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.23.attn.k_proj.Add.bias.preformat", + "model.layers.23.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.23.attn.k_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.23/attn/k_proj/Add/output_0.out27_11_93" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.23/attn/q_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.23/input_layernorm/output_0.out26_1_45" + ], + "const_args": [ + "model.layers.23.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.23.attn.q_proj.Add.bias.preformat", + "model.layers.23.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.23.attn.q_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.23/attn/q_proj/Add/output_0.out27_11_92" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.23/attn/v_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.23/input_layernorm/output_0.out26_1_45" + ], + "const_args": [ + "model.layers.23.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.23.attn.v_proj.Add.bias.preformat", + "model.layers.23.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.23.attn.v_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "present.23.value" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "128" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "total_seq_len": { + "type": "int", + "value": [ + "4096" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "5", + "0", + "95", + "47" + ] + }, + "head_num": { + "type": "int", + "value": [ + "2" + ] + }, + "use_gm": { + "type": "int", + "value": [ + "1" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "5", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.23/attn/GroupQueryAttention", + "type": "FLATMHA", + "in_args": [ + "/model/layers.23/attn/q_proj/Add/output_0.out27_11_92", + "/model/layers.23/attn/k_proj/Add/output_0.out27_11_93", + "past_key_values.23.key", + "past_key_values.23.value", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ], + "const_args": [], + "out_args": [ + "/model/layers.23/attn/GroupQueryAttention/output_0.out24_0_23", + "present.23.key" + ], + "attrs": { + "local_window_size": { + "type": "int", + "value": [ + "-1" + ] + }, + "qk_output": { + "type": "int", + "value": [ + "0" + ] + }, + "num_heads": { + "type": "int", + "value": [ + "14" + ] + }, + "kv_num_heads": { + "type": "int", + "value": [ + "2" + ] + }, + "softcap": { + "type": "float", + "value": [ + "0.0" + ] + }, + "scale": { + "type": "float", + "value": [ + "0.125" + ] + }, + "rotary_interleaved": { + "type": "int", + "value": [ + "0" + ] + }, + "smooth_softmax": { + "type": "int", + "value": [ + "-1" + ] + }, + "k_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "v_quant_type": { + "type": "str", + "value": [ + "NONE" + ] + }, + "do_rotary": { + "type": "int", + "value": [ + "1" + ] + }, + "sliding_window": { + "type": "int", + "value": [ + "0" + ] + }, + "input_shape": { + "type": "str", + "value": [ + "2", + "14", + "1", + "attention_mask_padded", + "64", + "4096" + ] + }, + "split_qk": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "split_ext_buf": { + "type": "int", + "value": [ + "0" + ] + }, + "external_buffers": { + "type": "int", + "value": [ + "2", + "0", + "92", + "46", + "3", + "0", + "93", + "47", + "8", + "0", + "94", + "46" + ] + }, + "update_tensor_offsets": { + "type": "int", + "value": [ + "8", + "0", + "0", + "128" + ] + } + } + }, + { + "name": "/model/layers.23/attn/o_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.23/attn/GroupQueryAttention/output_0.out24_0_23" + ], + "const_args": [ + "model.layers.23.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.23.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.23.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.23.attn.o_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.23/attn/o_proj/MatMulNBits/output_0.out27_11_94" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.23/post_attention_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.23/input_layernorm/output_3.out26_1_45", + "/model/layers.23/attn/o_proj/MatMulNBits/output_0.out27_11_94" + ], + "const_args": [ + "model.layers.23.post_attention_layernorm.weight.bf", + "eps_26_1_46" + ], + "out_args": [ + "/model/layers.23/post_attention_layernorm/output_3.out26_1_46", + "/model/layers.23/post_attention_layernorm/output_0.out26_1_46" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.23/mlp/FlatMLP", + "type": "FlatMLP", + "in_args": [ + "/model/layers.23/post_attention_layernorm/output_0.out26_1_46" + ], + "const_args": [ + "model.layers.23.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.23.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.23.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.23.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.23.mlp.up_proj.MatMulNBits.qweight", + "model.layers.23.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.23.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.23.mlp.up_proj.MatMulNBits.bias.f" + ], + "out_args": [ + "/model/layers.23/mlp/Mul/output_0.out25_0_23" + ], + "attrs": { + "input_shape": { + "type": "int", + "value": [ + "1", + "896", + "4864" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float", + "uint8", + "float" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/model/layers.23/mlp/down_proj/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.23/mlp/Mul/output_0.out25_0_23" + ], + "const_args": [ + "model.layers.23.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.23.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.23.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.23.mlp.down_proj.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "/model/layers.23/mlp/down_proj/MatMulNBits/output_0.out27_11_95" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "896" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "4864" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + }, + { + "name": "/model/layers.24/final_norm_layernorm/FlatRMSAdd", + "type": "FlatRMSAdd", + "in_args": [ + "/model/layers.23/post_attention_layernorm/output_3.out26_1_46", + "/model/layers.23/mlp/down_proj/MatMulNBits/output_0.out27_11_95" + ], + "const_args": [ + "model.layers.24.final_norm_layernorm.weight.bf", + "eps_26_1_47" + ], + "out_args": [ + "/model/layers.24/final_norm_layernorm/output_0.dummy", + "/model/layers.24/final_norm_layernorm/output_0.out26_1_47" + ], + "attrs": { + "a_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "in_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "out_dtypes": { + "type": "str", + "value": [ + "bfloat16", + "bfloat16" + ] + }, + "c_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "b_shape": { + "type": "int", + "value": [ + "1", + "1", + "896" + ] + }, + "is_gamma_ifm": { + "type": "int", + "value": [ + "1" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + } + } + }, + { + "name": "/lm_head/MatMulNBits", + "type": "MladfMatMul", + "in_args": [ + "/model/layers.24/final_norm_layernorm/output_0.out26_1_47" + ], + "const_args": [ + "lm_head.MatMulNBits.qweight.preformat", + "lm_head.MatMulNBits.bias.preformat", + "lm_head.MatMulNBits.scales.preformat", + "lm_head.MatMulNBits.qzeros.preformat" + ], + "out_args": [ + "logits.out27_11_96" + ], + "attrs": { + "accuracy_level": { + "type": "int", + "value": [ + "0" + ] + }, + "bits": { + "type": "int", + "value": [ + "4" + ] + }, + "N": { + "type": "int", + "value": [ + "151936" + ] + }, + "block_size": { + "type": "int", + "value": [ + "128" + ] + }, + "K": { + "type": "int", + "value": [ + "896" + ] + }, + "enable_ctrl_pkt": { + "type": "int", + "value": [ + "1" + ] + }, + "op_version": { + "type": "str", + "value": [ + "flat" + ] + }, + "default_shape": { + "type": "int", + "value": [ + "1" + ] + }, + "offload_npu": { + "type": "int", + "value": [ + "1" + ] + }, + "group_size": { + "type": "int", + "value": [ + "128" + ] + } + } + } + ], + "fused_tensors": { + "in": { + "buffer_size": 1932, + "xrt_arg_id": 0, + "packed_tensors": [ + "/model/embed_tokens/Gather/output_0.out27_14_0", + "attention_mask_const_uint", + "sin_cache_cos_cache_sliced", + "attention_mask_padded" + ] + }, + "out": { + "buffer_size": 305664, + "xrt_arg_id": 1, + "packed_tensors": [ + "/model/layers.24/final_norm_layernorm/output_0.dummy", + "logits.out27_11_96" + ] + }, + "scratch": { + "buffer_size": 583680, + "xrt_arg_id": 2, + "packed_tensors": [ + "/model/layers.0/input_layernorm/output_0.out27_14_0", + "/model/layers.0/attn/k_proj/Add/output_0.out27_11_1", + "/model/layers.0/attn/q_proj/Add/output_0.out27_11_0", + "/model/layers.0/attn/GroupQueryAttention/output_0.out24_0_0", + "/model/layers.0/attn/o_proj/MatMulNBits/output_0.out27_11_2", + "/model/layers.0/post_attention_layernorm/output_3.out26_1_0", + "/model/layers.0/post_attention_layernorm/output_0.out26_1_0", + "/model/layers.0/mlp/Mul/output_0.out25_0_0", + "/model/layers.0/mlp/down_proj/MatMulNBits/output_0.out27_11_3", + "/model/layers.1/input_layernorm/output_3.out26_1_1", + "/model/layers.1/input_layernorm/output_0.out26_1_1", + "/model/layers.1/attn/k_proj/Add/output_0.out27_11_5", + "/model/layers.1/attn/q_proj/Add/output_0.out27_11_4", + "/model/layers.1/attn/GroupQueryAttention/output_0.out24_0_1", + "/model/layers.1/attn/o_proj/MatMulNBits/output_0.out27_11_6", + "/model/layers.1/post_attention_layernorm/output_3.out26_1_2", + "/model/layers.1/post_attention_layernorm/output_0.out26_1_2", + "/model/layers.1/mlp/Mul/output_0.out25_0_1", + "/model/layers.1/mlp/down_proj/MatMulNBits/output_0.out27_11_7", + "/model/layers.2/input_layernorm/output_3.out26_1_3", + "/model/layers.2/input_layernorm/output_0.out26_1_3", + "/model/layers.2/attn/k_proj/Add/output_0.out27_11_9", + "/model/layers.2/attn/q_proj/Add/output_0.out27_11_8", + "/model/layers.2/attn/GroupQueryAttention/output_0.out24_0_2", + "/model/layers.2/attn/o_proj/MatMulNBits/output_0.out27_11_10", + "/model/layers.2/post_attention_layernorm/output_3.out26_1_4", + "/model/layers.2/post_attention_layernorm/output_0.out26_1_4", + "/model/layers.2/mlp/Mul/output_0.out25_0_2", + "/model/layers.2/mlp/down_proj/MatMulNBits/output_0.out27_11_11", + "/model/layers.3/input_layernorm/output_3.out26_1_5", + "/model/layers.3/input_layernorm/output_0.out26_1_5", + "/model/layers.3/attn/k_proj/Add/output_0.out27_11_13", + "/model/layers.3/attn/q_proj/Add/output_0.out27_11_12", + "/model/layers.3/attn/GroupQueryAttention/output_0.out24_0_3", + "/model/layers.3/attn/o_proj/MatMulNBits/output_0.out27_11_14", + "/model/layers.3/post_attention_layernorm/output_3.out26_1_6", + "/model/layers.3/post_attention_layernorm/output_0.out26_1_6", + "/model/layers.3/mlp/Mul/output_0.out25_0_3", + "/model/layers.3/mlp/down_proj/MatMulNBits/output_0.out27_11_15", + "/model/layers.4/input_layernorm/output_3.out26_1_7", + "/model/layers.4/input_layernorm/output_0.out26_1_7", + "/model/layers.4/attn/k_proj/Add/output_0.out27_11_17", + "/model/layers.4/attn/q_proj/Add/output_0.out27_11_16", + "/model/layers.4/attn/GroupQueryAttention/output_0.out24_0_4", + "/model/layers.4/attn/o_proj/MatMulNBits/output_0.out27_11_18", + "/model/layers.4/post_attention_layernorm/output_3.out26_1_8", + "/model/layers.4/post_attention_layernorm/output_0.out26_1_8", + "/model/layers.4/mlp/Mul/output_0.out25_0_4", + "/model/layers.4/mlp/down_proj/MatMulNBits/output_0.out27_11_19", + "/model/layers.5/input_layernorm/output_3.out26_1_9", + "/model/layers.5/input_layernorm/output_0.out26_1_9", + "/model/layers.5/attn/k_proj/Add/output_0.out27_11_21", + "/model/layers.5/attn/q_proj/Add/output_0.out27_11_20", + "/model/layers.5/attn/GroupQueryAttention/output_0.out24_0_5", + "/model/layers.5/attn/o_proj/MatMulNBits/output_0.out27_11_22", + "/model/layers.5/post_attention_layernorm/output_3.out26_1_10", + "/model/layers.5/post_attention_layernorm/output_0.out26_1_10", + "/model/layers.5/mlp/Mul/output_0.out25_0_5", + "/model/layers.5/mlp/down_proj/MatMulNBits/output_0.out27_11_23", + "/model/layers.6/input_layernorm/output_3.out26_1_11", + "/model/layers.6/input_layernorm/output_0.out26_1_11", + "/model/layers.6/attn/k_proj/Add/output_0.out27_11_25", + "/model/layers.6/attn/q_proj/Add/output_0.out27_11_24", + "/model/layers.6/attn/GroupQueryAttention/output_0.out24_0_6", + "/model/layers.6/attn/o_proj/MatMulNBits/output_0.out27_11_26", + "/model/layers.6/post_attention_layernorm/output_3.out26_1_12", + "/model/layers.6/post_attention_layernorm/output_0.out26_1_12", + "/model/layers.6/mlp/Mul/output_0.out25_0_6", + "/model/layers.6/mlp/down_proj/MatMulNBits/output_0.out27_11_27", + "/model/layers.7/input_layernorm/output_3.out26_1_13", + "/model/layers.7/input_layernorm/output_0.out26_1_13", + "/model/layers.7/attn/k_proj/Add/output_0.out27_11_29", + "/model/layers.7/attn/q_proj/Add/output_0.out27_11_28", + "/model/layers.7/attn/GroupQueryAttention/output_0.out24_0_7", + "/model/layers.7/attn/o_proj/MatMulNBits/output_0.out27_11_30", + "/model/layers.7/post_attention_layernorm/output_3.out26_1_14", + "/model/layers.7/post_attention_layernorm/output_0.out26_1_14", + "/model/layers.7/mlp/Mul/output_0.out25_0_7", + "/model/layers.7/mlp/down_proj/MatMulNBits/output_0.out27_11_31", + "/model/layers.8/input_layernorm/output_3.out26_1_15", + "/model/layers.8/input_layernorm/output_0.out26_1_15", + "/model/layers.8/attn/k_proj/Add/output_0.out27_11_33", + "/model/layers.8/attn/q_proj/Add/output_0.out27_11_32", + "/model/layers.8/attn/GroupQueryAttention/output_0.out24_0_8", + "/model/layers.8/attn/o_proj/MatMulNBits/output_0.out27_11_34", + "/model/layers.8/post_attention_layernorm/output_3.out26_1_16", + "/model/layers.8/post_attention_layernorm/output_0.out26_1_16", + "/model/layers.8/mlp/Mul/output_0.out25_0_8", + "/model/layers.8/mlp/down_proj/MatMulNBits/output_0.out27_11_35", + "/model/layers.9/input_layernorm/output_3.out26_1_17", + "/model/layers.9/input_layernorm/output_0.out26_1_17", + "/model/layers.9/attn/k_proj/Add/output_0.out27_11_37", + "/model/layers.9/attn/q_proj/Add/output_0.out27_11_36", + "/model/layers.9/attn/GroupQueryAttention/output_0.out24_0_9", + "/model/layers.9/attn/o_proj/MatMulNBits/output_0.out27_11_38", + "/model/layers.9/post_attention_layernorm/output_3.out26_1_18", + "/model/layers.9/post_attention_layernorm/output_0.out26_1_18", + "/model/layers.9/mlp/Mul/output_0.out25_0_9", + "/model/layers.9/mlp/down_proj/MatMulNBits/output_0.out27_11_39", + "/model/layers.10/input_layernorm/output_3.out26_1_19", + "/model/layers.10/input_layernorm/output_0.out26_1_19", + "/model/layers.10/attn/k_proj/Add/output_0.out27_11_41", + "/model/layers.10/attn/q_proj/Add/output_0.out27_11_40", + "/model/layers.10/attn/GroupQueryAttention/output_0.out24_0_10", + "/model/layers.10/attn/o_proj/MatMulNBits/output_0.out27_11_42", + "/model/layers.10/post_attention_layernorm/output_3.out26_1_20", + "/model/layers.10/post_attention_layernorm/output_0.out26_1_20", + "/model/layers.10/mlp/Mul/output_0.out25_0_10", + "/model/layers.10/mlp/down_proj/MatMulNBits/output_0.out27_11_43", + "/model/layers.11/input_layernorm/output_3.out26_1_21", + "/model/layers.11/input_layernorm/output_0.out26_1_21", + "/model/layers.11/attn/k_proj/Add/output_0.out27_11_45", + "/model/layers.11/attn/q_proj/Add/output_0.out27_11_44", + "/model/layers.11/attn/GroupQueryAttention/output_0.out24_0_11", + "/model/layers.11/attn/o_proj/MatMulNBits/output_0.out27_11_46", + "/model/layers.11/post_attention_layernorm/output_3.out26_1_22", + "/model/layers.11/post_attention_layernorm/output_0.out26_1_22", + "/model/layers.11/mlp/Mul/output_0.out25_0_11", + "/model/layers.11/mlp/down_proj/MatMulNBits/output_0.out27_11_47", + "/model/layers.12/input_layernorm/output_3.out26_1_23", + "/model/layers.12/input_layernorm/output_0.out26_1_23", + "/model/layers.12/attn/k_proj/Add/output_0.out27_11_49", + "/model/layers.12/attn/q_proj/Add/output_0.out27_11_48", + "/model/layers.12/attn/GroupQueryAttention/output_0.out24_0_12", + "/model/layers.12/attn/o_proj/MatMulNBits/output_0.out27_11_50", + "/model/layers.12/post_attention_layernorm/output_3.out26_1_24", + "/model/layers.12/post_attention_layernorm/output_0.out26_1_24", + "/model/layers.12/mlp/Mul/output_0.out25_0_12", + "/model/layers.12/mlp/down_proj/MatMulNBits/output_0.out27_11_51", + "/model/layers.13/input_layernorm/output_3.out26_1_25", + "/model/layers.13/input_layernorm/output_0.out26_1_25", + "/model/layers.13/attn/k_proj/Add/output_0.out27_11_53", + "/model/layers.13/attn/q_proj/Add/output_0.out27_11_52", + "/model/layers.13/attn/GroupQueryAttention/output_0.out24_0_13", + "/model/layers.13/attn/o_proj/MatMulNBits/output_0.out27_11_54", + "/model/layers.13/post_attention_layernorm/output_3.out26_1_26", + "/model/layers.13/post_attention_layernorm/output_0.out26_1_26", + "/model/layers.13/mlp/Mul/output_0.out25_0_13", + "/model/layers.13/mlp/down_proj/MatMulNBits/output_0.out27_11_55", + "/model/layers.14/input_layernorm/output_3.out26_1_27", + "/model/layers.14/input_layernorm/output_0.out26_1_27", + "/model/layers.14/attn/k_proj/Add/output_0.out27_11_57", + "/model/layers.14/attn/q_proj/Add/output_0.out27_11_56", + "/model/layers.14/attn/GroupQueryAttention/output_0.out24_0_14", + "/model/layers.14/attn/o_proj/MatMulNBits/output_0.out27_11_58", + "/model/layers.14/post_attention_layernorm/output_3.out26_1_28", + "/model/layers.14/post_attention_layernorm/output_0.out26_1_28", + "/model/layers.14/mlp/Mul/output_0.out25_0_14", + "/model/layers.14/mlp/down_proj/MatMulNBits/output_0.out27_11_59", + "/model/layers.15/input_layernorm/output_3.out26_1_29", + "/model/layers.15/input_layernorm/output_0.out26_1_29", + "/model/layers.15/attn/k_proj/Add/output_0.out27_11_61", + "/model/layers.15/attn/q_proj/Add/output_0.out27_11_60", + "/model/layers.15/attn/GroupQueryAttention/output_0.out24_0_15", + "/model/layers.15/attn/o_proj/MatMulNBits/output_0.out27_11_62", + "/model/layers.15/post_attention_layernorm/output_3.out26_1_30", + "/model/layers.15/post_attention_layernorm/output_0.out26_1_30", + "/model/layers.15/mlp/Mul/output_0.out25_0_15", + "/model/layers.15/mlp/down_proj/MatMulNBits/output_0.out27_11_63", + "/model/layers.16/input_layernorm/output_3.out26_1_31", + "/model/layers.16/input_layernorm/output_0.out26_1_31", + "/model/layers.16/attn/k_proj/Add/output_0.out27_11_65", + "/model/layers.16/attn/q_proj/Add/output_0.out27_11_64", + "/model/layers.16/attn/GroupQueryAttention/output_0.out24_0_16", + "/model/layers.16/attn/o_proj/MatMulNBits/output_0.out27_11_66", + "/model/layers.16/post_attention_layernorm/output_3.out26_1_32", + "/model/layers.16/post_attention_layernorm/output_0.out26_1_32", + "/model/layers.16/mlp/Mul/output_0.out25_0_16", + "/model/layers.16/mlp/down_proj/MatMulNBits/output_0.out27_11_67", + "/model/layers.17/input_layernorm/output_3.out26_1_33", + "/model/layers.17/input_layernorm/output_0.out26_1_33", + "/model/layers.17/attn/k_proj/Add/output_0.out27_11_69", + "/model/layers.17/attn/q_proj/Add/output_0.out27_11_68", + "/model/layers.17/attn/GroupQueryAttention/output_0.out24_0_17", + "/model/layers.17/attn/o_proj/MatMulNBits/output_0.out27_11_70", + "/model/layers.17/post_attention_layernorm/output_3.out26_1_34", + "/model/layers.17/post_attention_layernorm/output_0.out26_1_34", + "/model/layers.17/mlp/Mul/output_0.out25_0_17", + "/model/layers.17/mlp/down_proj/MatMulNBits/output_0.out27_11_71", + "/model/layers.18/input_layernorm/output_3.out26_1_35", + "/model/layers.18/input_layernorm/output_0.out26_1_35", + "/model/layers.18/attn/k_proj/Add/output_0.out27_11_73", + "/model/layers.18/attn/q_proj/Add/output_0.out27_11_72", + "/model/layers.18/attn/GroupQueryAttention/output_0.out24_0_18", + "/model/layers.18/attn/o_proj/MatMulNBits/output_0.out27_11_74", + "/model/layers.18/post_attention_layernorm/output_3.out26_1_36", + "/model/layers.18/post_attention_layernorm/output_0.out26_1_36", + "/model/layers.18/mlp/Mul/output_0.out25_0_18", + "/model/layers.18/mlp/down_proj/MatMulNBits/output_0.out27_11_75", + "/model/layers.19/input_layernorm/output_3.out26_1_37", + "/model/layers.19/input_layernorm/output_0.out26_1_37", + "/model/layers.19/attn/k_proj/Add/output_0.out27_11_77", + "/model/layers.19/attn/q_proj/Add/output_0.out27_11_76", + "/model/layers.19/attn/GroupQueryAttention/output_0.out24_0_19", + "/model/layers.19/attn/o_proj/MatMulNBits/output_0.out27_11_78", + "/model/layers.19/post_attention_layernorm/output_3.out26_1_38", + "/model/layers.19/post_attention_layernorm/output_0.out26_1_38", + "/model/layers.19/mlp/Mul/output_0.out25_0_19", + "/model/layers.19/mlp/down_proj/MatMulNBits/output_0.out27_11_79", + "/model/layers.20/input_layernorm/output_3.out26_1_39", + "/model/layers.20/input_layernorm/output_0.out26_1_39", + "/model/layers.20/attn/k_proj/Add/output_0.out27_11_81", + "/model/layers.20/attn/q_proj/Add/output_0.out27_11_80", + "/model/layers.20/attn/GroupQueryAttention/output_0.out24_0_20", + "/model/layers.20/attn/o_proj/MatMulNBits/output_0.out27_11_82", + "/model/layers.20/post_attention_layernorm/output_3.out26_1_40", + "/model/layers.20/post_attention_layernorm/output_0.out26_1_40", + "/model/layers.20/mlp/Mul/output_0.out25_0_20", + "/model/layers.20/mlp/down_proj/MatMulNBits/output_0.out27_11_83", + "/model/layers.21/input_layernorm/output_3.out26_1_41", + "/model/layers.21/input_layernorm/output_0.out26_1_41", + "/model/layers.21/attn/k_proj/Add/output_0.out27_11_85", + "/model/layers.21/attn/q_proj/Add/output_0.out27_11_84", + "/model/layers.21/attn/GroupQueryAttention/output_0.out24_0_21", + "/model/layers.21/attn/o_proj/MatMulNBits/output_0.out27_11_86", + "/model/layers.21/post_attention_layernorm/output_3.out26_1_42", + "/model/layers.21/post_attention_layernorm/output_0.out26_1_42", + "/model/layers.21/mlp/Mul/output_0.out25_0_21", + "/model/layers.21/mlp/down_proj/MatMulNBits/output_0.out27_11_87", + "/model/layers.22/input_layernorm/output_3.out26_1_43", + "/model/layers.22/input_layernorm/output_0.out26_1_43", + "/model/layers.22/attn/k_proj/Add/output_0.out27_11_89", + "/model/layers.22/attn/q_proj/Add/output_0.out27_11_88", + "/model/layers.22/attn/GroupQueryAttention/output_0.out24_0_22", + "/model/layers.22/attn/o_proj/MatMulNBits/output_0.out27_11_90", + "/model/layers.22/post_attention_layernorm/output_3.out26_1_44", + "/model/layers.22/post_attention_layernorm/output_0.out26_1_44", + "/model/layers.22/mlp/Mul/output_0.out25_0_22", + "/model/layers.22/mlp/down_proj/MatMulNBits/output_0.out27_11_91", + "/model/layers.23/input_layernorm/output_3.out26_1_45", + "/model/layers.23/input_layernorm/output_0.out26_1_45", + "/model/layers.23/attn/k_proj/Add/output_0.out27_11_93", + "/model/layers.23/attn/q_proj/Add/output_0.out27_11_92", + "/model/layers.23/attn/GroupQueryAttention/output_0.out24_0_23", + "/model/layers.23/attn/o_proj/MatMulNBits/output_0.out27_11_94", + "/model/layers.23/post_attention_layernorm/output_3.out26_1_46", + "/model/layers.23/post_attention_layernorm/output_0.out26_1_46", + "/model/layers.23/mlp/Mul/output_0.out25_0_23", + "/model/layers.23/mlp/down_proj/MatMulNBits/output_0.out27_11_95", + "/model/layers.24/final_norm_layernorm/output_0.out26_1_47" + ] + }, + "const": { + "buffer_size": 409872964, + "xrt_arg_id": 3, + "packed_tensors": [ + "model.layers.0.input_layernorm.weight", + "eps_27_14_0", + "model.layers.0.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.0.attn.k_proj.Add.bias.preformat", + "model.layers.0.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.0.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.0.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.0.attn.q_proj.Add.bias.preformat", + "model.layers.0.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.0.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.0.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.0.attn.v_proj.Add.bias.preformat", + "model.layers.0.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.0.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.0.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.0.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.0.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.0.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.0.post_attention_layernorm.weight.bf", + "eps_26_1_0", + "model.layers.0.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.0.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.0.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.0.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.0.mlp.up_proj.MatMulNBits.qweight", + "model.layers.0.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.0.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.0.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.0.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.0.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.0.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.0.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.1.input_layernorm.weight.bf", + "eps_26_1_1", + "model.layers.1.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.1.attn.k_proj.Add.bias.preformat", + "model.layers.1.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.1.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.1.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.1.attn.q_proj.Add.bias.preformat", + "model.layers.1.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.1.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.1.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.1.attn.v_proj.Add.bias.preformat", + "model.layers.1.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.1.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.1.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.1.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.1.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.1.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.1.post_attention_layernorm.weight.bf", + "eps_26_1_2", + "model.layers.1.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.1.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.1.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.1.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.1.mlp.up_proj.MatMulNBits.qweight", + "model.layers.1.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.1.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.1.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.1.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.1.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.1.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.1.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.2.input_layernorm.weight.bf", + "eps_26_1_3", + "model.layers.2.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.2.attn.k_proj.Add.bias.preformat", + "model.layers.2.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.2.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.2.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.2.attn.q_proj.Add.bias.preformat", + "model.layers.2.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.2.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.2.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.2.attn.v_proj.Add.bias.preformat", + "model.layers.2.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.2.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.2.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.2.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.2.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.2.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.2.post_attention_layernorm.weight.bf", + "eps_26_1_4", + "model.layers.2.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.2.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.2.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.2.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.2.mlp.up_proj.MatMulNBits.qweight", + "model.layers.2.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.2.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.2.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.2.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.2.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.2.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.2.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.3.input_layernorm.weight.bf", + "eps_26_1_5", + "model.layers.3.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.3.attn.k_proj.Add.bias.preformat", + "model.layers.3.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.3.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.3.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.3.attn.q_proj.Add.bias.preformat", + "model.layers.3.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.3.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.3.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.3.attn.v_proj.Add.bias.preformat", + "model.layers.3.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.3.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.3.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.3.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.3.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.3.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.3.post_attention_layernorm.weight.bf", + "eps_26_1_6", + "model.layers.3.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.3.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.3.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.3.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.3.mlp.up_proj.MatMulNBits.qweight", + "model.layers.3.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.3.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.3.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.3.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.3.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.3.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.3.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.4.input_layernorm.weight.bf", + "eps_26_1_7", + "model.layers.4.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.4.attn.k_proj.Add.bias.preformat", + "model.layers.4.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.4.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.4.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.4.attn.q_proj.Add.bias.preformat", + "model.layers.4.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.4.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.4.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.4.attn.v_proj.Add.bias.preformat", + "model.layers.4.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.4.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.4.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.4.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.4.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.4.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.4.post_attention_layernorm.weight.bf", + "eps_26_1_8", + "model.layers.4.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.4.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.4.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.4.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.4.mlp.up_proj.MatMulNBits.qweight", + "model.layers.4.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.4.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.4.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.4.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.4.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.4.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.4.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.5.input_layernorm.weight.bf", + "eps_26_1_9", + "model.layers.5.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.5.attn.k_proj.Add.bias.preformat", + "model.layers.5.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.5.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.5.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.5.attn.q_proj.Add.bias.preformat", + "model.layers.5.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.5.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.5.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.5.attn.v_proj.Add.bias.preformat", + "model.layers.5.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.5.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.5.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.5.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.5.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.5.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.5.post_attention_layernorm.weight.bf", + "eps_26_1_10", + "model.layers.5.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.5.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.5.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.5.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.5.mlp.up_proj.MatMulNBits.qweight", + "model.layers.5.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.5.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.5.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.5.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.5.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.5.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.5.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.6.input_layernorm.weight.bf", + "eps_26_1_11", + "model.layers.6.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.6.attn.k_proj.Add.bias.preformat", + "model.layers.6.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.6.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.6.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.6.attn.q_proj.Add.bias.preformat", + "model.layers.6.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.6.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.6.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.6.attn.v_proj.Add.bias.preformat", + "model.layers.6.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.6.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.6.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.6.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.6.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.6.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.6.post_attention_layernorm.weight.bf", + "eps_26_1_12", + "model.layers.6.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.6.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.6.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.6.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.6.mlp.up_proj.MatMulNBits.qweight", + "model.layers.6.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.6.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.6.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.6.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.6.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.6.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.6.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.7.input_layernorm.weight.bf", + "eps_26_1_13", + "model.layers.7.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.7.attn.k_proj.Add.bias.preformat", + "model.layers.7.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.7.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.7.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.7.attn.q_proj.Add.bias.preformat", + "model.layers.7.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.7.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.7.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.7.attn.v_proj.Add.bias.preformat", + "model.layers.7.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.7.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.7.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.7.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.7.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.7.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.7.post_attention_layernorm.weight.bf", + "eps_26_1_14", + "model.layers.7.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.7.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.7.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.7.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.7.mlp.up_proj.MatMulNBits.qweight", + "model.layers.7.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.7.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.7.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.7.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.7.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.7.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.7.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.8.input_layernorm.weight.bf", + "eps_26_1_15", + "model.layers.8.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.8.attn.k_proj.Add.bias.preformat", + "model.layers.8.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.8.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.8.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.8.attn.q_proj.Add.bias.preformat", + "model.layers.8.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.8.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.8.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.8.attn.v_proj.Add.bias.preformat", + "model.layers.8.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.8.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.8.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.8.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.8.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.8.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.8.post_attention_layernorm.weight.bf", + "eps_26_1_16", + "model.layers.8.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.8.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.8.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.8.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.8.mlp.up_proj.MatMulNBits.qweight", + "model.layers.8.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.8.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.8.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.8.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.8.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.8.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.8.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.9.input_layernorm.weight.bf", + "eps_26_1_17", + "model.layers.9.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.9.attn.k_proj.Add.bias.preformat", + "model.layers.9.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.9.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.9.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.9.attn.q_proj.Add.bias.preformat", + "model.layers.9.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.9.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.9.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.9.attn.v_proj.Add.bias.preformat", + "model.layers.9.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.9.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.9.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.9.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.9.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.9.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.9.post_attention_layernorm.weight.bf", + "eps_26_1_18", + "model.layers.9.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.9.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.9.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.9.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.9.mlp.up_proj.MatMulNBits.qweight", + "model.layers.9.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.9.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.9.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.9.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.9.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.9.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.9.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.10.input_layernorm.weight.bf", + "eps_26_1_19", + "model.layers.10.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.10.attn.k_proj.Add.bias.preformat", + "model.layers.10.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.10.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.10.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.10.attn.q_proj.Add.bias.preformat", + "model.layers.10.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.10.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.10.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.10.attn.v_proj.Add.bias.preformat", + "model.layers.10.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.10.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.10.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.10.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.10.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.10.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.10.post_attention_layernorm.weight.bf", + "eps_26_1_20", + "model.layers.10.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.10.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.10.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.10.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.10.mlp.up_proj.MatMulNBits.qweight", + "model.layers.10.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.10.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.10.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.10.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.10.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.10.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.10.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.11.input_layernorm.weight.bf", + "eps_26_1_21", + "model.layers.11.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.11.attn.k_proj.Add.bias.preformat", + "model.layers.11.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.11.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.11.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.11.attn.q_proj.Add.bias.preformat", + "model.layers.11.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.11.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.11.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.11.attn.v_proj.Add.bias.preformat", + "model.layers.11.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.11.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.11.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.11.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.11.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.11.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.11.post_attention_layernorm.weight.bf", + "eps_26_1_22", + "model.layers.11.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.11.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.11.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.11.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.11.mlp.up_proj.MatMulNBits.qweight", + "model.layers.11.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.11.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.11.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.11.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.11.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.11.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.11.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.12.input_layernorm.weight.bf", + "eps_26_1_23", + "model.layers.12.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.12.attn.k_proj.Add.bias.preformat", + "model.layers.12.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.12.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.12.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.12.attn.q_proj.Add.bias.preformat", + "model.layers.12.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.12.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.12.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.12.attn.v_proj.Add.bias.preformat", + "model.layers.12.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.12.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.12.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.12.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.12.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.12.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.12.post_attention_layernorm.weight.bf", + "eps_26_1_24", + "model.layers.12.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.12.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.12.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.12.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.12.mlp.up_proj.MatMulNBits.qweight", + "model.layers.12.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.12.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.12.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.12.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.12.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.12.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.12.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.13.input_layernorm.weight.bf", + "eps_26_1_25", + "model.layers.13.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.13.attn.k_proj.Add.bias.preformat", + "model.layers.13.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.13.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.13.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.13.attn.q_proj.Add.bias.preformat", + "model.layers.13.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.13.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.13.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.13.attn.v_proj.Add.bias.preformat", + "model.layers.13.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.13.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.13.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.13.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.13.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.13.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.13.post_attention_layernorm.weight.bf", + "eps_26_1_26", + "model.layers.13.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.13.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.13.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.13.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.13.mlp.up_proj.MatMulNBits.qweight", + "model.layers.13.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.13.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.13.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.13.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.13.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.13.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.13.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.14.input_layernorm.weight.bf", + "eps_26_1_27", + "model.layers.14.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.14.attn.k_proj.Add.bias.preformat", + "model.layers.14.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.14.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.14.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.14.attn.q_proj.Add.bias.preformat", + "model.layers.14.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.14.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.14.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.14.attn.v_proj.Add.bias.preformat", + "model.layers.14.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.14.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.14.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.14.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.14.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.14.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.14.post_attention_layernorm.weight.bf", + "eps_26_1_28", + "model.layers.14.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.14.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.14.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.14.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.14.mlp.up_proj.MatMulNBits.qweight", + "model.layers.14.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.14.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.14.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.14.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.14.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.14.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.14.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.15.input_layernorm.weight.bf", + "eps_26_1_29", + "model.layers.15.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.15.attn.k_proj.Add.bias.preformat", + "model.layers.15.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.15.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.15.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.15.attn.q_proj.Add.bias.preformat", + "model.layers.15.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.15.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.15.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.15.attn.v_proj.Add.bias.preformat", + "model.layers.15.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.15.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.15.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.15.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.15.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.15.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.15.post_attention_layernorm.weight.bf", + "eps_26_1_30", + "model.layers.15.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.15.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.15.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.15.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.15.mlp.up_proj.MatMulNBits.qweight", + "model.layers.15.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.15.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.15.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.15.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.15.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.15.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.15.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.16.input_layernorm.weight.bf", + "eps_26_1_31", + "model.layers.16.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.16.attn.k_proj.Add.bias.preformat", + "model.layers.16.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.16.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.16.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.16.attn.q_proj.Add.bias.preformat", + "model.layers.16.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.16.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.16.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.16.attn.v_proj.Add.bias.preformat", + "model.layers.16.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.16.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.16.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.16.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.16.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.16.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.16.post_attention_layernorm.weight.bf", + "eps_26_1_32", + "model.layers.16.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.16.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.16.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.16.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.16.mlp.up_proj.MatMulNBits.qweight", + "model.layers.16.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.16.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.16.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.16.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.16.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.16.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.16.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.17.input_layernorm.weight.bf", + "eps_26_1_33", + "model.layers.17.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.17.attn.k_proj.Add.bias.preformat", + "model.layers.17.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.17.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.17.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.17.attn.q_proj.Add.bias.preformat", + "model.layers.17.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.17.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.17.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.17.attn.v_proj.Add.bias.preformat", + "model.layers.17.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.17.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.17.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.17.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.17.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.17.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.17.post_attention_layernorm.weight.bf", + "eps_26_1_34", + "model.layers.17.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.17.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.17.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.17.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.17.mlp.up_proj.MatMulNBits.qweight", + "model.layers.17.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.17.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.17.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.17.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.17.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.17.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.17.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.18.input_layernorm.weight.bf", + "eps_26_1_35", + "model.layers.18.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.18.attn.k_proj.Add.bias.preformat", + "model.layers.18.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.18.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.18.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.18.attn.q_proj.Add.bias.preformat", + "model.layers.18.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.18.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.18.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.18.attn.v_proj.Add.bias.preformat", + "model.layers.18.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.18.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.18.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.18.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.18.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.18.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.18.post_attention_layernorm.weight.bf", + "eps_26_1_36", + "model.layers.18.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.18.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.18.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.18.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.18.mlp.up_proj.MatMulNBits.qweight", + "model.layers.18.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.18.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.18.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.18.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.18.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.18.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.18.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.19.input_layernorm.weight.bf", + "eps_26_1_37", + "model.layers.19.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.19.attn.k_proj.Add.bias.preformat", + "model.layers.19.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.19.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.19.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.19.attn.q_proj.Add.bias.preformat", + "model.layers.19.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.19.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.19.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.19.attn.v_proj.Add.bias.preformat", + "model.layers.19.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.19.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.19.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.19.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.19.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.19.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.19.post_attention_layernorm.weight.bf", + "eps_26_1_38", + "model.layers.19.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.19.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.19.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.19.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.19.mlp.up_proj.MatMulNBits.qweight", + "model.layers.19.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.19.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.19.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.19.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.19.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.19.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.19.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.20.input_layernorm.weight.bf", + "eps_26_1_39", + "model.layers.20.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.20.attn.k_proj.Add.bias.preformat", + "model.layers.20.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.20.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.20.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.20.attn.q_proj.Add.bias.preformat", + "model.layers.20.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.20.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.20.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.20.attn.v_proj.Add.bias.preformat", + "model.layers.20.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.20.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.20.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.20.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.20.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.20.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.20.post_attention_layernorm.weight.bf", + "eps_26_1_40", + "model.layers.20.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.20.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.20.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.20.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.20.mlp.up_proj.MatMulNBits.qweight", + "model.layers.20.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.20.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.20.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.20.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.20.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.20.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.20.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.21.input_layernorm.weight.bf", + "eps_26_1_41", + "model.layers.21.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.21.attn.k_proj.Add.bias.preformat", + "model.layers.21.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.21.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.21.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.21.attn.q_proj.Add.bias.preformat", + "model.layers.21.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.21.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.21.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.21.attn.v_proj.Add.bias.preformat", + "model.layers.21.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.21.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.21.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.21.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.21.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.21.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.21.post_attention_layernorm.weight.bf", + "eps_26_1_42", + "model.layers.21.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.21.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.21.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.21.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.21.mlp.up_proj.MatMulNBits.qweight", + "model.layers.21.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.21.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.21.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.21.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.21.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.21.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.21.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.22.input_layernorm.weight.bf", + "eps_26_1_43", + "model.layers.22.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.22.attn.k_proj.Add.bias.preformat", + "model.layers.22.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.22.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.22.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.22.attn.q_proj.Add.bias.preformat", + "model.layers.22.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.22.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.22.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.22.attn.v_proj.Add.bias.preformat", + "model.layers.22.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.22.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.22.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.22.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.22.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.22.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.22.post_attention_layernorm.weight.bf", + "eps_26_1_44", + "model.layers.22.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.22.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.22.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.22.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.22.mlp.up_proj.MatMulNBits.qweight", + "model.layers.22.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.22.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.22.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.22.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.22.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.22.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.22.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.23.input_layernorm.weight.bf", + "eps_26_1_45", + "model.layers.23.attn.k_proj.MatMulNBits.qweight.preformat", + "model.layers.23.attn.k_proj.Add.bias.preformat", + "model.layers.23.attn.k_proj.MatMulNBits.scales.preformat", + "model.layers.23.attn.k_proj.MatMulNBits.qzeros.preformat", + "model.layers.23.attn.q_proj.MatMulNBits.qweight.preformat", + "model.layers.23.attn.q_proj.Add.bias.preformat", + "model.layers.23.attn.q_proj.MatMulNBits.scales.preformat", + "model.layers.23.attn.q_proj.MatMulNBits.qzeros.preformat", + "model.layers.23.attn.v_proj.MatMulNBits.qweight.preformat", + "model.layers.23.attn.v_proj.Add.bias.preformat", + "model.layers.23.attn.v_proj.MatMulNBits.scales.preformat", + "model.layers.23.attn.v_proj.MatMulNBits.qzeros.preformat", + "model.layers.23.attn.o_proj.MatMulNBits.qweight.preformat", + "model.layers.23.attn.o_proj.MatMulNBits.bias.preformat", + "model.layers.23.attn.o_proj.MatMulNBits.scales.preformat", + "model.layers.23.attn.o_proj.MatMulNBits.qzeros.preformat", + "model.layers.23.post_attention_layernorm.weight.bf", + "eps_26_1_46", + "model.layers.23.mlp.gate_proj.MatMulNBits.qweight", + "model.layers.23.mlp.gate_proj.MatMulNBits.scales.f", + "model.layers.23.mlp.gate_proj.MatMulNBits.qzeros", + "model.layers.23.mlp.gate_proj.MatMulNBits.bias.f", + "model.layers.23.mlp.up_proj.MatMulNBits.qweight", + "model.layers.23.mlp.up_proj.MatMulNBits.scales.f", + "model.layers.23.mlp.up_proj.MatMulNBits.qzeros", + "model.layers.23.mlp.up_proj.MatMulNBits.bias.f", + "model.layers.23.mlp.down_proj.MatMulNBits.qweight.preformat", + "model.layers.23.mlp.down_proj.MatMulNBits.bias.preformat", + "model.layers.23.mlp.down_proj.MatMulNBits.scales.preformat", + "model.layers.23.mlp.down_proj.MatMulNBits.qzeros.preformat", + "model.layers.24.final_norm_layernorm.weight.bf", + "eps_26_1_47", + "lm_head.MatMulNBits.qweight.preformat", + "lm_head.MatMulNBits.bias.preformat", + "lm_head.MatMulNBits.scales.preformat", + "lm_head.MatMulNBits.qzeros.preformat" + ] + }, + "super_instr": { + "buffer_size": 0, + "xrt_arg_id": 4, + "packed_tensors": [] + }, + "ext_buf_0": { + "buffer_size": 50331648, + "xrt_arg_id": 5, + "packed_tensors": [ + "past_key_values.0.key", + "past_key_values.0.value", + "present.0.key", + "present.0.value", + "past_key_values.1.key", + "past_key_values.1.value", + "present.1.key", + "present.1.value", + "past_key_values.2.key", + "past_key_values.2.value", + "present.2.key", + "present.2.value", + "past_key_values.3.key", + "past_key_values.3.value", + "present.3.key", + "present.3.value", + "past_key_values.4.key", + "past_key_values.4.value", + "present.4.key", + "present.4.value", + "past_key_values.5.key", + "past_key_values.5.value", + "present.5.key", + "present.5.value", + "past_key_values.6.key", + "past_key_values.6.value", + "present.6.key", + "present.6.value", + "past_key_values.7.key", + "past_key_values.7.value", + "present.7.key", + "present.7.value", + "past_key_values.8.key", + "past_key_values.8.value", + "present.8.key", + "present.8.value", + "past_key_values.9.key", + "past_key_values.9.value", + "present.9.key", + "present.9.value", + "past_key_values.10.key", + "past_key_values.10.value", + "present.10.key", + "present.10.value", + "past_key_values.11.key", + "past_key_values.11.value", + "present.11.key", + "present.11.value", + "past_key_values.12.key", + "past_key_values.12.value", + "present.12.key", + "present.12.value", + "past_key_values.13.key", + "past_key_values.13.value", + "present.13.key", + "present.13.value", + "past_key_values.14.key", + "past_key_values.14.value", + "present.14.key", + "present.14.value", + "past_key_values.15.key", + "past_key_values.15.value", + "present.15.key", + "present.15.value", + "past_key_values.16.key", + "past_key_values.16.value", + "present.16.key", + "present.16.value", + "past_key_values.17.key", + "past_key_values.17.value", + "present.17.key", + "present.17.value", + "past_key_values.18.key", + "past_key_values.18.value", + "present.18.key", + "present.18.value", + "past_key_values.19.key", + "past_key_values.19.value", + "present.19.key", + "present.19.value", + "past_key_values.20.key", + "past_key_values.20.value", + "present.20.key", + "present.20.value", + "past_key_values.21.key", + "past_key_values.21.value", + "present.21.key", + "present.21.value", + "past_key_values.22.key", + "past_key_values.22.value", + "present.22.key", + "present.22.value", + "past_key_values.23.key", + "past_key_values.23.value", + "present.23.key", + "present.23.value" + ] + } + }, + "tensor_map": { + "/model/embed_tokens/Gather/output_0.out27_14_0": { + "packed_buffer_label": "in", + "xrt_arg_id": 0, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 0 + }, + "attention_mask_const_uint": { + "packed_buffer_label": "in", + "xrt_arg_id": 0, + "dtype": "uint32", + "shape": [ + 1 + ], + "size_in_bytes": 4, + "op_tensor_size": 4, + "offset": 1792 + }, + "sin_cache_cos_cache_sliced": { + "packed_buffer_label": "in", + "xrt_arg_id": 0, + "dtype": "bfloat16", + "shape": [ + 1, + 64 + ], + "size_in_bytes": 128, + "op_tensor_size": 128, + "offset": 1796 + }, + "attention_mask_padded": { + "packed_buffer_label": "in", + "xrt_arg_id": 0, + "dtype": "int64", + "shape": [ + 1, + 1 + ], + "size_in_bytes": 8, + "op_tensor_size": 8, + "dynamic_shapes": [ + "False", + "attention_mask_padded" + ], + "offset": 1924 + }, + "/model/layers.24/final_norm_layernorm/output_0.dummy": { + "packed_buffer_label": "out", + "xrt_arg_id": 1, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 0 + }, + "logits.out27_11_96": { + "packed_buffer_label": "out", + "xrt_arg_id": 1, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 151936 + ], + "size_in_bytes": 303872, + "op_tensor_size": 303872, + "offset": 1792 + }, + "/model/layers.0/input_layernorm/output_0.out27_14_0": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 0 + }, + "/model/layers.0/attn/k_proj/Add/output_0.out27_11_1": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 1792 + }, + "/model/layers.0/attn/q_proj/Add/output_0.out27_11_0": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 2048 + }, + "/model/layers.0/attn/GroupQueryAttention/output_0.out24_0_0": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 3840 + }, + "/model/layers.0/attn/o_proj/MatMulNBits/output_0.out27_11_2": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 5632 + }, + "/model/layers.0/post_attention_layernorm/output_3.out26_1_0": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 7424 + }, + "/model/layers.0/post_attention_layernorm/output_0.out26_1_0": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 9216 + }, + "/model/layers.0/mlp/Mul/output_0.out25_0_0": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 11008 + }, + "/model/layers.0/mlp/down_proj/MatMulNBits/output_0.out27_11_3": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 20736 + }, + "/model/layers.1/input_layernorm/output_3.out26_1_1": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 22528 + }, + "/model/layers.1/input_layernorm/output_0.out26_1_1": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 24320 + }, + "/model/layers.1/attn/k_proj/Add/output_0.out27_11_5": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 26112 + }, + "/model/layers.1/attn/q_proj/Add/output_0.out27_11_4": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 26368 + }, + "/model/layers.1/attn/GroupQueryAttention/output_0.out24_0_1": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 28160 + }, + "/model/layers.1/attn/o_proj/MatMulNBits/output_0.out27_11_6": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 29952 + }, + "/model/layers.1/post_attention_layernorm/output_3.out26_1_2": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 31744 + }, + "/model/layers.1/post_attention_layernorm/output_0.out26_1_2": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 33536 + }, + "/model/layers.1/mlp/Mul/output_0.out25_0_1": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 35328 + }, + "/model/layers.1/mlp/down_proj/MatMulNBits/output_0.out27_11_7": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 45056 + }, + "/model/layers.2/input_layernorm/output_3.out26_1_3": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 46848 + }, + "/model/layers.2/input_layernorm/output_0.out26_1_3": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 48640 + }, + "/model/layers.2/attn/k_proj/Add/output_0.out27_11_9": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 50432 + }, + "/model/layers.2/attn/q_proj/Add/output_0.out27_11_8": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 50688 + }, + "/model/layers.2/attn/GroupQueryAttention/output_0.out24_0_2": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 52480 + }, + "/model/layers.2/attn/o_proj/MatMulNBits/output_0.out27_11_10": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 54272 + }, + "/model/layers.2/post_attention_layernorm/output_3.out26_1_4": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 56064 + }, + "/model/layers.2/post_attention_layernorm/output_0.out26_1_4": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 57856 + }, + "/model/layers.2/mlp/Mul/output_0.out25_0_2": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 59648 + }, + "/model/layers.2/mlp/down_proj/MatMulNBits/output_0.out27_11_11": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 69376 + }, + "/model/layers.3/input_layernorm/output_3.out26_1_5": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 71168 + }, + "/model/layers.3/input_layernorm/output_0.out26_1_5": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 72960 + }, + "/model/layers.3/attn/k_proj/Add/output_0.out27_11_13": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 74752 + }, + "/model/layers.3/attn/q_proj/Add/output_0.out27_11_12": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 75008 + }, + "/model/layers.3/attn/GroupQueryAttention/output_0.out24_0_3": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 76800 + }, + "/model/layers.3/attn/o_proj/MatMulNBits/output_0.out27_11_14": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 78592 + }, + "/model/layers.3/post_attention_layernorm/output_3.out26_1_6": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 80384 + }, + "/model/layers.3/post_attention_layernorm/output_0.out26_1_6": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 82176 + }, + "/model/layers.3/mlp/Mul/output_0.out25_0_3": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 83968 + }, + "/model/layers.3/mlp/down_proj/MatMulNBits/output_0.out27_11_15": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 93696 + }, + "/model/layers.4/input_layernorm/output_3.out26_1_7": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 95488 + }, + "/model/layers.4/input_layernorm/output_0.out26_1_7": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 97280 + }, + "/model/layers.4/attn/k_proj/Add/output_0.out27_11_17": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 99072 + }, + "/model/layers.4/attn/q_proj/Add/output_0.out27_11_16": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 99328 + }, + "/model/layers.4/attn/GroupQueryAttention/output_0.out24_0_4": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 101120 + }, + "/model/layers.4/attn/o_proj/MatMulNBits/output_0.out27_11_18": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 102912 + }, + "/model/layers.4/post_attention_layernorm/output_3.out26_1_8": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 104704 + }, + "/model/layers.4/post_attention_layernorm/output_0.out26_1_8": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 106496 + }, + "/model/layers.4/mlp/Mul/output_0.out25_0_4": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 108288 + }, + "/model/layers.4/mlp/down_proj/MatMulNBits/output_0.out27_11_19": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 118016 + }, + "/model/layers.5/input_layernorm/output_3.out26_1_9": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 119808 + }, + "/model/layers.5/input_layernorm/output_0.out26_1_9": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 121600 + }, + "/model/layers.5/attn/k_proj/Add/output_0.out27_11_21": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 123392 + }, + "/model/layers.5/attn/q_proj/Add/output_0.out27_11_20": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 123648 + }, + "/model/layers.5/attn/GroupQueryAttention/output_0.out24_0_5": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 125440 + }, + "/model/layers.5/attn/o_proj/MatMulNBits/output_0.out27_11_22": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 127232 + }, + "/model/layers.5/post_attention_layernorm/output_3.out26_1_10": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 129024 + }, + "/model/layers.5/post_attention_layernorm/output_0.out26_1_10": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 130816 + }, + "/model/layers.5/mlp/Mul/output_0.out25_0_5": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 132608 + }, + "/model/layers.5/mlp/down_proj/MatMulNBits/output_0.out27_11_23": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 142336 + }, + "/model/layers.6/input_layernorm/output_3.out26_1_11": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 144128 + }, + "/model/layers.6/input_layernorm/output_0.out26_1_11": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 145920 + }, + "/model/layers.6/attn/k_proj/Add/output_0.out27_11_25": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 147712 + }, + "/model/layers.6/attn/q_proj/Add/output_0.out27_11_24": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 147968 + }, + "/model/layers.6/attn/GroupQueryAttention/output_0.out24_0_6": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 149760 + }, + "/model/layers.6/attn/o_proj/MatMulNBits/output_0.out27_11_26": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 151552 + }, + "/model/layers.6/post_attention_layernorm/output_3.out26_1_12": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 153344 + }, + "/model/layers.6/post_attention_layernorm/output_0.out26_1_12": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 155136 + }, + "/model/layers.6/mlp/Mul/output_0.out25_0_6": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 156928 + }, + "/model/layers.6/mlp/down_proj/MatMulNBits/output_0.out27_11_27": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 166656 + }, + "/model/layers.7/input_layernorm/output_3.out26_1_13": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 168448 + }, + "/model/layers.7/input_layernorm/output_0.out26_1_13": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 170240 + }, + "/model/layers.7/attn/k_proj/Add/output_0.out27_11_29": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 172032 + }, + "/model/layers.7/attn/q_proj/Add/output_0.out27_11_28": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 172288 + }, + "/model/layers.7/attn/GroupQueryAttention/output_0.out24_0_7": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 174080 + }, + "/model/layers.7/attn/o_proj/MatMulNBits/output_0.out27_11_30": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 175872 + }, + "/model/layers.7/post_attention_layernorm/output_3.out26_1_14": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 177664 + }, + "/model/layers.7/post_attention_layernorm/output_0.out26_1_14": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 179456 + }, + "/model/layers.7/mlp/Mul/output_0.out25_0_7": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 181248 + }, + "/model/layers.7/mlp/down_proj/MatMulNBits/output_0.out27_11_31": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 190976 + }, + "/model/layers.8/input_layernorm/output_3.out26_1_15": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 192768 + }, + "/model/layers.8/input_layernorm/output_0.out26_1_15": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 194560 + }, + "/model/layers.8/attn/k_proj/Add/output_0.out27_11_33": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 196352 + }, + "/model/layers.8/attn/q_proj/Add/output_0.out27_11_32": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 196608 + }, + "/model/layers.8/attn/GroupQueryAttention/output_0.out24_0_8": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 198400 + }, + "/model/layers.8/attn/o_proj/MatMulNBits/output_0.out27_11_34": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 200192 + }, + "/model/layers.8/post_attention_layernorm/output_3.out26_1_16": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 201984 + }, + "/model/layers.8/post_attention_layernorm/output_0.out26_1_16": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 203776 + }, + "/model/layers.8/mlp/Mul/output_0.out25_0_8": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 205568 + }, + "/model/layers.8/mlp/down_proj/MatMulNBits/output_0.out27_11_35": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 215296 + }, + "/model/layers.9/input_layernorm/output_3.out26_1_17": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 217088 + }, + "/model/layers.9/input_layernorm/output_0.out26_1_17": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 218880 + }, + "/model/layers.9/attn/k_proj/Add/output_0.out27_11_37": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 220672 + }, + "/model/layers.9/attn/q_proj/Add/output_0.out27_11_36": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 220928 + }, + "/model/layers.9/attn/GroupQueryAttention/output_0.out24_0_9": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 222720 + }, + "/model/layers.9/attn/o_proj/MatMulNBits/output_0.out27_11_38": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 224512 + }, + "/model/layers.9/post_attention_layernorm/output_3.out26_1_18": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 226304 + }, + "/model/layers.9/post_attention_layernorm/output_0.out26_1_18": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 228096 + }, + "/model/layers.9/mlp/Mul/output_0.out25_0_9": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 229888 + }, + "/model/layers.9/mlp/down_proj/MatMulNBits/output_0.out27_11_39": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 239616 + }, + "/model/layers.10/input_layernorm/output_3.out26_1_19": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 241408 + }, + "/model/layers.10/input_layernorm/output_0.out26_1_19": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 243200 + }, + "/model/layers.10/attn/k_proj/Add/output_0.out27_11_41": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 244992 + }, + "/model/layers.10/attn/q_proj/Add/output_0.out27_11_40": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 245248 + }, + "/model/layers.10/attn/GroupQueryAttention/output_0.out24_0_10": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 247040 + }, + "/model/layers.10/attn/o_proj/MatMulNBits/output_0.out27_11_42": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 248832 + }, + "/model/layers.10/post_attention_layernorm/output_3.out26_1_20": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 250624 + }, + "/model/layers.10/post_attention_layernorm/output_0.out26_1_20": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 252416 + }, + "/model/layers.10/mlp/Mul/output_0.out25_0_10": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 254208 + }, + "/model/layers.10/mlp/down_proj/MatMulNBits/output_0.out27_11_43": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 263936 + }, + "/model/layers.11/input_layernorm/output_3.out26_1_21": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 265728 + }, + "/model/layers.11/input_layernorm/output_0.out26_1_21": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 267520 + }, + "/model/layers.11/attn/k_proj/Add/output_0.out27_11_45": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 269312 + }, + "/model/layers.11/attn/q_proj/Add/output_0.out27_11_44": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 269568 + }, + "/model/layers.11/attn/GroupQueryAttention/output_0.out24_0_11": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 271360 + }, + "/model/layers.11/attn/o_proj/MatMulNBits/output_0.out27_11_46": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 273152 + }, + "/model/layers.11/post_attention_layernorm/output_3.out26_1_22": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 274944 + }, + "/model/layers.11/post_attention_layernorm/output_0.out26_1_22": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 276736 + }, + "/model/layers.11/mlp/Mul/output_0.out25_0_11": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 278528 + }, + "/model/layers.11/mlp/down_proj/MatMulNBits/output_0.out27_11_47": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 288256 + }, + "/model/layers.12/input_layernorm/output_3.out26_1_23": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 290048 + }, + "/model/layers.12/input_layernorm/output_0.out26_1_23": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 291840 + }, + "/model/layers.12/attn/k_proj/Add/output_0.out27_11_49": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 293632 + }, + "/model/layers.12/attn/q_proj/Add/output_0.out27_11_48": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 293888 + }, + "/model/layers.12/attn/GroupQueryAttention/output_0.out24_0_12": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 295680 + }, + "/model/layers.12/attn/o_proj/MatMulNBits/output_0.out27_11_50": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 297472 + }, + "/model/layers.12/post_attention_layernorm/output_3.out26_1_24": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 299264 + }, + "/model/layers.12/post_attention_layernorm/output_0.out26_1_24": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 301056 + }, + "/model/layers.12/mlp/Mul/output_0.out25_0_12": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 302848 + }, + "/model/layers.12/mlp/down_proj/MatMulNBits/output_0.out27_11_51": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 312576 + }, + "/model/layers.13/input_layernorm/output_3.out26_1_25": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 314368 + }, + "/model/layers.13/input_layernorm/output_0.out26_1_25": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 316160 + }, + "/model/layers.13/attn/k_proj/Add/output_0.out27_11_53": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 317952 + }, + "/model/layers.13/attn/q_proj/Add/output_0.out27_11_52": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 318208 + }, + "/model/layers.13/attn/GroupQueryAttention/output_0.out24_0_13": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 320000 + }, + "/model/layers.13/attn/o_proj/MatMulNBits/output_0.out27_11_54": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 321792 + }, + "/model/layers.13/post_attention_layernorm/output_3.out26_1_26": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 323584 + }, + "/model/layers.13/post_attention_layernorm/output_0.out26_1_26": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 325376 + }, + "/model/layers.13/mlp/Mul/output_0.out25_0_13": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 327168 + }, + "/model/layers.13/mlp/down_proj/MatMulNBits/output_0.out27_11_55": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 336896 + }, + "/model/layers.14/input_layernorm/output_3.out26_1_27": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 338688 + }, + "/model/layers.14/input_layernorm/output_0.out26_1_27": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 340480 + }, + "/model/layers.14/attn/k_proj/Add/output_0.out27_11_57": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 342272 + }, + "/model/layers.14/attn/q_proj/Add/output_0.out27_11_56": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 342528 + }, + "/model/layers.14/attn/GroupQueryAttention/output_0.out24_0_14": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 344320 + }, + "/model/layers.14/attn/o_proj/MatMulNBits/output_0.out27_11_58": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 346112 + }, + "/model/layers.14/post_attention_layernorm/output_3.out26_1_28": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 347904 + }, + "/model/layers.14/post_attention_layernorm/output_0.out26_1_28": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 349696 + }, + "/model/layers.14/mlp/Mul/output_0.out25_0_14": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 351488 + }, + "/model/layers.14/mlp/down_proj/MatMulNBits/output_0.out27_11_59": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 361216 + }, + "/model/layers.15/input_layernorm/output_3.out26_1_29": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 363008 + }, + "/model/layers.15/input_layernorm/output_0.out26_1_29": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 364800 + }, + "/model/layers.15/attn/k_proj/Add/output_0.out27_11_61": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 366592 + }, + "/model/layers.15/attn/q_proj/Add/output_0.out27_11_60": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 366848 + }, + "/model/layers.15/attn/GroupQueryAttention/output_0.out24_0_15": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 368640 + }, + "/model/layers.15/attn/o_proj/MatMulNBits/output_0.out27_11_62": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 370432 + }, + "/model/layers.15/post_attention_layernorm/output_3.out26_1_30": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 372224 + }, + "/model/layers.15/post_attention_layernorm/output_0.out26_1_30": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 374016 + }, + "/model/layers.15/mlp/Mul/output_0.out25_0_15": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 375808 + }, + "/model/layers.15/mlp/down_proj/MatMulNBits/output_0.out27_11_63": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 385536 + }, + "/model/layers.16/input_layernorm/output_3.out26_1_31": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 387328 + }, + "/model/layers.16/input_layernorm/output_0.out26_1_31": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 389120 + }, + "/model/layers.16/attn/k_proj/Add/output_0.out27_11_65": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 390912 + }, + "/model/layers.16/attn/q_proj/Add/output_0.out27_11_64": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 391168 + }, + "/model/layers.16/attn/GroupQueryAttention/output_0.out24_0_16": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 392960 + }, + "/model/layers.16/attn/o_proj/MatMulNBits/output_0.out27_11_66": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 394752 + }, + "/model/layers.16/post_attention_layernorm/output_3.out26_1_32": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 396544 + }, + "/model/layers.16/post_attention_layernorm/output_0.out26_1_32": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 398336 + }, + "/model/layers.16/mlp/Mul/output_0.out25_0_16": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 400128 + }, + "/model/layers.16/mlp/down_proj/MatMulNBits/output_0.out27_11_67": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 409856 + }, + "/model/layers.17/input_layernorm/output_3.out26_1_33": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 411648 + }, + "/model/layers.17/input_layernorm/output_0.out26_1_33": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 413440 + }, + "/model/layers.17/attn/k_proj/Add/output_0.out27_11_69": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 415232 + }, + "/model/layers.17/attn/q_proj/Add/output_0.out27_11_68": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 415488 + }, + "/model/layers.17/attn/GroupQueryAttention/output_0.out24_0_17": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 417280 + }, + "/model/layers.17/attn/o_proj/MatMulNBits/output_0.out27_11_70": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 419072 + }, + "/model/layers.17/post_attention_layernorm/output_3.out26_1_34": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 420864 + }, + "/model/layers.17/post_attention_layernorm/output_0.out26_1_34": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 422656 + }, + "/model/layers.17/mlp/Mul/output_0.out25_0_17": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 424448 + }, + "/model/layers.17/mlp/down_proj/MatMulNBits/output_0.out27_11_71": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 434176 + }, + "/model/layers.18/input_layernorm/output_3.out26_1_35": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 435968 + }, + "/model/layers.18/input_layernorm/output_0.out26_1_35": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 437760 + }, + "/model/layers.18/attn/k_proj/Add/output_0.out27_11_73": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 439552 + }, + "/model/layers.18/attn/q_proj/Add/output_0.out27_11_72": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 439808 + }, + "/model/layers.18/attn/GroupQueryAttention/output_0.out24_0_18": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 441600 + }, + "/model/layers.18/attn/o_proj/MatMulNBits/output_0.out27_11_74": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 443392 + }, + "/model/layers.18/post_attention_layernorm/output_3.out26_1_36": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 445184 + }, + "/model/layers.18/post_attention_layernorm/output_0.out26_1_36": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 446976 + }, + "/model/layers.18/mlp/Mul/output_0.out25_0_18": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 448768 + }, + "/model/layers.18/mlp/down_proj/MatMulNBits/output_0.out27_11_75": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 458496 + }, + "/model/layers.19/input_layernorm/output_3.out26_1_37": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 460288 + }, + "/model/layers.19/input_layernorm/output_0.out26_1_37": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 462080 + }, + "/model/layers.19/attn/k_proj/Add/output_0.out27_11_77": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 463872 + }, + "/model/layers.19/attn/q_proj/Add/output_0.out27_11_76": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 464128 + }, + "/model/layers.19/attn/GroupQueryAttention/output_0.out24_0_19": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 465920 + }, + "/model/layers.19/attn/o_proj/MatMulNBits/output_0.out27_11_78": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 467712 + }, + "/model/layers.19/post_attention_layernorm/output_3.out26_1_38": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 469504 + }, + "/model/layers.19/post_attention_layernorm/output_0.out26_1_38": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 471296 + }, + "/model/layers.19/mlp/Mul/output_0.out25_0_19": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 473088 + }, + "/model/layers.19/mlp/down_proj/MatMulNBits/output_0.out27_11_79": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 482816 + }, + "/model/layers.20/input_layernorm/output_3.out26_1_39": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 484608 + }, + "/model/layers.20/input_layernorm/output_0.out26_1_39": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 486400 + }, + "/model/layers.20/attn/k_proj/Add/output_0.out27_11_81": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 488192 + }, + "/model/layers.20/attn/q_proj/Add/output_0.out27_11_80": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 488448 + }, + "/model/layers.20/attn/GroupQueryAttention/output_0.out24_0_20": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 490240 + }, + "/model/layers.20/attn/o_proj/MatMulNBits/output_0.out27_11_82": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 492032 + }, + "/model/layers.20/post_attention_layernorm/output_3.out26_1_40": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 493824 + }, + "/model/layers.20/post_attention_layernorm/output_0.out26_1_40": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 495616 + }, + "/model/layers.20/mlp/Mul/output_0.out25_0_20": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 497408 + }, + "/model/layers.20/mlp/down_proj/MatMulNBits/output_0.out27_11_83": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 507136 + }, + "/model/layers.21/input_layernorm/output_3.out26_1_41": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 508928 + }, + "/model/layers.21/input_layernorm/output_0.out26_1_41": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 510720 + }, + "/model/layers.21/attn/k_proj/Add/output_0.out27_11_85": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 512512 + }, + "/model/layers.21/attn/q_proj/Add/output_0.out27_11_84": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 512768 + }, + "/model/layers.21/attn/GroupQueryAttention/output_0.out24_0_21": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 514560 + }, + "/model/layers.21/attn/o_proj/MatMulNBits/output_0.out27_11_86": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 516352 + }, + "/model/layers.21/post_attention_layernorm/output_3.out26_1_42": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 518144 + }, + "/model/layers.21/post_attention_layernorm/output_0.out26_1_42": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 519936 + }, + "/model/layers.21/mlp/Mul/output_0.out25_0_21": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 521728 + }, + "/model/layers.21/mlp/down_proj/MatMulNBits/output_0.out27_11_87": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 531456 + }, + "/model/layers.22/input_layernorm/output_3.out26_1_43": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 533248 + }, + "/model/layers.22/input_layernorm/output_0.out26_1_43": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 535040 + }, + "/model/layers.22/attn/k_proj/Add/output_0.out27_11_89": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 536832 + }, + "/model/layers.22/attn/q_proj/Add/output_0.out27_11_88": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 537088 + }, + "/model/layers.22/attn/GroupQueryAttention/output_0.out24_0_22": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 538880 + }, + "/model/layers.22/attn/o_proj/MatMulNBits/output_0.out27_11_90": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 540672 + }, + "/model/layers.22/post_attention_layernorm/output_3.out26_1_44": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 542464 + }, + "/model/layers.22/post_attention_layernorm/output_0.out26_1_44": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 544256 + }, + "/model/layers.22/mlp/Mul/output_0.out25_0_22": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 546048 + }, + "/model/layers.22/mlp/down_proj/MatMulNBits/output_0.out27_11_91": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 555776 + }, + "/model/layers.23/input_layernorm/output_3.out26_1_45": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 557568 + }, + "/model/layers.23/input_layernorm/output_0.out26_1_45": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 559360 + }, + "/model/layers.23/attn/k_proj/Add/output_0.out27_11_93": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 128 + ], + "size_in_bytes": 256, + "op_tensor_size": 256, + "offset": 561152 + }, + "/model/layers.23/attn/q_proj/Add/output_0.out27_11_92": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 561408 + }, + "/model/layers.23/attn/GroupQueryAttention/output_0.out24_0_23": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 563200 + }, + "/model/layers.23/attn/o_proj/MatMulNBits/output_0.out27_11_94": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 564992 + }, + "/model/layers.23/post_attention_layernorm/output_3.out26_1_46": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 566784 + }, + "/model/layers.23/post_attention_layernorm/output_0.out26_1_46": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 568576 + }, + "/model/layers.23/mlp/Mul/output_0.out25_0_23": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 4864 + ], + "size_in_bytes": 9728, + "op_tensor_size": 9728, + "offset": 570368 + }, + "/model/layers.23/mlp/down_proj/MatMulNBits/output_0.out27_11_95": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 580096 + }, + "/model/layers.24/final_norm_layernorm/output_0.out26_1_47": { + "packed_buffer_label": "scratch", + "xrt_arg_id": 2, + "dtype": "bfloat16", + "shape": [ + 1, + 1, + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 581888 + }, + "model.layers.0.input_layernorm.weight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 0, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_0.const", + "file_size": 1792 + }, + "eps_27_14_0": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 1792, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_1.const", + "file_size": 2 + }, + "model.layers.0.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 1796, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_2.const", + "file_size": 114688 + }, + "model.layers.0.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 116484, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_3.const", + "file_size": 512 + }, + "model.layers.0.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 116996, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_4.const", + "file_size": 3584 + }, + "model.layers.0.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 120580, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_5.const", + "file_size": 896 + }, + "model.layers.0.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 121476, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_6.const", + "file_size": 802816 + }, + "model.layers.0.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 924292, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_7.const", + "file_size": 3584 + }, + "model.layers.0.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 927876, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_8.const", + "file_size": 25088 + }, + "model.layers.0.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 952964, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_9.const", + "file_size": 6272 + }, + "model.layers.0.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 959236, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_10.const", + "file_size": 114688 + }, + "model.layers.0.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 1073924, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_11.const", + "file_size": 512 + }, + "model.layers.0.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 1074436, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_12.const", + "file_size": 3584 + }, + "model.layers.0.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 1078020, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_13.const", + "file_size": 896 + }, + "model.layers.0.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 1078916, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_14.const", + "file_size": 802816 + }, + "model.layers.0.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 1881732, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_15.const", + "file_size": 3584 + }, + "model.layers.0.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 1885316, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_16.const", + "file_size": 25088 + }, + "model.layers.0.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 1910404, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_17.const", + "file_size": 6272 + }, + "model.layers.0.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 1916676, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_18.const", + "file_size": 1792 + }, + "eps_26_1_0": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 1918468, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_19.const", + "file_size": 2 + }, + "model.layers.0.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 1918472, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_20.const", + "file_size": 2179072 + }, + "model.layers.0.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 4097544, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_21.const", + "file_size": 136192 + }, + "model.layers.0.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 4233736, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_22.const", + "file_size": 19456 + }, + "model.layers.0.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 4253192, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_23.const", + "file_size": 19456 + }, + "model.layers.0.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 4272648, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_24.const", + "file_size": 2179072 + }, + "model.layers.0.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 6451720, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_25.const", + "file_size": 136192 + }, + "model.layers.0.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 6587912, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_26.const", + "file_size": 19456 + }, + "model.layers.0.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 6607368, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_27.const", + "file_size": 19456 + }, + "model.layers.0.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 6626824, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_28.const", + "file_size": 4358144 + }, + "model.layers.0.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 10984968, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_29.const", + "file_size": 3584 + }, + "model.layers.0.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 10988552, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_30.const", + "file_size": 136192 + }, + "model.layers.0.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 11124744, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_31.const", + "file_size": 34048 + }, + "model.layers.1.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 11158792, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_32.const", + "file_size": 1792 + }, + "eps_26_1_1": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 11160584, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_33.const", + "file_size": 2 + }, + "model.layers.1.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 11160588, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_34.const", + "file_size": 114688 + }, + "model.layers.1.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 11275276, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_35.const", + "file_size": 512 + }, + "model.layers.1.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 11275788, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_36.const", + "file_size": 3584 + }, + "model.layers.1.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 11279372, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_37.const", + "file_size": 896 + }, + "model.layers.1.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 11280268, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_38.const", + "file_size": 802816 + }, + "model.layers.1.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 12083084, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_39.const", + "file_size": 3584 + }, + "model.layers.1.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 12086668, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_40.const", + "file_size": 25088 + }, + "model.layers.1.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 12111756, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_41.const", + "file_size": 6272 + }, + "model.layers.1.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 12118028, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_42.const", + "file_size": 114688 + }, + "model.layers.1.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 12232716, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_43.const", + "file_size": 512 + }, + "model.layers.1.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 12233228, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_44.const", + "file_size": 3584 + }, + "model.layers.1.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 12236812, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_45.const", + "file_size": 896 + }, + "model.layers.1.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 12237708, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_46.const", + "file_size": 802816 + }, + "model.layers.1.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 13040524, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_47.const", + "file_size": 3584 + }, + "model.layers.1.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 13044108, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_48.const", + "file_size": 25088 + }, + "model.layers.1.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 13069196, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_49.const", + "file_size": 6272 + }, + "model.layers.1.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 13075468, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_50.const", + "file_size": 1792 + }, + "eps_26_1_2": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 13077260, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_51.const", + "file_size": 2 + }, + "model.layers.1.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 13077264, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_52.const", + "file_size": 2179072 + }, + "model.layers.1.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 15256336, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_53.const", + "file_size": 136192 + }, + "model.layers.1.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 15392528, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_54.const", + "file_size": 19456 + }, + "model.layers.1.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 15411984, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_55.const", + "file_size": 19456 + }, + "model.layers.1.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 15431440, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_56.const", + "file_size": 2179072 + }, + "model.layers.1.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 17610512, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_57.const", + "file_size": 136192 + }, + "model.layers.1.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 17746704, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_58.const", + "file_size": 19456 + }, + "model.layers.1.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 17766160, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_59.const", + "file_size": 19456 + }, + "model.layers.1.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 17785616, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_60.const", + "file_size": 4358144 + }, + "model.layers.1.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 22143760, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_61.const", + "file_size": 3584 + }, + "model.layers.1.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 22147344, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_62.const", + "file_size": 136192 + }, + "model.layers.1.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 22283536, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_63.const", + "file_size": 34048 + }, + "model.layers.2.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 22317584, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_64.const", + "file_size": 1792 + }, + "eps_26_1_3": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 22319376, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_65.const", + "file_size": 2 + }, + "model.layers.2.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 22319380, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_66.const", + "file_size": 114688 + }, + "model.layers.2.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 22434068, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_67.const", + "file_size": 512 + }, + "model.layers.2.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 22434580, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_68.const", + "file_size": 3584 + }, + "model.layers.2.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 22438164, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_69.const", + "file_size": 896 + }, + "model.layers.2.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 22439060, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_70.const", + "file_size": 802816 + }, + "model.layers.2.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 23241876, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_71.const", + "file_size": 3584 + }, + "model.layers.2.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 23245460, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_72.const", + "file_size": 25088 + }, + "model.layers.2.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 23270548, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_73.const", + "file_size": 6272 + }, + "model.layers.2.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 23276820, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_74.const", + "file_size": 114688 + }, + "model.layers.2.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 23391508, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_75.const", + "file_size": 512 + }, + "model.layers.2.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 23392020, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_76.const", + "file_size": 3584 + }, + "model.layers.2.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 23395604, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_77.const", + "file_size": 896 + }, + "model.layers.2.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 23396500, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_78.const", + "file_size": 802816 + }, + "model.layers.2.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 24199316, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_79.const", + "file_size": 3584 + }, + "model.layers.2.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 24202900, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_80.const", + "file_size": 25088 + }, + "model.layers.2.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 24227988, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_81.const", + "file_size": 6272 + }, + "model.layers.2.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 24234260, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_82.const", + "file_size": 1792 + }, + "eps_26_1_4": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 24236052, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_83.const", + "file_size": 2 + }, + "model.layers.2.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 24236056, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_84.const", + "file_size": 2179072 + }, + "model.layers.2.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 26415128, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_85.const", + "file_size": 136192 + }, + "model.layers.2.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 26551320, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_86.const", + "file_size": 19456 + }, + "model.layers.2.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 26570776, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_87.const", + "file_size": 19456 + }, + "model.layers.2.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 26590232, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_88.const", + "file_size": 2179072 + }, + "model.layers.2.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 28769304, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_89.const", + "file_size": 136192 + }, + "model.layers.2.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 28905496, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_90.const", + "file_size": 19456 + }, + "model.layers.2.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 28924952, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_91.const", + "file_size": 19456 + }, + "model.layers.2.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 28944408, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_92.const", + "file_size": 4358144 + }, + "model.layers.2.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 33302552, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_93.const", + "file_size": 3584 + }, + "model.layers.2.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 33306136, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_94.const", + "file_size": 136192 + }, + "model.layers.2.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 33442328, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_95.const", + "file_size": 34048 + }, + "model.layers.3.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 33476376, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_96.const", + "file_size": 1792 + }, + "eps_26_1_5": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 33478168, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_97.const", + "file_size": 2 + }, + "model.layers.3.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 33478172, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_98.const", + "file_size": 114688 + }, + "model.layers.3.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 33592860, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_99.const", + "file_size": 512 + }, + "model.layers.3.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 33593372, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_100.const", + "file_size": 3584 + }, + "model.layers.3.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 33596956, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_101.const", + "file_size": 896 + }, + "model.layers.3.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 33597852, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_102.const", + "file_size": 802816 + }, + "model.layers.3.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 34400668, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_103.const", + "file_size": 3584 + }, + "model.layers.3.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 34404252, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_104.const", + "file_size": 25088 + }, + "model.layers.3.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 34429340, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_105.const", + "file_size": 6272 + }, + "model.layers.3.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 34435612, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_106.const", + "file_size": 114688 + }, + "model.layers.3.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 34550300, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_107.const", + "file_size": 512 + }, + "model.layers.3.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 34550812, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_108.const", + "file_size": 3584 + }, + "model.layers.3.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 34554396, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_109.const", + "file_size": 896 + }, + "model.layers.3.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 34555292, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_110.const", + "file_size": 802816 + }, + "model.layers.3.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 35358108, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_111.const", + "file_size": 3584 + }, + "model.layers.3.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 35361692, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_112.const", + "file_size": 25088 + }, + "model.layers.3.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 35386780, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_113.const", + "file_size": 6272 + }, + "model.layers.3.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 35393052, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_114.const", + "file_size": 1792 + }, + "eps_26_1_6": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 35394844, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_115.const", + "file_size": 2 + }, + "model.layers.3.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 35394848, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_116.const", + "file_size": 2179072 + }, + "model.layers.3.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 37573920, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_117.const", + "file_size": 136192 + }, + "model.layers.3.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 37710112, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_118.const", + "file_size": 19456 + }, + "model.layers.3.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 37729568, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_119.const", + "file_size": 19456 + }, + "model.layers.3.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 37749024, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_120.const", + "file_size": 2179072 + }, + "model.layers.3.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 39928096, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_121.const", + "file_size": 136192 + }, + "model.layers.3.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 40064288, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_122.const", + "file_size": 19456 + }, + "model.layers.3.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 40083744, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_123.const", + "file_size": 19456 + }, + "model.layers.3.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 40103200, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_124.const", + "file_size": 4358144 + }, + "model.layers.3.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 44461344, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_125.const", + "file_size": 3584 + }, + "model.layers.3.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 44464928, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_126.const", + "file_size": 136192 + }, + "model.layers.3.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 44601120, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_127.const", + "file_size": 34048 + }, + "model.layers.4.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 44635168, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_128.const", + "file_size": 1792 + }, + "eps_26_1_7": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 44636960, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_129.const", + "file_size": 2 + }, + "model.layers.4.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 44636964, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_130.const", + "file_size": 114688 + }, + "model.layers.4.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 44751652, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_131.const", + "file_size": 512 + }, + "model.layers.4.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 44752164, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_132.const", + "file_size": 3584 + }, + "model.layers.4.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 44755748, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_133.const", + "file_size": 896 + }, + "model.layers.4.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 44756644, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_134.const", + "file_size": 802816 + }, + "model.layers.4.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 45559460, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_135.const", + "file_size": 3584 + }, + "model.layers.4.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 45563044, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_136.const", + "file_size": 25088 + }, + "model.layers.4.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 45588132, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_137.const", + "file_size": 6272 + }, + "model.layers.4.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 45594404, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_138.const", + "file_size": 114688 + }, + "model.layers.4.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 45709092, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_139.const", + "file_size": 512 + }, + "model.layers.4.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 45709604, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_140.const", + "file_size": 3584 + }, + "model.layers.4.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 45713188, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_141.const", + "file_size": 896 + }, + "model.layers.4.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 45714084, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_142.const", + "file_size": 802816 + }, + "model.layers.4.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 46516900, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_143.const", + "file_size": 3584 + }, + "model.layers.4.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 46520484, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_144.const", + "file_size": 25088 + }, + "model.layers.4.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 46545572, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_145.const", + "file_size": 6272 + }, + "model.layers.4.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 46551844, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_146.const", + "file_size": 1792 + }, + "eps_26_1_8": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 46553636, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_147.const", + "file_size": 2 + }, + "model.layers.4.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 46553640, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_148.const", + "file_size": 2179072 + }, + "model.layers.4.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 48732712, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_149.const", + "file_size": 136192 + }, + "model.layers.4.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 48868904, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_150.const", + "file_size": 19456 + }, + "model.layers.4.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 48888360, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_151.const", + "file_size": 19456 + }, + "model.layers.4.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 48907816, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_152.const", + "file_size": 2179072 + }, + "model.layers.4.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 51086888, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_153.const", + "file_size": 136192 + }, + "model.layers.4.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 51223080, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_154.const", + "file_size": 19456 + }, + "model.layers.4.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 51242536, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_155.const", + "file_size": 19456 + }, + "model.layers.4.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 51261992, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_156.const", + "file_size": 4358144 + }, + "model.layers.4.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 55620136, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_157.const", + "file_size": 3584 + }, + "model.layers.4.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 55623720, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_158.const", + "file_size": 136192 + }, + "model.layers.4.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 55759912, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_159.const", + "file_size": 34048 + }, + "model.layers.5.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 55793960, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_160.const", + "file_size": 1792 + }, + "eps_26_1_9": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 55795752, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_161.const", + "file_size": 2 + }, + "model.layers.5.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 55795756, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_162.const", + "file_size": 114688 + }, + "model.layers.5.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 55910444, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_163.const", + "file_size": 512 + }, + "model.layers.5.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 55910956, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_164.const", + "file_size": 3584 + }, + "model.layers.5.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 55914540, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_165.const", + "file_size": 896 + }, + "model.layers.5.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 55915436, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_166.const", + "file_size": 802816 + }, + "model.layers.5.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 56718252, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_167.const", + "file_size": 3584 + }, + "model.layers.5.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 56721836, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_168.const", + "file_size": 25088 + }, + "model.layers.5.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 56746924, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_169.const", + "file_size": 6272 + }, + "model.layers.5.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 56753196, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_170.const", + "file_size": 114688 + }, + "model.layers.5.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 56867884, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_171.const", + "file_size": 512 + }, + "model.layers.5.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 56868396, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_172.const", + "file_size": 3584 + }, + "model.layers.5.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 56871980, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_173.const", + "file_size": 896 + }, + "model.layers.5.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 56872876, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_174.const", + "file_size": 802816 + }, + "model.layers.5.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 57675692, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_175.const", + "file_size": 3584 + }, + "model.layers.5.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 57679276, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_176.const", + "file_size": 25088 + }, + "model.layers.5.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 57704364, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_177.const", + "file_size": 6272 + }, + "model.layers.5.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 57710636, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_178.const", + "file_size": 1792 + }, + "eps_26_1_10": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 57712428, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_179.const", + "file_size": 2 + }, + "model.layers.5.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 57712432, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_180.const", + "file_size": 2179072 + }, + "model.layers.5.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 59891504, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_181.const", + "file_size": 136192 + }, + "model.layers.5.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 60027696, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_182.const", + "file_size": 19456 + }, + "model.layers.5.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 60047152, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_183.const", + "file_size": 19456 + }, + "model.layers.5.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 60066608, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_184.const", + "file_size": 2179072 + }, + "model.layers.5.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 62245680, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_185.const", + "file_size": 136192 + }, + "model.layers.5.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 62381872, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_186.const", + "file_size": 19456 + }, + "model.layers.5.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 62401328, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_187.const", + "file_size": 19456 + }, + "model.layers.5.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 62420784, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_188.const", + "file_size": 4358144 + }, + "model.layers.5.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 66778928, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_189.const", + "file_size": 3584 + }, + "model.layers.5.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 66782512, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_190.const", + "file_size": 136192 + }, + "model.layers.5.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 66918704, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_191.const", + "file_size": 34048 + }, + "model.layers.6.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 66952752, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_192.const", + "file_size": 1792 + }, + "eps_26_1_11": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 66954544, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_193.const", + "file_size": 2 + }, + "model.layers.6.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 66954548, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_194.const", + "file_size": 114688 + }, + "model.layers.6.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 67069236, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_195.const", + "file_size": 512 + }, + "model.layers.6.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 67069748, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_196.const", + "file_size": 3584 + }, + "model.layers.6.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 67073332, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_197.const", + "file_size": 896 + }, + "model.layers.6.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 67074228, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_198.const", + "file_size": 802816 + }, + "model.layers.6.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 67877044, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_199.const", + "file_size": 3584 + }, + "model.layers.6.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 67880628, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_200.const", + "file_size": 25088 + }, + "model.layers.6.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 67905716, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_201.const", + "file_size": 6272 + }, + "model.layers.6.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 67911988, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_202.const", + "file_size": 114688 + }, + "model.layers.6.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 68026676, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_203.const", + "file_size": 512 + }, + "model.layers.6.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 68027188, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_204.const", + "file_size": 3584 + }, + "model.layers.6.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 68030772, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_205.const", + "file_size": 896 + }, + "model.layers.6.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 68031668, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_206.const", + "file_size": 802816 + }, + "model.layers.6.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 68834484, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_207.const", + "file_size": 3584 + }, + "model.layers.6.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 68838068, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_208.const", + "file_size": 25088 + }, + "model.layers.6.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 68863156, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_209.const", + "file_size": 6272 + }, + "model.layers.6.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 68869428, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_210.const", + "file_size": 1792 + }, + "eps_26_1_12": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 68871220, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_211.const", + "file_size": 2 + }, + "model.layers.6.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 68871224, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_212.const", + "file_size": 2179072 + }, + "model.layers.6.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 71050296, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_213.const", + "file_size": 136192 + }, + "model.layers.6.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 71186488, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_214.const", + "file_size": 19456 + }, + "model.layers.6.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 71205944, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_215.const", + "file_size": 19456 + }, + "model.layers.6.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 71225400, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_216.const", + "file_size": 2179072 + }, + "model.layers.6.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 73404472, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_217.const", + "file_size": 136192 + }, + "model.layers.6.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 73540664, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_218.const", + "file_size": 19456 + }, + "model.layers.6.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 73560120, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_219.const", + "file_size": 19456 + }, + "model.layers.6.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 73579576, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_220.const", + "file_size": 4358144 + }, + "model.layers.6.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 77937720, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_221.const", + "file_size": 3584 + }, + "model.layers.6.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 77941304, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_222.const", + "file_size": 136192 + }, + "model.layers.6.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 78077496, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_223.const", + "file_size": 34048 + }, + "model.layers.7.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 78111544, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_224.const", + "file_size": 1792 + }, + "eps_26_1_13": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 78113336, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_225.const", + "file_size": 2 + }, + "model.layers.7.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 78113340, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_226.const", + "file_size": 114688 + }, + "model.layers.7.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 78228028, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_227.const", + "file_size": 512 + }, + "model.layers.7.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 78228540, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_228.const", + "file_size": 3584 + }, + "model.layers.7.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 78232124, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_229.const", + "file_size": 896 + }, + "model.layers.7.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 78233020, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_230.const", + "file_size": 802816 + }, + "model.layers.7.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 79035836, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_231.const", + "file_size": 3584 + }, + "model.layers.7.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 79039420, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_232.const", + "file_size": 25088 + }, + "model.layers.7.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 79064508, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_233.const", + "file_size": 6272 + }, + "model.layers.7.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 79070780, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_234.const", + "file_size": 114688 + }, + "model.layers.7.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 79185468, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_235.const", + "file_size": 512 + }, + "model.layers.7.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 79185980, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_236.const", + "file_size": 3584 + }, + "model.layers.7.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 79189564, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_237.const", + "file_size": 896 + }, + "model.layers.7.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 79190460, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_238.const", + "file_size": 802816 + }, + "model.layers.7.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 79993276, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_239.const", + "file_size": 3584 + }, + "model.layers.7.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 79996860, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_240.const", + "file_size": 25088 + }, + "model.layers.7.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 80021948, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_241.const", + "file_size": 6272 + }, + "model.layers.7.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 80028220, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_242.const", + "file_size": 1792 + }, + "eps_26_1_14": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 80030012, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_243.const", + "file_size": 2 + }, + "model.layers.7.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 80030016, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_244.const", + "file_size": 2179072 + }, + "model.layers.7.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 82209088, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_245.const", + "file_size": 136192 + }, + "model.layers.7.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 82345280, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_246.const", + "file_size": 19456 + }, + "model.layers.7.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 82364736, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_247.const", + "file_size": 19456 + }, + "model.layers.7.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 82384192, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_248.const", + "file_size": 2179072 + }, + "model.layers.7.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 84563264, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_249.const", + "file_size": 136192 + }, + "model.layers.7.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 84699456, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_250.const", + "file_size": 19456 + }, + "model.layers.7.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 84718912, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_251.const", + "file_size": 19456 + }, + "model.layers.7.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 84738368, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_252.const", + "file_size": 4358144 + }, + "model.layers.7.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 89096512, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_253.const", + "file_size": 3584 + }, + "model.layers.7.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 89100096, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_254.const", + "file_size": 136192 + }, + "model.layers.7.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 89236288, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_255.const", + "file_size": 34048 + }, + "model.layers.8.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 89270336, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_256.const", + "file_size": 1792 + }, + "eps_26_1_15": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 89272128, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_257.const", + "file_size": 2 + }, + "model.layers.8.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 89272132, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_258.const", + "file_size": 114688 + }, + "model.layers.8.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 89386820, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_259.const", + "file_size": 512 + }, + "model.layers.8.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 89387332, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_260.const", + "file_size": 3584 + }, + "model.layers.8.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 89390916, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_261.const", + "file_size": 896 + }, + "model.layers.8.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 89391812, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_262.const", + "file_size": 802816 + }, + "model.layers.8.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 90194628, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_263.const", + "file_size": 3584 + }, + "model.layers.8.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 90198212, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_264.const", + "file_size": 25088 + }, + "model.layers.8.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 90223300, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_265.const", + "file_size": 6272 + }, + "model.layers.8.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 90229572, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_266.const", + "file_size": 114688 + }, + "model.layers.8.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 90344260, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_267.const", + "file_size": 512 + }, + "model.layers.8.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 90344772, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_268.const", + "file_size": 3584 + }, + "model.layers.8.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 90348356, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_269.const", + "file_size": 896 + }, + "model.layers.8.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 90349252, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_270.const", + "file_size": 802816 + }, + "model.layers.8.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 91152068, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_271.const", + "file_size": 3584 + }, + "model.layers.8.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 91155652, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_272.const", + "file_size": 25088 + }, + "model.layers.8.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 91180740, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_273.const", + "file_size": 6272 + }, + "model.layers.8.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 91187012, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_274.const", + "file_size": 1792 + }, + "eps_26_1_16": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 91188804, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_275.const", + "file_size": 2 + }, + "model.layers.8.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 91188808, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_276.const", + "file_size": 2179072 + }, + "model.layers.8.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 93367880, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_277.const", + "file_size": 136192 + }, + "model.layers.8.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 93504072, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_278.const", + "file_size": 19456 + }, + "model.layers.8.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 93523528, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_279.const", + "file_size": 19456 + }, + "model.layers.8.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 93542984, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_280.const", + "file_size": 2179072 + }, + "model.layers.8.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 95722056, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_281.const", + "file_size": 136192 + }, + "model.layers.8.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 95858248, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_282.const", + "file_size": 19456 + }, + "model.layers.8.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 95877704, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_283.const", + "file_size": 19456 + }, + "model.layers.8.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 95897160, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_284.const", + "file_size": 4358144 + }, + "model.layers.8.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 100255304, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_285.const", + "file_size": 3584 + }, + "model.layers.8.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 100258888, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_286.const", + "file_size": 136192 + }, + "model.layers.8.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 100395080, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_287.const", + "file_size": 34048 + }, + "model.layers.9.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 100429128, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_288.const", + "file_size": 1792 + }, + "eps_26_1_17": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 100430920, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_289.const", + "file_size": 2 + }, + "model.layers.9.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 100430924, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_290.const", + "file_size": 114688 + }, + "model.layers.9.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 100545612, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_291.const", + "file_size": 512 + }, + "model.layers.9.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 100546124, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_292.const", + "file_size": 3584 + }, + "model.layers.9.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 100549708, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_293.const", + "file_size": 896 + }, + "model.layers.9.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 100550604, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_294.const", + "file_size": 802816 + }, + "model.layers.9.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 101353420, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_295.const", + "file_size": 3584 + }, + "model.layers.9.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 101357004, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_296.const", + "file_size": 25088 + }, + "model.layers.9.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 101382092, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_297.const", + "file_size": 6272 + }, + "model.layers.9.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 101388364, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_298.const", + "file_size": 114688 + }, + "model.layers.9.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 101503052, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_299.const", + "file_size": 512 + }, + "model.layers.9.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 101503564, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_300.const", + "file_size": 3584 + }, + "model.layers.9.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 101507148, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_301.const", + "file_size": 896 + }, + "model.layers.9.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 101508044, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_302.const", + "file_size": 802816 + }, + "model.layers.9.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 102310860, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_303.const", + "file_size": 3584 + }, + "model.layers.9.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 102314444, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_304.const", + "file_size": 25088 + }, + "model.layers.9.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 102339532, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_305.const", + "file_size": 6272 + }, + "model.layers.9.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 102345804, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_306.const", + "file_size": 1792 + }, + "eps_26_1_18": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 102347596, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_307.const", + "file_size": 2 + }, + "model.layers.9.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 102347600, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_308.const", + "file_size": 2179072 + }, + "model.layers.9.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 104526672, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_309.const", + "file_size": 136192 + }, + "model.layers.9.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 104662864, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_310.const", + "file_size": 19456 + }, + "model.layers.9.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 104682320, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_311.const", + "file_size": 19456 + }, + "model.layers.9.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 104701776, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_312.const", + "file_size": 2179072 + }, + "model.layers.9.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 106880848, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_313.const", + "file_size": 136192 + }, + "model.layers.9.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 107017040, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_314.const", + "file_size": 19456 + }, + "model.layers.9.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 107036496, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_315.const", + "file_size": 19456 + }, + "model.layers.9.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 107055952, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_316.const", + "file_size": 4358144 + }, + "model.layers.9.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 111414096, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_317.const", + "file_size": 3584 + }, + "model.layers.9.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 111417680, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_318.const", + "file_size": 136192 + }, + "model.layers.9.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 111553872, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_319.const", + "file_size": 34048 + }, + "model.layers.10.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 111587920, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_320.const", + "file_size": 1792 + }, + "eps_26_1_19": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 111589712, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_321.const", + "file_size": 2 + }, + "model.layers.10.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 111589716, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_322.const", + "file_size": 114688 + }, + "model.layers.10.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 111704404, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_323.const", + "file_size": 512 + }, + "model.layers.10.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 111704916, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_324.const", + "file_size": 3584 + }, + "model.layers.10.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 111708500, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_325.const", + "file_size": 896 + }, + "model.layers.10.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 111709396, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_326.const", + "file_size": 802816 + }, + "model.layers.10.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 112512212, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_327.const", + "file_size": 3584 + }, + "model.layers.10.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 112515796, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_328.const", + "file_size": 25088 + }, + "model.layers.10.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 112540884, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_329.const", + "file_size": 6272 + }, + "model.layers.10.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 112547156, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_330.const", + "file_size": 114688 + }, + "model.layers.10.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 112661844, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_331.const", + "file_size": 512 + }, + "model.layers.10.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 112662356, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_332.const", + "file_size": 3584 + }, + "model.layers.10.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 112665940, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_333.const", + "file_size": 896 + }, + "model.layers.10.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 112666836, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_334.const", + "file_size": 802816 + }, + "model.layers.10.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 113469652, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_335.const", + "file_size": 3584 + }, + "model.layers.10.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 113473236, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_336.const", + "file_size": 25088 + }, + "model.layers.10.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 113498324, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_337.const", + "file_size": 6272 + }, + "model.layers.10.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 113504596, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_338.const", + "file_size": 1792 + }, + "eps_26_1_20": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 113506388, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_339.const", + "file_size": 2 + }, + "model.layers.10.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 113506392, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_340.const", + "file_size": 2179072 + }, + "model.layers.10.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 115685464, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_341.const", + "file_size": 136192 + }, + "model.layers.10.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 115821656, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_342.const", + "file_size": 19456 + }, + "model.layers.10.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 115841112, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_343.const", + "file_size": 19456 + }, + "model.layers.10.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 115860568, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_344.const", + "file_size": 2179072 + }, + "model.layers.10.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 118039640, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_345.const", + "file_size": 136192 + }, + "model.layers.10.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 118175832, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_346.const", + "file_size": 19456 + }, + "model.layers.10.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 118195288, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_347.const", + "file_size": 19456 + }, + "model.layers.10.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 118214744, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_348.const", + "file_size": 4358144 + }, + "model.layers.10.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 122572888, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_349.const", + "file_size": 3584 + }, + "model.layers.10.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 122576472, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_350.const", + "file_size": 136192 + }, + "model.layers.10.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 122712664, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_351.const", + "file_size": 34048 + }, + "model.layers.11.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 122746712, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_352.const", + "file_size": 1792 + }, + "eps_26_1_21": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 122748504, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_353.const", + "file_size": 2 + }, + "model.layers.11.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 122748508, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_354.const", + "file_size": 114688 + }, + "model.layers.11.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 122863196, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_355.const", + "file_size": 512 + }, + "model.layers.11.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 122863708, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_356.const", + "file_size": 3584 + }, + "model.layers.11.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 122867292, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_357.const", + "file_size": 896 + }, + "model.layers.11.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 122868188, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_358.const", + "file_size": 802816 + }, + "model.layers.11.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 123671004, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_359.const", + "file_size": 3584 + }, + "model.layers.11.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 123674588, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_360.const", + "file_size": 25088 + }, + "model.layers.11.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 123699676, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_361.const", + "file_size": 6272 + }, + "model.layers.11.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 123705948, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_362.const", + "file_size": 114688 + }, + "model.layers.11.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 123820636, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_363.const", + "file_size": 512 + }, + "model.layers.11.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 123821148, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_364.const", + "file_size": 3584 + }, + "model.layers.11.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 123824732, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_365.const", + "file_size": 896 + }, + "model.layers.11.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 123825628, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_366.const", + "file_size": 802816 + }, + "model.layers.11.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 124628444, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_367.const", + "file_size": 3584 + }, + "model.layers.11.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 124632028, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_368.const", + "file_size": 25088 + }, + "model.layers.11.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 124657116, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_369.const", + "file_size": 6272 + }, + "model.layers.11.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 124663388, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_370.const", + "file_size": 1792 + }, + "eps_26_1_22": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 124665180, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_371.const", + "file_size": 2 + }, + "model.layers.11.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 124665184, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_372.const", + "file_size": 2179072 + }, + "model.layers.11.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 126844256, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_373.const", + "file_size": 136192 + }, + "model.layers.11.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 126980448, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_374.const", + "file_size": 19456 + }, + "model.layers.11.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 126999904, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_375.const", + "file_size": 19456 + }, + "model.layers.11.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 127019360, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_376.const", + "file_size": 2179072 + }, + "model.layers.11.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 129198432, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_377.const", + "file_size": 136192 + }, + "model.layers.11.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 129334624, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_378.const", + "file_size": 19456 + }, + "model.layers.11.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 129354080, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_379.const", + "file_size": 19456 + }, + "model.layers.11.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 129373536, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_380.const", + "file_size": 4358144 + }, + "model.layers.11.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 133731680, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_381.const", + "file_size": 3584 + }, + "model.layers.11.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 133735264, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_382.const", + "file_size": 136192 + }, + "model.layers.11.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 133871456, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_383.const", + "file_size": 34048 + }, + "model.layers.12.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 133905504, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_384.const", + "file_size": 1792 + }, + "eps_26_1_23": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 133907296, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_385.const", + "file_size": 2 + }, + "model.layers.12.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 133907300, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_386.const", + "file_size": 114688 + }, + "model.layers.12.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 134021988, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_387.const", + "file_size": 512 + }, + "model.layers.12.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 134022500, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_388.const", + "file_size": 3584 + }, + "model.layers.12.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 134026084, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_389.const", + "file_size": 896 + }, + "model.layers.12.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 134026980, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_390.const", + "file_size": 802816 + }, + "model.layers.12.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 134829796, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_391.const", + "file_size": 3584 + }, + "model.layers.12.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 134833380, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_392.const", + "file_size": 25088 + }, + "model.layers.12.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 134858468, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_393.const", + "file_size": 6272 + }, + "model.layers.12.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 134864740, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_394.const", + "file_size": 114688 + }, + "model.layers.12.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 134979428, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_395.const", + "file_size": 512 + }, + "model.layers.12.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 134979940, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_396.const", + "file_size": 3584 + }, + "model.layers.12.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 134983524, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_397.const", + "file_size": 896 + }, + "model.layers.12.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 134984420, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_398.const", + "file_size": 802816 + }, + "model.layers.12.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 135787236, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_399.const", + "file_size": 3584 + }, + "model.layers.12.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 135790820, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_400.const", + "file_size": 25088 + }, + "model.layers.12.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 135815908, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_401.const", + "file_size": 6272 + }, + "model.layers.12.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 135822180, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_402.const", + "file_size": 1792 + }, + "eps_26_1_24": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 135823972, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_403.const", + "file_size": 2 + }, + "model.layers.12.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 135823976, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_404.const", + "file_size": 2179072 + }, + "model.layers.12.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 138003048, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_405.const", + "file_size": 136192 + }, + "model.layers.12.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 138139240, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_406.const", + "file_size": 19456 + }, + "model.layers.12.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 138158696, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_407.const", + "file_size": 19456 + }, + "model.layers.12.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 138178152, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_408.const", + "file_size": 2179072 + }, + "model.layers.12.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 140357224, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_409.const", + "file_size": 136192 + }, + "model.layers.12.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 140493416, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_410.const", + "file_size": 19456 + }, + "model.layers.12.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 140512872, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_411.const", + "file_size": 19456 + }, + "model.layers.12.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 140532328, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_412.const", + "file_size": 4358144 + }, + "model.layers.12.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 144890472, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_413.const", + "file_size": 3584 + }, + "model.layers.12.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 144894056, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_414.const", + "file_size": 136192 + }, + "model.layers.12.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 145030248, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_415.const", + "file_size": 34048 + }, + "model.layers.13.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 145064296, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_416.const", + "file_size": 1792 + }, + "eps_26_1_25": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 145066088, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_417.const", + "file_size": 2 + }, + "model.layers.13.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 145066092, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_418.const", + "file_size": 114688 + }, + "model.layers.13.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 145180780, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_419.const", + "file_size": 512 + }, + "model.layers.13.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 145181292, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_420.const", + "file_size": 3584 + }, + "model.layers.13.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 145184876, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_421.const", + "file_size": 896 + }, + "model.layers.13.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 145185772, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_422.const", + "file_size": 802816 + }, + "model.layers.13.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 145988588, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_423.const", + "file_size": 3584 + }, + "model.layers.13.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 145992172, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_424.const", + "file_size": 25088 + }, + "model.layers.13.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 146017260, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_425.const", + "file_size": 6272 + }, + "model.layers.13.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 146023532, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_426.const", + "file_size": 114688 + }, + "model.layers.13.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 146138220, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_427.const", + "file_size": 512 + }, + "model.layers.13.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 146138732, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_428.const", + "file_size": 3584 + }, + "model.layers.13.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 146142316, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_429.const", + "file_size": 896 + }, + "model.layers.13.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 146143212, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_430.const", + "file_size": 802816 + }, + "model.layers.13.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 146946028, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_431.const", + "file_size": 3584 + }, + "model.layers.13.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 146949612, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_432.const", + "file_size": 25088 + }, + "model.layers.13.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 146974700, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_433.const", + "file_size": 6272 + }, + "model.layers.13.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 146980972, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_434.const", + "file_size": 1792 + }, + "eps_26_1_26": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 146982764, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_435.const", + "file_size": 2 + }, + "model.layers.13.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 146982768, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_436.const", + "file_size": 2179072 + }, + "model.layers.13.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 149161840, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_437.const", + "file_size": 136192 + }, + "model.layers.13.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 149298032, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_438.const", + "file_size": 19456 + }, + "model.layers.13.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 149317488, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_439.const", + "file_size": 19456 + }, + "model.layers.13.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 149336944, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_440.const", + "file_size": 2179072 + }, + "model.layers.13.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 151516016, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_441.const", + "file_size": 136192 + }, + "model.layers.13.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 151652208, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_442.const", + "file_size": 19456 + }, + "model.layers.13.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 151671664, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_443.const", + "file_size": 19456 + }, + "model.layers.13.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 151691120, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_444.const", + "file_size": 4358144 + }, + "model.layers.13.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 156049264, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_445.const", + "file_size": 3584 + }, + "model.layers.13.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 156052848, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_446.const", + "file_size": 136192 + }, + "model.layers.13.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 156189040, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_447.const", + "file_size": 34048 + }, + "model.layers.14.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 156223088, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_448.const", + "file_size": 1792 + }, + "eps_26_1_27": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 156224880, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_449.const", + "file_size": 2 + }, + "model.layers.14.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 156224884, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_450.const", + "file_size": 114688 + }, + "model.layers.14.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 156339572, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_451.const", + "file_size": 512 + }, + "model.layers.14.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 156340084, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_452.const", + "file_size": 3584 + }, + "model.layers.14.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 156343668, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_453.const", + "file_size": 896 + }, + "model.layers.14.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 156344564, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_454.const", + "file_size": 802816 + }, + "model.layers.14.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 157147380, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_455.const", + "file_size": 3584 + }, + "model.layers.14.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 157150964, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_456.const", + "file_size": 25088 + }, + "model.layers.14.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 157176052, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_457.const", + "file_size": 6272 + }, + "model.layers.14.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 157182324, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_458.const", + "file_size": 114688 + }, + "model.layers.14.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 157297012, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_459.const", + "file_size": 512 + }, + "model.layers.14.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 157297524, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_460.const", + "file_size": 3584 + }, + "model.layers.14.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 157301108, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_461.const", + "file_size": 896 + }, + "model.layers.14.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 157302004, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_462.const", + "file_size": 802816 + }, + "model.layers.14.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 158104820, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_463.const", + "file_size": 3584 + }, + "model.layers.14.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 158108404, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_464.const", + "file_size": 25088 + }, + "model.layers.14.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 158133492, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_465.const", + "file_size": 6272 + }, + "model.layers.14.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 158139764, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_466.const", + "file_size": 1792 + }, + "eps_26_1_28": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 158141556, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_467.const", + "file_size": 2 + }, + "model.layers.14.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 158141560, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_468.const", + "file_size": 2179072 + }, + "model.layers.14.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 160320632, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_469.const", + "file_size": 136192 + }, + "model.layers.14.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 160456824, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_470.const", + "file_size": 19456 + }, + "model.layers.14.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 160476280, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_471.const", + "file_size": 19456 + }, + "model.layers.14.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 160495736, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_472.const", + "file_size": 2179072 + }, + "model.layers.14.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 162674808, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_473.const", + "file_size": 136192 + }, + "model.layers.14.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 162811000, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_474.const", + "file_size": 19456 + }, + "model.layers.14.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 162830456, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_475.const", + "file_size": 19456 + }, + "model.layers.14.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 162849912, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_476.const", + "file_size": 4358144 + }, + "model.layers.14.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 167208056, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_477.const", + "file_size": 3584 + }, + "model.layers.14.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 167211640, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_478.const", + "file_size": 136192 + }, + "model.layers.14.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 167347832, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_479.const", + "file_size": 34048 + }, + "model.layers.15.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 167381880, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_480.const", + "file_size": 1792 + }, + "eps_26_1_29": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 167383672, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_481.const", + "file_size": 2 + }, + "model.layers.15.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 167383676, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_482.const", + "file_size": 114688 + }, + "model.layers.15.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 167498364, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_483.const", + "file_size": 512 + }, + "model.layers.15.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 167498876, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_484.const", + "file_size": 3584 + }, + "model.layers.15.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 167502460, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_485.const", + "file_size": 896 + }, + "model.layers.15.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 167503356, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_486.const", + "file_size": 802816 + }, + "model.layers.15.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 168306172, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_487.const", + "file_size": 3584 + }, + "model.layers.15.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 168309756, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_488.const", + "file_size": 25088 + }, + "model.layers.15.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 168334844, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_489.const", + "file_size": 6272 + }, + "model.layers.15.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 168341116, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_490.const", + "file_size": 114688 + }, + "model.layers.15.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 168455804, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_491.const", + "file_size": 512 + }, + "model.layers.15.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 168456316, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_492.const", + "file_size": 3584 + }, + "model.layers.15.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 168459900, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_493.const", + "file_size": 896 + }, + "model.layers.15.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 168460796, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_494.const", + "file_size": 802816 + }, + "model.layers.15.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 169263612, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_495.const", + "file_size": 3584 + }, + "model.layers.15.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 169267196, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_496.const", + "file_size": 25088 + }, + "model.layers.15.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 169292284, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_497.const", + "file_size": 6272 + }, + "model.layers.15.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 169298556, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_498.const", + "file_size": 1792 + }, + "eps_26_1_30": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 169300348, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_499.const", + "file_size": 2 + }, + "model.layers.15.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 169300352, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_500.const", + "file_size": 2179072 + }, + "model.layers.15.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 171479424, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_501.const", + "file_size": 136192 + }, + "model.layers.15.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 171615616, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_502.const", + "file_size": 19456 + }, + "model.layers.15.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 171635072, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_503.const", + "file_size": 19456 + }, + "model.layers.15.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 171654528, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_504.const", + "file_size": 2179072 + }, + "model.layers.15.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 173833600, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_505.const", + "file_size": 136192 + }, + "model.layers.15.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 173969792, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_506.const", + "file_size": 19456 + }, + "model.layers.15.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 173989248, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_507.const", + "file_size": 19456 + }, + "model.layers.15.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 174008704, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_508.const", + "file_size": 4358144 + }, + "model.layers.15.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 178366848, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_509.const", + "file_size": 3584 + }, + "model.layers.15.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 178370432, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_510.const", + "file_size": 136192 + }, + "model.layers.15.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 178506624, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_511.const", + "file_size": 34048 + }, + "model.layers.16.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 178540672, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_512.const", + "file_size": 1792 + }, + "eps_26_1_31": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 178542464, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_513.const", + "file_size": 2 + }, + "model.layers.16.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 178542468, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_514.const", + "file_size": 114688 + }, + "model.layers.16.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 178657156, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_515.const", + "file_size": 512 + }, + "model.layers.16.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 178657668, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_516.const", + "file_size": 3584 + }, + "model.layers.16.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 178661252, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_517.const", + "file_size": 896 + }, + "model.layers.16.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 178662148, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_518.const", + "file_size": 802816 + }, + "model.layers.16.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 179464964, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_519.const", + "file_size": 3584 + }, + "model.layers.16.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 179468548, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_520.const", + "file_size": 25088 + }, + "model.layers.16.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 179493636, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_521.const", + "file_size": 6272 + }, + "model.layers.16.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 179499908, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_522.const", + "file_size": 114688 + }, + "model.layers.16.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 179614596, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_523.const", + "file_size": 512 + }, + "model.layers.16.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 179615108, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_524.const", + "file_size": 3584 + }, + "model.layers.16.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 179618692, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_525.const", + "file_size": 896 + }, + "model.layers.16.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 179619588, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_526.const", + "file_size": 802816 + }, + "model.layers.16.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 180422404, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_527.const", + "file_size": 3584 + }, + "model.layers.16.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 180425988, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_528.const", + "file_size": 25088 + }, + "model.layers.16.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 180451076, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_529.const", + "file_size": 6272 + }, + "model.layers.16.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 180457348, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_530.const", + "file_size": 1792 + }, + "eps_26_1_32": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 180459140, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_531.const", + "file_size": 2 + }, + "model.layers.16.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 180459144, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_532.const", + "file_size": 2179072 + }, + "model.layers.16.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 182638216, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_533.const", + "file_size": 136192 + }, + "model.layers.16.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 182774408, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_534.const", + "file_size": 19456 + }, + "model.layers.16.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 182793864, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_535.const", + "file_size": 19456 + }, + "model.layers.16.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 182813320, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_536.const", + "file_size": 2179072 + }, + "model.layers.16.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 184992392, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_537.const", + "file_size": 136192 + }, + "model.layers.16.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 185128584, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_538.const", + "file_size": 19456 + }, + "model.layers.16.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 185148040, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_539.const", + "file_size": 19456 + }, + "model.layers.16.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 185167496, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_540.const", + "file_size": 4358144 + }, + "model.layers.16.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 189525640, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_541.const", + "file_size": 3584 + }, + "model.layers.16.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 189529224, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_542.const", + "file_size": 136192 + }, + "model.layers.16.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 189665416, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_543.const", + "file_size": 34048 + }, + "model.layers.17.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 189699464, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_544.const", + "file_size": 1792 + }, + "eps_26_1_33": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 189701256, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_545.const", + "file_size": 2 + }, + "model.layers.17.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 189701260, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_546.const", + "file_size": 114688 + }, + "model.layers.17.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 189815948, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_547.const", + "file_size": 512 + }, + "model.layers.17.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 189816460, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_548.const", + "file_size": 3584 + }, + "model.layers.17.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 189820044, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_549.const", + "file_size": 896 + }, + "model.layers.17.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 189820940, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_550.const", + "file_size": 802816 + }, + "model.layers.17.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 190623756, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_551.const", + "file_size": 3584 + }, + "model.layers.17.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 190627340, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_552.const", + "file_size": 25088 + }, + "model.layers.17.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 190652428, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_553.const", + "file_size": 6272 + }, + "model.layers.17.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 190658700, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_554.const", + "file_size": 114688 + }, + "model.layers.17.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 190773388, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_555.const", + "file_size": 512 + }, + "model.layers.17.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 190773900, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_556.const", + "file_size": 3584 + }, + "model.layers.17.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 190777484, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_557.const", + "file_size": 896 + }, + "model.layers.17.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 190778380, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_558.const", + "file_size": 802816 + }, + "model.layers.17.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 191581196, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_559.const", + "file_size": 3584 + }, + "model.layers.17.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 191584780, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_560.const", + "file_size": 25088 + }, + "model.layers.17.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 191609868, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_561.const", + "file_size": 6272 + }, + "model.layers.17.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 191616140, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_562.const", + "file_size": 1792 + }, + "eps_26_1_34": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 191617932, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_563.const", + "file_size": 2 + }, + "model.layers.17.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 191617936, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_564.const", + "file_size": 2179072 + }, + "model.layers.17.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 193797008, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_565.const", + "file_size": 136192 + }, + "model.layers.17.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 193933200, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_566.const", + "file_size": 19456 + }, + "model.layers.17.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 193952656, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_567.const", + "file_size": 19456 + }, + "model.layers.17.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 193972112, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_568.const", + "file_size": 2179072 + }, + "model.layers.17.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 196151184, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_569.const", + "file_size": 136192 + }, + "model.layers.17.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 196287376, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_570.const", + "file_size": 19456 + }, + "model.layers.17.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 196306832, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_571.const", + "file_size": 19456 + }, + "model.layers.17.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 196326288, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_572.const", + "file_size": 4358144 + }, + "model.layers.17.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 200684432, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_573.const", + "file_size": 3584 + }, + "model.layers.17.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 200688016, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_574.const", + "file_size": 136192 + }, + "model.layers.17.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 200824208, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_575.const", + "file_size": 34048 + }, + "model.layers.18.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 200858256, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_576.const", + "file_size": 1792 + }, + "eps_26_1_35": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 200860048, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_577.const", + "file_size": 2 + }, + "model.layers.18.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 200860052, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_578.const", + "file_size": 114688 + }, + "model.layers.18.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 200974740, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_579.const", + "file_size": 512 + }, + "model.layers.18.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 200975252, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_580.const", + "file_size": 3584 + }, + "model.layers.18.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 200978836, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_581.const", + "file_size": 896 + }, + "model.layers.18.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 200979732, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_582.const", + "file_size": 802816 + }, + "model.layers.18.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 201782548, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_583.const", + "file_size": 3584 + }, + "model.layers.18.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 201786132, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_584.const", + "file_size": 25088 + }, + "model.layers.18.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 201811220, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_585.const", + "file_size": 6272 + }, + "model.layers.18.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 201817492, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_586.const", + "file_size": 114688 + }, + "model.layers.18.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 201932180, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_587.const", + "file_size": 512 + }, + "model.layers.18.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 201932692, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_588.const", + "file_size": 3584 + }, + "model.layers.18.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 201936276, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_589.const", + "file_size": 896 + }, + "model.layers.18.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 201937172, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_590.const", + "file_size": 802816 + }, + "model.layers.18.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 202739988, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_591.const", + "file_size": 3584 + }, + "model.layers.18.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 202743572, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_592.const", + "file_size": 25088 + }, + "model.layers.18.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 202768660, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_593.const", + "file_size": 6272 + }, + "model.layers.18.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 202774932, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_594.const", + "file_size": 1792 + }, + "eps_26_1_36": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 202776724, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_595.const", + "file_size": 2 + }, + "model.layers.18.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 202776728, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_596.const", + "file_size": 2179072 + }, + "model.layers.18.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 204955800, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_597.const", + "file_size": 136192 + }, + "model.layers.18.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 205091992, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_598.const", + "file_size": 19456 + }, + "model.layers.18.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 205111448, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_599.const", + "file_size": 19456 + }, + "model.layers.18.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 205130904, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_600.const", + "file_size": 2179072 + }, + "model.layers.18.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 207309976, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_601.const", + "file_size": 136192 + }, + "model.layers.18.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 207446168, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_602.const", + "file_size": 19456 + }, + "model.layers.18.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 207465624, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_603.const", + "file_size": 19456 + }, + "model.layers.18.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 207485080, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_604.const", + "file_size": 4358144 + }, + "model.layers.18.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 211843224, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_605.const", + "file_size": 3584 + }, + "model.layers.18.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 211846808, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_606.const", + "file_size": 136192 + }, + "model.layers.18.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 211983000, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_607.const", + "file_size": 34048 + }, + "model.layers.19.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 212017048, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_608.const", + "file_size": 1792 + }, + "eps_26_1_37": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 212018840, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_609.const", + "file_size": 2 + }, + "model.layers.19.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 212018844, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_610.const", + "file_size": 114688 + }, + "model.layers.19.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 212133532, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_611.const", + "file_size": 512 + }, + "model.layers.19.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 212134044, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_612.const", + "file_size": 3584 + }, + "model.layers.19.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 212137628, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_613.const", + "file_size": 896 + }, + "model.layers.19.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 212138524, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_614.const", + "file_size": 802816 + }, + "model.layers.19.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 212941340, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_615.const", + "file_size": 3584 + }, + "model.layers.19.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 212944924, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_616.const", + "file_size": 25088 + }, + "model.layers.19.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 212970012, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_617.const", + "file_size": 6272 + }, + "model.layers.19.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 212976284, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_618.const", + "file_size": 114688 + }, + "model.layers.19.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 213090972, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_619.const", + "file_size": 512 + }, + "model.layers.19.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 213091484, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_620.const", + "file_size": 3584 + }, + "model.layers.19.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 213095068, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_621.const", + "file_size": 896 + }, + "model.layers.19.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 213095964, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_622.const", + "file_size": 802816 + }, + "model.layers.19.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 213898780, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_623.const", + "file_size": 3584 + }, + "model.layers.19.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 213902364, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_624.const", + "file_size": 25088 + }, + "model.layers.19.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 213927452, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_625.const", + "file_size": 6272 + }, + "model.layers.19.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 213933724, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_626.const", + "file_size": 1792 + }, + "eps_26_1_38": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 213935516, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_627.const", + "file_size": 2 + }, + "model.layers.19.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 213935520, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_628.const", + "file_size": 2179072 + }, + "model.layers.19.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 216114592, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_629.const", + "file_size": 136192 + }, + "model.layers.19.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 216250784, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_630.const", + "file_size": 19456 + }, + "model.layers.19.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 216270240, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_631.const", + "file_size": 19456 + }, + "model.layers.19.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 216289696, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_632.const", + "file_size": 2179072 + }, + "model.layers.19.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 218468768, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_633.const", + "file_size": 136192 + }, + "model.layers.19.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 218604960, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_634.const", + "file_size": 19456 + }, + "model.layers.19.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 218624416, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_635.const", + "file_size": 19456 + }, + "model.layers.19.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 218643872, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_636.const", + "file_size": 4358144 + }, + "model.layers.19.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 223002016, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_637.const", + "file_size": 3584 + }, + "model.layers.19.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 223005600, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_638.const", + "file_size": 136192 + }, + "model.layers.19.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 223141792, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_639.const", + "file_size": 34048 + }, + "model.layers.20.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 223175840, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_640.const", + "file_size": 1792 + }, + "eps_26_1_39": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 223177632, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_641.const", + "file_size": 2 + }, + "model.layers.20.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 223177636, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_642.const", + "file_size": 114688 + }, + "model.layers.20.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 223292324, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_643.const", + "file_size": 512 + }, + "model.layers.20.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 223292836, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_644.const", + "file_size": 3584 + }, + "model.layers.20.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 223296420, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_645.const", + "file_size": 896 + }, + "model.layers.20.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 223297316, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_646.const", + "file_size": 802816 + }, + "model.layers.20.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 224100132, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_647.const", + "file_size": 3584 + }, + "model.layers.20.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 224103716, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_648.const", + "file_size": 25088 + }, + "model.layers.20.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 224128804, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_649.const", + "file_size": 6272 + }, + "model.layers.20.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 224135076, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_650.const", + "file_size": 114688 + }, + "model.layers.20.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 224249764, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_651.const", + "file_size": 512 + }, + "model.layers.20.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 224250276, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_652.const", + "file_size": 3584 + }, + "model.layers.20.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 224253860, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_653.const", + "file_size": 896 + }, + "model.layers.20.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 224254756, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_654.const", + "file_size": 802816 + }, + "model.layers.20.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 225057572, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_655.const", + "file_size": 3584 + }, + "model.layers.20.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 225061156, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_656.const", + "file_size": 25088 + }, + "model.layers.20.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 225086244, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_657.const", + "file_size": 6272 + }, + "model.layers.20.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 225092516, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_658.const", + "file_size": 1792 + }, + "eps_26_1_40": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 225094308, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_659.const", + "file_size": 2 + }, + "model.layers.20.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 225094312, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_660.const", + "file_size": 2179072 + }, + "model.layers.20.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 227273384, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_661.const", + "file_size": 136192 + }, + "model.layers.20.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 227409576, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_662.const", + "file_size": 19456 + }, + "model.layers.20.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 227429032, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_663.const", + "file_size": 19456 + }, + "model.layers.20.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 227448488, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_664.const", + "file_size": 2179072 + }, + "model.layers.20.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 229627560, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_665.const", + "file_size": 136192 + }, + "model.layers.20.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 229763752, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_666.const", + "file_size": 19456 + }, + "model.layers.20.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 229783208, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_667.const", + "file_size": 19456 + }, + "model.layers.20.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 229802664, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_668.const", + "file_size": 4358144 + }, + "model.layers.20.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 234160808, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_669.const", + "file_size": 3584 + }, + "model.layers.20.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 234164392, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_670.const", + "file_size": 136192 + }, + "model.layers.20.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 234300584, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_671.const", + "file_size": 34048 + }, + "model.layers.21.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 234334632, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_672.const", + "file_size": 1792 + }, + "eps_26_1_41": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 234336424, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_673.const", + "file_size": 2 + }, + "model.layers.21.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 234336428, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_674.const", + "file_size": 114688 + }, + "model.layers.21.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 234451116, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_675.const", + "file_size": 512 + }, + "model.layers.21.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 234451628, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_676.const", + "file_size": 3584 + }, + "model.layers.21.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 234455212, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_677.const", + "file_size": 896 + }, + "model.layers.21.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 234456108, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_678.const", + "file_size": 802816 + }, + "model.layers.21.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 235258924, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_679.const", + "file_size": 3584 + }, + "model.layers.21.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 235262508, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_680.const", + "file_size": 25088 + }, + "model.layers.21.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 235287596, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_681.const", + "file_size": 6272 + }, + "model.layers.21.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 235293868, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_682.const", + "file_size": 114688 + }, + "model.layers.21.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 235408556, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_683.const", + "file_size": 512 + }, + "model.layers.21.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 235409068, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_684.const", + "file_size": 3584 + }, + "model.layers.21.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 235412652, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_685.const", + "file_size": 896 + }, + "model.layers.21.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 235413548, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_686.const", + "file_size": 802816 + }, + "model.layers.21.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 236216364, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_687.const", + "file_size": 3584 + }, + "model.layers.21.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 236219948, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_688.const", + "file_size": 25088 + }, + "model.layers.21.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 236245036, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_689.const", + "file_size": 6272 + }, + "model.layers.21.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 236251308, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_690.const", + "file_size": 1792 + }, + "eps_26_1_42": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 236253100, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_691.const", + "file_size": 2 + }, + "model.layers.21.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 236253104, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_692.const", + "file_size": 2179072 + }, + "model.layers.21.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 238432176, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_693.const", + "file_size": 136192 + }, + "model.layers.21.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 238568368, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_694.const", + "file_size": 19456 + }, + "model.layers.21.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 238587824, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_695.const", + "file_size": 19456 + }, + "model.layers.21.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 238607280, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_696.const", + "file_size": 2179072 + }, + "model.layers.21.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 240786352, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_697.const", + "file_size": 136192 + }, + "model.layers.21.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 240922544, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_698.const", + "file_size": 19456 + }, + "model.layers.21.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 240942000, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_699.const", + "file_size": 19456 + }, + "model.layers.21.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 240961456, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_700.const", + "file_size": 4358144 + }, + "model.layers.21.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 245319600, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_701.const", + "file_size": 3584 + }, + "model.layers.21.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 245323184, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_702.const", + "file_size": 136192 + }, + "model.layers.21.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 245459376, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_703.const", + "file_size": 34048 + }, + "model.layers.22.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 245493424, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_704.const", + "file_size": 1792 + }, + "eps_26_1_43": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 245495216, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_705.const", + "file_size": 2 + }, + "model.layers.22.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 245495220, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_706.const", + "file_size": 114688 + }, + "model.layers.22.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 245609908, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_707.const", + "file_size": 512 + }, + "model.layers.22.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 245610420, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_708.const", + "file_size": 3584 + }, + "model.layers.22.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 245614004, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_709.const", + "file_size": 896 + }, + "model.layers.22.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 245614900, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_710.const", + "file_size": 802816 + }, + "model.layers.22.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 246417716, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_711.const", + "file_size": 3584 + }, + "model.layers.22.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 246421300, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_712.const", + "file_size": 25088 + }, + "model.layers.22.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 246446388, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_713.const", + "file_size": 6272 + }, + "model.layers.22.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 246452660, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_714.const", + "file_size": 114688 + }, + "model.layers.22.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 246567348, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_715.const", + "file_size": 512 + }, + "model.layers.22.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 246567860, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_716.const", + "file_size": 3584 + }, + "model.layers.22.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 246571444, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_717.const", + "file_size": 896 + }, + "model.layers.22.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 246572340, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_718.const", + "file_size": 802816 + }, + "model.layers.22.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 247375156, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_719.const", + "file_size": 3584 + }, + "model.layers.22.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 247378740, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_720.const", + "file_size": 25088 + }, + "model.layers.22.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 247403828, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_721.const", + "file_size": 6272 + }, + "model.layers.22.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 247410100, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_722.const", + "file_size": 1792 + }, + "eps_26_1_44": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 247411892, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_723.const", + "file_size": 2 + }, + "model.layers.22.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 247411896, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_724.const", + "file_size": 2179072 + }, + "model.layers.22.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 249590968, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_725.const", + "file_size": 136192 + }, + "model.layers.22.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 249727160, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_726.const", + "file_size": 19456 + }, + "model.layers.22.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 249746616, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_727.const", + "file_size": 19456 + }, + "model.layers.22.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 249766072, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_728.const", + "file_size": 2179072 + }, + "model.layers.22.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 251945144, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_729.const", + "file_size": 136192 + }, + "model.layers.22.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 252081336, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_730.const", + "file_size": 19456 + }, + "model.layers.22.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 252100792, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_731.const", + "file_size": 19456 + }, + "model.layers.22.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 252120248, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_732.const", + "file_size": 4358144 + }, + "model.layers.22.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 256478392, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_733.const", + "file_size": 3584 + }, + "model.layers.22.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 256481976, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_734.const", + "file_size": 136192 + }, + "model.layers.22.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 256618168, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_735.const", + "file_size": 34048 + }, + "model.layers.23.input_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 256652216, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_736.const", + "file_size": 1792 + }, + "eps_26_1_45": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 256654008, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_737.const", + "file_size": 2 + }, + "model.layers.23.attn.k_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 256654012, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_738.const", + "file_size": 114688 + }, + "model.layers.23.attn.k_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 256768700, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_739.const", + "file_size": 512 + }, + "model.layers.23.attn.k_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 256769212, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_740.const", + "file_size": 3584 + }, + "model.layers.23.attn.k_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 256772796, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_741.const", + "file_size": 896 + }, + "model.layers.23.attn.q_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 256773692, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_742.const", + "file_size": 802816 + }, + "model.layers.23.attn.q_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 257576508, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_743.const", + "file_size": 3584 + }, + "model.layers.23.attn.q_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 257580092, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_744.const", + "file_size": 25088 + }, + "model.layers.23.attn.q_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 257605180, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_745.const", + "file_size": 6272 + }, + "model.layers.23.attn.v_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 128 + ], + "size_in_bytes": 114688, + "op_tensor_size": 114688, + "offset": 257611452, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_746.const", + "file_size": 114688 + }, + "model.layers.23.attn.v_proj.Add.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 128 + ], + "size_in_bytes": 512, + "op_tensor_size": 512, + "offset": 257726140, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_747.const", + "file_size": 512 + }, + "model.layers.23.attn.v_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 257726652, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_748.const", + "file_size": 3584 + }, + "model.layers.23.attn.v_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896 + ], + "size_in_bytes": 896, + "op_tensor_size": 896, + "offset": 257730236, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_749.const", + "file_size": 896 + }, + "model.layers.23.attn.o_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 896 + ], + "size_in_bytes": 802816, + "op_tensor_size": 802816, + "offset": 257731132, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_750.const", + "file_size": 802816 + }, + "model.layers.23.attn.o_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 258533948, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_751.const", + "file_size": 3584 + }, + "model.layers.23.attn.o_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 6272 + ], + "size_in_bytes": 25088, + "op_tensor_size": 25088, + "offset": 258537532, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_752.const", + "file_size": 25088 + }, + "model.layers.23.attn.o_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 6272 + ], + "size_in_bytes": 6272, + "op_tensor_size": 6272, + "offset": 258562620, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_753.const", + "file_size": 6272 + }, + "model.layers.23.post_attention_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 258568892, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_754.const", + "file_size": 1792 + }, + "eps_26_1_46": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 258570684, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_755.const", + "file_size": 2 + }, + "model.layers.23.mlp.gate_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 258570688, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_756.const", + "file_size": 2179072 + }, + "model.layers.23.mlp.gate_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 260749760, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_757.const", + "file_size": 136192 + }, + "model.layers.23.mlp.gate_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 260885952, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_758.const", + "file_size": 19456 + }, + "model.layers.23.mlp.gate_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 260905408, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_759.const", + "file_size": 19456 + }, + "model.layers.23.mlp.up_proj.MatMulNBits.qweight": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 4864, + 7, + 64 + ], + "size_in_bytes": 2179072, + "op_tensor_size": 2179072, + "offset": 260924864, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_760.const", + "file_size": 2179072 + }, + "model.layers.23.mlp.up_proj.MatMulNBits.scales.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 263103936, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_761.const", + "file_size": 136192 + }, + "model.layers.23.mlp.up_proj.MatMulNBits.qzeros": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "uint8", + "shape": [ + 19456 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 263240128, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_762.const", + "file_size": 19456 + }, + "model.layers.23.mlp.up_proj.MatMulNBits.bias.f": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 4864 + ], + "size_in_bytes": 19456, + "op_tensor_size": 19456, + "offset": 263259584, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_763.const", + "file_size": 19456 + }, + "model.layers.23.mlp.down_proj.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 4864, + 896 + ], + "size_in_bytes": 4358144, + "op_tensor_size": 4358144, + "offset": 263279040, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_764.const", + "file_size": 4358144 + }, + "model.layers.23.mlp.down_proj.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 896 + ], + "size_in_bytes": 3584, + "op_tensor_size": 3584, + "offset": 267637184, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_765.const", + "file_size": 3584 + }, + "model.layers.23.mlp.down_proj.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 34048 + ], + "size_in_bytes": 136192, + "op_tensor_size": 136192, + "offset": 267640768, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_766.const", + "file_size": 136192 + }, + "model.layers.23.mlp.down_proj.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 34048 + ], + "size_in_bytes": 34048, + "op_tensor_size": 34048, + "offset": 267776960, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_767.const", + "file_size": 34048 + }, + "model.layers.24.final_norm_layernorm.weight.bf": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 896 + ], + "size_in_bytes": 1792, + "op_tensor_size": 1792, + "offset": 267811008, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_768.const", + "file_size": 1792 + }, + "eps_26_1_47": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "bfloat16", + "shape": [ + 1 + ], + "size_in_bytes": 2, + "op_tensor_size": 2, + "offset": 267812800, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_769.const", + "file_size": 2 + }, + "lm_head.MatMulNBits.qweight.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 896, + 151936 + ], + "size_in_bytes": 136134656, + "op_tensor_size": 136134656, + "offset": 267812804, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_770.const", + "file_size": 136134656 + }, + "lm_head.MatMulNBits.bias.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 151936 + ], + "size_in_bytes": 607744, + "op_tensor_size": 607744, + "offset": 403947460, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_771.const", + "file_size": 607744 + }, + "lm_head.MatMulNBits.scales.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "float", + "shape": [ + 1063552 + ], + "size_in_bytes": 4254208, + "op_tensor_size": 4254208, + "offset": 404555204, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_772.const", + "file_size": 4254208 + }, + "lm_head.MatMulNBits.qzeros.preformat": { + "packed_buffer_label": "const", + "xrt_arg_id": 3, + "dtype": "int8", + "shape": [ + 1063552 + ], + "size_in_bytes": 1063552, + "op_tensor_size": 1063552, + "offset": 408809412, + "file_name": "cache/Token_modellayers.0input_layernormMLADFRMSNORM_773.const", + "file_size": 1063552 + }, + "past_key_values.0.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 0 + }, + "past_key_values.0.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 1048576 + }, + "present.0.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 0 + }, + "present.0.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 1048576 + }, + "past_key_values.1.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 2097152 + }, + "past_key_values.1.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 3145728 + }, + "present.1.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 2097152 + }, + "present.1.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 3145728 + }, + "past_key_values.2.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 4194304 + }, + "past_key_values.2.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 5242880 + }, + "present.2.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 4194304 + }, + "present.2.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 5242880 + }, + "past_key_values.3.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 6291456 + }, + "past_key_values.3.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 7340032 + }, + "present.3.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 6291456 + }, + "present.3.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 7340032 + }, + "past_key_values.4.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 8388608 + }, + "past_key_values.4.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 9437184 + }, + "present.4.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 8388608 + }, + "present.4.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 9437184 + }, + "past_key_values.5.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 10485760 + }, + "past_key_values.5.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 11534336 + }, + "present.5.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 10485760 + }, + "present.5.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 11534336 + }, + "past_key_values.6.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 12582912 + }, + "past_key_values.6.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 13631488 + }, + "present.6.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 12582912 + }, + "present.6.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 13631488 + }, + "past_key_values.7.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 14680064 + }, + "past_key_values.7.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 15728640 + }, + "present.7.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 14680064 + }, + "present.7.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 15728640 + }, + "past_key_values.8.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 16777216 + }, + "past_key_values.8.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 17825792 + }, + "present.8.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 16777216 + }, + "present.8.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 17825792 + }, + "past_key_values.9.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 18874368 + }, + "past_key_values.9.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 19922944 + }, + "present.9.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 18874368 + }, + "present.9.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 19922944 + }, + "past_key_values.10.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 20971520 + }, + "past_key_values.10.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 22020096 + }, + "present.10.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 20971520 + }, + "present.10.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 22020096 + }, + "past_key_values.11.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 23068672 + }, + "past_key_values.11.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 24117248 + }, + "present.11.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 23068672 + }, + "present.11.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 24117248 + }, + "past_key_values.12.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 25165824 + }, + "past_key_values.12.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 26214400 + }, + "present.12.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 25165824 + }, + "present.12.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 26214400 + }, + "past_key_values.13.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 27262976 + }, + "past_key_values.13.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 28311552 + }, + "present.13.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 27262976 + }, + "present.13.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 28311552 + }, + "past_key_values.14.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 29360128 + }, + "past_key_values.14.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 30408704 + }, + "present.14.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 29360128 + }, + "present.14.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 30408704 + }, + "past_key_values.15.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 31457280 + }, + "past_key_values.15.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 32505856 + }, + "present.15.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 31457280 + }, + "present.15.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 32505856 + }, + "past_key_values.16.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 33554432 + }, + "past_key_values.16.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 34603008 + }, + "present.16.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 33554432 + }, + "present.16.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 34603008 + }, + "past_key_values.17.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 35651584 + }, + "past_key_values.17.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 36700160 + }, + "present.17.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 35651584 + }, + "present.17.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 36700160 + }, + "past_key_values.18.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 37748736 + }, + "past_key_values.18.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 38797312 + }, + "present.18.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 37748736 + }, + "present.18.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 38797312 + }, + "past_key_values.19.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 39845888 + }, + "past_key_values.19.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 40894464 + }, + "present.19.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 39845888 + }, + "present.19.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 40894464 + }, + "past_key_values.20.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 41943040 + }, + "past_key_values.20.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 42991616 + }, + "present.20.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 41943040 + }, + "present.20.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 42991616 + }, + "past_key_values.21.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 44040192 + }, + "past_key_values.21.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 45088768 + }, + "present.21.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 44040192 + }, + "present.21.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 45088768 + }, + "past_key_values.22.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 46137344 + }, + "past_key_values.22.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 47185920 + }, + "present.22.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 46137344 + }, + "present.22.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 47185920 + }, + "past_key_values.23.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 48234496 + }, + "past_key_values.23.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 49283072 + }, + "present.23.key": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 48234496 + }, + "present.23.value": { + "packed_buffer_label": "ext_buf_0", + "xrt_arg_id": 5, + "dtype": "bfloat16", + "shape": [ + 1, + 2, + 4096, + 64 + ], + "size_in_bytes": 1048576, + "op_tensor_size": 1048576, + "offset": 49283072 + } + }, + "dynamic_shape_subgraph": true, + "dynamic_shape_list": [ + { + "attention_mask_padded": 256 + }, + { + "attention_mask_padded": 512 + }, + { + "attention_mask_padded": 1024 + }, + { + "attention_mask_padded": 2048 + }, + { + "attention_mask_padded": 3072 + }, + { + "attention_mask_padded": 4096 + } + ], + "aux_info": { + "is_llm": true, + "states": [ + [ + 0, + 1, + 1 + ] + ] + } +} \ No newline at end of file