profiles: - id: agents_a1_iq4_xs_mtp_graft_headq6_c1_speed status: recommended_single_user_tps artifact: artifacts/quant/agents-a1-IQ4_XS-MTP-graft-headQ6.gguf quant: IQ4_XS_body_Q6_MTP_head tensor_parallel_size: 1 command_env: REASONING: "off" CTX_SIZE: "8192" PARALLEL: "1" BATCH_SIZE: "4096" UBATCH_SIZE: "512" SPEC_TYPE: draft-mtp SPEC_DRAFT_N_MAX: "2" SPEC_DRAFT_N_MIN: "0" SPEC_DRAFT_BACKEND_SAMPLING: "1" LLAMA_SPEC_MAX_DRAFTING_SLOTS: "1" LLAMA_MTP_FAST_BACKEND_SAMPLE: "1" LLAMA_MTP_DRAFT_TOP_K: "1" LLAMA_MTP_DRAFT_TOP_P: "1" LLAMA_MTP_DRAFT_TEMP: "1" measured: aggregate_output_tok_s: 275.03 target_only_tok_s: 224.59 speedup_vs_target_only: 1.2246 draft_acceptance: 0.7651 mean_accepted_length: 2.52 acceptance_by_position: [0.830, 0.692] benchmark_shape: c1_12x128_streaming_chat evidence: summary_json: runs/agents-a1-mtp-q4-profile-summary.json summary_markdown: runs/agents-a1-mtp-q4-profile-summary.md benchmark_jsonl: runs/bench-openai-agents-a1-IQ4_XS-MTP-graft-headQ6-c1.jsonl server_log: runs/server-agents-a1-IQ4_XS-MTP-graft-headQ6.log - id: agents_a1_iq4_xs_mtp_graft_headq6_c1_acceptance status: high_acceptance_fallback artifact: artifacts/quant/agents-a1-IQ4_XS-MTP-graft-headQ6.gguf quant: IQ4_XS_body_Q6_MTP_head tensor_parallel_size: 1 command_env: REASONING: "off" CTX_SIZE: "8192" PARALLEL: "1" BATCH_SIZE: "4096" UBATCH_SIZE: "512" SPEC_TYPE: draft-mtp SPEC_DRAFT_N_MAX: "1" SPEC_DRAFT_N_MIN: "0" SPEC_DRAFT_BACKEND_SAMPLING: "1" LLAMA_SPEC_MAX_DRAFTING_SLOTS: "1" LLAMA_MTP_FAST_BACKEND_SAMPLE: "1" LLAMA_MTP_DRAFT_TOP_K: "1" LLAMA_MTP_DRAFT_TOP_P: "1" LLAMA_MTP_DRAFT_TEMP: "1" measured: aggregate_output_tok_s: 259.58 target_only_tok_s: 224.59 speedup_vs_target_only: 1.1558 draft_acceptance: 0.8647 mean_accepted_length: 1.86 acceptance_by_position: [0.865] benchmark_shape: c1_12x128_streaming_chat evidence: summary_json: runs/agents-a1-mtp-q4-profile-summary.json summary_markdown: runs/agents-a1-mtp-q4-profile-summary.md benchmark_jsonl: runs/bench-openai-agents-a1-IQ4_XS-MTP-graft-headQ6-nmax1-c1.jsonl server_log: runs/server-agents-a1-IQ4_XS-MTP-graft-headQ6-nmax1.log - id: agents_a1_q4_k_m_mtp_graft_headq6_c1_speed status: recommended_single_user_tps artifact: artifacts/quant/agents-a1-Q4_K_M-MTP-graft-headQ6.gguf quant: Q4_K_M_body_Q6_MTP_head tensor_parallel_size: 1 command_env: REASONING: "off" CTX_SIZE: "8192" PARALLEL: "1" BATCH_SIZE: "4096" UBATCH_SIZE: "512" SPEC_TYPE: draft-mtp SPEC_DRAFT_N_MAX: "2" SPEC_DRAFT_N_MIN: "0" SPEC_DRAFT_BACKEND_SAMPLING: "1" LLAMA_SPEC_MAX_DRAFTING_SLOTS: "1" LLAMA_MTP_FAST_BACKEND_SAMPLE: "1" LLAMA_MTP_DRAFT_TOP_K: "1" LLAMA_MTP_DRAFT_TOP_P: "1" LLAMA_MTP_DRAFT_TEMP: "1" measured: aggregate_output_tok_s: 273.80 target_only_tok_s: 230.48 speedup_vs_target_only: 1.1880 draft_acceptance: 0.7718 mean_accepted_length: 2.53 acceptance_by_position: [0.847, 0.687] benchmark_shape: c1_12x128_streaming_chat evidence: summary_json: runs/agents-a1-mtp-q4-profile-summary.json summary_markdown: runs/agents-a1-mtp-q4-profile-summary.md benchmark_jsonl: runs/bench-openai-agents-a1-Q4_K_M-MTP-graft-headQ6-c1.jsonl server_log: runs/server-agents-a1-Q4_K_M-MTP-graft-headQ6.log - id: agents_a1_q4_k_m_mtp_graft_headq6_c1_acceptance status: high_acceptance_fallback artifact: artifacts/quant/agents-a1-Q4_K_M-MTP-graft-headQ6.gguf quant: Q4_K_M_body_Q6_MTP_head tensor_parallel_size: 1 command_env: REASONING: "off" CTX_SIZE: "8192" PARALLEL: "1" BATCH_SIZE: "4096" UBATCH_SIZE: "512" SPEC_TYPE: draft-mtp SPEC_DRAFT_N_MAX: "1" SPEC_DRAFT_N_MIN: "0" SPEC_DRAFT_BACKEND_SAMPLING: "1" LLAMA_SPEC_MAX_DRAFTING_SLOTS: "1" LLAMA_MTP_FAST_BACKEND_SAMPLE: "1" LLAMA_MTP_DRAFT_TOP_K: "1" LLAMA_MTP_DRAFT_TOP_P: "1" LLAMA_MTP_DRAFT_TEMP: "1" measured: aggregate_output_tok_s: 264.88 target_only_tok_s: 230.48 speedup_vs_target_only: 1.1493 draft_acceptance: 0.9146 mean_accepted_length: 1.91 acceptance_by_position: [0.915] benchmark_shape: c1_12x128_streaming_chat evidence: summary_json: runs/agents-a1-mtp-q4-profile-summary.json summary_markdown: runs/agents-a1-mtp-q4-profile-summary.md benchmark_jsonl: runs/bench-openai-agents-a1-Q4_K_M-MTP-graft-headQ6-nmax1-c1.jsonl server_log: runs/server-agents-a1-Q4_K_M-MTP-graft-headQ6-nmax1.log - id: agents_a1_q5_k_m_mtp_graft_headq6_c1_speed status: available_unbenchmarked artifact: artifacts/quant/agents-a1-Q5_K_M-MTP-graft-headQ6.gguf quant: Q5_K_M_body_Q6_MTP_head tensor_parallel_size: 1 command_env: REASONING: "off" CTX_SIZE: "8192" PARALLEL: "1" BATCH_SIZE: "4096" UBATCH_SIZE: "512" SPEC_TYPE: draft-mtp SPEC_DRAFT_N_MAX: "2" SPEC_DRAFT_N_MIN: "0" SPEC_DRAFT_BACKEND_SAMPLING: "1" LLAMA_SPEC_MAX_DRAFTING_SLOTS: "1" LLAMA_MTP_FAST_BACKEND_SAMPLE: "1" LLAMA_MTP_DRAFT_TOP_K: "1" LLAMA_MTP_DRAFT_TOP_P: "1" LLAMA_MTP_DRAFT_TEMP: "1" measured: aggregate_output_tok_s: target_only_tok_s: speedup_vs_target_only: draft_acceptance: mean_accepted_length: acceptance_by_position: [] benchmark_shape: c1_12x128_streaming_chat evidence: build_report: reports/agents-a1-q5-mtp-build-report.json donor_artifact: artifacts/quant/agents-a1-Q4_K_M-MTP-graft-headQ6.gguf - id: agents_a1_q5_k_m_mtp_graft_headq6_c1_acceptance status: available_unbenchmarked artifact: artifacts/quant/agents-a1-Q5_K_M-MTP-graft-headQ6.gguf quant: Q5_K_M_body_Q6_MTP_head tensor_parallel_size: 1 command_env: REASONING: "off" CTX_SIZE: "8192" PARALLEL: "1" BATCH_SIZE: "4096" UBATCH_SIZE: "512" SPEC_TYPE: draft-mtp SPEC_DRAFT_N_MAX: "1" SPEC_DRAFT_N_MIN: "0" SPEC_DRAFT_BACKEND_SAMPLING: "1" LLAMA_SPEC_MAX_DRAFTING_SLOTS: "1" LLAMA_MTP_FAST_BACKEND_SAMPLE: "1" LLAMA_MTP_DRAFT_TOP_K: "1" LLAMA_MTP_DRAFT_TOP_P: "1" LLAMA_MTP_DRAFT_TEMP: "1" measured: aggregate_output_tok_s: target_only_tok_s: speedup_vs_target_only: draft_acceptance: mean_accepted_length: acceptance_by_position: [] benchmark_shape: c1_12x128_streaming_chat evidence: build_report: reports/agents-a1-q5-mtp-build-report.json donor_artifact: artifacts/quant/agents-a1-Q4_K_M-MTP-graft-headQ6.gguf