version: 1 description: Centralized perf/accuracy targets for models CI. # Default tolerance used by validation is 0.15 (+/-15% around expected) # Entry status semantics: # - active: target is enforced by validation and used for perf/accuracy checks. # - TODO: explicit gap placeholder; excluded from normal target resolution and treated as missing/TODO by validator. # `prefill_time_to_first_token` values are in milliseconds (ms). targets: # Consolidation: additive central-accuracy targets for models whose TTTv2 demo has no committed CI e2e # leg (qwen2_7b / phi4 / deepseek); consumed by the demo's CI=true central accuracy gate. Values are # measured floors and are refreshed per-model during the re-measure sweep. deepseek-r1-distill-qwen-14b: aliases: ["DeepSeek-R1-Distill-Qwen-14B", "deepseek-ai/DeepSeek-R1-Distill-Qwen-14B"] skus: wh_n300: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 87 top5: 99 wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 87 top5: 98 phi-4: aliases: ["Phi-4", "microsoft/phi-4"] skus: wh_n300: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 94 top5: 99 qwen2-7b: aliases: ["Qwen2-7B", "Qwen/Qwen2-7B-Instruct"] skus: wh_n300: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 88 top5: 97 distil-whisper-large-v3: aliases: ["distil-whisper/distil-large-v3"] skus: wh_n300: entries: - batch_size: null seq_len: null status: active perf: prefill_time_to_first_token: 130 decode_t/s/u: 124.0 decode_t/s: 248.0 prefill_time_to_first_token_tolerance: 0.2 decode_t_s_tolerance: 0.2 decode_t_s_u_tolerance: 0.2 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models wh_llmbox_perf: entries: - batch_size: null seq_len: null status: active perf: prefill_time_to_first_token: 140 decode_t/s/u: 105.0 decode_t/s: 840.0 prefill_time_to_first_token_tolerance: 0.2 decode_t_s_tolerance: 0.2 decode_t_s_u_tolerance: 0.2 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models wh_galaxy_perf: entries: - batch_size: null seq_len: null status: active perf: prefill_time_to_first_token: 220 decode_t/s/u: 77.5 decode_t/s: 2480.0 prefill_time_to_first_token_tolerance: 0.2 decode_t_s_tolerance: 0.2 decode_t_s_u_tolerance: 0.2 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models bh_p100: entries: - batch_size: null seq_len: null status: active perf: prefill_time_to_first_token: 60 decode_t/s/u: 310.0 decode_t/s: 310.0 prefill_time_to_first_token_tolerance: 0.2 decode_t_s_tolerance: 0.2 decode_t_s_u_tolerance: 0.2 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models bh_p150: entries: - batch_size: null seq_len: null status: active perf: prefill_time_to_first_token: 50 decode_t/s/u: 530.0 decode_t/s: 530.0 prefill_time_to_first_token_tolerance: 0.2 decode_t_s_tolerance: 0.2 decode_t_s_u_tolerance: 0.2 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models falcon-40b: aliases: ["falcon-40b"] skus: wh_llmbox_perf: entries: - batch_size: 32 seq_len: 128 status: active perf: decode_t/s/u: 36.0 decode_t/s: 1152.0 accuracy: {} owner_id: U053W15B6JF # Djordje Ivanovic team: models falcon-7b: aliases: ["falcon-7b"] skus: wh_n150: entries: - batch_size: 32 seq_len: 128 status: active perf: prefill_t/s: 1990.0 decode_t/s/u: 17.16 decode_t/s: 549.0 accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models - batch_size: 32 seq_len: 1024 status: active perf: prefill_t/s: 2855.0 decode_t/s/u: 15.24 decode_t/s: 487.0 accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models - batch_size: 32 seq_len: 2048 status: active perf: prefill_t/s: 2450.0 decode_t/s/u: 13.91 decode_t/s: 445.0 accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models wh_llmbox_perf: entries: - batch_size: 256 seq_len: 128 status: active perf: prefill_t/s: 15500.0 decode_t/s/u: 14.6 decode_t/s: 3737.0 accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models - batch_size: 256 seq_len: 1024 status: active perf: prefill_t/s: 22100.0 decode_t/s/u: 13.2 decode_t/s: 3379.0 accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models - batch_size: 256 seq_len: 2048 status: active perf: prefill_t/s: 19150.0 decode_t/s/u: 12.4 decode_t/s: 3174.0 accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models wh_galaxy_perf: entries: - batch_size: 1024 seq_len: 128 status: active perf: prefill_t/s: 24100.0 decode_t/s/u: 6.9 decode_t/s: 7065.6 accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models - batch_size: 1024 seq_len: 1024 status: active perf: prefill_t/s: 21200.0 decode_t/s/u: 6.3 decode_t/s: 6451.2 accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models - batch_size: 1024 seq_len: 2048 status: active perf: prefill_t/s: 20400.0 decode_t/s/u: 6.6 decode_t/s: 6758.4 accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models gemma-2-2b: aliases: ["gemma-2-2b", "google/gemma-2-2b-it"] skus: # Seeded from measured ci-token-matching on N300 (90.0 / 100.0). # N150 uses the same model path; CI will re-calibrate if it drifts. wh_n150: entries: - batch_size: 1 seq_len: null status: active # Tier 3: exempt from perf targets; accuracy still gated. perf: {} accuracy: top1: 90 top5: 100 owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models wh_n300: entries: - batch_size: 1 seq_len: null status: active perf: {} accuracy: top1: 90 top5: 100 owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models gemma-2-9b: aliases: ["gemma-2-9b", "google/gemma-2-9b-it"] skus: # Seeded from measured ci-token-matching on N300 (91.8 / 99.8). wh_n300: entries: - batch_size: 1 seq_len: null status: active # Tier 3: exempt from perf targets; accuracy still gated. perf: {} accuracy: top1: 92 top5: 100 owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models gemma-3-27b: aliases: ["gemma-3-27b", "google/gemma-3-27b-it", "gemma-3-27b-vision"] skus: wh_llmbox_perf: entries: - batch_size: 1 seq_len: 2089 status: active perf: # prefill_t/s deliberately omitted: observed values vary 49 → 76 # across runs, well beyond the validator's 15% upper-tolerance # band. The metric needs to be anchored to a fixed seq_len # (see PR #43982 description) before it can gate CI usefully. decode_t/s/u: 14.2 decode_t/s: 14.3 decode_t_s_tolerance: 0.25 decode_t_s_u_tolerance: 0.25 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models gemma-3-4b: aliases: ["gemma-3-4b", "google/gemma-3-4b-it", "gemma-3-4b-vision"] skus: wh_n150: entries: - batch_size: 1 seq_len: 2089 status: active perf: prefill_t/s: 84.0 # decode targets lowered from 27 → 24 after observed regression # to 24.95 t/s in scheduled run 25778148259 (2026-05-13). decode_t/s/u: 24.0 decode_t/s: 24.0 prefill_t_s_tolerance: 0.3 decode_t_s_tolerance: 0.2 decode_t_s_u_tolerance: 0.2 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models gemma-4-26b-a4b: aliases: ["gemma-4-26b-a4b", "google/gemma-4-26B-A4B-it"] skus: wh_llmbox_perf: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} p300x2: entries: - batch_size: 1 seq_len: null status: TODO perf: {} accuracy: {} owner_id: U08TJ70UFRT # Harry Andrews team: models note: "TODO gap tracked from active models_e2e_tests.yaml combo" gemma-4-31b: aliases: ["gemma-4-31b", "google/gemma-4-31B-it"] skus: wh_llmbox_perf: entries: - batch_size: 1 seq_len: null status: TODO perf: {} accuracy: {} p300x2: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} gemma-4-e2b: aliases: ["gemma-4-e2b", "google/gemma-4-E2B-it"] skus: wh_n150: entries: - batch_size: 1 seq_len: null status: TODO perf: {} accuracy: {} owner_id: U08TJ70UFRT # Harry Andrews team: models note: "TODO gap tracked from active models_e2e_tests.yaml combo" bh_p150: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} gemma-4-e4b: aliases: ["gemma-4-e4b", "google/gemma-4-E4B-it"] skus: wh_n150: entries: - batch_size: 1 seq_len: null status: TODO perf: {} accuracy: {} bh_p300: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} p300x2: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} gpt-oss-120b: aliases: ["gpt-oss-120b", "openai/gpt-oss-120b", "T3K_gpt-oss-120b", "GLX_gpt-oss-120b"] skus: p300x2: entries: # Seeded from scheduled run 26930964774 (ci eval, batch-1). - batch_size: 1 seq_len: 128 status: active perf: prefill_time_to_first_token: 893 decode_t/s/u: 24.46 decode_t/s: 24.46 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models bh_galaxy_perf: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} note: "TODO gap tracked from active models_e2e_tests.yaml combo" owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models wh_llmbox_perf: entries: - batch_size: 1 seq_len: 128 status: active perf: prefill_time_to_first_token: 1700 decode_t/s/u: 8.0 decode_t/s: 2200.0 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models wh_galaxy_perf: entries: - batch_size: 128 seq_len: 128 status: active perf: prefill_time_to_first_token: 400 decode_t/s/u: 3.0 decode_t/s: 380.0 accuracy: {} owner_id: U08TJ70UFRT # Stuti Raizada team: models bh_galaxy: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} gpt-oss-20b: aliases: ["gpt-oss-20b", "openai/gpt-oss-20b", "T3K_gpt-oss-20b", "GLX_gpt-oss-20b"] skus: bh_p150: entries: # Seeded from scheduled run 26930964774 (ci eval, batch-1). - batch_size: 1 seq_len: 128 status: active perf: prefill_time_to_first_token: 541 decode_t/s/u: 14.73 decode_t/s: 14.73 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models bh_loudbox: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} note: "TODO gap tracked from active models_e2e_tests.yaml combo" owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models p300x2: entries: # Seeded from scheduled run 26930964774 (ci eval, batch-1). - batch_size: 1 seq_len: 128 status: active perf: prefill_time_to_first_token: 171 decode_t/s/u: 35.63 decode_t/s: 35.63 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models wh_llmbox_perf: entries: - batch_size: 1 seq_len: 128 status: active perf: prefill_time_to_first_token: 400 decode_t/s/u: 12.0 decode_t/s: 2200.0 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models wh_galaxy_perf: entries: - batch_size: 128 seq_len: 128 status: active perf: prefill_time_to_first_token: 250 decode_t/s/u: 5.0 decode_t/s: 650.0 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models llama3.1-70b: aliases: ["Llama-3.1-70B", "meta-llama/Llama-3.1-70B-Instruct"] skus: wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 96 top5: 100 wh_galaxy_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 95 top5: 100 llama3.1-8b: aliases: ["Llama-3.1-8B", "meta-llama/Llama-3.1-8B-Instruct"] skus: wh_n150: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 90 top5: 97 - batch_size: 32 seq_len: 712 status: active perf: # Greedy decode now takes the on-device argmax fast-path on single-chip # (allow_force_argmax gated on num_devices == 1). Measured 23.50-23.60 t/s/u # over 3 CI runs, was 8.8; TTFT 113.2-113.5 ms, was 195. decode_t/s/u: 23.5 prefill_time_to_first_token: 113 accuracy: {} wh_n300: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 90 top5: 97 bh_p100: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 90 top5: 98 bh_p150: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 90 top5: 98 - batch_size: 32 seq_len: 712 status: active perf: # Greedy decode uses the on-device argmax fast-path and two DRAM readers per bank # for the decode projections. Measured 42.08-42.09 t/s/u locally and in CI. decode_t/s/u: 42.0 prefill_time_to_first_token: 53 accuracy: {} bh_p300: entries: - batch_size: 64 seq_len: 131072 status: TODO perf: decode_t/s/u: 20.3 prefill_time_to_first_token: 75 accuracy: top1: 90 top5: 98 wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 89.5 top5: 97.5 - batch_size: 32 seq_len: 712 status: active perf: decode_t/s/u: 64.0 prefill_time_to_first_token: 40 accuracy: {} wh_galaxy_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 88 top5: 97 p300x2: entries: # Catch-all (no batch_size/seq_len): serves the data-parallel b4 perf run and the # token-matching (b1) accuracy check. Accuracy is the token-matching (b1) target. # Perf re-seeded 2026-08-07 from the b4 data-parallel run 31146018726 after #48886 # (0a0510e7fc1, "Enable force_argmax fast-path on non-Galaxy") landed: measured # 37.34 t/s/u / 149.35 t/s / 14.44 ms vs the old 20.88 / 83.5 / 19 targets, which # tripped the validator's "much better than expected" upper bound. The 14 scheduled # runs before #48886 sat at 18.9-21.3 t/s/u with no runner correlation, so this is a # genuine step change from that optimization, not machine variance. - status: active perf: prefill_time_to_first_token: 14.4 decode_t/s/u: 37.3 decode_t/s: 149.4 accuracy: top1: 90.0 top5: 97.6 owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models # batch-32 ci-eval-32 DRAM-prefetcher run (single repeat batch) on bh_quietbox_2. # Specific entry so it wins over the catch-all for this config without disturbing # the b4/b1 runs above. Seeded from the prefetcher run once the job could pass (#47820). - batch_size: 32 seq_len: 712 status: active perf: prefill_time_to_first_token: 28 decode_t/s/u: 57 decode_t/s: 1824 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models # Active e2e combinations with no centralized target yet must remain explicit TODOs. llama3.1-8b-dp: aliases: ["llama3.1-8b-dp"] skus: wh_llmbox_perf: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} bh_p300: entries: - batch_size: 1 seq_len: null status: TODO perf: {} accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models llama3.1-8b-dp-galaxy: aliases: ["llama3.1-8b-dp-galaxy"] skus: wh_galaxy_perf: entries: - batch_size: 1 seq_len: null status: TODO perf: {} accuracy: {} owner_id: U08BH66EXAL # Radoica Draskic team: models llama3.2-11b-vision: aliases: ["Llama-3.2-11B", "Llama-3.2-11B-Vision", "meta-llama/Llama-3.2-11B-Vision-Instruct"] skus: wh_n150: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 90 top5: 98 wh_n300: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 90 top5: 98 wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 90 top5: 98 wh_galaxy_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 87 top5: 97 llama3.2-1b: aliases: ["Llama-3.2-1B", "meta-llama/Llama-3.2-1B-Instruct"] skus: wh_n150: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 79 top5: 97 wh_n300: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 79 top5: 97 wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 80 top5: 97 wh_galaxy_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 77 top5: 96 llama3.2-3b: aliases: ["Llama-3.2-3B", "meta-llama/Llama-3.2-3B-Instruct"] skus: wh_n150: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 89 top5: 98 wh_n300: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 89 top5: 98 wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 91 top5: 99 wh_galaxy_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 87 top5: 97 llama3.3-70b: aliases: ["Llama-3.3-70B", "meta-llama/Llama-3.3-70B-Instruct"] skus: wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 96 top5: 100 wh_galaxy_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 95 top5: 100 # ci-eval-1 (batch-1, 6 repeat batches) on models/demos/llama3_70b_galaxy. # As with the b32/s507 entry below, seq_len is the *character* length of the prompt the # demo resolves with, not a token count: 419 is the first prompt in # eval_repeat_prompts_batch1.json, which is the batch the metrics are measured from. # Measured on wh_galaxy_perf: local at 8baa100a684 (TTFT 725.6-730.4ms, 62.8-63.0 t/s/u), # CI runs 31196515869 / 31196536004 (TTFT 671.5-674.0ms, 64.6-65.0 t/s/u). - batch_size: 1 seq_len: 419 status: active perf: prefill_time_to_first_token: 730 decode_t/s/u: 62.9 decode_t/s: 62.9 prefill_time_to_first_token_tolerance: 0.3 accuracy: {} # p300x2 == bh_quietbox_2. Measured on workflow run 26785408151 # (ci-eval-32 invocation). p300x2: entries: - batch_size: 1 seq_len: 512 status: active # b1/s512 is the token-matching (accuracy) config; perf is validated # only on eval runs, so this entry carries accuracy only. perf: {} # Token-matching (b1) accuracy; seeded from run 27016350392. accuracy: top1: 96 top5: 100 owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models llama3.3-70b-galaxy: aliases: ["llama3.3-70b-galaxy"] skus: wh_galaxy_perf: entries: - batch_size: 32 seq_len: 507 status: active perf: prefill_time_to_first_token: 99.0 decode_t/s/u: 71.5 decode_t/s: 2288.0 prefill_time_to_first_token_tolerance: 0.3 decode_t_s_tolerance: 0.5 decode_t_s_u_tolerance: 0.5 accuracy: {} owner_id: U053W15B6JF # Djordje Ivanovic team: models llama90b-vl: aliases: ["Llama-3.2-90B", "llama3.2-90b-vision", "Llama-3.2-90B-Vision", "meta-llama/Llama-3.2-90B-Vision-Instruct"] skus: wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 96 top5: 100 mamba-2.8b: aliases: ["mamba-2.8b-slimpj", "state-spaces/mamba-2.8b-slimpj"] skus: wh_n150: entries: - batch_size: 32 seq_len: 32 status: active perf: prefill_t/s: 135.0 decode_t/s: 346.0 decode_t/s/u: 10.8 prefill_t_s_tolerance: 0.2 decode_t_s_tolerance: 0.2 decode_t_s_u_tolerance: 0.2 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models - batch_size: 32 seq_len: 128 status: active perf: prefill_t/s: 380.0 decode_t/s: 346.0 decode_t/s/u: 10.8 prefill_t_s_tolerance: 0.2 decode_t_s_tolerance: 0.2 decode_t_s_u_tolerance: 0.2 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models mistral-7b: aliases: ["Mistral-7B", "mistralai/Mistral-7B-Instruct-v0.3"] skus: wh_n150: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 95 top5: 99 - batch_size: 32 seq_len: 744 status: active perf: decode_t/s/u: 25.3 prefill_time_to_first_token: 118 prefill_time_to_first_token_tolerance: 0.2 accuracy: {} wh_n300: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 95 top5: 100 wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 95 top5: 100 mistral-small-3.1-24b: aliases: ["Mistral-Small-3.1-24B", "mistralai/Mistral-Small-3.1-24B-Instruct-2503"] skus: # Ported from blackhole_demo_tests.yaml / t3k_unit_tests.yaml — neither the # e2e pipeline_tests nor the vision unit tests emit a benchmark payload yet, # so there are no measured numbers to target. Populate after the first # tiered run reports perf / accuracy. wh_llmbox_perf: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} owner_id: U08GEGWHQJF # Adam Roberge team: models bh_quietbox_2: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} owner_id: U08GEGWHQJF # Adam Roberge team: models mixtral-8x7b: aliases: ["Mixtral-8x7B", "mistralai/Mixtral-8x7B-Instruct-v0.1"] skus: wh_llmbox_perf: entries: - batch_size: 32 seq_len: 1024 status: active perf: prefill_time_to_first_token: 138.7 decode_t/s/u: 16.3 decode_t/s: 516.1 prefill_time_to_first_token_tolerance: 0.4 decode_t_s_tolerance: 0.4 decode_t_s_u_tolerance: 0.4 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 98.0 top5: 100.0 owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models - batch_size: 1 seq_len: 1024 status: active perf: {} accuracy: top1: 98.0 top5: 100.0 owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models phi-3-mini: aliases: ["Phi-3-mini", "Phi-3-mini-128k-instruct", "microsoft/Phi-3-mini-128k-instruct"] skus: wh_n150: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 89 top5: 99 wh_n300: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 89 top5: 99 qwen2.5-32b: aliases: ["Qwen2.5-32B", "Qwen/Qwen2.5-32B-Instruct"] skus: wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 98 top5: 99 - batch_size: 32 seq_len: 707 status: active # Seeded from scheduled run 27016350392 (ci eval-32). perf: decode_t/s/u: 22.66 prefill_time_to_first_token: 101 accuracy: {} p300x2: entries: # Token-matching (b1) accuracy; seeded from run 27016350392. - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 98 top5: 99 - batch_size: 32 seq_len: 707 status: active # Re-seeded from run 27036772693 (ci eval-32); the earlier 7.63/189 # were the token-matching teacher-forcing artifacts, not eval perf. perf: decode_t/s/u: 22.1 prefill_time_to_first_token: 82 accuracy: {} qwen2.5-72b: aliases: ["Qwen2.5-72B", "Qwen/Qwen2.5-72B-Instruct"] skus: wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 99 top5: 100 - batch_size: 32 seq_len: 707 status: active perf: decode_t/s/u: 15.2 prefill_time_to_first_token: 225 accuracy: {} qwen2.5-vl-72b: aliases: ["Qwen2.5-VL-72B", "Qwen/Qwen2.5-VL-72B-Instruct"] skus: wh_llmbox_perf: entries: # Seeded from scheduled run 26930964774 (ci eval). decode_t/s on the # b32 entry is the aggregate batch throughput (32 x t/s/u). - batch_size: 1 seq_len: 128 status: active perf: prefill_time_to_first_token: 419 # b1 decode is run-to-run unstable: measured ~10.9-13.2 t/s across # recent scheduled runs + 3 targeted reruns. The old 13.77 center # sat at the top of the range, so its tol-0.15 floor (11.70) tripped # on the low observations (10.87 on 06-30, 10.96 on a rerun). # Recenter on the observed mean ~12 with a wide 0.2 tolerance # ([9.6, 14.4]) to absorb the swing while still catching gross # regressions (the one-off 07-01 collapse to 6.2 would still fail). decode_t/s/u: 12.0 decode_t_s_u_tolerance: 0.2 decode_t/s: 12.0 decode_t_s_tolerance: 0.2 accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models - batch_size: 32 seq_len: 2048 status: active perf: # b32 decode throughput is strongly host-correlated on the # wh_llmbox_perf T3K pool, not commit-correlated: healthy hosts # (f10cs06/07/08/09) measure 311-406 t/s across the last 10 # scheduled runs, while the degraded host f10cs05 runs ~142-185 # (tracked as infra, INFRA-3). The default 0.15 tolerance made the # gate flip between "too fast" (f10cs09 ~400 > 399.7 upper bound, # 07-03) and "regression" (f10cs05, 07-01) purely by which box the # scheduler picked. Keep center 347.6 and widen tolerance to 0.20 # ([278, 417]) to bracket the healthy-host spread; f10cs05's ~⅔ # readings still fail by design (do not lower the gate to mask a # degraded host). prefill_time_to_first_token: 1163 decode_t/s/u: 10.86 decode_t_s_u_tolerance: 0.2 decode_t/s: 347.6 decode_t_s_tolerance: 0.2 accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models p300x2: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models qwen2.5-7b: aliases: ["Qwen2.5-7B", "Qwen/Qwen2.5-7B-Instruct"] skus: wh_n300: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 84 top5: 96 qwen2.5-coder-32b: aliases: ["Qwen2.5-Coder-32B", "Qwen/Qwen2.5-Coder-32B-Instruct"] skus: wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 96 top5: 99 - batch_size: 32 seq_len: 707 status: active # decode_t/s/u is run-to-run unstable (see issue #46214). Recent # scheduled runs have shifted up (measured 20.6-21.9 across the last # week), so center raised 16.0 -> 20.0 while keeping the wide 0.2 # tolerance ([16, 24]) to absorb the run-to-run swing. perf: decode_t/s/u: 20.0 decode_t_s_u_tolerance: 0.2 prefill_time_to_first_token: 105 accuracy: {} bh_quietbox_2: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 96 top5: 99 - batch_size: 32 seq_len: 1024 status: active perf: decode_t/s: 665.9 decode_t/s/u: 20.81 decode_t_s_u_tolerance: 0.2 prefill_time_to_first_token: 83.3 prefill_time_to_first_token_tolerance: 0.2 accuracy: {} qwen2.5-vl-32b: aliases: ["Qwen2.5-VL-32B", "Qwen/Qwen2.5-VL-32B-Instruct"] skus: wh_llmbox_perf: entries: # Seeded from scheduled run 26868040883 (ci eval). decode_t/s on the # b32 entry is the aggregate batch throughput (32 x t/s/u). - batch_size: 1 seq_len: 128 status: active perf: # TTFT is run-to-run unstable and bimodal (observed ~205ms and ~327ms # clusters across the last 8 scheduled runs, mean ~296ms). The old 351 # center left the tol-0.4 floor at 211ms, which the ~205ms cluster kept # tripping ("much better than expected"). Center on the observed mean so # the [178, 414] band spans both clusters and still gates gross regressions. prefill_time_to_first_token: 296 prefill_time_to_first_token_tolerance: 0.4 # decode_t/s/u is run-to-run unstable (see #48510). The earlier 18.0 center # (set pre-data, post-#48037) is now stale-low: the last 8 scheduled runs # measured ~19-25 t/s (mean ~22), so 22+ readings tripped the old 21.6 # ceiling ("better than expected"). Recenter on the observed mean ~22 with a # wide 0.2 tolerance ([17.6, 26.4]) to absorb the swing while still catching # gross regressions (e.g. a halving to ~11). decode_t/s/u: 22.0 decode_t_s_u_tolerance: 0.2 decode_t/s: 22.0 decode_t_s_tolerance: 0.2 accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models - batch_size: 32 seq_len: 2048 status: active perf: prefill_time_to_first_token: 616 decode_t/s/u: 17.76 decode_t/s: 568.19 accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models p300x2: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} owner_id: U08JFM3CFA4 # Gongyu Wang team: models qwen3-32b: aliases: ["Qwen3-32B", "Qwen/Qwen3-32B"] skus: wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 89 top5: 97 - batch_size: 32 seq_len: 686 status: active # decode_t/s/u corrected 14.4 -> 22.0 (measured in run 27016350392). perf: decode_t/s/u: 22.0 prefill_time_to_first_token: 123 accuracy: {} # p300x2 == bh_quietbox_2. Measured on workflow run 26785408151 # (ci-eval-32 invocation; decode_t/s = batch_size * decode_t/s/u). p300x2: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 91.0 top5: 99.0 owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models - batch_size: 32 seq_len: 686 status: active # Targets raised to match sustained perf on bh_quietbox_2 (p300x2): # 6/6 recent scheduled runs measured decode_t/s ~680-703, t/s/u ~21.8, # TTFT ~86-88ms (old targets 278 / 18.7 / 207 were stale). decode_t/s # = batch_size * decode_t/s/u (32 * 21.6 ~= 691). perf: prefill_time_to_first_token: 87 prefill_time_to_first_token_tolerance: 0.2 decode_t/s/u: 21.6 decode_t/s: 691 accuracy: {} owner_id: U03PUAKE719 # Miguel Tairum Cruz team: models qwen3-32b-galaxy: aliases: ["qwen3-32b-galaxy"] skus: wh_galaxy_perf: entries: - batch_size: 32 seq_len: 507 status: active perf: prefill_time_to_first_token: 700.0 decode_t/s/u: 60.0 decode_t/s: 1920.0 accuracy: {} owner_id: U053W15B6JF # Djordje Ivanovic team: models - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 89 top5: 97 owner_id: U053W15B6JF # Djordje Ivanovic team: models # Blackhole Galaxy runs the same Qwen3-32B model on the no-prefetcher path. # Perf seeded from measured batch-32 (seq_len=507) e2e demo runs on BH Galaxy # (TTFT ~1226 ms, ~33.9 tok/s/user); accuracy floor mirrors the WH Galaxy entry # (measured token accuracy 95.6% top-1 / 99.8% top-5 clears it with margin). bh_galaxy_perf: entries: - batch_size: 32 seq_len: 507 status: active perf: prefill_time_to_first_token: 1225.0 decode_t/s/u: 34.0 decode_t/s: 1088.0 accuracy: {} owner_id: U053W15B6JF # Djordje Ivanovic team: models - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 89 top5: 97 owner_id: U053W15B6JF # Djordje Ivanovic team: models qwen3.6-27b: aliases: ["Qwen3.6-27B", "Qwen/Qwen3.6-27B"] skus: # bh_quietbox_2 normalizes to canonical p300x2 (2x P300 = 4 dies, labeled P150x4). # TTFT (prefill_time_to_first_token) is in ms; decode is batch_size=1 so t/s == t/s/u. bh_quietbox_2: entries: - batch_size: 1 seq_len: 128 status: active perf: prefill_time_to_first_token: 174.0 decode_t/s: 26.0 decode_t/s/u: 26.0 tolerance: 0.2 accuracy: {} owner_id: U08HL8X1ECD # Aniruddha Tupe team: models - batch_size: 1 seq_len: 4096 status: active perf: prefill_time_to_first_token: 2030.0 decode_t/s: 18.13 decode_t/s/u: 18.13 tolerance: 0.2 accuracy: {} owner_id: U08HL8X1ECD # Aniruddha Tupe team: models - batch_size: 1 seq_len: 8192 status: active perf: prefill_time_to_first_token: 4620.0 decode_t/s: 18.01 decode_t/s/u: 18.01 tolerance: 0.2 accuracy: {} owner_id: U08HL8X1ECD # Aniruddha Tupe team: models - batch_size: 1 seq_len: 16384 status: active perf: prefill_time_to_first_token: 9470.0 decode_t/s: 17.87 decode_t/s/u: 17.87 tolerance: 0.2 accuracy: {} owner_id: U08HL8X1ECD # Aniruddha Tupe team: models - batch_size: 1 seq_len: 32768 status: active perf: prefill_time_to_first_token: 19960.0 decode_t/s: 17.57 decode_t/s/u: 17.57 tolerance: 0.2 accuracy: {} owner_id: U08HL8X1ECD # Aniruddha Tupe team: models - batch_size: 1 seq_len: 65536 status: active perf: prefill_time_to_first_token: 44160.0 decode_t/s: 17.1 decode_t/s/u: 17.1 tolerance: 0.2 accuracy: {} owner_id: U08HL8X1ECD # Aniruddha Tupe team: models - batch_size: 1 seq_len: 131072 status: active perf: prefill_time_to_first_token: 77610.0 decode_t/s: 16.65 decode_t/s/u: 16.65 tolerance: 0.2 accuracy: {} owner_id: U08HL8X1ECD # Aniruddha Tupe team: models - batch_size: 1 seq_len: 262144 status: active perf: prefill_time_to_first_token: 286340.0 decode_t/s: 14.67 decode_t/s/u: 14.67 tolerance: 0.2 accuracy: {} owner_id: U08HL8X1ECD # Aniruddha Tupe team: models qwq-32b: aliases: ["QwQ-32B", "Qwen/QwQ-32B"] skus: wh_llmbox_perf: entries: - batch_size: 1 seq_len: 512 status: active perf: {} accuracy: top1: 96 top5: 100 whisper: aliases: ["whisper"] skus: wh_n150: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} bh_p150: entries: - batch_size: null seq_len: null status: TODO perf: {} accuracy: {} owner_id: U0491KT8MNY # Mohamed Bahnas team: models vit: aliases: ["ViT", "vit", "google/vit-base-patch16-224"] skus: # Placeholder coverage for the tier-2 (vit, wh_n150) / (vit, wh_n300) combos # registered in models_e2e_tests.yaml. The demo asserts samples/sec against its # own hardcoded expectation and emits no benchmark payload, and samples/sec is not # in ALLOWED_TARGET_METRIC_NAMES in validate_perf_targets.py, so there is no # measurable metric to declare here yet. Kept as TODO so the combo is on the books. wh_n150: entries: - batch_size: 8 seq_len: null status: TODO perf: {} accuracy: {} owner_id: U088413NP0Q # Ashai Reddy Ginuga team: models wh_n300: entries: - batch_size: 8 seq_len: null status: TODO perf: {} accuracy: {} owner_id: U088413NP0Q # Ashai Reddy Ginuga team: models # ResNet-50 is tier 1, so unlike the tier-3 models migrated alongside it, it is not # exempt from targets. The values below are measured, not placeholders: # wh_n150 batch 16 top1 75.375 fps 28.43 (run 31590127406) # wh_llmbox_perf batch 128 top1 75.456 fps 29.05 (run 31606078371) # # Only the two e2e SKUs are listed -- validate_perf_targets.py's # _collect_active_test_combos walks models_e2e_tests.yaml alone, so unit and device-perf # combos are never gap-checked. # # top1 and fps come from the vision measurement pair registered in # validate_perf_targets.py: ("inference", "top1_accuracy") and ("inference", "fps"), # emitted by run_resnet_imagenet_inference. Accuracy is held tight at 1% because it was # reproducible to three decimals across both SKUs; fps gets 15% because throughput on # shared CI hosts is far noisier than correctness. resnet50: aliases: ["ResNet-50", "resnet50", "ResNet50", "microsoft/resnet-50"] skus: wh_n150: entries: - batch_size: 16 seq_len: null status: active perf: fps: 28.43 fps_tolerance: 0.15 accuracy: top1: 75.375 top1_tolerance: 0.01 owner_id: U085GDCAZTN # Pavle Josipovic team: ttnn wh_llmbox_perf: entries: - batch_size: 128 seq_len: null status: active perf: fps: 29.05 fps_tolerance: 0.15 accuracy: top1: 75.456 top1_tolerance: 0.01 owner_id: U085GDCAZTN # Pavle Josipovic team: ttnn # Tier 3 is exempt from perf targets, but these three e2e tests measure throughput and # previously gated nothing: prep_perf_report writes a CSV and asserts nothing, so an # arbitrarily large regression stayed invisible. Each fps below is the first CI # measurement on wh_n150 (run 32378297838), held to a 10% band, and each reproduced # within 1.3% on run 32388317325. There is no accuracy block: only MobileNetV3 # validates outputs on the e2e path (PCC 0.98, in-test), and none of the three emit an # accuracy metric. # # EfficientDet-D0 is deliberately absent. Its pipeline is use_trace=False with a single # command queue, so throughput is host-dispatch bound and swung 7.49 -> 6.52 fps (13%) # between runs 32378297838 and 32388317325 while these three moved 0.2-1.3%. Tier 3 # carries no perf obligation, so it is left unmeasured rather than gated on runner CPU # contention. # # fps resolves to the ("inference", "fps") measurement pair, emitted by # report_vision_fps in models/demos/utils/common_demo_utils.py. prep_perf_report's own # payload writes throughput_iter_per_s under an "end_to_end_perf" step, which no target # metric maps to -- do not add its model_name strings (efficientdet_d0-notrace-1cq, # ssd512-trace-2cq, ttnn_mobilenetV3_trace_2cqs_batch_size1) as aliases here, or that # payload resolves to this entry and hard-fails on a missing fps measurement. mobilenetv3: aliases: ["mobilenetv3", "MobileNetV3", "mobileNetV3"] skus: wh_n150: entries: - batch_size: 1 seq_len: null status: active perf: fps: 263.29 fps_tolerance: 0.10 owner_id: U07J5E9NA9G # Dalar Vartanians team: models ssd512: aliases: ["ssd512", "SSD512"] skus: wh_n150: entries: - batch_size: 1 seq_len: null status: active perf: fps: 40.83 fps_tolerance: 0.10 owner_id: U07J5E9NA9G # Dalar Vartanians team: models yunet: aliases: ["yunet", "YuNet", "YUNet"] skus: wh_n150: entries: - batch_size: 1 seq_len: null status: active perf: fps: 76.0 fps_tolerance: 0.10 owner_id: U0978501V8B # CODEOWNERS @tvardhineniTT team: models