Compare commits

...

227 Commits

Author SHA1 Message Date
e0811e95b4 Record code-trace canary fidelity gates 2026-07-24 02:42:29 +08:00
e1f2557a0c Allow long-context SSE events in exact replay 2026-07-24 02:10:18 +08:00
14ed991059 Record code-trace calibration and canary gates 2026-07-24 02:03:11 +08:00
d5bb9745f8 Reuse host-local caches for long-context replay 2026-07-24 01:50:17 +08:00
da480a8761 Extend serving rotary cache for code context 2026-07-24 01:14:15 +08:00
766f09d3ed Enable gated long context real trace replay 2026-07-24 00:59:40 +08:00
7aed90f9e6 Support long code traces in Frontier replay 2026-07-24 00:57:08 +08:00
b80d3f03de Freeze code long context profile v6 2026-07-24 00:52:23 +08:00
89d5ebbbc0 Add code trace max model length smoke 2026-07-24 00:49:12 +08:00
8462afa56f Encode zero cacheable blocks explicitly 2026-07-24 00:40:15 +08:00
5c45944388 Restore unrelated documentation paths 2026-07-24 00:36:29 +08:00
5a4011ca12 Exclude partial prompt blocks from prefix cache replay 2026-07-24 00:35:48 +08:00
ab952b47e7 Make synthetic trace block identities parent-sensitive 2026-07-24 00:11:48 +08:00
029c8991b6 Generalize trace remapping for code workloads 2026-07-24 00:09:50 +08:00
a9e88f14de Freeze profile v5 base for code long-context extension 2026-07-24 00:08:56 +08:00
e251046c30 Gate long-context profile override explicitly 2026-07-24 00:02:53 +08:00
288f7b239f Add code long-context attention profiling grid 2026-07-23 23:58:49 +08:00
1d9182f305 Handle zero-token source rows in code trace audit 2026-07-23 23:52:03 +08:00
fbaa909723 Prepare Frontier code trace fidelity campaign 2026-07-23 23:48:28 +08:00
cd7665d882 Ignore generated experiment SVGs 2026-07-23 18:10:04 +08:00
08921193a1 Record structured attention experiment verdict 2026-07-23 18:09:24 +08:00
4f22688bfd Record TP2 prefill serving-path verdict 2026-07-23 18:08:32 +08:00
cf610003ed Close BC8 decode curve counterfactual 2026-07-23 18:08:00 +08:00
ecc5599381 Profile decode batch grid repeats 2026-07-23 17:37:18 +08:00
1126d9be7d Pass TP4 decode stability gate 2026-07-23 17:09:49 +08:00
c1c200b7cd Add decode batch-grid stability experiment 2026-07-23 16:55:55 +08:00
cb67ac8621 Reuse validated FlashInfer cache for TP2 smoke 2026-07-23 16:24:20 +08:00
fd859bc52c Use empty scp sync for TP2 fleet job 2026-07-23 16:12:46 +08:00
be523b1c07 Isolate TP2 smoke from dirty remote checkout 2026-07-23 16:11:13 +08:00
2c3220c2be Add TP2 prefill serving-path smoke experiment 2026-07-23 16:09:52 +08:00
9c1175a434 experiment: pin cu129 real pilot runtime 2026-07-20 19:02:40 +08:00
788270183d experiment: gate per-gpu sweep on control completion 2026-07-20 18:58:43 +08:00
e651ecc923 experiment: reuse predictors across load contracts 2026-07-20 18:51:31 +08:00
809ad9ffef experiment: add per-gpu workload control 2026-07-20 18:48:22 +08:00
a033a72195 analysis: summarize workload simulator regimes 2026-07-20 18:17:03 +08:00
dfe3f345d8 research: record workload sweep launch 2026-07-20 18:15:03 +08:00
157bf3668d fix: isolate simulator predictor cache 2026-07-20 18:11:33 +08:00
f727cbcf76 fix: sweep simulator families independently 2026-07-20 18:04:23 +08:00
f38e260639 fix: parse simulator group arguments 2026-07-20 18:01:38 +08:00
3453fbe522 experiment: add parallel workload simulator sweep 2026-07-20 17:54:20 +08:00
b302954dcb docs: correct Q30 MNS surface 2026-07-20 17:39:59 +08:00
ca999c4e49 fix: derive complete trace blocks from private artifact 2026-07-20 17:38:32 +08:00
75946d9d73 experiment: add workload regime taxonomy 2026-07-20 17:36:02 +08:00
39766141fb Decompose good/bad selection split across frozen surfaces
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-20 12:05:20 +08:00
18c0b25ae7 Conclude Qwen30 admission diagnosis 2026-07-19 20:56:36 +08:00
2970f74d67 Analyze Qwen30 admission amplification 2026-07-19 20:52:48 +08:00
a6c9beaa89 Diagnose Qwen30 Fixed-PD admission state 2026-07-19 20:41:20 +08:00
409d83a876 Record Qwen235 state diagnosis results 2026-07-19 19:18:57 +08:00
e602d41c80 Report exact state composition verdict 2026-07-19 19:16:20 +08:00
27b2daf7ce Distinguish proxy and exact state verdicts 2026-07-19 19:12:51 +08:00
3be6cd04ad Reschedule clean Qwen235 TP4 iteration state 2026-07-19 19:05:05 +08:00
51f2072d30 Use critical path totals for TP8 breakdown 2026-07-19 19:03:49 +08:00
6b80266aa7 Fix Qwen235 state reweighting 2026-07-19 19:01:36 +08:00
9c9479c313 Match Frontier to exact iteration composition 2026-07-19 19:00:18 +08:00
3349b23290 Analyze corrected Frontier op traces 2026-07-19 18:53:12 +08:00
d5c3c93577 Schedule Qwen235 exact iteration state runs 2026-07-19 18:51:15 +08:00
b49c072502 Add exact state observation modes 2026-07-19 18:49:06 +08:00
670eca176d Include mixed steps in Qwen235 state analysis 2026-07-19 18:39:36 +08:00
b6dfbfcad7 Analyze Qwen235 fixed PD simulator state 2026-07-19 18:37:18 +08:00
5927b6bfc3 Add Qwen235 state replay diagnosis 2026-07-19 18:31:47 +08:00
10567da523 Freeze Qwen235 ablation operator profiles 2026-07-19 17:11:26 +08:00
4ca295af0b Match FlashInfer TP8 workspace initialization 2026-07-19 16:54:24 +08:00
80ab724608 Run Qwen235 profiling from clean remote worktree 2026-07-19 16:47:12 +08:00
506e63a633 Add measured collective profile gate for Qwen235 2026-07-19 16:45:35 +08:00
f4a75aa8e4 Update prefix trace parser test contract 2026-07-19 15:35:17 +08:00
9c34650746 Merge simulator fidelity work into feature/sim 2026-07-19 15:34:28 +08:00
d17a2d4ab0 Merge fixed-PD pressure probe into feature/sim 2026-07-19 15:34:23 +08:00
3d3878c5aa Summarize Frontier selection regret 2026-07-19 15:31:16 +08:00
4c8d581a5b Track simulator fidelity experiment artifacts 2026-07-19 15:31:09 +08:00
e0ea7e9961 Support scp GPU fleet sync 2026-07-19 15:30:54 +08:00
fbf0f7c50b Set Qwen235 profile KV cache layout 2026-07-19 11:47:36 +08:00
81f3a5c76d Profile Qwen235 true mixed attention 2026-07-19 11:45:08 +08:00
71dbf80ed4 Match Qwen235 Frontier chunked prefill 2026-07-19 11:35:27 +08:00
5f48f7ec8b Avoid stale exact-trace HTTP connections 2026-07-19 10:58:38 +08:00
42e4ae3422 Avoid occupied Qwen235 trace ports 2026-07-19 08:10:29 +08:00
b72de0fd15 Wait for Qwen235 GPU memory cleanup 2026-07-19 07:07:08 +08:00
b2f97927de Expose nvcc to Qwen235 serving workers 2026-07-19 06:42:28 +08:00
979f179a47 Fix Qwen235 real surface output expansion 2026-07-19 06:29:47 +08:00
dfd9646d2c Freeze Qwen235 KV cache layout for profiling 2026-07-19 06:24:02 +08:00
922d66c5c1 Align Qwen235 attention profile batch limit 2026-07-19 06:21:33 +08:00
ac5061fa5d Align Qwen235 attention profile sequence limit 2026-07-19 06:19:08 +08:00
d563c30b42 Set Qwen3 model type for vLLM profile dispatch 2026-07-19 06:12:11 +08:00
a21382c8aa Raise Q235 profiler token limit 2026-07-19 05:45:37 +08:00
5067bc2cb1 Expose nvcc to Q235 FlashInfer profiling 2026-07-19 05:42:26 +08:00
39e4719b28 Run Qwen235 profilers from Frontier source 2026-07-19 03:00:04 +08:00
79e9870975 Defer Qwen235 campaign until Qwen30 completion 2026-07-19 02:54:32 +08:00
80a067e3a5 Orchestrate Qwen235 four-case fidelity campaign 2026-07-19 02:52:54 +08:00
3e32ea609f Cover Qwen235 MNS128 graph buckets 2026-07-19 02:48:51 +08:00
40bac6dbf4 Add Qwen235 same-stack Frontier profiling campaign 2026-07-19 02:46:58 +08:00
e631f6a269 Add Qwen235 Frontier latency surface runner 2026-07-19 02:44:55 +08:00
a3c8cb5808 Add Qwen235 Frontier MoE portability smoke 2026-07-19 02:42:28 +08:00
f4813cf537 Add Qwen235 vLLM 0.20 real surface runner 2026-07-19 02:39:16 +08:00
c8383c9c4c Launch Qwen30 fixed pressure comparison 2026-07-19 02:25:15 +08:00
7ea9635878 Record fixed PD pressure calibration result 2026-07-19 01:45:17 +08:00
33b73afe9b Narrow fixed PD pressure sweep around load knee 2026-07-19 01:26:09 +08:00
909a80f0a6 Reuse validated FlashInfer cache for pressure probe 2026-07-19 01:13:50 +08:00
9837fa5133 Match Qwen30 probe file descriptor limit 2026-07-19 01:07:57 +08:00
6726318792 Profile fixed PD workload pressure against trace anchor 2026-07-19 00:54:48 +08:00
e98608911e Resume valid Qwen30 latency cells safely 2026-07-18 17:27:12 +08:00
f11daa6776 Fix Qwen30 exact trace served-model routing 2026-07-18 11:56:43 +08:00
b9523cef5c Pin DeepGEMM nvcc for Qwen235 smoke 2026-07-18 11:15:22 +08:00
b7f9cef9c5 Keep Qwen30 JIT cache scoped to experiment 2026-07-18 11:11:10 +08:00
ecb45cf762 Extend Qwen30 JIT readiness for latency surface 2026-07-18 11:09:22 +08:00
686b050517 Isolate Qwen235 DeepGEMM JIT cache 2026-07-18 10:57:48 +08:00
28ffb22bee Create Qwen30 case output before launch logging 2026-07-18 10:50:11 +08:00
a01c1206ab Launch frozen Qwen30 latency case surfaces 2026-07-18 10:48:45 +08:00
5fc0b48fa1 Extend Qwen235 smoke readiness deadline 2026-07-18 10:32:21 +08:00
7a437b4d91 Allow fixed runtime preflight resume 2026-07-18 01:11:37 +08:00
12705411f9 Record Qwen235 portability experiment gate 2026-07-18 01:08:54 +08:00
ddacc5f7a6 Add Qwen235 vLLM 0.20 compatibility gate 2026-07-18 01:08:08 +08:00
a7e3c0fdf3 Record fixed-case vLLM runtime state 2026-07-18 01:05:46 +08:00
44104bd96e Prepare remaining Qwen30 latency cases 2026-07-18 01:03:06 +08:00
65fee8450a Record graph-aligned Frontier result 2026-07-18 00:30:59 +08:00
e6f3e4a690 Audit graph-aligned Frontier surface 2026-07-18 00:09:06 +08:00
e2cd43808d Bound graph predictor grid to runtime 2026-07-17 23:41:01 +08:00
27a2a3468a Use verified FlashInfer profile cache 2026-07-17 23:27:07 +08:00
2deb53cb72 Allow graph linear profile smoke 2026-07-17 23:25:15 +08:00
bdc357dc6c Align Frontier piecewise graph profiles 2026-07-17 23:22:42 +08:00
47355a9411 Record corrected Frontier liveness probe 2026-07-17 22:50:00 +08:00
d3be91dd58 Audit Frontier prefix-cache trace contract 2026-07-17 22:42:57 +08:00
e7b482658e Prepare cache-consistent T1 real smoke 2026-07-17 13:29:18 +08:00
8b054ffcb1 Bound Frontier predictor training parallelism 2026-07-17 13:02:08 +08:00
903cbe682d Record Frontier trace stalls without ranking them 2026-07-17 12:46:57 +08:00
03aa794448 Isolate FlashInfer cache for trace replay 2026-07-17 12:44:50 +08:00
2aa713f5d6 Prepare sealed exact-trace real smoke 2026-07-17 11:55:12 +08:00
e9ae04d852 Add exact production-trace real replay client 2026-07-17 10:40:36 +08:00
ec775cdafd Correct F1 to a steady fixed-QPS workload 2026-07-17 10:36:28 +08:00
10afe4e2e8 Freeze long-context Frontier profile 2026-07-17 10:30:37 +08:00
b6ef6eeae7 Add long-context attention profile closure 2026-07-17 10:26:21 +08:00
202ae718b3 Freeze batch-aware Frontier ablation 2026-07-17 10:22:56 +08:00
a2cf361ffe Generalize Qwen30 fixed-shape real runner 2026-07-17 10:17:21 +08:00
6e8704d525 Add fixed and exact-trace Frontier surfaces 2026-07-17 10:13:25 +08:00
0c747448b6 Normalize frozen profile CSV line endings 2026-07-17 09:58:26 +08:00
1fa203384f Add batch-aware profile and exact trace preparation 2026-07-17 09:55:09 +08:00
95f4af3d99 Add Frontier fidelity envelope campaign 2026-07-17 09:44:49 +08:00
3a59d5df96 Evaluate Qwen30 prefill simulator fidelity 2026-07-17 03:11:45 +08:00
97c2f34700 Add Qwen30 prefill fidelity experiment 2026-07-17 00:32:01 +08:00
76107d3e87 Evaluate vLLM 0.20 profiles against Frontier 2026-07-16 23:59:10 +08:00
008324e70c Summarize vLLM scheduler step profiles 2026-07-16 23:20:04 +08:00
630de9f573 Freeze vLLM 0.20 profiles and capture trace routing 2026-07-16 23:14:27 +08:00
07b1eb4b75 Initialize TP context for router profiling 2026-07-16 22:54:17 +08:00
414838a799 Profile replicated router without TP group 2026-07-16 22:49:31 +08:00
fcca259475 Profile Qwen MoE router on vLLM 0.20 2026-07-16 22:43:55 +08:00
b8f5f1dcf3 Adapt Frontier RMSNorm profiling to vLLM 0.20 2026-07-16 22:34:23 +08:00
885db8527a Initialize vLLM config for Frontier profiling 2026-07-16 22:29:10 +08:00
9b54075e75 Adapt Frontier RoPE profiling to vLLM 0.20 2026-07-16 22:23:56 +08:00
0e9b64893f Add full Frontier linear profile sweep 2026-07-16 22:18:30 +08:00
48d2c2fb80 Add Frontier linear profiler smoke 2026-07-16 22:11:04 +08:00
32ab584789 Initialize collective profiler with model config 2026-07-16 21:59:58 +08:00
3121e35b0e Profile collective dispatch across tensor sizes 2026-07-16 21:57:44 +08:00
4bf9bdf28f Add full Qwen30 MoE profiling matrix 2026-07-16 21:52:43 +08:00
1fb777fc11 Set vLLM config for collective profiling 2026-07-16 21:48:52 +08:00
4fed4329cf Profile MoE TP-local shards without collectives 2026-07-16 21:41:38 +08:00
9bbfb87a85 Keep KV smoke artifacts separate 2026-07-16 21:40:07 +08:00
9bc38d8851 Profile FA3 KV-cache updates separately 2026-07-16 21:37:01 +08:00
a9aed87518 Add FlashInfer TRTLLM all-reduce profile smoke 2026-07-16 21:34:20 +08:00
5a958077a7 Add exact vLLM 0.20 MoE profile smoke 2026-07-16 21:27:35 +08:00
adaf68badc Add full per-TP FlashAttention profile sweep 2026-07-16 21:24:11 +08:00
a70a1fac71 Retry profile smoke with cu129 vLLM wheel 2026-07-16 21:19:22 +08:00
b3e437acb0 Serialize vLLM attention profile results 2026-07-16 21:17:27 +08:00
2f04971c90 Freeze Qwen30 trace profile support 2026-07-16 21:09:51 +08:00
fb12e3b502 Use local cache for vLLM profile smoke 2026-07-16 21:07:24 +08:00
e9a4be6153 Add vLLM 0.20 per-TP FlashAttention profile smoke 2026-07-16 21:00:27 +08:00
13f0713bf2 Document Frontier fixed-cohort fidelity results 2026-07-16 06:21:40 +08:00
344af3a428 Prepare targeted disputed-boundary trial 2026-07-16 04:16:49 +08:00
def9e24c4d Allow targeted SLO boundary repeats 2026-07-16 04:02:29 +08:00
3db632552b Compare Frontier fidelity across real trials 2026-07-16 03:41:41 +08:00
501ceb8171 Skip null SLOs in boundary repeats 2026-07-16 03:09:19 +08:00
163fb3d9e8 Repeat boundaries for every registered SLO 2026-07-16 02:35:41 +08:00
993cce6143 Refine Frontier rank boundaries before repeats 2026-07-16 01:59:37 +08:00
f7052edd14 Report request-level SLO fidelity for Frontier 2026-07-16 01:42:13 +08:00
678388084c Add fleet job for Frontier boundary repeats 2026-07-16 01:28:23 +08:00
84584c719c Add reverse boundary repeats for Frontier comparison 2026-07-16 00:52:13 +08:00
e47bbf3e76 Compare Frontier rank under explicit objectives 2026-07-15 23:32:07 +08:00
ef5c17e6ec Use writable fleet root on dash0 2026-07-15 23:20:01 +08:00
91869c531c Add blind Qwen235B fleet job 2026-07-15 23:17:43 +08:00
33c91109d7 Require batched Frontier EP equivalence 2026-07-15 22:45:40 +08:00
1fabea26ef Batch exact Frontier EP lane predictions 2026-07-15 22:44:10 +08:00
dadcdbd351 Warm community rank runs before measurement 2026-07-15 22:32:40 +08:00
5dbcd9f51c Parallelize independent Frontier configs 2026-07-15 22:11:44 +08:00
ae933f2256 Record Frontier cache equivalence evidence 2026-07-15 22:10:36 +08:00
5fadc86edd Cache exact Frontier EP lane predictions 2026-07-15 22:07:40 +08:00
ed500fa016 Add fixed-cohort Frontier rank protocol 2026-07-15 21:47:25 +08:00
ca88c06d06 Preserve trace arrivals in Frontier grid 2026-07-15 21:08:19 +08:00
6307620e20 Prevent client-side clipping in vLLM grid 2026-07-15 20:36:10 +08:00
2f730417e0 Revert "Preserve global token count in Frontier EP profiles"
This reverts commit d64f12eb5b.
2026-07-15 20:26:27 +08:00
d64f12eb5b Preserve global token count in Frontier EP profiles 2026-07-15 20:23:21 +08:00
76e0de33ee Model critical EP lane for Frontier prefill 2026-07-15 20:15:09 +08:00
9275619984 Add reproducible community vLLM prefill grid 2026-07-15 20:00:41 +08:00
6e619b75d2 Replay raw completion prompts without chat wrapping 2026-07-15 19:53:00 +08:00
584af7b253 Fingerprint Frontier source snapshots 2026-07-15 19:41:07 +08:00
a6f101b09e Add reproducible Frontier prefill grid freeze 2026-07-15 19:39:30 +08:00
a54f69352d Add Frontier MoE gating context closure 2026-07-15 19:28:39 +08:00
c71f379110 Add Qwen235B NCCL profile runner 2026-07-15 19:10:52 +08:00
97e66ae276 Add validated Frontier profile assembly 2026-07-15 19:04:40 +08:00
e6e6fef41a Skip dense FFN surrogate for MoE profiles 2026-07-15 18:51:52 +08:00
c9a5f0a0c3 Profile exact vLLM MoE serving entrypoint 2026-07-15 18:41:23 +08:00
d9df3003dd Match in-place vLLM MoE serving path 2026-07-15 18:37:26 +08:00
684a2de413 Align Frontier FP8 profiling with vLLM runtime 2026-07-15 18:34:13 +08:00
9c8570f36b Report held-out active intervention result 2026-07-15 03:34:49 +08:00
39b767e384 Correct active intervention engine provenance 2026-07-15 02:07:27 +08:00
e5fd463f05 Add prospective active intervention experiment 2026-07-15 02:03:38 +08:00
d229f2a85e Add action-conditioned intervention feasibility model 2026-07-15 01:42:37 +08:00
0d16838097 Audit tuning cost and core challenges 2026-07-15 01:41:25 +08:00
8c930ba3a1 Report action-aware constraint pilot results 2026-07-14 22:34:26 +08:00
2af22dbce4 Audit action-response mechanism telemetry 2026-07-14 22:29:07 +08:00
3facb18bcf Fix async telemetry coverage audit 2026-07-14 22:23:43 +08:00
c5ab073af5 Fix action-aware burn-in gate 2026-07-14 20:47:23 +08:00
1db737e641 Revise action-aware pilot after token overload 2026-07-14 20:39:04 +08:00
823c550e53 Add crossed-constraint action-aware pilot 2026-07-14 20:26:54 +08:00
26c2cdab2b Record phase-aware telemetry pilot result 2026-07-14 19:18:57 +08:00
7fd9563550 Replace undrainable telemetry load with fresh rerun 2026-07-14 17:52:21 +08:00
c0b40af24f Enforce phase-stable telemetry pilot gates 2026-07-14 17:35:58 +08:00
2afc6eeb8d Add dry run gate for long telemetry pilot 2026-07-14 17:25:45 +08:00
52a9dc13dd Give long replay bands stable request identities 2026-07-14 17:22:56 +08:00
650f54b35e Merge disjoint bands for long replay pilot 2026-07-14 17:21:30 +08:00
0515ad8ecc Make telemetry audit replay-phase aware 2026-07-14 17:18:25 +08:00
791f7a8889 Audit telemetry intervention response for tuning 2026-07-14 16:39:38 +08:00
7a3631b528 Audit telemetry residual tuning premise 2026-07-14 15:44:37 +08:00
f01819680d Report failed fidelity pilot gate 2026-07-14 13:57:12 +08:00
24a9c27b10 Add fidelity pilot shortlist replay 2026-07-14 13:53:14 +08:00
2261818994 Account for failed fidelity pilot attempts 2026-07-14 13:41:46 +08:00
1f32ae217e Harden strong fidelity pilot validation 2026-07-14 13:39:38 +08:00
4ad699ef97 Report fidelity pilot covariate shift 2026-07-14 13:38:32 +08:00
12d1d4ad02 Make fidelity simulator tools self-contained 2026-07-14 13:29:38 +08:00
a3b25f4a92 Add simulator-aware fidelity pilot audit 2026-07-14 13:28:16 +08:00
23142aa359 Strengthen fidelity calibration baseline 2026-07-14 13:08:45 +08:00
684 changed files with 240367 additions and 50 deletions

22
.gitignore vendored
View File

@@ -19,3 +19,25 @@ runs/**/*.jsonl
.ruff_cache/
# Recovered dash1 interaction-run stores (100 MB raw tune logs, kept on disk only)
recovered-stores/
# Local reference material and accidental shell output.
/AITuner系统优化与挑战.pdf
/16
/docs/assets/simulator-fidelity/*.svg
# Generated experiment state. Protocols, analysis code, compact result tables,
# and frozen manifests remain tracked next to these directories.
/runs/frontier-phase-factorial-v0/fleet-artifacts*/
/runs/frontier-phase-factorial-v0/fleet-state*/
/runs/frontier-phase-factorial-v0/invalid-overlap-*/
/runs/frontier-phase-factorial-v0/simulator-smoke/
/runs/frontier-phase-factorial-v0/simulator-*/cache
/runs/frontier-phase-factorial-v0/simulator-*/runs/
/runs/frontier-phase-factorial-v0/simulator-*/traces/
/runs/frontier-phase-factorial-v0/results/final/qwen30-prefill-ranking.png
/runs/frontier-qwen30-vllm020-profile-v1/comparison/
/runs/frontier-qwen30-vllm020-profile-v1/fleet-artifacts/
/runs/frontier-qwen30-vllm020-profile-v1/fleet-state/
/runs/frontier-multicase-sufficiency-v1/fleet-artifacts/
/runs/frontier-multicase-sufficiency-v1/fleet-state/
/runs/frontier-multicase-sufficiency-v1/frontier-smoke-failure/

View File

@@ -0,0 +1,60 @@
# 实验 S0good/bad case 分裂的统一分解margin vs differential residual
> **状态:** 已完成2026-07-20含 S0b 方向化修正与一轮 strict review 修复)
>
> 用户指令:核心要务是分析为什么部分 case 下 Frontier work、部分不 work 的 system 根因;本 card 是该诊断 campaign 的第一个 slice仅使用 frozen artifacts零 GPU 成本。人工 review 由用户的直接指令("只有做好这个分析我们才能推进下一步")满足。
## Claim 与决策
- **Parent claim** ongoing.md H2——误差机制是 action-conditioned residual本实验把它细化为"分裂从哪来"。
- **现象(已冻结):** 同一 best-effort Frontier 栈上Q30/Q235 的 Trace-PD 与多数 PO 面 selection 近优,而 Fixed-PD 面 1458% regret失败 objective 随负载档切换Q30 低压 TPOT/E2E 全反、高压 TTFT 5658%A1 collective profile 修复了 Q235 Trace-PO p90 21.2%→0.3% 但对 Fixed-PD 33% 完全无效(本 card frozen-inputs/q235-ablation-a1
- **Competing hypotheses**
- **H-SCALE** 分裂完全由「config-differential residual vs 真机 decision margin」的关系解释good case 的 sim/real 比值跨 config 近似均匀乘性偏移argmin 不变bad case 的比值跨 config 分散且超过 margin。workload shape 本身不需要出现在解释里。
- **H-THRESH** 绝对 service-time 高估近似均匀但与离散机制MNS admission cap、MoE token-bucket、graph bucket交互后被转换为 config-differential 误差fixed uniform workload 把所有请求同步到同一 state 轨迹,使阈值交叉对整个 cell 相干生效trace 的长度/到达异质性把阈值效应摊平。
- **H-STATE** 失败由 simulator 闭环 batch state 分布漂移主导Q235sim decode batch 13.5 vs real 3.9 + B4→B5 profile cliff 正反馈);即使打破 workload 同步性,闭环漂移仍可翻转排序。
- 三者关系H-SCALE 是现象层必要条件H-THRESH/H-STATE 是 differential residual 的两种产生机制,可共存但可判别(见事前预测)。
- **事前预测:**
- H-SCALE 成立 ⟺ 对每个 case×objectivefailure 恰好发生在「top 邻域 log-ratio spread > log1p(真机相对 margin)」处(两侧同为 log-space 尺度),无反例。
- H-THRESH 独有bad case 的 differential 误差集中于阈值语义分量first-scheduling wait、bucket 跳变段),且 sim-only 反事实(去阈值/加 jitter恢复排序——Q30 高压 TTFT 的 admission 反事实已支持一例。
- H-STATE 独有:差异化误差在去掉阈值分量后仍在 execution 项内Q235 Fixed-PD 的 own-composition 20.07 vs exact-state +10.90 已支持一例)。
- **判定规则:** S0 只裁决 H-SCALE 与「分量定位」queue vs executionH-THRESH/H-STATE 的干预判别属 S1+sim-only 反事实)与 GPU 实验(需另行 review。若 H-SCALE 出现反例good case 有 spread>margin 仍选对,或 bad case spread<margin必须原样报告不得平滑
## Setup
- **输入全部 frozenruns/frontier-split-rootcause-v0/frozen-inputs/** q30-trace-pdgraph-piecewise comparison)、q30-fixed-hifixed-pd/fixed-po 高压面)、q30-expansion-lo低压 fixed-pd/fixed-po/trace-po)、q235-fourcase-a0q235-ablation-a1+provenance)、q235-state-diagq30-admission-diag来源 cpfs 路径与 SHA 见各目录内 manifest/launch 记录
- **计算 case×objective** per-config 比值 r_c=sim_c/real_cconfig-uniform scale=geomean(r_c)differential residual=log-ratio spread surface 与真机 top-3 邻域各一真机 relative marginbest 2nd-bestbest sim-winner 的真机值差failure flag=regret>5%H-SCALE 检验=failure ⟺ 邻域 spread>log1p(margin)review 修正:初版直接以 ln 差比较普通 relative margin尺度不一致修正后 70 行 verdict 不变)。
- **分量定位:** q30-admission-diag 提供 TTFT=first-scheduling wait+prefill execution 分解q235-state-diag 提供 own-composition vs exact-state contrast把这些已知分量证据合并进统一表。
- **交叉核对:** 重算的 regret 必须与各 frozen comparison.md 表一致(抽查 58.0%、33.0%、0.0%A0 vs A1 的 Q235 对比必须复现 trace-po p90 21.2%→0.3%、fixed-pd 四项不变。
## 预期产物与 review
- runs/frontier-split-rootcause-v0/analyze_split_decomposition.py只读 frozen-inputs确定性输出
- runs/frontier-split-rootcause-v0/results/decomposition.{json,md}:统一表,每行 case×objective列出 winner、regret、scale、spread全/邻域、margin、H-SCALE verdict、已知分量归因
- runs/frontier-split-rootcause-v0/results/margin-vs-residual.pngx=真机 marginy=邻域 differential residual点色=selection 对错H-SCALE 成立则对错点被对角线分离
- 人工验收:编排者亲自重跑脚本、抽查交叉核对数字、亲自查看渲染图
## 复现信息
- **Code** AITuner branch feature/sim自 HEAD 18c0b25 起worker/reviewer job-id 见下方「Review 与 provenance 补记」;脚本与产物随本 card 同一 commit 入库(含 frozen-inputs 本地拷贝)。
- **Environment** 本地 workstationCPU-onlypython3+matplotlib不访问远端。
- **已知 deviation** frozen-inputs 是 cpfs 原件的本地拷贝scp2026-07-20q30-expansion-lo 的低压 Fixed-PD 面已被高压面取代为 primary本分析将两档并列为独立观测不混合。
## 结果
- **观察事实:**
- 70 行14 个 case surface全部算出无数据缺口四组硬性交叉核对通过连续运行产物 SHA 一致。
- **H-SCALE 判为必要非充分**23 个 material failureregret>5%全部满足「top-3 邻域 log spread > log1p(margin)」,无一例失败发生在 residual 小于 margin 处;但另有 40/70 行同样满足该条件却均非 material failure其中仅 16 行 exact winner match其余 24 行是小 regret 的 winner 错位——14 个 MNS 精确 tie 与 10 个 strict reversal——无方向 spread 不携带决策信息。
- **S0b 方向化后的机制普查**23 个 material failure 的 winner-deciding pair 分布为 tp-axis 11、mixed 10、mns-axis 2Q235 A0/A1 Fixed-PD 的 8 个 TPOT/E2E failure 全为 tp-axisQ30 Fixed-PD 高低压为 tp/mixed仅有的 2 个 mns-axis failure 是 Q235 A0/A1 Trace-PD E2E p90regret 6.2%,勉强越过 5% 门槛。trace 面严格反序中 tp-axis 为 0q30 trace-pd 三轴全 0
- **A1 对照的轴分解**serving-matched collective profile 把 Q235 两个 PO 面的 tp-axis 反序从 4/4 清零trace-po p90 regret 21.2%→0.3%),但 Fixed-PD 仅 5→4、四项 regret 一位小数不动——prefill 路径的 TP-differential 误差源=collective profile可修decode 耦合的 TP-differential 误差另有来源。
- **MNS 不敏感缺陷**14 个 winner-label mismatch 是 simulator 逐位相等的 tie全部 mns-axis如 q30 fixed-po 的 MNS16↔32、q235 fixed-pd 的 MNS64↔128tie 计入后 MNS 边界误差 31 与 TP 严格反序 32 相当,但 MNS 侧 regret 小。
- **成功的鲁棒性**23 个 exact-winner success 中 17 个 margin-robustmargin≥1%6 个 fragile含 q30 trace-pd E2E p90 的 0.1% margin 与 q235 A1 fixed-po 四项)。
- **面级 scale 对照**prefill-only 面 geomean scale 0.961.37×(绝对预测基本准确),含 decode 的面 4.3130×——绝对误差灾难集中于 decode。
- **异常:** 无数据异常。strict reviewFAIL3 Major/1 Minor指出 H-SCALE 尺度混用log spread vs relative margin、tie 轴普查缺失、card 状态过期、Q235 一致性表述过强;全部修复,修复后 70 行 verdict 逐行不变。
- **含义:** 分裂的现象层解释是「margin 保护 + config-differential 误差」。机制层上21/23 个 material failure 由 TP/mixed pair 决定,且所有大 regret≥13%failure 都发生在含 decode 的面上:其中 Q235 Fixed-PD 有 state-drift 直接证据、Q30 高压 TTFT 有 admission 反事实证据,而 **Q30 低压 TPOT/E2E 反转的机制尚未诊断**S1 目标。「decode 耦合的 TP-differential 误差是主要载体」是当前最强归纳,不是对全部 failure 的已证机制归因2 个 mns-axis 边缘 failure6.2%在该归纳之外。trace 面成功伴随「TP 反序为零 + TP margin 宽」但「误差小」与「margin 宽」谁是主因仍未判——这正是 H-THRESH vs H-STATE 的判别缺口。轴标签与机制不一一对应Q30 admission 是 MNS 阈值机制但 deciding pair 为 tp/mixed因 TP 改变到达压力)。
- **Claim update** H2action-conditioned residualsupported 且被细化residual 的决策相关分量集中在 TP 轴、由 decode 状态耦合产生H-SCALE 降级为必要条件H-THRESH/H-STATE 保持 competing待 S1 判别。
- **下一步:** S1sim-only 反事实Q30 低压 Fixed-PD TPOT 反转的分量定位——这是唯一无机制解释的 material failurefixed workload jitter 判别 H-THRESH vs H-STATEGPU 判别实验(加压 Trace-PD、jittered Fixed-PD 真机面dash14另行出 card 供 review。
## Review 与 provenance 补记
- S0 workercodex `task-mrsn1s9c-z6ha0w`S0b`task-mrsnjzn6-k18x4n`resumestrict reviewerfresh 只读):`task-mrsocmlb-386btc`verdict FAIL修复轮`task-mrsop06n-pwxo0b`fresh writable。编排者独立验收脚本重跑、SHA 比对、两图目视检查。
- 产物 SHA修复后decomposition.json `81ea56b2…`、decomposition.md `4d19af54…`、margin-vs-residual.png `df2110c4…`、decision-pair-axis.png `2e787301…`

View File

@@ -0,0 +1,183 @@
# Frontier workload-regime taxonomy
- Date: 2026-07-20
- Status: proposed; awaiting review before workload generation or GPU runs
- Scope: explain when Frontier preserves the real-system config ranking, rather than merely comparing Fixed with Trace
## Claim under test
Frontier reliability is controlled by three quantities:
1. the latency-model residual between simulator and real execution;
2. the closed-loop gain from timing to scheduler state (batch, MoE routing, CUDA-graph bucket, MNS occupancy, admission/KV pressure);
3. the real decision margin between configurations.
For a config pair `a,b`, define
```text
D_real(a,b) = log L_real(a) - log L_real(b)
delta(a,b) = [log L_sim(a)-log L_real(a)]
- [log L_sim(b)-log L_real(b)]
slack(a,b) = sign(D_real) * [D_real + delta]
```
`slack < 0` means the simulator reverses the real pairwise ordering. The primary hypothesis is that reversals occur when simulator and real execution land on different sides of a scheduler-state knee, or when the real decision margin is too small to absorb the differential residual. `Fixed` and `Trace` are not themselves the causal classes.
## Existing evidence motivating the experiment
- Q30 Trace-PD preserves all six objective winners, but many pairwise residuals oppose the real winner. Its success is therefore often margin protection, not zero residual.
- Q235 Trace-PD preserves TTFT/TPOT winners but misses E2E p90 by 6.2%; Trace is not universally safe.
- Q30/Q235 Fixed-PD decode objectives show negative minimum signed slack and 13--37% regret.
- In Q30 low-load Fixed-PD, Frontier's batch-1 TP ordering is correct, while the closed-loop simulator increases TP4's effective batch and changes the MoE cost enough to reverse the ordering. This identifies a concrete state knee, but does not yet establish a general rule.
## Workload families
All comparisons use the same request multiset where applicable, the same total observation window, and the same normalized offered decode load
```text
rho = request_rate * E[output_tokens] / measured_reference_decode_capacity.
```
This avoids equating equal request rates with equal load.
| ID | Shape / request lengths | Arrival process | Prefix/session state | Isolated effect |
|---|---|---|---|---|
| W0 | short fixed `2048 -> 128` | uniform | off | known low-residence failure anchor |
| W1 | trace-mean fixed ISL/OSL | uniform | off | homogeneous baseline |
| W2 | trace-mean fixed ISL/OSL | trace timestamps | off | arrival burst only |
| W3 | exact trace ISL/OSL multiset | uniform | off | length heterogeneity only |
| W4 | exact trace ISL/OSL multiset | trace timestamps | off | length + burst |
| W5 | exact trace prompts/ISL/OSL | uniform | exact prefix/session identity | prefix state without burst |
| W6 | exact trace prompts/ISL/OSL | trace timestamps | exact prefix/session identity | full production trace |
Prefix is intentionally a nested factor: enabling a synthetic prefix graph on fixed identical requests would introduce a different workload rather than isolate production prefix reuse. Therefore this is not presented as a full `2^3` factorial.
## Load sweep and expected patterns
Simulator discovery sweep: `rho in {0.05, 0.25, 0.50, 0.90, 1.20}`. The points mean deep low load, light batching, moderate batching, capacity knee, and overload; their request rates are derived independently for every workload family.
| Pattern | Observable state | Prediction for Frontier |
|---|---|---|
| P1 singleton-linear | real and sim stay below the first batch/graph knee | works if the batch-1 operator ordering is correct |
| P2 knee-straddling | real and sim occupy opposite sides of a batch/MoE/graph/MNS knee | fails systematically; Fixed-PD is the current example |
| P3 same-side batched | both systems cross the same knee and remain below admission pressure | works if batch-conditioned operator ordering is correct |
| P4 capacity/admission aligned | both systems are governed by the same capacity bottleneck | TTFT/config winner may work despite large absolute error; E2E/MNS can remain fragile |
| P5 heterogeneity-smoothed | broad lengths reduce coherent threshold occupancy at matched `rho` | may work; this is a hypothesis, not an established explanation |
| P6 burst-sensitive | same request multiset, but transient queue/MNS occupancy differs | mean ranking may work while TTFT/E2E tail ranking fails |
| P7 prefix-state-sensitive | hit/eviction and reused-token distributions differ | TTFT ranking fails unless prefix-state transitions are modeled; decode TPOT may remain stable |
| P8 decision-boundary | real config margin is comparable to run variance/residual | fragile; an exact winner match is not reliable evidence |
## Hypotheses and distinguishing tests
### H1: state-regime hypothesis (primary)
I believe config-ranking failures occur when the latency residual moves a workload across a scheduler-state knee, because the residual is then amplified into a different batch/resource trajectory. I will verify this by checking whether signed-slack zero crossings co-locate with measured real/simulator state-knee crossings.
### H2: heterogeneity-smoothing hypothesis
I believe length heterogeneity can reduce coherent threshold amplification, because requests reach scheduler boundaries at dispersed times. I will verify it with W1 vs W3 and W2 vs W4 at matched `rho`, requiring a smaller real/sim state-distribution gap rather than merely a correct winner.
### H3: bottleneck/margin-protection alternative
Trace success may instead be explained entirely by a large real decision margin or a shared capacity bottleneck. This hypothesis wins over H2 if W3/W4 do not reduce state-distribution error after matching load and margin, while ranking correctness remains predicted by margin alone.
### H4: burst and prefix are independent failure channels
I believe arrival bursts primarily affect waiting/admission and tail TTFT/E2E, whereas prefix mismatch primarily affects prefill/TTFT state. I will verify this with W1/W2, W3/W4, and W3/W5 paired comparisons.
## Configuration and model scope
Discovery uses Qwen30B because its 12-cell `TP x MNS` surface already has simulator and real anchors:
- TP: `{1, 2, 4}`
- MNS: `{8, 16, 32, 64}`
- objectives: mean/p90 TTFT, TPOT, E2E
Qwen235B is a held-out confirmation, not pooled into discovery:
- existing four feasible TP/MNS configurations;
- only the workload/load patterns that discriminate H1--H4 after Q30 converges.
## Measurements
End-to-end:
- completed/failed requests and achieved request/token rate;
- TTFT, TPOT, E2E mean/p50/p90/p95;
- config regret, pairwise agreement, signed decision slack;
- run-to-run winner stability.
Closed-loop state:
- prefill/decode batch-size histograms and time-weighted batch;
- Running/Waiting distributions and admission delay;
- MNS active-token occupancy and KV/context pressure;
- CUDA-graph bucket residency and fallback frequency;
- prefix hit/reused-token/eviction distributions for W5/W6.
## Decision rules
A workload/load region is:
- **reliable** if regret is at most 5%, pairwise agreement is at least 0.8 at two adjacent load points, and the winner is stable across confirmation trials;
- **fragile** if regret is at most 5% but the real margin overlaps run uncertainty, or a small rate/timing perturbation changes the winner;
- **failed** if regret exceeds 5% or a decision-critical pair has negative signed slack;
- **mechanistically explained by H1** only if the ranking transition co-locates with an observed state-regime transition. Correlation with the Fixed/Trace label is insufficient.
H2 is supported only if the heterogeneous member of a matched pair reduces state-distribution error and shifts the failure boundary in repeated trials. A correct winner alone does not support smoothing.
## Execution plan after review
1. Materialize W0--W6 with one manifest recording request multiset, arrival timestamps, prefix identity, rate contract, and hashes.
2. Run the simulator sweep across `rho` and the Q30 surface; emit a per-stage state ledger.
3. Select real-machine pilot points only around the predicted knees plus one safe-side control. Use guard configs `TP1/MNS64`, `TP4/MNS8`, and `TP4/MNS64`; add `TP2/MNS32` only if the transition is not bracketed.
4. Use only `dash1`, `dash2`, `dash3`, and `dash4`, each verified as an 8×H20 host. `dash0` is excluded from probing, synchronization, and execution. Pin one independent experiment group to each host so at most four groups run in parallel; do not split one trial across hosts.
5. Run one pilot trial per selected point. Confirm only hypothesis-discriminating points with three fresh-server trials and rotated order.
6. Apply the resulting classifier unchanged to the Q235 held-out cases.
Provisional four-way allocation after the simulator identifies the discriminating points:
| Host | Experiment group | Primary contrast |
|---|---|---|
| dash1 | homogeneous controls | W0/W1 across safe side and first knee |
| dash2 | arrival effect | W1 vs W2 and W3 vs W4 |
| dash3 | length heterogeneity | W1 vs W3 and W2 vs W4 |
| dash4 | prefix/full trace | W3 vs W5 and W4 vs W6 |
The groups are logical queues, not permanent ownership: if a host probe fails, that host is excluded and its group waits or moves to another permitted idle host. Cross-host latency values are not pooled until a common canary config verifies that host effects are within run uncertainty.
No GPU run is authorized by this card yet. The review decision is whether the workload decomposition and decision rules are sufficient to implement the materializer and launch Phase 1.
## Expected figure
The accompanying mock figure is schematic, not data. Panel A shows the state knee that real and simulator trajectories may cross at different loads. Panel B shows the corresponding minimum signed decision slack; a negative value denotes a ranking reversal. The claim is supported only if measured zero crossings and state knees align across workload families.
## Risks and controls
- Equal `rho` does not guarantee equal prefill pressure; report both prefill and decode offered work and stratify if necessary.
- Full-trace overload can collapse all configs to similarly poor latency. Such points identify a capacity-limited region but cannot validate fine-grained ranking.
- MNS ties and censored/failed requests can create false winners; exclude invalid cells before calculating regret and report the exclusion.
- One trace cannot establish generality. The initial result is a mechanism boundary for this trace/model/hardware, followed by held-out Q235 validation.
## Execution log
### 2026-07-20: materialization and simulator launch
- Code baseline: `feature/sim@157bf36` for the valid v4 sweep.
- Hosts probed: `dash1`, `dash2`, `dash3`, `dash4`; each exposed 8 NVIDIA H20 GPUs with 0 MiB used at probe time. `dash0` was not probed or used.
- Source cohort: 129 Q30 Trace-PD requests. The private artifact supplies exact prompts, lengths, outputs, timestamps, sessions, and runtime block identities; the simulator projection retains only the first `floor(ISL/16)` complete block identities.
- Materialized: 35 cases = W0--W6 × `rho {0.05,0.25,0.50,0.90,1.20}`. Audit passed request count, exact decode offered load, empirical arrival rate, prefix block count, and prefix-off empty identity vectors.
- Simulator smoke: W0 / `rho=0.05` / TP4-MNS64 completed 129/129. Simulator TTFT mean/p90 was 109.81/124.16 ms and TPOT mean/p90 was 36.26/36.79 ms. This is a harness check, not real-system fidelity evidence.
- Invalid attempts retained for audit: v1 had a Bash argument-expansion error; v2 mixed multiple workload families into a runner that requires strictly increasing anchors from one family; v3 exposed a scikit-learn cache-version mismatch. None is used as scientific evidence.
- Valid v4 controls: isolated output/predictor cache per TP/prefix group; scikit-learn 1.9.0 matching the predictor cache format; per-family five-point runner invocations; stage batch ledger enabled; TP1 exempted from the collective fallback gate because a single rank has no all-reduce.
- Active v4 allocation: dash1=TP1 prefix off/on, dash2=TP2 prefix off/on, dash3=TP4 prefix off, dash4=TP4 prefix on. The four fleet jobs are running from fresh `sim-v4` output roots. First-process audit found the explicit isolated `--metrics_config_cache_dir` on all hosts and zero cross-version warnings.
- First valid v4 tranche: 16/16 observed cells completed, each with 129 requests, request metrics, and a stage-batch ledger; no traceback, fallback, or version warning was found. The tranche covers all five W0 load points at TP1/TP2/TP4-MNS8 plus the first W5 prefix points at TP4-MNS8.
- Early load-boundary observation: W0 at `rho=0.05` is low-latency for TP4-MNS8 (simulator TTFT mean 109.25 ms) but already queues for TP1-MNS8 (25.70 s); at `rho=0.25`, even TP4-MNS8 reaches 27.13 s mean TTFT. Because `rho` normalizes decode tokens only, high-rate short-output W0 also raises prefill and active-sequence pressure. These points map the overload boundary and are not eligible as reasonable-latency real pilots.
- Real-runtime gate: a stock vLLM 0.20.0 environment passed import/H20 checks but used CUDA 13.0, so it is excluded from comparison with the historical CUDA 12.9 baseline. The replacement environment `vllm-0.20.0-cu129-workload-regime-v2` passes `vllm CLI=0.20.0+cu129`, torch `2.11.0+cu129`, CUDA runtime 12.9, H20 visibility, and all 179 package dependency checks. The first CPFS install used file copies and was stopped after download because it was still copying roughly 7 GB after 12 minutes; its incomplete directory is retained with an `invalid-copy-incomplete` suffix, while v2 uses same-filesystem hardlinks from the validated cache.
- Load-contract correction: the original Fixed-PD surface held request rate per GPU constant, so global arrival rate scaled with TP. The v4 sweep holds global arrival rate constant and is retained as the control that isolates service-topology changes. A matched per-GPU sweep is now required to reproduce the original closed-loop intervention: TP1/TP2/TP4 receive `1x/2x/4x` global arrival rate at the same per-GPU `rho`.
- Per-GPU low-load materialization: 105 cases = W0--W6 × `rho {0.0025,0.005,0.01,0.02,0.05}` × TP `{1,2,4}` were generated under `traces-per-gpu-low`. Audit passed 105 unique paths, 129 public/private rows per case, digests, arrival alignment, and exact `global_rate / TP = per_gpu_rate`. W0 `rho=0.01` is 0.239375 req/s/GPU, bracketing the original 0.215 req/s/GPU Fixed-PD point with `rho=0.005`.
- The per-GPU sweep writes to a separate `sim-per-gpu-v1` result root but reuses the completed v4 predictor cache for the same TP/prefix/config. Predictor cache provenance is explicit in every surface manifest; workload results and state ledgers are never shared.
- `wait_and_dispatch_per_gpu.sh` is active locally as a serial gate. It requires all four exact v4 run directories to contain `finished_at` and exit code zero before probing dash1--dash4 and dispatching the four per-GPU jobs; it does not launch a second sweep while v4 is still consuming CPU.
- A first materialization attempt rounded both `rho=0.005` and `rho=0.01` to the same `rho0p01` directory. Digest validation stopped before simulator launch; the invalid directories were retained with an `invalid-rho-label-collision` suffix. The label function now preserves up to 12 significant digits and has a regression test.
Current decision: finish the v4 fixed-global-rate control, then reuse its trained predictors for the low-load per-GPU sweep before selecting discriminating real-machine pilot points. No real latency result from vLLM 0.20.2 will be compared with the historical vLLM 0.20.0 baseline until the runtime-version gate is resolved.

View File

@@ -0,0 +1,45 @@
# 实验Qwen235 Fixed-PD state-matched decode diagnosis
> **状态:** 已批准,执行中
>
> 本 card 记录 Qwen235 Fixed-PD 在真实 collective profile 后仍保留 30%+ selection regret 的下一层判因实验。
## Claim 与决策
- **Parent claim** Qwen235 Fixed-PD 的错误排序来自 action-conditioned decode residual而不是缺失的 TP8 all-reduce profile。
- **目的:** 区分 simulator 错在 state distribution还是相同 state 下的 conditional execution-time composition。
- **Competing hypotheses** H1Frontier 生成的 decode batch/context/graph state 与真机不同H2state 对齐后 Frontier 仍把 TP8/EP8 预测得更快,误差位于 MoE/EP、graph 或 attention 的 conditional stage model。
- **事前预测:** 真机 config contrast 为 `TP8-TP4=+6.95 ms/token`A1 simulator 为 `-20.07 ms/token`。若 H1 成立,用真实 state 重加权后 contrast 应翻正;若 H2 成立matched-state contrast 仍为负。
- **判定规则:** 先比较 frozen-real coarse state 与 full-ledger Frontier state。state 明显不匹配则补 iteration telemetrystate 支持重叠且 matched-state predictor 仍反序,才进入 stage breakdown。任何 stage 只有在 measured substitution 能使 winner 翻转时才称为 decision-bearing root cause。
## Setup
- **自变量:** state sourcefrozen real / Frontier后续 matched-state replay 中固定 decode batch、context-length、graph bucket 与 routing load。
- **控制变量:** Qwen235 FP8、vLLM 0.20.0、H20、Fixed-PD 4096→256、0.2 req/s/GPU、MBT8192、MNS64、TP4/EP1 与 TP8/EP8、Frontier commit、r2 operator profiles、A1 measured collective CSV 全部冻结。
- **选择 MNS64** A1 中 MNS64/128 的 TTFT/TPOT/E2E 完全相同;先去掉不提供判别力的重复维度。
- **第一阶段:** CPU-only 重放原 A1 commands只打开 `frontier_stage_batch_ledger` 与 individual batch metrics验证 request metrics 与原 A1 bitwise/score 等价。真机先复用 3 次 frozen server logs 的 10 秒 Running/Waiting/KV samples明确标为 coarse proxy不冒充 per-iteration batch。
- **第二阶段触发条件:** coarse proxy 不足以判断或 state mismatch 显著时,短窗口重跑真机并采集 per-iteration `decode_batch_size/context_length_hist/cudagraph bucket`;否则进入 matched-state whole-decode-step。
- **Metrics** decode batch/token distribution、prefill fraction、scheduler steps/s、graph bucket/padding、queue/KV proxyconfiguration contrast `TP8-TP4`stage measured-substitution 后的 winner。
## 预期产物与 review
- **预期数据:** 两个 Frontier full-ledger replays三次真机日志的 coarse state summarystate overlap/reweighting verdict必要时的 short-window iteration telemetry。
- **Figure prototype** `../../runs/frontier-fidelity-envelope-v1/qwen235-state-matched-diagnosis-mock.png`。左图对比 real/sim state右图展示 H1 与 H2 下 matched-state contrast 的可区分方向。全部数值标为 schematic/mock。
- **人工 review** 已批准(用户在分析方案后要求“推进”)。
- **Review 意见:** 先做最便宜的 state audit不直接启动完整 Nsight sweep每一步只在能改变下一决策时升级证据成本。
## 复现信息
- **Code** AITuner `feature/sim`;运行 commit 待冻结。Frontier `6e8e0d845bceff11b0b62cb29df3a1a93411fdd4`
- **Environment** dash0Frontier replay CPU-only后续真机才使用 4/8×H20。
- **输入:** `/home/admin/cpfs/wjh/aituner/qwen235-collective-profile-ablation-20260719-r1/sim/fixed-pd` 与 frozen real campaign `/home/admin/cpfs/wjh/aituner/qwen235-v020-fourcase-20260719-r1/real/fixed-pd`
- **产物路径:** `/home/admin/cpfs/wjh/aituner/qwen235-fixed-pd-state-diagnosis-20260719-r1`
- **已知 deviation** frozen real logs 的 Running 指标是 10 秒采样的 active-request proxy不是 scheduler iteration ledger不能单独支持 matched-state causal claim。
## 结果
- **观察事实:** 待运行。
- **异常:** 待运行。
- **含义:** 待运行。
- **Claim update** unchanged
- **下一步:** 运行两项 CPU state replay 与 coarse real-state analysis。

View File

@@ -0,0 +1,65 @@
# 实验 EXP-SIMFID-Q235-CC-TP8真实 TP4/TP8 collective profile 消融
> **状态:** review 通过,执行中
>
> 本 card 是 SHA、command、config、log 等 provenance 的唯一归宿;本轮只重跑 simulator不重跑已经冻结的 48 个真机 trial。
## Claim 与决策
- **Parent claim** Qwen235 Fixed-PD 的 30%+ TPOT/E2E selection regret是否主要由 TP8 collective profile 缺失及 TP4 profile 与真实 serving backend 不匹配造成。
- **目的:** 支持或反驳 mechanism hypothesis不是用同一 workload 的 E2E calibration 修正 simulator。
- **Competing hypotheses**
- H1collective profile coverage/backend mismatch 是排序反转的必要主因。换成与真机 serving 一致的 TP4/TP8 实测 profile 后Frontier 的 Fixed-PD TPOT winner 从 TP8 翻到 TP4mean/p90 TPOT 与 E2E selection regret 降到 10% 以内。
- H2collective mismatch 只解释部分误差。换 profile 后 TP8 仍是 Frontier winnerFixed-PD TPOT/E2E regret 仍超过 10%;下一主因应定位 decode batch/state-conditioned MoE composition。
- **事前预测:** 当前 Frontier 在 Fixed-PD 上预测 TP4/TP8 mean TPOT 为 87.77/61.59 msTP8 有 26.19 ms 优势;真实 TP4/TP8 为 21.04/27.99 ms。若新的 TP4/TP8 collective profile 使这个 26.19 ms 的 simulator margin 反转,则支持 H1若不能则支持 H2。
- **判定规则:** 只以 frozen simulator rerun 的 winner 与真实 frozen surface 计算 selection regret。绝对 latency ratio 作为 secondary metric不用它替代 selection verdict。
## Setup
- **自变量:**
- A0当前 `measured-allreduce.csv`TP4 是 Qwen30 hidden=2048 的旧实测,且 profiler 只检查 FlashInfer 可用、没有证明每个 payload 的实际 dispatchTP8 无行并静默 analytical fallback。
- A1Qwen235 serving-matched piecewise collective profileTP4/TP8 都在 dash0 H20、vLLM 0.20.0 commit `88d34c640...` 上实测。Frozen server logs 证明真机同时使用 `disable_custom_all_reduce=true` 与 FlashInfer-TRTLLM `allreduce_rms` fusionprofile 对 fusion-eligible payload 测同一 FlashInfer communicator对阈值外 payload 测真实 PyNCCL/symmetric fallback。
- **控制变量:** Frontier commit、Qwen235 operator profiles、runtime contract、四类 frozen traces、候选配置、MNS/MBT、prefix policy、real results 与分析脚本全部不变。
- **System context** Qwen3-235B-A22B-FP8vLLM 0.20.0+cu129dash0 8×H20`{TP4/EP1, TP8/EP8} × MNS{64,128}`MBT=8192Frontier piecewise graph path。
- **Workload 或 trace** 重跑四类 simulator surfaceFixed-PD 4096→256 @ 0.2 req/s/GPU、Fixed-PO 4096→1、Trace-PD、Trace-PO每 cell 沿用原 129-request trace。Fixed-PD 是 primary另外三类检查 profile 替换是否引入新的 selection regression。
- **Profile protocol** payload 覆盖所有 Qwen235 decode graph buckets1--256含真实 capture sizes、fusion 阈值两侧 `{63,64,65}` / `{255,256,257}`,以及 512--8192 prefill sizes每个 TP 与 payload 先 warmup再保留 3×20 个 per-rank CUDA-event samples。raw JSON 记录实际 backend dispatch、fusion byte limit、world size、dtype、payload bytes、GPU/runtime/commit 与 source hashes。TP4/TP8 使用相同 payload grid不把 microbenchmark 直接当作 E2E 结论。
- **Profile contract** H20/SM90 上 vLLM 0.20 的 fusion limit 是 TP4 2 MiB、TP8 0.5 MiB即 Q235 BF16 hidden=4096 时分别为 256/64 tokens。simulator runner 启动前解析 CSV要求所选 configs 的每个 `TP>1` 都有有限、正值的 measured rows缺覆盖立即失败。结果 manifest 写入 CSV SHA-256、TP coverage、row counts 与 piecewise backend 集合。决策实验禁止 analytical fallback。
- **Baselines** A0 current Frontier、A1 measured-profile Frontier、frozen real hardware surface。
- **Metrics** profile latency median/p90 与跨 rank spreadsimulated mean/p90 TTFT/TPOT/E2Ewinner、selection regret、tau-b可定义时每个 TP 的 measured-profile hit/fallback counters。
## 预期产物与 review
- **预期数据:** TP4/TP8 raw collective JSONmaterialized Frontier CSV + manifest四类 A1 simulator surfaceA0/A1/real comparison JSON/Markdownprofile cost ledger。
- **Figure prototype** `../../runs/frontier-fidelity-envelope-v1/qwen235-collective-ablation-mock.png`;左图固定真实与 A0 TPOT并为 A1 留待测 series右图明确“winner flip→0% regret / unchanged→33% regret”的判定。它回答 profile 修复是否足以改变配置选择。
- **人工 review** 通过2026-07-19用户明确要求“推进实验”
- **Review 意见:** 保留 frozen real surface只补真实 TP4/TP8 profile 后重跑 simulator每个 simulator 实验必须使用真实 profile缺失 coverage 或运行时 analytical fallback 立即失败。
## Benchmark design auditexperiment-design-review
| Crime | Verdict | Severity | Evidence | Fix / gate |
|---|---|---|---|---|
| 用 microbenchmark 代替 E2E | PASS | — | collective profile 只作为自变量;结论来自完整 simulator surface 对 frozen real surface 的 selection regret | 保留 A0/A1/real 三方结果 |
| calibration set 等于 evaluation set | PASS | — | A1 只测 collective operator不使用 real E2E latency 拟合参数 | 禁止 E2E scale/calibration |
| selective benchmarking | PASS | — | primary Fixed-PD 外,同时重跑另外三类 workload | 报告所有 16 个 simulator cells |
| 缺失平台/版本 | PASS | — | raw/manifest 绑定 H20、vLLM commit、model、backend 与 hashes | 任一 provenance 缺失则 profile 不可采纳 |
| 缺失方差 | NEEDS EVIDENCE | Major | 尚未执行 profile repeats | raw artifact 必须保留 per-rank repeated samples并报告 spread |
| profile 覆盖静默降级 | FAILA0 | Blocking | TP8 无 measured rowsFrontier 使用 analytical fallback | A1 runner fail-fastfallback count 必须为 0 |
| backend/fusion 阈值未对齐 | FAILA0 | Blocking | 真机日志启用 FlashInfer `allreduce_rms`vLLM 源码规定 H20 TP4/TP8 fusion limit 为 2/0.5 MiB旧 CSV 未记录这条 piecewise contract | A1 在阈值两侧实测并记录每行 dispatch |
**总体建议:** 已批准执行coverage gate、backend match 与 provenance gate 任一不通过则 Block。
## 复现信息
- **Code** AITuner branch `feature/sim`;本 card 创建时 HEAD `f4a75aa8e400ead4eb6d305178192e85940c6de6`,后续运行 commit 待填。vLLM source commit `88d34c6409e9fb3c7b8ca0c04756f061d2099eb1`
- **Environment** dash0 8×H20`/tmp/wjh/venvs/vllm-0.20.0-cu129-profiler-v1`model `/home/admin/cpfs/wjh/models/Qwen/Qwen3-235B-A22B-FP8`
- **产物路径:** 待 review 后冻结;不得覆盖旧 campaign `/home/admin/cpfs/wjh/aituner/qwen235-v020-fourcase-20260719-r1`
- **已知 deviation** `--disable-custom-all-reduce` 只关闭 vLLM custom AR不关闭编译器的 FlashInfer `allreduce_rms` fusion。真机 TP4/TP8 日志都显示自动选择 `trtllm` workspace旧 TP4 profiler 没有记录/执行真实 fusion-limit piecewise dispatch。A1 因此必须同时重测 TP4 与 TP8不能只追加 TP8 行。
- **执行异常:** 首次 simulator launch 误把 frozen operator profile root 写成 r1原 A0 campaign 实际使用 r2。四个 Fixed-PD cells 因缺少 `attn_decode_in_mixed` predictor 均 fail-fast未产生可用 metric。command diff 确认后停止后续运行,恢复 r2 并从 failed cells 重新执行;这些失败不计入 A1 surface。
## 结果
- **观察事实:** 待运行。
- **异常:** 待运行。
- **含义:** 待运行。
- **Claim update** unchanged
- **下一步:** 依次完成 collective profiling、profile materialization、CPU simulator rerun 与 analysis。

View File

@@ -0,0 +1,46 @@
# 实验Qwen30 Fixed-PD TTFT admission diagnosis
> **状态:** 已完成
>
> 用户要求分析 Frontier 在 Qwen30 Fixed-PD 高压 case 上 56--58% TTFT
> selection regret 的根因,并给出简洁结论。
## Claim 与决策
- **Parent claim** Frontier 在 capacity knee 附近的配置排序是否会因 state transition error 失效。
- **目的:** 区分 TP2/TP4 conditional prefill-time 错误、mixed-step composition 错误与 admission queue feedback。
- **Competing hypotheses** H1Frontier 把 TP4 prefill execution 相对 TP2 算慢H2decode service time 的绝对误差使 `arrival_rate × residence_time` 越过 MNS cap首次调度等待被阈值放大H3即使固定 admission statemixed prefill/decode composition 仍反序。
- **事前预测:** H1 下去掉 queue 后 TP2 仍有更低 prefill timeH2 下去掉 queue 后 TP4 恢复更快,且只有 simulator 的 required concurrency 超过 MNSH3 下 state-matched stage contrast 仍支持 TP2。
- **判定规则:** 只有 state ledger/scorer 等价、queue counterfactual 和真机 Running/Waiting 同时支持时才归因 H2否则保留 H1/H3 并补最小 telemetry。
## Setup
- **自变量:** config 为 Frontier winner `TP2/MNS64`、real mean winner `TP4/MNS32` 与 real p90 winner `TP4/MNS64`
- **控制变量:** Qwen3-30B-A3B BF16、community vLLM 0.20、H20、Fixed-PD 4096→256、1.125 req/s/GPU、MBT8192、piecewise graph、原 measured profiles/collectives 和原 257-request traces全部冻结。
- **Workload** uniform open-loop arrivalglobal rate 随 TP 为 2.25/4.5 req/sprefix cache off每个真机 cell 三次 fresh-server。
- **Baselines** 完整 12-cell frozen real/sim surface真机三轮 pooled metrics。
- **Metrics** TTFT=`first scheduling delay + prefill execution`request execution/residence time`arrival_rate × service_time` 相对 MNSRunning/Waitingstage ledger composition。
## 预期产物与 review
- **预期数据:** 三个 scorer-equivalent Frontier state replaysservice/admission decompositionH1--H3 verdict。
- **Figure prototype** `../../runs/frontier-fidelity-envelope-v1/qwen30-fixed-pd-ttft-admission-mock.png`;左图区分 TTFT execution 与 queue右图显示 required concurrency 是否跨越 MNS。
- **人工 review** 已批准(用户要求直接分析清楚该 case
- **Review 意见:** 先复用 existing artifacts 和 CPU replay只有现有真机 periodic queue proxy 不足时才增加 GPU telemetry。
## 复现信息
- **Code** AITuner `2970f74d`Frontier `deadc4a321f0baaa534c6ebd17f974123733cdc2`
- **Environment** dash0Frontier replay CPU-onlyGPU visibility disabled。
- **输入:** `/home/admin/cpfs/wjh/aituner/qwen30-fixed-pressure-surface-20260719-r1`
- **产物路径:** `/home/admin/cpfs/wjh/aituner/qwen30-fixed-pd-ttft-diagnosis-20260719-r1`
- **已知 deviation** 原真机日志只有 10 秒 periodic Running/Waiting没有 per-iteration ledger它可验证 steady queue 是否积压,但不用于细粒度 stage timing。
## 结果
- **观察事实:** Frontier 把 TP2/MNS64 与 TP4/MNS64 的 TPOT 分别高估 8.09× 与 5.61×。按 frozen request execution time 计算TP2 需要 58.0 个并发槽,未超过 MNS64TP4 需要 79.8 个,超过 MNS64。真机 TP4 的 E2E-based required-slot upper bound 只有 14.7MNS16/32/64 三组 periodic logs 的 Waiting max 均为 0。
- **关键反事实:** observed Frontier 中 `TP4/MNS32 - TP2/MNS64` TTFT 为 `+27087.0 ms`;减去每请求的 first-scheduling delay 后变为 `-55.4 ms`,与真机 `-71.4 ms` 同方向。TP4/MNS32 的 27.3 秒 simulated TTFT 中27.18 秒来自首次调度前等待,而不是 prefill execution。
- **异常:** 计划中的 state-ledger replay 会重新训练 frozen no-cache predictors8 分钟后仍停留在 predictor training。由于原 request metrics 已精确提供 TTFT=first-scheduling wait+prefill execution且 MNS sweep 已构成 controlled intervention继续 ledger 不改变判定故主动停止partial output 保留在产物根目录但不进入结果。
- **含义:** H2 supportedH1/H3 对“TP topology 排序从哪里被反转”均 rejected。Frontier 的无排队 prefill 仍预测 TP4 比 TP2 快,错误由 decode service-time 绝对高估使 TP4 独自跨过 admission cap再经 queue feedback 放大产生。该反事实恢复的是 TP4 topology 方向,不声称恢复 exact MNS winner现有真机日志也不能把最初的 service-time overprediction 继续归因到某个单独 operator。
- **Claim update** supported
- **下一步:** 若研究问题升级为“为什么 TPOT 绝对值高估 4--8×需增加真机 stage timing/overlap 证据;它不是解释本次 TTFT selection reversal 所必需。

79
.research/ongoing.md Normal file
View File

@@ -0,0 +1,79 @@
# AITuner 研究当前状态
> 2026-07-17写给未参与项目的读者可直接作为 presentation 讲稿。历史过程与复现信息见 `../runs/*/` 各 experiment card、`../docs/` 各 campaign 文档。
>
> **2026-07-19 update** Qwen235 Fixed-PD 的错误排序在 exact real state composition 下已经翻正,主因是 simulator closed-loop batch state而不是 collective。Qwen30 Fixed-PD 的 56--58% TTFT regret 也已定位Frontier 将 decode service time 高估 4--8×使 TP4 的 modeled concurrency 越过 MNS admission cap并产生虚假排队去掉该等待后 Frontier 与真机都判定 TP4 topology 更快。详见 [`experiments/qwen30-fixed-pd-ttft-admission-diagnosis-20260719.md`](experiments/qwen30-fixed-pd-ttft-admission-diagnosis-20260719.md)。
>
> **2026-07-20 update** 对全部 14 个 frozen case surface70 个 case×objective做了统一的 margin-vs-residual 分解与方向化机制普查([`experiments/frontier-split-rootcause-s0-20260720.md`](experiments/frontier-split-rootcause-s0-20260720.md))。三个要点:(1) 「residual 超过 margin」是失败的必要条件但远非充分——good/bad 分裂不能用无方向误差量解释;(2) 23 个 material failure 的 winner-deciding pair 中 21 个落在 TP 轴或 mixed其余 2 个是 6.2% regret 的边缘 mns-axis casetrace 面的 TP 反序为零A1 measured collective 把 Qwen235 两个 prefill-only 面的 TP 反序清零trace-PO p90 regret 21.2%→0.3%)却对 Fixed-PD 完全无效——prefill 路径的 TP 差异化误差源是 collective profile可修decode 耦合的 TP 差异化误差是当前所有 material failure 的载体;(3) 「Fixed-PD 失败因为高压」被否证:失败 Fixed-PD 的真机 in-flight14.05)低于全对的 Trace-PD38.69),且低压 Fixed-PD 同样失败、失败 objective 随负载切换。另有次要缺陷14 个 winner 错位来自 simulator 对 MNS 逐位不敏感的精确 tie。
>
> **2026-07-20 root-cause update** Q30 低压 Fixed-PD 的 exact stage ledger 关闭了最后一个未解释的 material failure。相同 batch=1 state 下 Frontier full predictor 给 TP4 `18.3515 ms/step`、TP1 `19.3506 ms/step`,方向正确;但 per-GPU 固定到达率使 cluster arrival 随 TP 增长,叠加 decode residence 高估后TP4 在 simulator 内自激到 time-weighted batch `3.0437`96.13% decode 时间 batch≥3own-state step 变为 `28.1712 ms`。其中相对 batch=1 的 `+9.8197 ms` 有 `+8.9297 ms` 来自 batch-conditioned MoEcollective 仅 `+0.0121 ms`。因此 Fixed-PD 的根因不是“固定 workload”或“高压力”本身而是 **execution-time residual 进入离散事件时钟后改变 future scheduler state该 state 再通过 MoE/profile/graph 或 MNS admission 非线性放大,形成 action-dependent signed residual 并穿过 decision margin**。Q30 低压是平滑 state-feedbackQ30 高压是跨 MNS cap 的 threshold amplificationQ235 是 composition drift三者为同一闭环机制族。
>
> **2026-07-20 load-audit update** Trace-PD overload 不是 Fixed/Trace good-bad 分裂的统一解释。旧 Q30 Trace-PD decode offered/observed-peak throughput≈`1.00×`、peak Running/Waiting=`47/0`;降到 `0.10 req/s/GPU` 后 TTFT `245.95/685.51 → 228.14/835.38 ms`mean/p90不出现 tail collapseTPOT `13.18/15.39 → 7.91/8.90 ms`。旧 Q235 则是 `3.44×` 明确过载、peak=`116/3`;降到 `0.035 req/s/GPU` 后 TTFT `1141.54/2616.69 → 478.14/1347.75 ms`TPOT `61.89/78.62 → 24.00/28.49 ms`。旧 surface 仍有 `417×/32.6×` mean-TTFT spread否定“所有配置一样差”。八 case baseline 与 claim boundary 见 [`experiments/frontier-eightcase-load-audit-20260720.md`](experiments/frontier-eightcase-load-audit-20260720.md)。
## 一眼看懂
- **Topic / problem** LLM serving 的自动、低成本配置调优AITuner。当前主线问题用 simulator 给部署配置(并行度、批量上限等)排序,什么时候可信?需要补多少真机证据?算上这些成本还划算吗?
- **Central claim** simulator 要能帮助配置调优,必须先满足 scheduler transition 的 liveness/coverage再满足「配置相关残差小于真机 decision margin」前者决定 capacity 是否有定义,后者决定排序是否正确。(ID: C0)
- **当前结论:** 早先 35 个 trace stall 不是 Frontier scheduler liveness failureadapter 为不满 16-token 的 prefix block 错误生成了 cache identityFrontier 又没有 fail-fast。修正为完整 block、使用真实 graph buckets/KV blocks 和 `piecewise`/`KERNEL_ONLY` profile 后Qwen30 Trace-PD 的全部 12 个 cell 完成 129/129 requestFrontier 对 TTFT/TPOT/E2E 的 6 个 argmin 均与三次 fresh-server 真机一致;但绝对 latency 仍高估 4--511×。这只证明一个 MoE Trace-PD surface 的 selection fidelity不能外推到 prefill-only、fixed workload 或 235B。
- **最大 uncertainty / risk** 根因已收敛,且 overload 已被排除为统一解释但可信域边界仍未画清trace 面的 heterogeneity 是否让 closed-loop state residual 变小,还是当前 success 主要由 capacity/MNS margin 保护?两个降载点只建立 reference-config latency baseline不能证明新负载下全 surface 仍选对。
- **下一项 critical action** 不再做无锚点的 jitter 猜测;保持 request shape 不变,在预测的 MoE/MNS knee 两侧做小规模 rate sweep并用少量真机 state/batch anchor 验证 `λR(B)` fixed point。成功标准是同时预测 state-regime、排名与 knee而不只是某个点的 regret。
- **停止条件:** T1 出 verdict 且成本账本建立后pass 且摊销论证成立 → 转向「sim 剪枝 + 真机终选」的 hybrid 机制设计fail → 转入失败机制归因;两条路都无 insight 增量 → 收敛写作。
## 核心概念
- **Frontier** 本项目使用的 simulator属 Vidur 系(直接使用 vidur backend加自研 FP8/MoE/EP/decode-profile 兼容补丁。
- **Regret** 按 simulator 排序选配置,相对真机最优配置的性能损失百分比(以每 GPU capacity 计。primary metric排序选对则 regret=0。
- **τ-bKendall tau-b** simulator 排序与真机排序的秩相关1 = 完全一致1 = 完全反序0 = 无关tie-aware。
- **Decision margin** 真机上头部配置之间的性能差距,即 simulator 误差的容忍带。
- **Action-differential residual** simulator 误差中随配置action不同而不同的部分。Why needed所有配置统一偏移不影响排序只有差异化残差才可能穿过 margin 改变选择——这解释了「绝对误差 33%」与「排序全对」为何可以同时成立。
- **Capacity bracket** 真机 anchor 为候选配置的 capacity 划出的上下界「bracket 不反转」指未测的负载点不可能推翻 top 选择。
- **Decision-valid coverage** simulator 能从初始状态推进到所有请求完成,并为 config×workload cell 产生合法 SLO metric 的比例。若 reachable nonterminal state 没有 enabled transition/future eventcapacity 与 rank 都没有定义,不能把该 cell 当作 infeasible。
- **Workload realism 阶梯:** prefill-only无 decode→ fixed-shape mixed固定输入输出长度的混合负载→ trace-faithful mixed生产 trace 忠实回放。fidelity 结论不能向更高一级外推。
## Claim 层级
- **Central claim** 见「一眼看懂」。(ID: C0)
- **Subclaim** zero-shot 排序失败是真实现象。(ID: C1supported)
- 30B 纯 profile 驱动的 regret 为 25.63%(τ-b=0另一 throughput-proxy 评测口径下为 30.46%。Boundary均发生在 capacity-point + SLO-gated selection——恰是 Vidur 论文自己声明预测误差会爆炸、评测刻意回避的 regime见 claim map
- **Subclaim** 少量结构化的真机证据可以恢复低 regret 排序。(ID: C2)
- **Hypothesisdecision-bearing** trace-faithful 回放下,同栈 profile + 真机 KV capacity + 兼容补丁、且不做逐案例端到端校准的 Frontier能满足 gateregret ≤5% ∧ τ-b ≥0.8 ∧ bracket 不反转。(ID: H1weakened)
- **Supporting** 235B prefill-only regret=0235B fixed-shape mixed 的 top set 全中30B 加 per-TP 校准后 regret 0.76%(但这是外部端到端 scale 给出的上界,不是原生 profile 保真度)。
- **Counterevidence** 修正 prefix trace contract 后的 TP2/MNS16 `none`-graph run 完成但 p50 TPOT 约 96 ms真机为约 14 ms然而该比较尚未对齐 real vLLM 的 `FULL_AND_PIECEWISE` graph path。
- **下一项 discriminative experiment** 补齐 `KERNEL_ONLY` graph family并以 `piecewise` 重跑相同 trace若 full surface 仍错graph omission 不再是可用解释。
- **Hypothesis机制active** 误差机制是 action-conditioned residual——执行状态的转移并行拓扑、kernel family、graph mode、batch 组成)使按算子 profile 的组合预测跨配置不可复合;残差大于 margin 时排序失败。(ID: H2supported已细化)
- **Supporting** 三个 TP 档的端到端校准系数为 0.72/0.47/0.35残差确实随配置剧烈变化235B 的批量上限交互预测错误但被 2× margin 容忍30B prefill-only 在低负载近似对齐、饱和后按 TP 反向放大,最终 τ-b=1。
- **细化2026-07-20 统一普查):** 决策相关的残差分量集中在 TP 轴且由 decode 状态耦合产生——prefill-only 面的绝对 scale 仅 0.961.37× 且 measured collective 即可清除其 TP 反序,而含 decode 的面 scale 4.3130×、全部 material failure 都由 TP/mixed pair 决定。「residual>margin」只是必要条件失败还需要残差对准 winner-deciding pair。
- **机制 verdict2026-07-20** closed-loop state drift 是根因,离散阈值是其放大器而非 competing explanation。Q30 低压 exact ledger 显示同 state 的 TP 方向正确,但 TP4 被模拟 residence 反馈推到 batch 3--4MoE step 增长后反序Q30 高压进一步跨过 MNS admission capQ235 换成 exact real composition 后排序翻正。下一步从“找根因”转为测量 state-regime/knee 的可信边界。
- **Subclaim** 成本论证只有在摊销前提下成立。(ID: C3)
- **Hypothesisactive** 每个 model×硬件×runtime 的一次性对齐成本,摊销到大配置面、频繁重调(引擎版本 churn 的频率证据见 claim map或禁止在线实验的场景后低于重复真机调优。(ID: H3untested——分母已实测分子未入账)
- **下一步:** 建 cost ledger见「下一步」
## 当前 critical experiment
- **Question** 生产 trace 忠实回放prefix 打开、原始到达时间与会话结构best-effort Frontier 能否满足 low-regret gate
- **为什么现在做:** 这是 H1 的判决实验;所有已完成的机制分解都在人工 workload 上,不能替代这个 verdict。
- **当前状态:** Trace-PD 的 graph-aligned surface 已通过原负载 selection gate但绝对 latency 不通过 calibrationFixed-PD 的 failure 已定位为 closed-loop state drift。两个降载 Trace-PD anchor 已通过完成率/admission/backlog gate下一步需要 full surface rate sweep 才能检验 ranking 是否跨 load regime 保持。
- **Result → decision** 若其它 surface 排序失败,保留 Trace-PD success 为条件化 envelope并按 fixed/trace/prefill/decode 的差异定位 state composition若都通过才扩大到 Q235 或寻找 simulator 已解决范围之外的新问题。
- **Experiment card** [`../runs/frontier-fidelity-envelope-v1/experiment-card.md`](../runs/frontier-fidelity-envelope-v1/experiment-card.md)
## Key evidence最多 3 条)
- **E1否证「prefill-only 是充分 easy condition」支持 H2** 30B BF16、去掉 decode/prefix/混合 batch 后,真机最优是 TP48 vs 7 req/s/GPUsimulator 却把 TP4 排最差6 vs 8top set 无交集regret 12.5%,τ-b=1。产物`../runs/frontier-phase-factorial-v0/results/final/`dash012.07 H20-GPUh
- **E2统一机制普查material failure 全部由 decode 耦合的 TP 差异化误差决定,支持 H2 细化):** 对 14 个 frozen surface、70 个 case×objective 的方向化分解显示23 个 material failure 中 21 个由 TP/mixed pair 决定(仅 2 个 6.2% 边缘 mns-axis case、trace 面 TP 反序为零measured collectiveA1把 Qwen235 两个 prefill-only 面的 TP 反序清零trace-PO p90 regret 21.2%→0.3%)但对 Fixed-PD 的 33% 无效「residual>margin」仅为失败的必要条件。产物[`../runs/frontier-split-rootcause-v0/results/`](../runs/frontier-split-rootcause-v0/results/decomposition.md)(实验 card[`experiments/frontier-split-rootcause-s0-20260720.md`](experiments/frontier-split-rootcause-s0-20260720.md))。
- **E3closed-loop state 是 Fixed-PD 根因,而非同 state predictor 反序):** Q30 低压相同 batch=1 state 下 TP4 比 TP1 快约 1.00 ms/step但 TP4 own state 的 time-weighted batch=3.0437,使 step 增加 9.8197 ms其中 MoE +8.9297 ms并反序Q235 用 exact real composition 重放也把 TP8TP4 从错向 20.07 ms 翻为正确 +10.90 ms。Q30 高压再由 MNS cap 将同族 state/residence 误差放大成约 27 s 排队。产物:[`experiments/frontier-split-rootcause-s1-20260720.md`](experiments/frontier-split-rootcause-s1-20260720.md)。
## 下一步(最多 3 项)
- [ ] **画可信域边界:** 固定 request shape在预测的 MoE/MNS knee 两侧做最小 rate sweep只在判别点补真机 batch/state anchor验证 `B≈min(MNS, λR(B))` 是否同时解释 state 与 ranking。
- [ ] **Q235 portability gate** 先验证 vLLM0.20 TP4/TP8 FP8 runtime 和 deadc4a profile provenance再决定是否允许其 Fixed-P sweep。
- [ ] **建 cost ledger** parent H3完成标准 = 每 case 一行profiling GPU-h、补丁工时、校准探测、sim CPU-h与已实测的真机调优成本同表随每个 case 更新。
## Blocker 或 anomaly
- **当前运行状态:** 八 case load audit 的新增真机 run 已完成;未启动 full-surface rate sweep避免把两个 single-config anchor 外推成 ranking claim。自 2026-07-20 起,本任务只允许使用 `dash1`--`dash4`(每台 8×H20、最多四组并行`dash0` 保留给其他同事,不做 probe、同步或运行。
- **Anomaly保留** 235B pilot 中 simulator 把 10/34 个 anchor 误判为不可行——false-infeasible 是 H1 的主要威胁模式T1 分析时须单独报告。
- **平台边界(已更新):** 历史结果仍来自其各自 card 记录的平台,不改写 provenance后续实验平台切换为 `dash1`--`dash4`。跨主机比较前必须跑相同 canary 并量化 host effect。fixed-shape pilot 的主 SLOTPOT 40ms无判别力150ms 是事后明示的敏感性分析,不得写成盲选的 primary。
## Related work
- Claim map[`../docs/simulator-claim-map-20260716.md`](../docs/simulator-claim-map-20260716.md)。核心缺口capacity-point + SLO-gated selection 的 regret 无人用真机 ground-truth 面验证过alignment 成本无人与真机调优成本放进同一张表比较。

View File

@@ -1,10 +1,11 @@
# Project Operating Notes
## Remote experiment host
## Remote experiment hosts
- Default experiment machine: `dash0`.
- Hardware expectation: 8 NVIDIA H20 GPUs.
- SSH check: use `ssh dash0` before scheduling or debugging remote runs.
- Experiment machines: `dash1`, `dash2`, `dash3`, and `dash4`.
- Do not use or probe `dash0`; it is reserved for other users.
- Hardware expectation: 8 NVIDIA H20 GPUs per host.
- Before scheduling, probe only `dash1`--`dash4` and confirm all eight GPUs are idle and healthy.
- Remote project path: `/home/admin/cpfs/wjh/aituner/aituner`.
- If remote downloads are slow or fail, start the proxy from the remote `wjh`
home directory with `./auto_proxy.sh`, then run downloads in a shell where
@@ -13,7 +14,8 @@
## Local/remote sync workflow
- Treat this local repository and the `dash0` repository as the same project checkout.
- Treat this local repository and the `dash1`--`dash4` repositories as the same project checkout.
- Synchronize code through Git using `commit`, `push`, and `pull`.
- For remote experiments, commit local changes, push to `origin`, then pull on `dash0` in `/home/admin/cpfs/wjh/aituner/aituner` before running.
- For remote experiments, commit local changes, push to `origin`, then pull on each assigned host in `/home/admin/cpfs/wjh/aituner/aituner` before running.
- Up to four independent 8-GPU experiment groups may run in parallel, one group per host; pin every job explicitly to one of `dash1`--`dash4`.
- Do not ask for the remote host or project path again unless the user explicitly changes them.

View File

@@ -0,0 +1,179 @@
# Action-aware constraint pilot v0 protocol
Status: **FROZEN BEFORE NEW GPU RUNS**.
Date: 2026-07-14 (Asia/Singapore).
## Headline question
Can telemetry from one complete initial-config benchmark identify which of two
competing knob families should be changed, before either target configuration
is evaluated?
This pilot tests a narrow prerequisite, not an end-to-end tuner claim. It
uses fields already present in the per-step OpProf stream to reconstruct exact
zero-slack conditions for `max_num_seqs` (MNS) and
`max_num_batched_tokens` (MBBT). No new vLLM instrumentation is justified
unless those action-conditioned conditions predict crossed real-system
intervention responses.
## Hypothesis
I believe config-normalized scheduler constraints provide a stronger tuning
signal than an aggregate queue symptom because the same waiting queue can be
blocked by different admission limits.
I will verify it by holding model, hardware, TP, request bands, arrival times,
and offered load fixed while constructing two source configurations with
different binding constraints. From each source run alone, the larger
exclusive binding fraction predicts the action family. Both candidate
actions are then measured on the same requests for the full 300-second replay.
## Frozen platform and workload
- Host: `dash0`, solo placement on GPUs 0-3, four NVIDIA H20 GPUs.
- Model: Qwen3-30B-A3B BF16.
- Engine: patched vLLM `0.24.1.dev3+opprof`, TP=4.
- Workload: the three disjoint `mid` bands from
`chat_w20260312_1000`, 2.125 requests/s/GPU, 300-second arrival window,
exactly 128 output tokens.
- SLO: the unchanged study TTFT/TPOT thresholds and 0.95 pass-rate target.
- Every config starts one fresh server, performs the accepted 16-request
warm-up and the existing burn-in, then runs all three disjoint measured
bands in its frozen order.
- SLO early stopping is disabled. A measured run must drain all selected
requests and finish within the 450-second client deadline.
## Frozen configuration and action matrix
| ID | MNS | MBBT | Role |
|---|---:|---:|---|
| `b_base` | 64 | 256 | token-budget-bound source; operational gate runs first |
| `a_base` | 16 | 8192 | MNS-bound source |
| `shared` | 64 | 8192 | MNS action from A; MBBT action from B |
| `b_mns` | 128 | 256 | competing MNS action from B |
| `a_mbbt` | 16 | 16384 | competing MBBT action from A |
The two decisions are therefore:
```text
Regime A: a_base -> {shared (increase MNS), a_mbbt (increase MBBT)}
Regime B: b_base -> {b_mns (increase MNS), shared (increase MBBT)}
```
The candidate magnitudes are intentionally large in this feasibility pilot so
that a missing crossed response is not explained by an imperceptibly small
intervention. This does not establish that these are production step sizes.
Frozen config order is `b_base`, `a_base`, `shared`, `b_mns`, `a_mbbt`.
Frozen repetition orders are respectively `123`, `231`, `312`, `132`, and
`213`, reducing band/time alignment without reusing a server across configs.
## Pre-action signal
For each source run, let `waiting` include the normal and deferred waiting
queues, and let `scheduled_tokens = prefill_tokens + decode_tokens`.
```text
mns_exclusive = waiting > 0
and running == configured MNS
and scheduled_tokens < configured MBBT
mbbt_exclusive = waiting > 0
and scheduled_tokens == configured MBBT
and running < configured MNS
both = waiting > 0
and running == configured MNS
and scheduled_tokens == configured MBBT
```
Each score is the fraction of all scheduler records in the measured interval
that satisfies the condition. The predicted action is the family with the
larger exclusive fraction. This uses no target telemetry or target outcome.
KV usage and preemptions are reported as possible alternative constraints but
are not silently reassigned to either score.
These conditions reproduce two scheduler loop boundaries, but they are still
a Level-0 proxy: they do not expose the exact request rejected at the boundary
or run a shadow schedule. The pilot explicitly tests whether that additional
engine patch is warranted.
## Outcomes and baselines
Primary intervention outcome:
```text
SLO-goodput = full-run SLO pass count / 300-second arrival window
```
Also report pass rate, TTFT p50/p95/p99, TPOT p50/p95/p99, drain elapsed time,
KV usage, preemptions, queue area, and CUDA-graph padding.
Required decision baselines:
1. always choose the MNS family;
2. always choose the MBBT family;
3. queue-pressure-only, which has no candidate-specific score and therefore
must use one frozen family for both regimes;
4. the pre-action exclusive-binding prediction.
This is a mechanism ablation. It does not compare against a trained black-box
tuner because two regimes are not a valid training surface.
## Gates and failure meanings
Data validity requires 15 uncensored measured runs, exact request/arrival/input
hashes across each repetition, full request accounting, one continuous OpProf
stream per config, zero dropped records, monotonic timestamps and step indices,
nonnegative counters, bounded ratios, clean GPU placement, and config values in
the result matching the server command.
The crossed-response gate passes only if, in all three repetitions:
- the MNS target has higher SLO-goodput than the MBBT target in Regime A;
- the MBBT target has higher SLO-goodput than the MNS target in Regime B;
- each winning target exceeds its competing target by at least 10% of the
source SLO-goodput. A source with zero goodput makes the run invalid for
this relative gate rather than changing the denominator.
The binding gate passes only if, in both regimes:
- the predicted family matches the measured winning family in all three
repetitions;
- the median winning-family exclusive fraction is at least 0.10;
- it is at least 5x the median competing-family exclusive fraction;
- the direction is unchanged under cumulative 25%, 50%, 75%, and 100%
checkpoints after the 25% checkpoint.
Decision meanings:
- `STOP_WORKLOAD_NOT_CROSSED`: candidate outcomes do not have different
winners; the experiment cannot test action selection.
- `STOP_BINDING_NOT_PREDICTIVE`: outcomes cross but source-only constraint
scores do not select them; do not implement shadow scheduling from this
hypothesis.
- `STOP_NO_NEW_INSTRUMENTATION_NEEDED`: the signal works but every required
field was already present; keep it as an analysis/tuner feature and do not
claim a new engine-instrumentation contribution.
- `OPEN_EXACT_ATTRIBUTION_ABLATION`: the signal works but unresolved/both/KV
cases are material enough that exact rejection reasons could change a
decision. Only this result authorizes a minimal vLLM attribution patch.
Ambiguity is material only when, in either regime, the median
`both + waiting_unresolved` fraction is at least the median absolute gap
between the two exclusive fractions, or when any source run records a
preemption or median source KV maximum is at least 0.90. Otherwise all fields
needed for the observed decision were already present and the result is
`STOP_NO_NEW_INSTRUMENTATION_NEEDED`.
No result from this development pilot is a paper-level E2E tuning claim.
## Cost and stopping discipline
- Hard cap: 8.0 H20-hours, including failed sessions.
- Expected: 6.0-7.2 H20-hours and 90-110 minutes wall time.
- `b_base` runs first. If its first measured band cannot drain by 450 seconds,
the controller stops before any comparative analysis; MBBT=256 is then an
operationally invalid source, not negative evidence.
- Any data red flag stops analysis before computing a tuning conclusion.

View File

@@ -0,0 +1,44 @@
# Action-aware constraint pilot v1 amendment
Status: **FROZEN AFTER V0 OPERATIONAL STOP AND BEFORE V1 GPU RUNS**.
Date: 2026-07-14 (Asia/Singapore).
The complete claim, workload, baselines, metrics, action matrix, analysis gates,
and data-validity requirements remain those in the v0 protocol. This amendment
changes only the token-bound source severity and adds an operational burn-in
gate.
## Why v0 produced no comparative evidence
The first v0 session used MNS=64 and MBBT=256. During the 510-request,
60-second burn-in, the client had run for 197 seconds and the engine still held
64 requests: 13 running and 51 waiting. The last step scheduled exactly 256
tokens, KV usage was 0.01151, and there were zero preemptions. No measured run
or target configuration had started.
The session was stopped and cleanly released all GPUs after consuming
0.3859868995 H20-hours. This is evidence that MBBT=256 is a real token-budget
bottleneck, but it is not an admissible source for the 2.125 requests/s/GPU
comparison because it cannot sustain the offered load. V0 contributes no
tuning label and none of its runtime data is reused by V1.
Authoritative failure artifact:
`/home/admin/cpfs/wjh/action-aware-constraint-v0-20260714/operational-stop-v0.json`.
## V1 changes
- `b_base`: MNS=64, MBBT **2048** instead of 256.
- `b_mns`: MNS=128, MBBT **2048** instead of 256.
- The B-family MBBT action is therefore 2048 -> 8192.
- All five configurations and all three repetitions run fresh under a new V1
run root.
- Before any measured run, every config's 510-request/60-second burn-in must
drain in at most **90 seconds**. A slower config is an operational failure;
the controller stops before comparative analysis.
- V1 incremental hard cap is 7.6140131005 H20-hours so that V0 plus V1 remains
within the original global 8.0 H20-hour cap.
The V0 protocol's crossed-response and source-only binding gates are unchanged.
In particular, V1 still requires three-of-three action-family predictions in
both regimes and a different real winner in Regime A versus Regime B.

View File

@@ -0,0 +1,35 @@
# Action-aware constraint pilot v2 amendment
Status: **FROZEN AFTER V1 CONTROLLER STOP AND BEFORE V2 GPU RUNS**.
Date: 2026-07-14 (Asia/Singapore).
The V1 configuration matrix and every scientific gate remain unchanged. V2
fixes one controller bug and reruns every configuration and repetition fresh.
## V1 controller failure
The MNS64/MBBT2048 burn-in completed all 510 requests in 61.259 seconds, below
the frozen 90-second operational limit. However, the controller accidentally
assigned the preceding 16-request warm-up result to `burnin_result`; its state
therefore recorded 4.376 seconds and evaluated the wrong object.
The first measured replay was terminated after 75 seconds, before it produced
a result. No target configuration had started. V1 consumed
0.3184109431 H20-hours and contributes no action label or telemetry to V2.
## V2 correction and regression gate
- The completed burn-in `run_client()` return value is assigned to
`burnin_result`.
- A dedicated `burnin_gate()` rejects any non-anchor object, any request count
other than 510, and elapsed time above 90 seconds.
- Unit tests explicitly pass a warm-up object and require rejection, then test
accepted and over-limit burn-ins.
- All five configs and 15 measured runs use a new run root; no V0/V1 runtime
artifact is reused.
V0 and V1 together consumed 0.7043978426 H20-hours. V2's incremental hard cap
is therefore 7.2956021574 H20-hours, preserving the original global 8.0
H20-hour cap. The authoritative accounting file is
`runs/action-aware-v0/prior-attempts-v2.json`.

View File

@@ -0,0 +1,260 @@
# Action-aware constraint pilot v2 results
Date: 2026-07-14 (Asia/Singapore).
Decision: **`STOP_WORKLOAD_NOT_CROSSED`**.
The pilot produced one valid positive regime and one invalid-for-effect-size
regime. It supports continuing a narrower action-response investigation, but
it does not justify an end-to-end telemetry-guided tuner claim or new engine
instrumentation yet.
## Question tested
Given only a completed source run, can existing engine telemetry distinguish
which of two one-knob interventions will improve SLO-goodput more?
The frozen score counted scheduler steps with backlog where either MNS or MBBT
was exclusively at its configured limit. It made two pre-intervention
predictions on the same workload and offered load:
- Regime A: source `(MNS=16, MBBT=8192)` predicts increasing MNS to 64 over
increasing MBBT to 16384.
- Regime B: source `(MNS=64, MBBT=2048)` predicts increasing MBBT to 8192 over
increasing MNS to 128.
The primary outcome was 300-second SLO-goodput. A predicted action had to beat
the alternative on every paired request band by at least 10% of that band's
source goodput. Telemetry direction also had to remain stable at 25%, 50%,
75%, and 100% of the replay.
## Setup
- Host: `dash0`, GPU 0-3 used exclusively; four NVIDIA H20 GPUs; TP=4. GPU
4-7 remained idle to avoid co-location effects.
- Model: Qwen3-30B-A3B BF16 at
`/home/admin/cpfs/wjh/models/Qwen/Qwen3-30B-A3B`.
- Runtime: patched vLLM `0.24.1.dev3+g668cfb7e2`, source commit `4b253fd`, with
OpProf Layer-1 telemetry.
- Workload: `chat_w20260312_1000`, 2.125 requests/s/GPU, 300-second arrival
window, 128 output tokens, three disjoint request bands.
- Five fresh-server configurations, three measured runs each, counter-rotated
repetition order, 16-request warm-up, and a 510-request/60-second burn-in.
SLO early stopping was disabled.
- Exact request-id, arrival-order, and input-length hashes matched for every
paired comparison.
The authoritative run root is
`/home/admin/cpfs/wjh/action-aware-constraint-v2-20260714`. The final audit is
`pilot-audit-final.json`, SHA256
`7ebe080fcc4970bef423bc587253d157e75aed1ea8b410bd37770c17708135ab`.
## End-to-end result
### Regime A: strong MNS constraint
| Rep | Source 16/8192 | MBBT action 16/16384 | MNS action 64/8192 | MNS-only source steps | MBBT-only source steps | `(MNS action - MBBT action) / source` |
|---:|---:|---:|---:|---:|---:|---:|
| 1 | 4.710 | 7.687 | 8.500 | 79.85% | 0.038% | 17.27% |
| 2 | 2.803 | 4.150 | 8.500 | 94.61% | 0.005% | 155.17% |
| 3 | 2.227 | 3.943 | 8.500 | 79.87% | 0.005% | 204.64% |
Units are SLO-goodput requests/s except the percentage columns. The source
prediction was stable at every phase checkpoint and correct in all three
paired bands. The predicted MNS action cleared the frozen 10% material-margin
gate in all three bands.
This is a real but limited positive result. The source was an extreme case:
MNS was full on nearly every scheduler step that retained backlog, TTFT p50 was
1.64-5.61 seconds, KV usage remained below 2.45%, and no preemption occurred.
An expert or a simple rule could identify this case without a learned tuner.
### Regime B: MBBT direction at an outcome ceiling
| Rep | Source 64/2048 | MNS action 128/2048 | MBBT action 64/8192 | MBBT-only source steps | `(MBBT action - MNS action) / source` |
|---:|---:|---:|---:|---:|
| 1 | 8.423 | 8.420 | 8.500 | 10.90% | 0.950% |
| 2 | 8.447 | 8.500 | 8.500 | 12.03% | 0% |
| 3 | 8.500 | 8.497 | 8.500 | 8.79% | 0.039% |
The telemetry direction was stable in all phase checkpoints. The predicted
action won twice and tied once, while the wrong MNS action left the MBBT-only
state intact. However, the source already delivered 99.10-100% of the offered
8.5 requests/s. Even a perfect action could not reach the preregistered 10%
margin. Regime B therefore does not test material weak-signal value; it is a
workload-selection failure, not evidence that telemetry does or does not help
near a decision boundary.
The missing preflight condition is mathematical. For an effect threshold
`delta` and offered goodput ceiling `G`, the source must satisfy
`source <= G / (1 + delta)`. Here `G=8.5` and `delta=0.10`, so any source above
7.727 requests/s cannot possibly pass before either target is measured.
## Why the exclusive-limit rule is incomplete
The alternative MBBT action improved Regime A by 48.0%, 63.2%, and 77.1% over
the source even though MBBT was almost never the exclusive backlog constraint.
This rules out the binary interpretation "a knob that is not exclusively at
its cap cannot help."
Existing richer telemetry provides a plausible mechanism:
| Rep | Split-prefill requests, source -> MBBT action | Prefill steps, source -> MBBT action | Prefill requests/step, source -> MBBT action | Prefix-hit rate, source -> MBBT action |
|---:|---:|---:|---:|---:|
| 1 | 41 -> 2 | 2324 -> 2022 | 1.071 -> 1.249 | 13.851% -> 13.747% |
| 2 | 7 -> 0 | 2410 -> 2291 | 1.012 -> 1.076 | 13.078% -> 12.988% |
| 3 | 12 -> 1 | 2377 -> 2270 | 1.018 -> 1.096 | 13.613% -> 13.604% |
Increasing MBBT allows more prefill work to be packed into one iteration and
nearly eliminates split prefills. Under MNS=16, this can reduce the number of
iterations for which long prompts occupy scarce running slots. Prefix-cache
hit rates differ by at most 0.104 percentage points, and exact workload hashes
match, so neither explains the gain. Step-duration p99 also remains similar;
one 1.127-second decode-step outlier appears in `a_mbbt/rep2`, but the same
action direction occurs in all three bands.
This is a mechanism-consistent explanation, not a completed causal
decomposition. MBBT simultaneously changes total per-iteration token budget,
per-request chunk size, and multi-request packing. Instrumentation observes
their joint response but cannot separate those effects without another
intervention.
## Instrumentation decision
Do **not** add a new engine patch for this mechanism yet. The existing OpProf
stream already records submit/complete timestamps, prefill/decode composition,
chunked-prefill categories, prefix hits, queues, KV usage, and CUDA graph mode.
Those fields are sufficient to identify the interaction missed by the initial
exclusive-limit rule.
The next narrow mechanism ablation is available in the current vLLM runtime:
1. `(MNS=16, MBBT=8192, long-prefill-threshold=0)` is the current source.
2. `(MNS=16, MBBT=16384, long-prefill-threshold=8192)` keeps individual long
chunks at 8192 while increasing total packing budget.
3. `(MNS=16, MBBT=16384, long-prefill-threshold=0)` is the current MBBT action.
The runtime exposes `--long-prefill-token-threshold`; with a threshold of 8192,
the second arm separates total packing headroom from the larger per-request
chunk allowed by the third arm. A formal test must rerun all three arms fresh
with counter-rotated order rather than reuse today's endpoints.
## Correct tuning-research route
The pilot does not support turning the frozen equality checks into a larger
rule tree. The supported route is **intervention-calibrated, action-conditioned
system identification**:
```text
source event sequence + normalized config delta
-> predicted distribution of Delta SLO-goodput and evaluation cost
-> uncertainty-aware next-config selection
```
The policy input should retain continuous distributions and phase evolution:
queue/running residency, MNS and token slack, prefill/decode composition,
partial-prefill occupancy, step time, KV state, and graph behavior. Human
bottleneck labels and hand-authored `if queue then increase MNS` mappings are
not policy inputs. Mechanism summaries remain audit and interpretation tools;
the action response is learned from paired real interventions.
The harness has a narrower, non-heuristic role:
- define legal configurations and exact paired workloads;
- reject source points without outcome headroom before a full sweep;
- randomize/counter-rotate execution order and preserve failures/cost;
- validate stream coverage, hashes, request accounting, and censoring;
- expose target outcomes only after a source-only prediction is frozen;
- evaluate fixed-budget regret and H20-hours, not explanation quality alone.
The next tuning experiment should use non-extreme, non-ceiling source points
and a local two-dimensional MNS/MBBT neighborhood. A short run may screen load
only; every inferential telemetry and outcome result remains a 300-second run.
At least one held-out workload must be reserved before choosing features or
thresholds.
Primary evaluation is H20-hours/trials to reach 95% of the real local oracle
and cost-normalized regret AUC. Required baselines are random search,
config/outcome-only sequential search, the current rule heuristic, and the
same action-response model with telemetry removed. Action-ranking accuracy is
supporting evidence only.
## What this pilot establishes and does not establish
Established:
- Long-window engine state can make a correct, phase-stable action-family
prediction in an extreme MNS-constrained regime.
- A naive `queue > 0 -> increase MNS` rule would choose an ineffective action
in Regime B; action-conditioned state distinguishes the mechanism direction,
although the measured effect is immaterial at the selected load.
- Binary exclusive-cap attribution misses a substantial MNS/MBBT interaction;
existing chunk/step telemetry reveals a plausible explanation.
- Source outcome headroom must be an explicit experiment admission gate.
Not established:
- telemetry improves an end-to-end tuner over an outcome-only baseline;
- weak or mixed constraints can be ranked with material gain;
- the response transfers across workloads, models, TP, or hardware;
- new engine instrumentation is necessary;
- the chunking/packing breakdown is causal rather than mechanism-consistent.
## Change and verification
Reproduction:
```bash
python3 runs/action-aware-v0/test_pilot.py
python3 runs/action-aware-v0/analyze_pilot.py \
--run-root /home/admin/cpfs/wjh/action-aware-constraint-v2-20260714/runs/pilot \
--manifest runs/action-aware-v0/pilot-manifest-v2.json \
--output /home/admin/cpfs/wjh/action-aware-constraint-v2-20260714/pilot-audit-final.json
```
The fresh GPU run used AITuner commit `c5ab073`; asynchronous coverage was
corrected in `3facb18`; reproducible mechanism summaries were added in
`2af22db`. The raw run is unchanged across those analyzer-only commits.
Change: added a crossed real-intervention controller and audit, fixed the
burn-in result gate, corrected asynchronous per-step coverage accounting, and
added reproducible step/chunk/prefix mechanism summaries.
Expected effect: distinguish descriptive telemetry from source-only action
predictions that survive paired real interventions.
Verification: local and remote action-aware test suites pass; all five sessions
completed; all stream/footer and request-accounting invariants pass; the final
analyzer was run twice and produced byte-identical output.
Result: Regime A passes; Regime B is invalid for the frozen effect-size test;
the global decision is `STOP_WORKLOAD_NOT_CROSSED`.
Remaining risk: one model, one TP, one trace family, three bands, two action
families, and deliberately constructed endpoints are development evidence
only. The strong positive regime is too obvious to support a paper claim.
## Data sanity
- Measured runs: n=15; elapsed 300.610-317.350 seconds; 15 distinct. Pass
rate min/max 0.2620/1.0 with 11 distinct values; SLO-goodput min/max
2.2267/8.5 requests/s with 11 distinct values.
- Telemetry intervals: n=15; records min/max 13,621/23,711; 15 distinct.
Start gaps min/max 0.0412/0.1227 seconds; end gaps 0.00053/0.0649; uncovered
internal gaps 0/0.3231. One submit gap reached 1.1190 seconds but was fully
covered by a 1.1269-second recorded execution; contiguous indices and zero
drops were preserved.
- Sessions: n=5; cost min/max 1.1691/1.2730 H20-hours; 5 distinct. V2 cost
was 6.0862 H20-hours; V0/V1/V2 total was 6.7906, below the 8.0 cap.
- Regime-A split-prefill observations: n=6; min/max 0/41 requests; 6 distinct.
Prefix-hit rates: n=6; min/max 0.12988/0.13851; 6 distinct.
- Checked invariants: non-negative counters and durations; ratios in `[0,1]`;
exact request, arrival, and length hashes; 2550/2550 request accounting per
measured run; uncensored outcomes; outcomes across configurations not all
identical;
five complete streams; monotonic timestamps; contiguous step indices; zero
drops; footer/sidecar agreement; chunk-token accounting; bounded prefix hits;
no OOM, controller error, or residual GPU allocation. No unresolved red
flag remains. The three identical shared goodputs are reported as the
offered-load ceiling, not treated as independent performance variation.

View File

@@ -0,0 +1,136 @@
# Active intervention + measurement v0 protocol
Date: 2026-07-15 (Asia/Singapore)
Status: **FROZEN BEFORE THE `chat_w20260313_1000` GPU RUN**.
## Research question
This experiment asks whether a tuner conditioned on direct engine-state
trajectories can choose both a measurement horizon and a coupled configuration
intervention with lower real-GPU cost than the same tuner using only external
prefix outcomes.
The contribution is not the controller, legality checks, telemetry collection,
or the ridge model. The route remains open only if engine state changes an
actual decision and reduces cost-to-near-oracle on unseen workloads.
## Development result that motivates, but does not pass, the route
The frozen trace-12 dataset contains 72 examples: six source decisions, four
measurement checkpoints, and `noop/MNS/MBBT` actions. Features are direct
continuous Layer-1 state summaries; cap-exclusive and bottleneck labels are
excluded. Leave-one-repetition-out sequential replay uses the same model,
candidate set, confidence rule, and checkpoint set for both modes.
The external-outcome policy and telemetry policy both put all six decisions
within 2% regret. Outcome-only selected a mean 262.5-second source measurement
and cost 3.750 replay H20-hours across the six replayed decisions; telemetry
selected 275 seconds and cost 3.833 H20-hours. Telemetry therefore increased
the replay lower-bound cost by 2.22%, with no regret reduction. This is a
negative result. It does not settle the question because the dataset has only
two source regimes, one source is at the offered ceiling, and there is no joint
MNS+MBBT action.
Sanity: n=6 decisions; regret min=0, max=0.009412, distinct=3; source cutoff
min=150s, max=300s, distinct=3 across the two policies; all costs are
non-negative, regrets are in `[0,1]`, target results are not all identical, and
the six decisions are complete exact-workload pairs.
## Frozen prospective setup
- Host: `dash0`, 8 NVIDIA H20 GPUs available; each TP4 server runs alone on
GPUs 0-3. Co-location is prohibited for SLO verdicts.
- Engine: patched vLLM `0.24.1.dev3+g668cfb7e2` from clean source commit
`4b253fd8619764b6971a7f2e3a3aa7545f6ace05` at
`/home/admin/cpfs/wjh/opprof-phase2-dash0-20260711/vllm-v0.24.0`, using
`/tmp/wjh-opprof-phase2-dash0-20260711/.venv`.
- Model: `/home/admin/cpfs/wjh/models/Qwen/Qwen3-30B-A3B`, BF16.
- Workload: unseen `chat_w20260313_1000`; input 0-8192; output exactly 128;
replay scale 0.5; 300-second arrival window.
- Three disjoint repetitions: source rows are assigned by a deterministic
SHA-256 modulo-3 partition before input filtering. Each repetition selects
approximately 3300 requests, or 2.75 requests/s/GPU at TP4.
- SLO: at least 95% pass; stepped TTFT 2/4/6 seconds; TPOT at most 50 ms.
- Checkpoints: 75, 150, 225, and 300 seconds.
- Full 2x2 surface:
- source: `MNS=32, MBBT=4096`;
- MNS action: `64,4096`;
- MBBT action: `32,8192`;
- joint action: `64,8192`;
- `noop` retains the source.
- Four config sessions are serialized. Each session uses a fresh server,
warm-up, burn-in, and counter-rotated repetition order.
- Expected campaign cost: 4.6-5.5 H20-hours; hard cap: 6.0 H20-hours;
expected wall time: 75-100 minutes.
The source is executed first. The frozen telemetry policy selects the next
real config session; all remaining cells are then measured only to construct
the exact finite-surface oracle. Oracle annotation after the selected action
is reported separately from tuner cost.
## Frozen policies
Both policies fit the paired treatment effect
```text
target normalized SLO-goodput - source normalized SLO-goodput
```
from source config, full config delta, offered load, and external prefix
outcomes. The telemetry policy additionally receives fixed direct Layer-1
summaries and their interactions with `delta_log2(MNS)` and
`delta_log2(MBBT)`. It does not receive a bottleneck label or a
diagnosis-to-knob rule.
At each checkpoint, jackknife models produce an effect distribution for
`noop`, MNS, MBBT, and joint actions. Measurement stops at the earliest second
consecutive checkpoint with the same confident best action; otherwise it uses
the full 300 seconds. Confidence requires a predicted margin of at least 0.02
and the best lower bound to exceed the second-best upper bound. If the final
choice is not confident, the next run is the positive-UCB action, explicitly
marked as a diagnostic intervention. The exact same rule is used for the
outcome-only baseline.
## Hypotheses and gates
### H1: action value
Engine state must change the selected intervention or its ranking and reduce
real action regret. Prediction error or bottleneck-label accuracy is not a
success metric.
### H2: measurement value
Engine state must select a shorter stable source measurement without increasing
action regret. A shorter reconstructed prefix is only a trigger; it is not an
actual GPU-cost claim until an early-terminated confirmation run measures
startup, warm-up, drain, and cleanup.
### H3: end-to-end cost
Primary development metric is H20-hours to first reach a configuration within
2% of the exact median-goodput oracle. The outcome-only and telemetry policies
use the same measured config costs and differ only in source information.
- At least 10% prospective replay cost reduction, telemetry regret at most 2%,
and no outcome-only-to-telemetry harm triggers an actual early-stop
confirmation.
- At least 20% measured all-in H20-hour reduction is required for a contribution
claim. This one task can only establish development feasibility; a paper
claim additionally requires task-held-out replication.
- Source median normalized goodput at or above 0.98 stops the surface before
target runs because the workload has no material improvement headroom.
- Any hash mismatch, missing/censored result, telemetry drop, non-monotonic
phase, negative cost, ratio outside `[0,1]`, or all-identical config outcomes
is a red flag and stops analysis.
If the 10% trigger fails, this route is closed for the current engine-state
representation. The experimental control plane is not retained as a fallback
research contribution.
Pre-run provenance amendment: the first controller dry-run on 2026-07-15
rejected two stale engine paths before starting a server. The paths and exact
runtime version above were recovered from the accepted trace-12 campaign and
corrected before any trace-13 GPU work. No scientific treatment or gate was
changed.

View File

@@ -0,0 +1,106 @@
# Active intervention v0: held-out trace-13 result
Date: 2026-07-15 (Asia/Singapore)
Decision: **close the passive-telemetry treatment-effect route**. The held-out
campaign produced no telemetry-induced action change, measurement reduction, or
GPU-cost reduction. It did show that the engine state contained the correct
action-specific mechanism; the current feature model failed to use it.
## Headline result
The outcome-only and telemetry policies both measured the source for 300
seconds, selected `joint=(MNS64,MBBT8192)`, and produced the same complete
acquisition order. Both reached the exact finite-surface oracle after the
first intervention at a reconstructed all-in lower-bound cost of 2.4284
H20-hours. Telemetry GPU-cost reduction was therefore exactly 0%, below the
10% confirmation trigger and 20% contribution gate. No actual early-stop
confirmation was launched.
The complete annotation campaign cost 5.0379 H20-hours, below the 6.0 H20-hour
hard cap. It ran 12 uncensored real-GPU outcomes: four configs, three disjoint
request partitions, and a fresh server per config.
## Exact response surface
Median normalized SLO-goodput was:
| Config | Rep values | Median |
|---|---|---:|
| `MNS32,MBBT4096` source | 0.40091 / 0.39788 / 0.42061 | 0.40091 |
| `MNS64,MBBT4096` | 1.00000 / 0.99970 / 1.00000 | 1.00000 |
| `MNS32,MBBT8192` | 0.44394 / 0.41515 / 0.42606 | 0.42606 |
| `MNS64,MBBT8192` joint | 1.00000 / 1.00000 / 1.00000 | 1.00000 |
Increasing MNS alone was sufficient and joint was redundant. Increasing MBBT
alone improved the median by only 0.02515, versus 0.59909 for MNS. This is a
strong non-additive action response, not a setting where independently tuning
the knobs and merging their improvements is valid.
## What the telemetry actually said
Across 41,086 source scheduler records, 93.12% of steps had waiting work,
85.36% were MNS-exclusive binding, 1.11% were MBBT-exclusive, mean running-slot
utilization was 97.39%, mean token-budget utilization was 15.69%, mean KV usage
was 2.75%, and there were no preemptions.
The intervention transition agreed with that state:
| Config | Waiting | MNS-exclusive | MBBT-exclusive | Median goodput |
|---|---:|---:|---:|---:|
| source | 93.12% | 85.36% | 1.11% | 0.40091 |
| MNS only | 5.38% | 0% | 5.38% | 1.00000 |
| MBBT only | 91.19% | 91.09% | 0.04% | 0.42606 |
| joint | 0.89% | 0% | 0.89% | 1.00000 |
Thus this experiment does **not** support the claim that engine telemetry lacks
tuning information. It rejects the narrower claim that adding passive state
summaries to the current small-data ridge policy converts that information into
lower tuning cost.
## Why the learned policy failed
At 300 seconds, the telemetry model predicted joint, MNS, and MBBT effects of
0.35190, 0.26118, and 0.09686. The actual median effects were 0.59909,
0.59909, and 0.02515. Telemetry therefore made the nonexistent joint-over-MNS
gap larger: 0.09072 predicted versus 0 actual; the outcome-only model predicted
0.03188.
The failure has three concrete causes:
1. The six training decisions contain no joint intervention. The
`delta_product` feature has no support, so joint ranking is extrapolation.
2. Passive raw summaries do not represent the counterfactual scheduler work
unlocked by each action. Capacity-normalized MNS pressure was visible, but
the model was not structurally required to map it to MNS marginal value.
3. The policy maximizes predicted effect. It does not identify the smallest
epsilon-optimal intervention or price unsupported action complexity.
## Research implication
Do not retain the harness or the passive telemetry model as a contribution.
The next defensible route is engine-native, action-conditional counterfactual
instrumentation: at a real scheduling state, shadow-replay the exact scheduler
decision under an MNS relaxation, MBBT relaxation, and their joint relaxation,
then expose the incremental queued work admitted by each action. Real paired
interventions calibrate how those one-step shadow effects map to E2E SLO
goodput. This is distinct from a hand-written cap-to-knob rule and from a
full-system simulator: it reuses the exact live queue, scheduler, and cache
state while simulating only the local decision boundary.
That route should be evaluated against outcome-only search, the present passive
telemetry model, a cap-hit expert rule, and a full simulator. The paper-level
gate remains at least 20% measured H20-hour reduction to a 2%-oracle config on
task-held-out workloads with at most 2% regret.
## Sanity
Surface outcomes: n=12, min=0.39788, max=1.0, distinct=8. Session costs: n=4,
min=1.1702, max=1.3566 H20-hours, distinct=4. Scheduler-record counts: n=4,
min=37,001, max=41,348, distinct=4. All counters and costs were non-negative;
all ratios were in `[0,1]`; request hashes matched; all 12 runs were uncensored;
the controller and four sessions completed; and config outcomes were not all
identical. No red flags were found.
Machine-readable summary: `runs/active-intervention-v0/trace13-results.json`.
Raw immutable root: `/home/admin/cpfs/wjh/active-intervention-prospective-20260715`.

View File

@@ -0,0 +1,60 @@
{
"schema": "simulator-fidelity-figure-data-v1",
"objective": "maximum_tested_slo_feasible_offered_request_rate_per_gpu",
"qwen30_mixed": {
"sources": {
"real": "recovered-stores/aituner-interaction-runs-dash1-20260710/interaction-mixed-qwen30b-tp-mns-surface-high1-dash1-d8899c5-20260701T095858Z",
"comparison": "/home/gahow/phd/replayserve/runs/simfid_s2rb/results/metrics.json"
},
"configs": [
{"name": "tp1_mns8", "tp": 1, "mns": 8, "real": 2.1, "frontier_profile_only": 1.1, "frontier_calibrated": 1.7166666666666666},
{"name": "tp1_mns16", "tp": 1, "mns": 16, "real": 2.35, "frontier_profile_only": 1.1, "frontier_calibrated": 2.3833333333333333},
{"name": "tp1_mns32", "tp": 1, "mns": 32, "real": 2.283333333333333, "frontier_profile_only": 1.1, "frontier_calibrated": 2.3833333333333333},
{"name": "tp1_mns64", "tp": 1, "mns": 64, "real": 2.283333333333333, "frontier_profile_only": 1.1, "frontier_calibrated": 2.3833333333333333},
{"name": "tp2_mns8", "tp": 2, "mns": 8, "real": 2.275, "frontier_profile_only": 0.0, "frontier_calibrated": 1.7416666666666667},
{"name": "tp2_mns16", "tp": 2, "mns": 16, "real": 2.275, "frontier_profile_only": 1.1916666666666667, "frontier_calibrated": 2.3},
{"name": "tp2_mns32", "tp": 2, "mns": 32, "real": 3.283333333333333, "frontier_profile_only": 0.0, "frontier_calibrated": 3.75},
{"name": "tp2_mns64", "tp": 2, "mns": 64, "real": 3.2583333333333333, "frontier_profile_only": 0.0, "frontier_calibrated": 3.75},
{"name": "tp4_mns8", "tp": 4, "mns": 8, "real": 1.2833333333333334, "frontier_profile_only": 0.0, "frontier_calibrated": 1.3208333333333333},
{"name": "tp4_mns16", "tp": 4, "mns": 16, "real": 2.441666666666667, "frontier_profile_only": 0.0, "frontier_calibrated": 2.5},
{"name": "tp4_mns32", "tp": 4, "mns": 32, "real": 2.441666666666667, "frontier_profile_only": 1.3208333333333333, "frontier_calibrated": 2.5},
{"name": "tp4_mns64", "tp": 4, "mns": 64, "real": 2.441666666666667, "frontier_profile_only": 1.3208333333333333, "frontier_calibrated": 2.5}
],
"profile_only_metrics": {
"kendall_tau_b": 0.0,
"pairwise_exact_sign_accuracy": 0.3787878787878788,
"simulator_top_set": ["tp4_mns32", "tp4_mns64"],
"real_top_set": ["tp2_mns32"],
"top1_regret_worst": 0.25634517766497456
},
"calibrated_metrics": {
"kendall_tau_b": 0.9668009539030813,
"pairwise_exact_sign_accuracy": 0.9393939393939394,
"simulator_top_set": ["tp2_mns32", "tp2_mns64"],
"real_top_set": ["tp2_mns32"],
"top1_regret_best": 0.0,
"top1_regret_worst": 0.0076142131979695165
}
},
"qwen235_prefill": {
"source": "runs/frontier-multicase-sufficiency-v0/best_effort/fixed_cohort_evidence/v2_refined_comparison.json",
"configs": [
{"name": "tp4_mns64_mbt8192", "tp": 4, "mns": 64, "mbt": 8192, "expert_parallel": false, "real": 0.05, "frontier": 0.0375},
{"name": "tp4_mns128_mbt8192", "tp": 4, "mns": 128, "mbt": 8192, "expert_parallel": false, "real": 0.05, "frontier": 0.0375},
{"name": "tp4_mns64_mbt16384", "tp": 4, "mns": 64, "mbt": 16384, "expert_parallel": false, "real": 0.075, "frontier": 0.0625},
{"name": "tp4_mns128_mbt16384", "tp": 4, "mns": 128, "mbt": 16384, "expert_parallel": false, "real": 0.075, "frontier": 0.0625},
{"name": "tp8_mns64_mbt8192", "tp": 8, "mns": 64, "mbt": 8192, "expert_parallel": true, "real": 0.05625, "frontier": 0.05},
{"name": "tp8_mns128_mbt8192", "tp": 8, "mns": 128, "mbt": 8192, "expert_parallel": true, "real": 0.05625, "frontier": 0.05},
{"name": "tp8_mns64_mbt16384", "tp": 8, "mns": 64, "mbt": 16384, "expert_parallel": true, "real": 0.05625, "frontier": 0.05625},
{"name": "tp8_mns128_mbt16384", "tp": 8, "mns": 128, "mbt": 16384, "expert_parallel": true, "real": 0.05625, "frontier": 0.05625}
],
"metrics": {
"spearman_rank_correlation": 0.9486832980505138,
"pairwise_non_tied_accuracy": 1.0,
"comparable_non_tied_pairs": 20,
"simulator_top_set": ["tp4_mns64_mbt16384", "tp4_mns128_mbt16384"],
"real_top_set": ["tp4_mns64_mbt16384", "tp4_mns128_mbt16384"],
"top1_regret_worst": 0.0
}
}
}

Binary file not shown.

After

Width:  |  Height:  |  Size: 133 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 152 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 183 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 106 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 216 KiB

View File

@@ -1,10 +1,16 @@
# Fidelity-aware harness headroom audit
Status: **PROMISING PREMISE, NO CONTRIBUTION CLAIM**.
Status: **HISTORICAL PREMISE DID NOT PASS PROSPECTIVE P1; NO CONTRIBUTION CLAIM**.
The audit answers whether engine instrumentation has enough incremental signal
to justify a prospective experiment. It does not establish generalization.
Post-run update: the exact-timestamp held-out P1 completed and failed the
registered gate. Under the stronger simulator-aware `k=2` end-to-end replay,
telemetry preserved zero regret but saved only 1.426% online H20-hours versus
sim top-k + real final. The current route is closed; see
`docs/fidelity-aware-harness-p1-report-20260714.md`.
## Simulator shortlist lower bound
On the frozen 12-cell SimFid task, the strongest calibrated SLO simulator
@@ -46,6 +52,35 @@ at 15 seconds it is 88.89% versus 91.67%; at 20 seconds it is 86.11% versus
91.67%, but both 0.95 policies make one false reject. Five seconds is therefore
a training-selected operating point, not a test result.
## Strong simulator-aware calibration baseline
The original nested comparison used the same simulator shortlist but did not
put Frontier's per-anchor prediction in either model. A stronger retrospective
audit now gives both models frozen-calibrated simulated throughput, simulated
SLO pass rate, and simulated feasibility. Under the same leave-one-cell-out
folds, 5-second cutoff, L2 logistic family, regularization 1.0, and threshold
0.95:
| Metric | Sim + outcome | Sim + outcome + instrumentation | Delta |
|---|---:|---:|---:|
| Accuracy | 81.08% | 89.19% | +8.11 pp |
| Balanced accuracy | 72.42% | 81.55% | +9.13 pp |
| Brier score | 0.1058 | 0.0957 | -0.0101 |
| Safe early decisions | 20/37 | 25/37 | +5 |
| Valid full-trial cost reduction | 50.89% | 68.98% | +18.09 pp |
| Residual verification H20-hours | 0.5240 | 0.3310 | -36.84% |
Both 0.95 policies have zero false accept and zero false reject on this
retrospective task. Only three 0.5-threshold classifications differ in favor
of instrumentation and none in favor of the strong baseline; McNemar's exact
two-sided p-value is 0.25. The cell-bootstrap accuracy-delta interval is
`[0.00,+18.18]` percentage points. The result is not robust to regularization:
at 0.1 the strong baseline is more accurate and the instrumentation policy
makes two unsafe decisions; at 10.0 the strong baseline is also more accurate.
Thus the stronger comparison still has enough point-estimate headroom for a
held-out test, but it materially weakens the evidence and makes a prospective
task-level result mandatory.
## Interpretation
There is enough headroom to run a held-out pilot, but not enough evidence to
@@ -73,6 +108,9 @@ with three full repetitions. The registered protocol is
- `runs/fidelity-headroom/prefix-metrics.json`
- `runs/fidelity-headroom/test_analysis.py`
- `runs/fidelity-headroom/test_prefix_analysis.py`
- `runs/fidelity-headroom/analyze_strong_baseline.py`
- `runs/fidelity-headroom/strong-baseline-metrics.json`
- `runs/fidelity-headroom/test_strong_baseline.py`
## Sanity block
@@ -86,6 +124,8 @@ with three full repetitions. The registered protocol is
| Outcome probability | 37 | in `[0,1]` | in `[0,1]` | >1 | Checked before metrics |
| Instrumentation probability | 37 | in `[0,1]` | in `[0,1]` | >1 | Checked before metrics |
| Layer-1 streams | 12 | 14,174 records | 58,725 records | 12 | Contiguous, zero drops |
| Matched frozen simulator anchors | 37 | pass rate 0.0688 | pass rate 1.0 | 12 pass-rate values | Every prefix matched exactly once |
| Frozen simulator anchor corpus | 92 | positive throughput | positive throughput | >1 | No duplicate cell/anchor run |
Checked invariants: same folds/model family and cutoff; no full verdict in a
feature; prefix-only Layer-1 slicing; non-negative costs/counters; bounded

View File

@@ -0,0 +1,242 @@
# Fidelity-aware harness P1 result
Status: **REGISTERED ROUTE REJECTED; DO NOT OPEN P2/P3 FOR THE CURRENT METHOD**.
Date: 2026-07-14 (Asia/Singapore).
## Outcome
The registered five-second instrumentation-aware verifier did not pass P1.
The stronger simulator-aware comparison also failed the independent
contribution bar. On the frozen `k=2` end-to-end replay:
- `sim top-k + real final` selected the real oracle with zero regret;
- instrumentation-aware also selected the oracle, but reduced online H20-hours
by only **1.426%** (1.329% when the prior failed attempt is added to both);
- the required reduction was 30% versus full real final and 20% versus a safe
outcome-only calibrator;
- the outcome-only calibrator was not safe: it rejected the true best cell, so
its apparent cost saving is not a deployable comparison.
This rejects the claim that the **current joint logistic verifier**, trained on
one historical workload, gives the harness an independent tuning contribution.
It does not prove that engine telemetry contains no useful signal. Telemetry
improved held-out classification and removed unsafe decisions, but did not turn
that signal into meaningful end-to-end tuning-cost reduction.
## Frozen setup
- Host: `dash0`, 8 NVIDIA H20 GPUs; cells were serialized and used TP1, TP2,
or TP4 without co-resident serving jobs.
- Engine/model: patched vLLM 0.24.1.dev3, Qwen3-30B-A3B BF16.
- Workload: held-out `chat_w20260312_1000`, seven disjoint repeat bands,
60-second replay after 0.1 time scaling, input `[0,8192]`, exactly 128 output
tokens.
- SLO: stepped TTFT 2/4/6 seconds, TPOT 50 ms, request pass rate at least 0.95.
- Cells: TP1/MNS8, TP1/MNS64, TP2/MNS8, TP2/MNS64, TP4/MNS16, TP4/MNS64.
- Per cell: burn-in, three low-rate repeats, and three high-rate repeats. The
first repeat supplied the five-second prefix; 2-of-3 supplied its label.
- Models: the registered pair used config/workload/outcome versus the same
vector plus Layer-1 engine telemetry. The strengthened pair additionally
gave both models identical frozen Frontier throughput, SLO pass-rate, and
feasibility predictions.
- Policy: accept at `p>=0.95`, reject at `p<=0.05`, otherwise continue the same
trial. Model, cutoff, threshold, role order, request hashes, and cap were
frozen before their applicable evaluation.
The first launch failed its warm-up input-count validation before a measured
anchor. It cost 0.020552 H20-hours. The corrected primary attempt cost
1.722112 H20-hours, so aggregate campaign cost was **1.742664 H20-hours**, below
the 3.5 cap. The fix changed only warm-up validation; formal request counts and
hash checks were unchanged.
## P1 labels are not an artificial easy split
The 12 adjudicated anchor labels contain 7 feasible and 5 infeasible examples.
They are not simply “low feasible, high infeasible”:
- TP2/MNS64 high was feasible in all three repeats;
- TP4/MNS64 low and high were feasible in all six repeats;
- TP4/MNS16 low and high were infeasible in all six repeats.
That last pair creates a large real MNS interaction under an otherwise matched
TP4 configuration. Frontier correctly predicted TP4/MNS64 high as feasible,
but incorrectly predicted TP4/MNS16 low as feasible. It also incorrectly
predicted TP1/MNS64 high as feasible. Overall simulator-only feasibility was
10/12 correct: 83.33% accuracy, with two false-feasible predictions and no
false-infeasible prediction.
The two false-feasible cases expose the intended latent-state problem. At five
seconds, all 26 completed TP4/MNS16-low requests and all 9 completed
TP1/MNS64-high requests still passed their SLO, although both full anchors were
infeasible. External outcomes had not yet exposed the future failure; queue,
running-batch, and scheduler state existed before the tail outcome. This is
mechanistic evidence that instrumentation can be useful, not evidence that the
current learned policy uses it well enough.
## Registered and strengthened prefix results
At the frozen 0.95 policy threshold:
| Comparison | Accuracy | Balanced acc. | Early decisions | False accept | False reject | Valid primary-trial saving |
|---|---:|---:|---:|---:|---:|---:|
| Registered outcome-only | 41.67% | 50.00% | 6/12 | 0 | 2 | invalid |
| Registered + telemetry | 66.67% | 71.43% | 4/12 | 0 | 0 | 11.44% |
| Strong sim + outcome | 66.67% | 68.57% | 5/12 | 0 | 1 | invalid |
| Strong sim + outcome + telemetry | 83.33% | 85.71% | 4/12 | 0 | 0 | 11.44% |
For the strong pair, telemetry was correct on two examples where the baseline
was wrong and lost none; McNemar's exact two-sided p-value is 0.5 at `n=12`.
This is a safety/classification improvement, not a cost contribution. The
registered instrumentation policy made two fewer early decisions than its
baseline, so it failed the registered `+3 decisions or +15 percentage points`
incremental gate.
The result is not robust to the frozen regularization sensitivity:
| L2 lambda | Sim+outcome acc. | +telemetry acc. | Base policy errors | Telemetry policy errors | Base saving | Telemetry saving |
|---:|---:|---:|---:|---:|---:|---:|
| 0.1 | 41.67% | 75.00% | 4 | 2 | invalid | invalid |
| 1.0 | 66.67% | 83.33% | 1 | 0 | invalid | 11.44% |
| 10.0 | 83.33% | 83.33% | 0 | 0 | 0.00% | 5.98% |
Consequently the positive classification delta is neither statistically nor
hyperparameter robust.
## End-to-end shortlist result
Frontier's simulator-feasible ranking on the tested P1 surface was:
| Rank | Cell / anchor | Sim throughput/GPU | Real feasible | Real offered goodput/GPU |
|---:|---|---:|---:|---:|
| 1 | TP4/MNS64 high | 3.0718 | yes | 3.1250 |
| 2 | TP1/MNS64 high | 2.8823 | no | 2.9833 |
| 3 | TP2/MNS64 high | 2.8096 | yes | 2.8750 |
| 4 | TP4/MNS16 low | 2.0866 | no | 2.1250 |
| 5 | TP1/MNS8 low | 1.9806 | yes | 2.0333 |
| 6 | TP2/MNS8 low | 1.8637 | yes | 1.9083 |
The real oracle was TP4/MNS64 high at 3.125 req/s/GPU. Cost includes an
inferred per-cell startup/warm-up/burn-in/cleanup component and the selected
anchor; benchmark-only 2-of-3 annotation intervals are removed. Gaps around
annotation runs remain in the shared setup term, making this a conservative
method-cost estimate.
| Frozen policy (`k=2`) | Online H20-hours | + prior failure | Real regret | Safety | Saving vs full |
|---|---:|---:|---:|---|---:|
| Sim top-2 + real final | 0.281383 | 0.301935 | 0.00% | valid | — |
| Sim + outcome prefix | 0.214664 | 0.235216 | no selected cell | 1 false reject | invalid |
| Sim + outcome + telemetry | 0.277370 | 0.297922 | 0.00% | valid | **1.426%** |
For `k=1`, the simulator already picked the oracle and telemetry abstained, so
its saving was exactly 0%. For `k=3`, telemetry remained correct but saved
1.007%. The result therefore does not depend on an unfavorable choice of `k`.
Across all six cells, a generous replay that removes benchmark-only repeat
intervals estimates 0.944244 H20-hours for the full online workflow. The
instrumentation policy saved 0.043467 H20-hours: 11.44% of primary trial time,
but only **4.603%** of end-to-end online cost. Even a post-hoc oracle symmetric
threshold cannot make the current telemetry model reach the contribution bar;
its best zero-error envelope saves at most 5.69% of inferred online cost. A
strong outcome-only model at another post-hoc regularization/threshold can save
16.13%, which further prevents attributing a unique advantage to telemetry.
These oracle-threshold numbers are diagnostics only and are not test evidence.
## Why the learned verifier did not generalize
The training corpus has only 37 anchors from one workload/SLO task. P1 shows
large covariate shift:
- sim+outcome: 12/192 feature values exceed 3 training standard deviations and
4 exceed 5; maximum absolute z-score is 10.36;
- sim+outcome+telemetry: 19/396 exceed 3 and 9 exceed 5;
- the largest shifts include admitted input-length mean (10.36), waiting state
(7.77), running maximum (6.38), and decode-batch maximum (6.08).
Coefficient attribution shows that the input-length feature dominates several
wrong feasible-anchor logits. Because all training examples share one task,
the joint classifier can learn incidental within-task correlation and override
a correct simulator prior on TP2/MNS64-high and TP4/MNS64-high. This is a
supported diagnosis of model/data insufficiency; it is not a causal proof that
one feature alone caused the P1 failure.
More importantly, retuning lambda, threshold, features, or cutoff on P1 and
then calling P1 a held-out result would violate calibration/evaluation
separation. P1 may now be used only as development data.
## Decision and the only defensible reopening condition
Do not run registered P2/P3 with the current model. It failed the predeclared
gate on the favorable primary-trial denominator and is even farther from the
bar under end-to-end cost. Spending six-task headline GPU budget on the same
method would be metric shopping, not replication.
A new route may be opened only as a new hypothesis:
1. Replace the joint classifier with a **simulator-residual verifier**. The
simulator prediction remains an explicit prior; nested outcome-only and
telemetry models learn when that prior is wrong, rather than freely
relearning feasibility and overriding it under workload shift.
2. Train on multiple complete workload/SLO tasks. SLO thresholds and target
pass rate must be explicit inputs; splits are by complete task.
3. Calibrate abstention with task-level risk control. No threshold is selected
on a headline task, and “never early decide” is included as the safe
outcome-only baseline.
4. Treat Phase 6 and P1 as development only, freeze the residual architecture,
features, cutoff, threshold, simulator reading, and `k`, then use entirely
new trace windows for a new gate.
This reopening is justified only if development data show both (a) the
simulator's errors are predictable from pre-outcome engine state and (b) a
simulator-preserving residual model does not corrupt correct simulator
predictions. It is a new project decision, not a continuation automatically
authorized by P1.
## Benchmark audit
| Audit item | Verdict | Severity | Evidence / disposition |
|---|---|---|---|
| Calibration set separate from P1 | PASS | — | Phase 6/0311 trained; P1/0312 tested |
| Strong simulator-aware baseline | PASS | — | Identical Frontier features in both nested models |
| Sim top-k + real-final E2E baseline | PASS | — | Frozen `k=2`, tie expansion, measured setup/continuation cost |
| Multiple independent headline tasks | NEEDS EVIDENCE | Blocking for a positive claim | P1 gate failed; P2 correctly not opened |
| Statistical significance | NEEDS EVIDENCE | Blocking for a positive claim | n=12 anchors from one task; McNemar p=0.5 |
| Hyperparameter robustness | FAIL | Blocking | Lambda sensitivity changes safety and relative result |
| Full resource accounting | PASS for P1 | — | Failures, startup/warm-up/burn-in, continuation and annotation separated |
| Avoid post-test retuning | PASS only if route stops | Blocking if violated | P1 is now development-only |
| Selective winning-workload reporting | PASS | — | Negative P1 and TP/MNS losing cases retained |
Overall recommendation: **Block the current independent harness contribution
claim.**
## Artifacts
- Registered protocol: `docs/fidelity-aware-harness-protocol-20260714.md`
- Historical headroom: `docs/fidelity-aware-harness-headroom-20260714.md`
- Registered P1 analysis: `runs/fidelity-headroom/analyze_pilot.py`
- Strong P1 analysis: `runs/fidelity-headroom/analyze_strong_pilot.py`
- E2E shortlist replay: `runs/fidelity-headroom/analyze_pilot_e2e.py`
- External immutable result root:
`/home/gahow/phd/replayserve/runs/fidelity_p1_frontier_committed_20260714`
## Data sanity block
| Data | n | Min | Max | Distinct | Invariant |
|---|---:|---:|---:|---:|---|
| P1 labels | 12 | 0 | 1 | 2 | 7 feasible / 5 infeasible |
| Primary elapsed seconds | 12 | 19.448 | 61.435 | 12 | Every five-second prefix is in range |
| Prefix Layer-1 records | 12 | 332 | 557 | 12 | Contiguous; zero drops |
| Exact timestamped outcomes | 12 anchors | 54 | 750 | 11 | Monotonic completion timestamps |
| Simulator pass rate | 12 | 0.1548 | 1.0 | 7 | Ratios in `[0,1]` |
| Strong nested probabilities | 24 | 0.000208 | 0.809422 | 24 | Ratios in `[0,1]` |
| E2E cost components | 36 | 0.001389 | 0.169653 H20-h | 21 | Non-negative |
| GPU attempts | 2 | 0.020552 | 1.722112 H20-h | 2 | Aggregate 1.742664 < 3.5 |
| Copied raw files | 191 | | 153,093,348 bytes total | | Remote/local aggregate SHA identical |
Checked invariants: six cells and twelve anchors; exact request count and
request-ID/arrival/length hashes; all cell validation flags true; both labels
present; probabilities bounded; costs and counters non-negative; simulator
results not all identical; committed simulator rerun 12/12 numerically
identical to the exploratory run; no prompt text in public simulator fixtures;
no co-resident serving process; final eight GPUs at 0 MiB and 0% utilization.
No red flag remains.

View File

@@ -1,9 +1,18 @@
# Fidelity-aware real-verification harness protocol
Status: **PRE-REGISTERED STAGED EVALUATION; CONTRIBUTION NOT YET ESTABLISHED**.
Status: **P1 FAILED; P2/P3 CLOSED FOR THE REGISTERED METHOD; CONTRIBUTION NOT ESTABLISHED**.
Date frozen: 2026-07-14 (Asia/Singapore).
Post-run disposition (2026-07-14): P1 completed with valid data but failed its
registered incremental gate. The strengthened simulator-aware comparison and
end-to-end `k=2` replay also failed: instrumentation was safe and retained zero
regret, but reduced online H20-hours by only 1.426% versus sim top-k + real
final, against the 30% bar. Outcome-only was unsafe. P2/P3 are therefore not
opened for this model. Full results and the permitted reopening condition are
in `docs/fidelity-aware-harness-p1-report-20260714.md`; the protocol below is
retained unchanged as the pre-run record.
## Research question and contribution bar
The harness has an independent systems contribution only if engine-internal
@@ -52,6 +61,35 @@ difference is Z. The initial family is intentionally simple: a positive result
then demonstrates value in the engine signal rather than capacity in a larger
learner. A sequence model is admissible only as a later, paired ablation.
### Amendment A1: strengthen the calibration baseline before P2
Frozen 2026-07-14 13:08 Asia/Singapore, after P1 launch but before P1
completion or analysis. A baseline audit found that the first frozen P1
models use the simulator only to define candidate order; their feature vectors
do not contain the simulator's per-anchor prediction. This is insufficient
for the stronger term **outcome-only calibration**. P1 therefore remains a
prospective test of the originally frozen cross-workload predictor, but cannot
by itself open a contribution claim.
For P2/P3, both nested models must additionally receive the identical frozen
simulator outputs available at that decision: predicted completed throughput
per GPU, predicted SLO pass rate, and predicted feasibility. The comparison
is consequently `sim + config + workload + real outcome prefix` versus that
exact vector plus real engine state. Simulator features, regularization,
cutoff, and thresholds are frozen before any P2 task. If telemetry does not
improve this stronger baseline, the harness has no independent contribution.
The same audit also separates algorithm cost from benchmark-oracle cost.
Headline method cost includes every action the method would execute online:
simulator profiling/calibration, model onboarding, server startup, warm-up,
real prefix, continuation after abstention, method-requested confirmation,
logging overhead, failures, and cleanup. Exhaustive real-oracle runs and the
extra repetitions used only to construct 2-of-3 evaluation labels are common
benchmark annotation cost; they are reported separately and charged to no
method. A second, deliberately conservative table adds that common cost to
all methods. This prevents both hiding real method cost and making the
percentage gate mathematically depend on offline ground-truth annotation.
The frozen first policy uses a 5-second prefix, L2 regularization 1.0, and a
two-sided abstaining threshold of 0.95: accept at `p(feasible)>=0.95`, reject at
`p(feasible)<=0.05`, otherwise continue the exact same trial to completion.
@@ -64,15 +102,15 @@ therefore not evidence; all claims come from subsequent held-out tasks.
|---|---:|---:|---:|---:|---:|
| Real-only oracle | no | no | full | optional diagnostic | every candidate/anchor |
| Sim top-k + real final | yes | included in full run | full | no decision use | every shortlisted candidate/anchor |
| Outcome-only calibration | yes | yes | yes | no | only on abstention |
| Instrumentation-aware | yes | yes | yes | yes | only on abstention |
| Outcome-only calibration | yes, including its prediction features | yes | yes | no | only on abstention |
| Instrumentation-aware | same prediction features | yes | yes | yes | only on abstention |
Tie buckets are expanded before top-k. `k` is selected on training tasks and
is fixed on held-out tasks; an oracle per-task k is forbidden. Outcome-only
receives all information available outside the engine, including config and
workload features. Instrumentation cannot use any record submitted after the
cutoff. The full label, confirmation votes, simulator error, and later
requests are never model features.
receives all information available outside the engine, including config,
workload, and frozen simulator-prediction features. Instrumentation cannot use
any record submitted after the cutoff. The full label, confirmation votes,
realized simulator error, and later requests are never model features.
## Staged experiment

View File

@@ -0,0 +1,71 @@
# Telemetry intervention-response v0 protocol
Status: **FROZEN BEFORE V0 ANALYSIS**.
Date: 2026-07-14 (Asia/Singapore).
## Claim boundary
The closed residual route asked whether one absolute engine-state snapshot can
predict unmeasured configurations. V0 asks a different, narrower question:
> Does an adjacent, controlled MNS intervention produce an early engine-state
> response that is distinguishable from same-config repeat noise?
Passing this gate only authorizes a matched real-GPU pilot. It does not prove
that telemetry improves tuning, that any metric is a causal mediator, or that
the response transfers to a new workload, topology, or knob family.
## Data and estimand
- Source: Phase 6 solo-authoritative Qwen3-30B-A3B/vLLM 0.24 Layer-1 streams.
- Action pairs: primary runs at identical study hash, TP, sampling anchor, and
request-order hash, with adjacent `MNS={8,16,32,64}` values.
- Noise pairs: primary versus confirmation at the same complete config,
anchor, and request-order hash. Only primary-to-confirmation pairs are used;
confirmations are not combined into pseudo-independent all-pairs.
- Fixed early windows: 5 seconds and 10 seconds from the measured interval
start. All runs exceed 10 seconds, so early-stop censoring cannot change the
telemetry window.
- Full-run pass rate and feasibility are descriptive only because an early
stop can make full elapsed durations differ.
The statistical unit is a run pair. Scheduler steps are summarized within a
run and are never counted as independent trials.
## Frozen response gate
The directly measured gate features are scheduler-step rate, decode-batch
mean, prefill-token fraction, waiting/running queue mean, KV-usage mean, and
CUDA-graph padding fraction.
A feature qualifies at one horizon only if:
1. at least 75% of nonzero action deltas have the same sign;
2. median absolute action delta is at least 2x the median absolute repeat
delta; and
3. at least 50% of action deltas exceed the repeat-noise absolute p95.
V0 opens a GPU pilot only if:
- there are exactly 17 frozen adjacent-MNS action pairs;
- there are at least 20 primary/confirmation repeat pairs;
- all identity, finite-value, counter, and ratio invariants pass; and
- at least two gate features qualify at both 5 and 10 seconds.
Any data red flag stops the analysis before interpreting the response.
## If V0 passes
Register a dash0 pilot around a known scaling knee. The pilot must use the
same request sequence and arrival times, one serving job at a time, one changed
knob, randomized `A/B` versus `B/A` order, common non-censored measurement
windows, and trial-level repetitions. It must compare a response-aware next
action against an outcome-only policy under complete startup, warm-up, and
H20-hour accounting.
## If V0 fails
Do not add telemetry fields or train a larger model. The current Layer-1 state
does not identify even an MNS intervention above repeat noise on this task, so
the telemetry-guided tuning route remains diagnostic only.

View File

@@ -0,0 +1,157 @@
# Telemetry intervention-response v0/v1 results
Date: 2026-07-14 (Asia/Singapore).
## Decision
**STOP before a new H20 pilot.** The current Layer-1 aggregate telemetry does
not identify a sufficiently general early response to an MNS intervention,
and it does not improve action-efficacy prediction over exact external prefix
outcomes on the available development tasks.
This is a negative result about the present state representation and
experiment design. It does not establish that engine telemetry is useless for
tuning, and it is not held-out evidence.
## Hypothesis and frozen test
The tested hypothesis was:
> With the workload and all non-MNS settings held fixed, increasing MNS causes
> a 5--10 second engine-state response that is larger than same-config repeat
> noise and that predicts whether the action makes the full run feasible.
A response feature had to satisfy all three frozen conditions at both 5 and
10 seconds: at least 0.75 sign consistency, median absolute action effect at
least 2x the repeat median, and at least 0.50 of action deltas above the repeat
absolute p95. At least two features had to pass. A telemetry feature was
decision-relevant only if its leave-one-repeat-out balanced accuracy was at
least 0.75 and at least 0.15 above the best exact external prefix outcome.
## What was implemented
- A common-window analyzer over the existing per-scheduler-step Layer-1 stream.
- Exact action pairing with request-order hash, offered load, TP, load role,
and repetition held fixed.
- Same-config repeat-noise estimation without treating scheduler steps as
independent samples.
- Exact 5/10-second request-prefix outcomes using monotonic completion times.
- A one-feature leave-one-repeat-out efficacy audit; no multivariate model was
fitted to the 12 examples.
- Input hashes, stream hashes, frozen thresholds, pair-level deltas, and sanity
invariants in machine-readable audit artifacts.
- Trial-by-trial validation against the P1 manifest, plus content hashes for
every result, request file, and Layer-1 stream.
## Experiment A: Phase-6 retrospective audit
Phase 6 supplied 17 adjacent-MNS actions and 29 same-config
primary/confirmation pairs. No feature passed at either horizon, producing
`STOP_NO_IDENTIFIABLE_RESPONSE`.
The confirmation sample is not a clean replication distribution: confirmations
were selectively run after disputed primary outcomes. Several same-config
pairs consequently followed radically different trajectories. This result
therefore remains a valid failure of the frozen v0 gate, but it cannot by itself
separate normal run variance from confirmation-selection bias.
## Experiment B: prospective-repeat confirmation
P1 supplied three pre-arranged, disjoint request bands for every cell/load.
Exact matched actions exist for TP1 `MNS 8 -> 64` and TP4 `MNS 16 -> 64`, at
low/high load and repetitions 1/2/3. This yields 12 action pairs and 24
same-config consecutive-repeat pairs.
The 24 adjacent repeat differences share their middle repetition within each
three-run group. They define a conservative empirical noise reference; they
are not used as 24 independent samples in an inferential test.
The result is `STOP_NO_PROSPECTIVE_RESPONSE`: zero features passed the response
gate at either horizon.
The strongest response was mean waiting-queue occupancy:
| Horizon | Sign consistency | Action/repeat median | Action above repeat p95 | Gate |
|---|---:|---:|---:|---|
| 5 s | 1.000 | 1.292x | 0.167 | fail |
| 10 s | 1.000 | 2.611x | 0.250 | fail |
The direction is real enough to merit diagnosis, but the effect is not broad
enough to guide a general action. It is large for TP4/high-load trials and
small or absent in other regimes.
Full-run transitions contain six beneficial actions (`false -> true`) and six
non-beneficial actions (three `false -> false`, three `true -> true`). The
beneficial label is also perfectly confounded with TP4 in this small dataset,
so it cannot support a topology-general claim.
| Horizon | Best telemetry delta | Balanced accuracy | Best external prefix delta | Balanced accuracy | Telemetry advantage |
|---|---|---:|---|---:|---:|
| 5 s | waiting queue | 0.750 | max TPOT / SLO | 0.833 | -0.083 |
| 10 s | waiting queue | 0.750 | outstanding / admitted | 0.750 | 0.000 |
No telemetry feature reaches the preregistered `+0.15` incremental threshold.
## What this rules out
It rules out using the current vector of 5/10-second global means as a solid
mechanism for choosing the next config. In particular, adding these aggregates
to an LLM prompt or fitting a larger predictor would currently hide, rather
than solve, the identifiability problem.
It does not rule out an instrumentation-aware tuner built around a deliberately
excited local system. The existing runs were designed for endpoint/fidelity
evaluation, not system identification: the MNS action is large, efficacy is
confounded with TP, repeat bands contain different requests, and global means
erase when queue buildup or service-rate changes occur.
## Required redesign before spending H20-hours
The next admissible experiment is a randomized, local A/B system-identification
pilot around one fixed TP and one load knee:
1. Replay the exact same request sequence and arrival times for both endpoints.
2. Use small adjacent actions and randomized `A/B` versus `B/A` order.
3. Record event-aligned response curves, including queue growth/drain rate,
prefill/decode service rate, and per-step service time, rather than only one
global mean.
4. Separate a mechanism gate (repeatable response) from the end-to-end gate:
fewer trials or H20-hours to select a feasible near-optimal config than an
outcome-only tuner.
5. Hold out a second load/workload for the final policy comparison.
Until that design is frozen, a wider sweep would only generate more correlated
observations and is not justified by the evidence above.
## Reproduction
```bash
python3 runs/intervention-response-v0/test_analysis.py
python3 runs/intervention-response-v0/test_p1_analysis.py
python3 runs/intervention-response-v0/analyze_phase6.py \
--metrics runs/opprof-phase6/phase6/metrics.json \
--raw-root runs/opprof-phase6/phase6/solo-authoritative/cells \
--output runs/intervention-response-v0/phase6-audit.json
python3 runs/intervention-response-v0/analyze_p1.py \
--run-root /home/gahow/phd/replayserve/runs/fidelity_p1_frontier_committed_20260714/real/p1b \
--manifest /home/gahow/phd/replayserve/runs/fidelity_p1_frontier_committed_20260714/real/p1b/pilot-manifest.json \
--output runs/intervention-response-v0/p1-audit.json
```
## Data sanity
- Phase 6: action pairs `n=17`, repeat pairs `n=29`, trials `n=66`; MNS
action size min/max `8/32`, `3` distinct; action-state vectors `n=17`, `17`
distinct; streams `n=12`, bytes min/max `12,745,297/52,957,710`, `12`
distinct.
- P1: action pairs `n=12`, repeat pairs `n=24`, trials `n=36`; MNS action
size min/max `48/56`, `2` distinct; efficacy labels `n=12`, min/max `0/1`,
`2` distinct; streams `n=6`, bytes min/max `17,449,143/29,431,988`, `6`
distinct.
- Checked invariants: exact action request hashes and offered loads match;
all `36/36` P1 trials match the manifest; expected pair counts hold; all
deltas are finite; non-negative counters and bounded ratios hold; per-config
state vectors are not all identical; both efficacy classes are present. No
red flags were observed.

View File

@@ -0,0 +1,58 @@
# Intervention-response v1 prospective-repeat confirmation
Status: **FROZEN AFTER PHASE-6 V0 FAILURE AND BEFORE P1 RESPONSE ANALYSIS**.
Date: 2026-07-14 (Asia/Singapore).
## Why this is a new confirmation, not a relaxed V0
Phase-6 V0 failed its frozen global response gate. Its 29 same-config
confirmations were triggered after disputed outcomes, and the resulting noise
sample contains extreme trajectory divergence by construction. V0 remains
failed and its thresholds are unchanged.
The already-completed P1 campaign supplies a distinct test: three
prospectively scheduled, disjoint repeat bands for every cell/load. TP1 and
TP4 use identical offered loads and exact request-order hashes across their MNS
endpoints. V1 asks whether an MNS response is identifiable against this
prospective workload-repeat noise, and whether that response predicts action
efficacy beyond exact external prefix outcomes.
P1 is now development data. No result here is held-out or paper-facing.
## Frozen pairs
- Action pairs: TP1 `MNS 8 -> 64` and TP4 `MNS 16 -> 64`, at low/high load and
repeat 1/2/3. Endpoints must have identical TP, offered rate, repeat role,
and request-order hash. Expected `n=12`.
- Repeat-noise pairs: consecutive pre-arranged repeat bands within each of six
cells and low/high load: `rep1 -> rep2`, `rep2 -> rep3`. Expected `n=24`.
Repeat bands intentionally contain different requests and therefore include
workload-sampling noise rather than pretending to be identical trials.
Adjacent differences share the middle run; the gate uses their empirical
magnitude only and does not treat the 24 differences as independent samples
for a p-value or confidence interval.
- Prefix horizons: 5 and 10 seconds. Exact monotonic request completion times
and the same Layer-1 intervals are used.
## Frozen gates
The response-identifiability thresholds are exactly the Phase-6 V0 thresholds:
75% sign consistency, 2x median effect/repeat noise, and at least 50% of action
deltas above repeat absolute p95. At least two response features must qualify
at both horizons.
Action efficacy is one only for an infeasible-to-feasible full-run transition.
The 12 action pairs must contain at least four examples of each class.
For decision relevance, each individual external-outcome response feature and
each individual telemetry-response feature is evaluated by leave-one-repeat-
band-out threshold fitting. This intentionally avoids a multivariate model on
12 examples. At least one telemetry feature must, at both horizons:
1. reach balanced accuracy at least 0.75; and
2. exceed the best external-outcome response feature by at least 0.15.
Only if data validity, response identifiability, and incremental decision
relevance all pass does V1 open a newly registered matched GPU pilot. No
threshold or feature is changed after observing V1.

View File

@@ -0,0 +1,94 @@
# Phase-aware telemetry intervention-response v2 protocol
Status: **INVALID OPERATIONAL ATTEMPT; SUPERSEDED BY V3**.
Date: 2026-07-14 (Asia/Singapore).
The first MNS=16 session timed out while draining the 3.125 requests/s/GPU
workload after its 300-second arrival window. It produced no high-load result,
and no MNS=64 endpoint was run. No comparative conclusion is drawn from this
attempt; see `intervention-response-v3-two-load-protocol-20260714.md`.
## Correction to v0/v1
The 5/10-second analyses tested an ultra-early verifier. They did not test
whether telemetry observed after the engine has developed queue, batch, and KV
state can guide tuning. The P1 replay lasts 60 seconds after time scaling, and
the 5/10-second prefixes contain only a small fraction of its requests.
V2 therefore replaces absolute cutoffs with replay phase. The old audits and
their negative decisions remain immutable, but their claim is narrowed to the
first 5/10 seconds.
## Historical corrective audit
The historical audit is development-only and cannot become confirmatory after
the horizon concern was observed.
- Infer each trial's intended replay duration as selected requests divided by
offered requests per second. All trials must agree.
- Find every complete 10% replay decile supported by every trial. Analyze all
such deciles; selecting only the best horizon is forbidden.
- At each decile report both:
- cumulative state from replay start to the checkpoint; and
- the non-overlapping 10%-wide state block ending at the checkpoint.
- Report admitted/completed request coverage, response-versus-repeat statistics,
telemetry versus external-outcome efficacy, and per-feature trajectory drift.
- Reuse the frozen v1 action and repeat pairs and the frozen response and
incremental-efficacy thresholds. These thresholds are descriptive in V2;
passing one post-hoc horizon does not open a contribution claim.
If early stopping prevents complete observation of the replay phases, the
historical decision is `REQUIRES_UNCENSORED_PHASE_AWARE_PILOT`, independent of
which early decile looks best.
## Uncensored matched pilot
The pilot is a mechanism gate, not paper evidence.
- Hardware/engine/model: solo placement on dash0, 4 NVIDIA H20 GPUs, patched
vLLM `0.24.1.dev3+opprof`, Qwen3-30B-A3B, fixed `TP=4`.
- Action: `MNS 16 -> 64`; topology, model, engine build, workload, arrival
sequence, offered load, and all other settings remain fixed.
- Workload: `chat_w20260312_1000` at replay-time scale `0.5`, hence 300 seconds.
- Offered loads per GPU: `1.5`, `2.125`, and `3.125` requests/s. These supply a
low control and the two already-observed P1 pressure regimes.
- Repetitions: three disjoint session bands, exact request sequence matched
across action endpoints. Endpoint order alternates `A/B`, `B/A`, `A/B`;
load order is counter-rotated across repetitions.
- A fresh server receives the accepted long-request warm-up and a bounded
burn-in before each measured session.
- SLO-unrecoverable early stop is disabled. Every run must observe the full
300-second arrival window; a separate 360-second safety deadline may mark a
run invalid but cannot manufacture a full-run label.
- Cumulative checkpoints: 10%, 25%, 50%, 75%, and 100%, or 30/75/150/225/300
seconds. Quarter blocks are analyzed separately from cumulative means.
- A measured Layer-1 interval is complete only when its start-boundary,
end-boundary, and maximum internal record gaps are each at most one second;
timestamps must be monotonic.
- Placement is serialized. Co-location remains forbidden because Phase 6
observed material co-location-induced outcome shifts.
- Hard cap: 8 H20-hours including startup, warm-up, burn-in, invalid attempts,
and cleanup.
## Gates
Data validity requires complete 300-second Layer-1 coverage, zero dropped
records, exact request/arrival/length hashes across action endpoints, monotonic
timestamps, full request accounting, idle GPUs before and after each session,
and no co-resident GPU process.
Mechanism evidence requires at least two telemetry features whose matched action
response exceeds same-config repeat noise at the same pair of consecutive
checkpoints under the unchanged v1 response thresholds. Those features must
also have a consistent direction in at least two of the three load regimes.
Decision evidence additionally requires both action-efficacy classes and at
least one of the phase-stable mechanism features to reach
leave-one-repetition-out balanced accuracy at least 0.75 and exceed the best
external prefix outcome by at least 0.15 at two adjacent predeclared
checkpoints from 25% onward. Without label balance the pilot can adjudicate
mechanism evidence only.
No H20 run is launched if the local analyzer/tests, manifest preflight, GPU
probe, command dry-run, projected cost, or cleanup plan fails.

View File

@@ -0,0 +1,160 @@
# Phase-aware telemetry intervention-response v3 results
Date: 2026-07-14 (Asia/Singapore).
Decision: **`STOP_NO_INCREMENTAL_TUNING_SIGNAL`**.
## Claim tested
After the replay has developed queue, batch, and KV state, does increasing MNS
from 16 to 64 create telemetry responses that exceed workload-repeat noise, and
does any such response identify whether the action repairs the full-run SLO
better than external prefix outcomes alone?
The first clause passed. The second clause failed. Long-window telemetry is
mechanistically informative, but this pilot does not support its necessity for
tuning this action.
## Setup
- dash0 GPU 0-3: four NVIDIA H20 GPUs; Qwen3-30B-A3B; patched vLLM
`0.24.1.dev3+opprof`; TP=4.
- Action: MNS `16 -> 64` with exact request, arrival, and input-length hashes.
- Trace: `chat_w20260312_1000`, replay-time scale 0.5, 300 seconds.
- Loads: 1.5 and 2.125 requests/s/GPU; three disjoint bands; endpoint order
A/B, B/A, A/B; load order low/mid, mid/low, low/mid.
- Five cumulative checkpoints: 30, 75, 150, 225, and 300 seconds; four
non-overlapping quarter blocks.
- Six fresh-server sessions, 12 measured runs, six action pairs, and eight
same-config repeat pairs.
The prior three-load attempt is not part of the result. Its MNS=16 workload at
3.125 requests/s/GPU could not drain by the 450-second client timeout and
produced no high-load result. V3 reran every retained point from scratch.
## End-to-end outcome
All three low-load pairs remained feasible (`true->true`, label 0). All three
pressure-load pairs changed from infeasible to feasible (`false->true`, label
1). MNS=16 pressure pass rates were 0.5604, 0.3145, and 0.2635; all three
MNS=64 pressure runs reached 1.0. This yielded a balanced 3/3 action label set.
## Mechanism result
No telemetry feature passed the action-versus-repeat gate at 10% or 25%. At
50%, graph padding first passed. At both 75% and 100%, graph padding and queue
waiting passed, satisfying the requirement for two features at the same pair
of adjacent checkpoints and with consistent directions in both load regimes.
| Feature | Direction for MNS 16->64 | 75% effect/repeat median | 75% above repeat p95 | 100% effect/repeat median | 100% above repeat p95 |
|---|---:|---:|---:|---:|---:|
| `queue_waiting_mean` | lower | 1346.22 | 3/6 | 898.70 | 3/6 |
| `graph_padding_fraction` | higher | 5.00 | 4/6 | 5.74 | 5/6 |
The very large queue effect/median-repeat ratios should not be read alone: its
repeat p95 was much larger than its repeat median, so the independent p95
coverage criterion remained binding. Full-window queue-waiting deltas were
-0.19 to -0.32 at low load and -20.99 to -32.16 at pressure load. Graph
padding increased in every pair, by 0.00133-0.00217 at low load and
0.00705-0.00873 at pressure load.
The mechanism is therefore a real tradeoff: larger MNS reduces queueing,
especially under pressure, while increasing CUDA-graph padding.
## Tuning-signal result
Leave-one-repetition-out balanced accuracy was evaluated against the best
external prefix-outcome feature at every predeclared checkpoint.
| Replay phase | Best external BA | Best telemetry BA | Incremental telemetry gate |
|---:|---:|---:|---|
| 10% | 0.833 | 0.833 | fail |
| 25% | 1.000 | 1.000 | fail |
| 50% | 0.833 | 1.000 | pass at this checkpoint only |
| 75% | 1.000 | 1.000 | fail |
| 100% | 1.000 | 1.000 | fail |
At 50%, several telemetry features exceeded the external baseline by at least
0.15, including the phase-stable mechanism feature
`graph_padding_fraction`. The advantage did not hold at either adjacent
checkpoint. Consequently no feature passed the frozen two-adjacent-phase
requirement.
The important ordering is that external TTFT already classified the action
perfectly at 25%, whereas the robust two-feature mechanism response did not
emerge until 75%-100%. In this setup telemetry explains *why* MNS helps, but it
does not provide earlier or more reliable action selection than direct prefix
outcomes.
## Research conclusion
The 5/10-second negative result was indeed too narrow. It only ruled out an
ultra-early telemetry verifier; it did not rule out engine-state information.
The 300-second pilot finds a clear and reproducible queueing-versus-padding
response.
However, this does not rescue the direct telemetry-guided tuning claim. For
this action and workload, the external signal is already as good or better
before the telemetry mechanism becomes stable. The project should therefore
not claim that engine instrumentation is necessary for tuning on this evidence,
and should not open an E2E policy test from this pilot.
The narrower simulator-residual route remains logically open: telemetry may
explain why a simulator misranks real configurations even when direct online
outcomes can guide a tuner. That is a different hypothesis and was not tested
here.
## Change and verification
Change: absolute 5/10-second prefixes were replaced by phase-aware 30/75/150/
225/300-second analysis; SLO early stop was disabled; full Layer-1 coverage,
hash, request-accounting, controller, and stream/footer gates were added. The
mechanism gate was corrected to require two features at the same adjacent
phase pair.
Expected effect: distinguish “telemetry has not developed yet” from “telemetry
does not identify or improve the action decision.”
Verification: five local analysis/controller test suites passed; remote
manifest preflight and command dry-run passed; six serialized sessions passed
all stream invariants; the analyzer was rerun and produced byte-identical
output.
Result: mechanism evidence passed; incremental tuning evidence failed. Audit
SHA256: `45f6f248712f9cbd3ed72036837ff6dc5b5c14c0f2eb6ba5cd5daceb1aa4ddb7`.
Remaining risk: this is a development pilot with one model, one TP, one action,
two retained loads, three request bands, and six action labels. It is adequate
to reject opening the next direct-policy stage, not to establish a universal
negative claim about telemetry.
## Research-validity audit
| Check | Verdict | Evidence / boundary |
|---|---|---|
| Real system and E2E outcome | PASS | Real H20/vLLM replay; full SLO outcome accompanies mechanism telemetry. |
| Matched action baseline | PASS | Exact request/arrival/length hashes for MNS 16 and 64; external prefix outcome is the decision baseline. |
| Repeats and order effects | PASS for pilot | Three disjoint bands; A/B, B/A, A/B endpoint order; counter-rotated load order. |
| Selective load removal | PASS with narrowed claim | The 3.125 load produced no result before any action comparison; the failure and cost are retained, and all kept points were freshly rerun. |
| Significance/generalization | NEEDS EVIDENCE for a paper claim | Only three bands, one model, one TP, one action, and six labels. This is explicitly a stage gate. |
| Calibration versus evaluation | NEEDS EVIDENCE for a positive policy claim | Frozen gates and leave-one-band-out folds reduce leakage, but a new workload/model hold-out is still required. |
| Platform/reproducibility | PASS | Commit, commands, manifest, controller state, platform fingerprint, raw remote paths, and audit hashes are recorded. |
## Data sanity
- Measured runs: n=12; elapsed 300.346-317.012 seconds; 12 distinct; pass rate
0.2635-1.0 with 4 distinct values; selected requests 1800-2550 with 2
distinct values.
- Sessions: n=6; 0.8413-0.8631 H20-hours; 6 distinct; Layer-1 records
58,465-64,776; 6 distinct. V3 cost was 5.0924 H20-hours; total including
the invalid attempt was 6.4505, below the 8.0 cap.
- Labels: n=6; min/max 0/1; 2 distinct. Action pairs were 6 and repeat pairs
were 8 at every checkpoint.
- Coverage-gap observations: n=60; start gaps 0.0427-0.1247 seconds; end gaps
0.00014-0.1705; maximum internal gaps 0.1695-0.6528, all below one second.
- Checked invariants: exact pair hashes and counts, all runs uncensored, full
request accounting, monotonic admitted/completed coverage, monotonic Layer-1
timestamps, nonnegative counters, bounded ratios, non-identical per-config
states, contiguous step indices, zero drops, footer/sidecar agreement, no
controller failures, all sessions complete, GPU idle after completion. No
red flags were found.

View File

@@ -0,0 +1,73 @@
# Phase-aware telemetry intervention-response v3 protocol
Status: **FROZEN AFTER A NON-COMPARATIVE OPERATIONAL FAILURE AND BEFORE V3 RUNS**.
Date: 2026-07-14 (Asia/Singapore).
## Why v2 was invalid
The first v2 session completed the 300-second arrival windows at 1.5 and 2.125
requests/s/GPU. At 3.125 requests/s/GPU, MNS=16 could not drain the admitted
requests before the 450-second client timeout. The session produced no result
and no MNS=64 action endpoint was run. V2 is therefore an invalid operational
attempt, not evidence for or against the telemetry hypothesis.
This failure was observed before any MNS action comparison. V3 excludes only
the unmeasurable overload point and reruns every retained point on fresh
servers; it does not reuse the completed v2 low/mid results.
## Question and hypothesis
Question: after enough replay time for queue, batch, and KV state to develop,
does an MNS intervention create telemetry responses that exceed workload-repeat
noise, and does any such response predict whether the intervention repairs the
full-run SLO outcome better than external prefix outcomes alone?
Hypothesis: increasing MNS from 16 to 64 has little value at the 1.5
requests/s/GPU control load but can repair the 2.125 requests/s/GPU pressure
load. Queue, running-set, batch, or KV telemetry should expose the difference
at stable replay phases. Label balance is an assumption to test, not a fact.
## Frozen setup
- Solo placement on dash0 GPU 0-3: 4 NVIDIA H20 GPUs, Qwen3-30B-A3B, patched
vLLM `0.24.1.dev3+opprof`, fixed TP=4.
- Action: MNS `16 -> 64`; all other engine and workload parameters fixed.
- Workload: `chat_w20260312_1000`, replay-time scale 0.5, hence 300 seconds.
- Loads per GPU: 1.5 control and 2.125 pressure requests/s. The failed 3.125
overload point is excluded from V3 and retained only as a failure artifact.
- Three disjoint request bands. Each MNS action pair has exact request,
arrival, and input-length hashes. Endpoint order is A/B, B/A, A/B; load
order is low/mid, mid/low, low/mid.
- Every session starts a fresh server, then runs the accepted 16-request long
warm-up and bounded burn-in before measured runs.
- SLO-unrecoverable early stop is disabled. Measured results must cover the
full 300-second arrival window and must not be early-stopped.
- Cumulative checkpoints are 10%, 25%, 50%, 75%, and 100%; non-overlapping
quarter blocks are also reported.
- A Layer-1 interval is complete only if timestamps are monotonic and its
start, end, and maximum internal record gaps are each at most one second.
- Incremental V3 cap is the unused portion of the original 8 H20-hour cap.
The exact prior cost and V3 cap are machine-recorded in the manifest.
## Frozen gates
Data validity requires six uncensored sessions, six action pairs, eight
same-config repeat pairs, exact action-pair hashes, full request accounting,
zero Layer-1 drops, continuous coverage, all stream/footer invariants, no
co-resident compute process, idle GPUs before and after sessions, nonnegative
counters, bounded ratios, non-identical per-config state, and monotonic request
coverage across checkpoints. Any red flag stops analysis.
Mechanism evidence requires at least two telemetry features to exceed the
unchanged v1 repeat-noise thresholds at the same pair of adjacent checkpoints.
Those features must have a direction consistent in both retained load regimes.
Decision evidence additionally requires at least two positive and two negative
full-run action-efficacy labels, valid leave-one-repetition-out folds, and at
least one phase-stable mechanism feature whose balanced accuracy is at least
0.75 and at least 0.15 above the best external prefix-outcome feature at two
adjacent predeclared checkpoints from 25% onward.
V3 remains a development mechanism pilot. Even `OPEN_E2E_POLICY_TEST` opens a
held-out tuning-policy experiment; it is not itself a paper performance claim.

View File

@@ -0,0 +1,75 @@
# Simulator-for-config-tuning related-work claim map
日期2026-07-16。目的为「Frontier/Vidur-class simulator 能否低成本解决 config tuning」这条主线建立 related-work 边界。Vidur 与 LLMServingSim 的条目基于原文PDF 全文核读SimAI 基于论文页与摘要口径。每项按 Context / Claim / Assumption / Mechanism / Evidence / Boundary / 与本 project 的关系提取。
## VidurMLSys 2024arXiv:2405.05465
| 维度 | 内容 |
|---|---|
| Context | MSR India。首个面向 LLM inference 的大规模模拟器。Motivation 与我们一致config search 复杂度 O(\|M\|·\|T\|),且 optimal config 是 (model, trace) 的函数——Fig 1b 显示跨 trace misconfiguration 代价最高 2×。 |
| Claim | (a) request-level 预测误差 <9%static trace P95 normalized execution latency 误差 3.33%4 模型 × 3 tracedynamic trace **85% capacity** 负载下误差 <5%。(b) Vidur-Search 用约 1 小时 96-core CPU$9.93/h LLaMA2-70B 找到最优 config对比 deployment-based exploration 估算 42K GPU-hours $218K。(c) what-if 全量探索 $125 模拟成本 vs 估算 $1.14M 真机成本 |
| Assumption | operator runtime 可由单 GPU profiling + 小型 ML 估计器random forest插值prefill attention 可用等效单序列 sqrt(Σp_i²) 近似decode attention runtime 只依赖总 KV 读量而非 per-request context 分布LLM 架构同质小算子集合跨模型共享)。 |
| Mechanism | 声明式 model spec 算子三分类token-level / sequence-level / communication)→ GPU CUPTI profiling RF runtime estimator event-driven simulator + 三层 hierarchical scheduler支持 vLLM/Orca+/Sarathi-Serve/FasterTransformer/LightLLM 策略)→ Vidur-Search 对每个 config 二分搜索 max QPS判据 P99 scheduling delay <5s目标 QPS/dollar |
| Evidence | LLaMA2-7B/70BInternLM-20BQwen-72B denseAzure A100/H100 4-GPU pairwise-NVLink 节点Chat-1M / Arxiv-4K / BWB-4K trace总长截断到 4096 tokens |
| Boundary | **作者明示**接近 capacity point 时小误差会因排队失控放大 fidelity 评测停在 85% capacity。**结构性** MoE FP8/量化 prefix-cache reuse多轮对话按独立请求处理)、 speculative decoding列为 future work)、PP 仅同步长上下文未覆盖4K 截断)。metric 口径为 normalized execution latencystatic 排除 scheduling delay)。**最关键**sim 选出的 config 在真机 ground-truth surface 上的 selection regret 从未被验证42K GPU-h/$218K 是反事实估算分母是穷举式 exploration 而非 strong sequential tuner |
| 与本 project 的关系 | Frontier Vidur-class代码直接使用 vidur backend+ 我们的 FP8/MoE/EP/decode-profile patches我们的所有实验恰好工作在 Vidur 声明误差爆炸并回避的 regimecapacity point + SLO gate补的正是它缺的 selection-regret ground truth我们的 zero-shot 失败2530% regret与其 <9% 不矛盾——不同 metric不同 load regime不同 stack alignment论文必须主动写明这一点 Fig 1b workload-conditioned 结论与我们 P4 sign-flipP6 churn 互为独立佐证 支持 retune 频率 / amortization 论证C3)。 |
## LLMServingSimIISWC 2024arXiv:2408.05499
| 维度 | 内容 |
|---|---|
| Context | KAISTscale-out LLM serving HW/SW co-simulation面向 NPU/PIM/异构加速器设计探索基于 ASTRA-sim |
| Claim | 对真实 multi-GPU vLLM serving 平均误差 14.7% 趋势一致」; mNPUsim/GeneSys/NeuPIMs 34.7491×摘要口径 91.5×)。 |
| Assumption | iteration-level 模拟 + decoder-block 冗余复用编译一个 block 复制展开attention/ attention 分离可在可行时间内保持足够精度硬件行为可由可插拔 accelerator compiler+simulator 栈表达GeneSys 原型)。 |
| Mechanism | iterationscheduleriteration-level batchingKV pagingoperator mapping)→ per-device 硬件模拟 graph converterChakra)→ ASTRA-sim 网络级模拟 循环 |
| Evidence | multi-GPU vLLM 真机对照变量为 LLM 架构并行方案NPU 数量异构度报告平均误差与趋势一致性 |
| Boundary | 定位是硬件/系统设计空间探索不是 engine-knob config tuningvalidation 口径是 trend-following SLO-gated capacity selection-regret14.7% 平均误差大于典型 config capacity margin我们 12-cell 面上 top-2 差距 0.76%故该精度不足以支撑近邻 config 选择 |
| 与本 project 的关系 | 说明模拟保 trend是社区通行 validation 标准;「trend selection这一缺口对它同样成立不构成直接 baseline但在 related work 中界定我们评测口径selection regret at capacity point的必要性 |
## SimAINSDI 2025Alibabaaliyun/SimAI
| 维度 | 内容 |
|---|---|
| Context | 大规模 LLM **training** 的架构设计与参数调优模拟生产背景Alibaba Cloud)。 |
| Claim | 各测试场景平均 98.1% 与真实结果对齐 host 设计与参数设置提供生产可用 guidance |
| Assumption | training 过程可由 framework + kernel computation + collective communication 的选择性高保真集成复现 |
| Mechanism | 高保真集成三层栈 + 多线程加速 + lock-free global context sharing |
| Evidence | 与生产 training 场景对齐论文口径未逐一核读实验细节)。 |
| Boundary | training-onlytraining iteration 均匀batch 组成静态——恰是 Vidur 指出 inference 所缺的性质因此 98.1% 不可外推到 serving capacity point |
| 与本 project 的关系 | simulator 指导 infra 决策的工业先例与动机背书不与 serving config tuning claim 竞争引用价值在 motivation不在 evaluation 对照 |
## Frontier本 project 被测对象,非 related work
内部 Vidur-class 实现vidur backend+ project FP8/MoE tuning-keyQwen MoE serving planTP/EP-aware cache keycritical-lanedecode/true-mixed profile 补丁我们全部 fidelity 结论限定于该实现与已声明的 patch `simulator-fidelity.md`
## Consensus / disagreement / uncovered regime
**Consensus三方一致或与我们互证**
1. operator/iteration profile + 调度复合的模拟器在中低负载下能达到 515% latency 误差模拟成本比真机低数个数量级
2. optimal config (model, workload) 的函数misconfiguration 代价可达 ~2×Vidur Fig 1b我们 P4 pattern sign-flip P6 engine-churn 独立复证)。
**Disagreement** 无直接冲突数字我们的 zero-shot 失败与 Vidur <9% 处于不同 metric/regime论文需主动解释防止被误读为矛盾或重复
**Uncovered regime本 project 的空间):**
1. **capacity-point + SLO-gated selection regret 无人用真机 ground-truth surface 验证。** Vidur 自认该 regime 误差爆炸并把评测停在 85% loadLLMServingSim 只验 trend config tuning 的决策恰好发生在 capacity point
2. MoEFP8prefix reusespeculative decodingEP topology长上下文均在已发表 fidelity envelope 之外
3. **alignment/profiling 成本从不与真机 tuning 成本同表比较。** Vidur $218K 对比用穷举做分母正确分母是 strong sequential tuner我们实测 0.270.45 H20h/task`runs/tuning-cost/metrics.json`)。
4. envelope 失效的低成本检测workload/runtime/topology 变化后何时还能信 simulator无人提出
## 对本 project claim 的直接影响
- **C1 定位句**不是Vidur 错了」,而是Vidur-class claim 停在 sub-capacity load prediction fidelity把它外推到 SLO-gated capacity selection 是社区的隐含用法我们证明该外推在 zero-shot 下失败2530% regret并给出恢复 ranking 所需的最小真机证据层级」。
- **C2**Vidur 没有 minimum-real-evidence 的概念要么全模拟要么全真机per-TP calibration / 同栈 profile + KV capacity + patches 的证据层级是新贡献面
- **C3**省钱叙事必须从数量级修正为仅在 amortization 下成立」,分母换成 strong tuner 实测值Vidur Fig 1b + 我们 P6 churn 共同支撑 retune 频率前提
## 待 triage 的相邻工作(未读原文,暂不写 claim
APEXarXiv:2411.17651并行执行计划模拟)、LLMServingSim 2.0arXiv:2602.23036异构+分离式)、CharonarXiv:2605.17164training+inference 统一)、inference-fleet-simarXiv:2603.16054排队论容量规划)、AgentServeSimarXiv:2606.09613多轮 agent serving)。若审稿风险评估需要按本表格式各补一行
## Sources
- Vidur: <https://arxiv.org/abs/2405.05465>全文核读版本mlsys24 PDF
- LLMServingSim: <https://arxiv.org/pdf/2408.05499>
- SimAI: <https://www.usenix.org/conference/nsdi25/presentation/wang-xizheng-simai><https://github.com/aliyun/SimAI>

View File

@@ -0,0 +1,14 @@
# Simulator tuning evaluation
This directory contains decision-level summaries for experiments that compare
a serving simulator's selected configuration with the best configuration on
real hardware.
Current report:
- [Frontier selection regret on Qwen3-30B and Qwen3-235B](frontier-selection-regret-qwen30-qwen235-20260719.md)
The primary quantity is **real-hardware selection regret**, not simulator
absolute-latency error. Raw commands, profiles, traces, and experiment-specific
audit records remain under `runs/` or in the immutable remote artifact roots
listed by each report.

View File

@@ -0,0 +1,75 @@
# Frontier selection regret: Qwen3-30B and Qwen3-235B
> Date: 2026-07-19
> Scope: H20, community vLLM 0.20, Frontier piecewise simulation, no SLO gate
## Question and metric
For each workload and latency objective, Frontier selects the configuration
with the lowest simulated latency. We then look up that configuration on the
complete real-hardware surface and compare it with the real-hardware optimum.
```text
selection regret = real_latency(Frontier winner) / real_latency(real winner) - 1
```
Lower is better. `0%` means Frontier selected the real winner. Positive values
mean that following Frontier produces slower real serving. Each objective is
selected independently; this table does not combine TTFT, TPOT, and E2E into a
single score.
## Qwen3-30B-A3B
Configuration surface: `TP in {1,2,4} x MNS in {8,16,32,64}`, with
`MBT=8192`. Each real cell uses three fresh-server trials.
| Workload | TTFT mean | TTFT p90 | TPOT mean | TPOT p90 | E2E mean | E2E p90 |
|---|---:|---:|---:|---:|---:|---:|
| Trace-PD | 0.0% | 0.0% | 0.0% | 0.0% | 0.0% | 0.0% |
| Fixed-PD, 4096->256, 1.125 req/s/GPU | **58.0%** | **56.2%** | 0.0% | 0.0% | 1.7% | 5.5% |
| Trace-PO, OSL=1 | 3.2% | 0.4% | N/A | N/A | 3.2% | 0.3% |
| Fixed-PO, 4096->1, 1.125 req/s/GPU | 0.3% | 0.5% | N/A | N/A | 0.3% | 0.5% |
Interpretation: Frontier is near-optimal for Trace-PD and both prefill-only
cases, but the high-pressure Fixed-PD TTFT choice is materially wrong: its
selected configuration is 56--58% slower than the real TTFT optimum.
## Qwen3-235B-A22B-FP8
Configuration surface: `{TP4/EP1, TP8/EP8} x MNS in {64,128}`, with
`MBT=8192`. Each workload has 129 requests per cell and each real cell uses
three fresh-server trials.
| Workload | TTFT mean | TTFT p90 | TPOT mean | TPOT p90 | E2E mean | E2E p90 |
|---|---:|---:|---:|---:|---:|---:|
| Trace-PD | 0.0% | 0.0% | 0.0% | 0.0% | 0.6% | 6.2% |
| Fixed-PD, 4096->256, 0.2 req/s/GPU | 4.2% | 0.2% | **33.0%** | **37.2%** | **30.7%** | **34.6%** |
| Trace-PO, OSL=1 | 7.0% | **21.2%** | N/A | N/A | 7.0% | **21.2%** |
| Fixed-PO, 4096->1, 0.2 req/s/GPU | 5.9% | 1.7% | N/A | N/A | 5.9% | 1.7% |
Interpretation: Trace-PD is mostly near-optimal. Fixed-PD reverses the real
decode/E2E preference between the tested parallel configurations and incurs
31--37% regret. Trace-PO also has a material p90 failure of 21.2%.
## Decision
The tested Frontier stack has **not** solved serving configuration tuning.
Its selected configuration can be near-optimal for one workload and materially
wrong for another on the same model and hardware. The strongest current
counterexamples are Qwen3-30B Fixed-PD TTFT and Qwen3-235B Fixed-PD TPOT/E2E.
This statement is limited to the two tested MoE models and Frontier. It is not
yet evidence about dense models, Vidur/APEX as separately reproduced systems,
other hardware, or SLO-constrained tuning.
## Provenance
Primary immutable analysis artifacts on `dash0`:
- Qwen3-30B Trace-PD: `/home/admin/cpfs/wjh/aituner/graph-piecewise-qwen30-20260717/simulator-piecewise-surface-v2/analysis/comparison.json`
- Qwen3-30B Fixed-PD/PO: `/home/admin/cpfs/wjh/aituner/qwen30-fixed-pressure-surface-20260719-r1/analysis/`
- Qwen3-30B Trace-PO: `/home/admin/cpfs/wjh/aituner/qwen30-latency-expansion-20260718-r2/analysis-r6/trace-po-comparison.json`
- Qwen3-235B four-case matrix: `/home/admin/cpfs/wjh/aituner/qwen235-v020-fourcase-20260719-r1/analysis/comparison.json`
The Qwen3-235B artifact root includes `provenance/artifacts.sha256`; the final
matrix contains 48/48 valid real trials and 16/16 complete simulator cells.

View File

@@ -0,0 +1,286 @@
# Telemetry-conditioned residual tuning roadmap
Status: **R0 COMPLETE / FAILED; R1 AND R2 CLOSED FOR THIS MODEL**.
Date: 2026-07-14 (Asia/Singapore).
## Research question and claim boundary
The question is whether a small number of real engine observations can correct
a simulator's task-specific error over **unmeasured configurations**, and
whether that correction reduces the real-GPU cost of finding a high
SLO-goodput serving configuration.
The intended headline claim, if the evidence supports it, is:
> An engine-state-conditioned residual model turns a simulator prediction into
> a task-specific posterior over unmeasured serving configurations, allowing a
> sequential tuner to reach near-oracle SLO-goodput with materially fewer
> H20-hours than simulator-only and outcome-only tuning.
Classification accuracy, simulator-error diagnosis, and telemetry overhead are
supporting evidence. None is an end-to-end tuning contribution by itself.
The following method is closed and will not be revived under another name:
per-candidate five-second accept/reject as the headline contribution. The P1
result showed only 1.426% cost reduction in the frozen `k=2` workflow.
## Two models, one evaluation
Both branches use the same legal candidate set, real measurements, task split,
cost accounting, and acquisition function.
### Simulator-residual branch (primary)
For measured anchor `c_t` and unmeasured candidate `c'`:
```text
y_hat(c') = y_real(c_t)
+ [y_sim(c') - y_sim(c_t)]
+ f(state_real(c_t) - state_sim(c_t), c' - c_t, workload, SLO)
```
The simulator delta is the prior. The learned model may correct it only with
training-supported state/config transitions; uncertainty or distribution shift
must shrink the correction back toward the simulator prior.
### Telemetry-only branch (mandatory)
```text
y_hat(c') = y_real(c_t)
+ g(state_real(c_t), c' - c_t, workload, SLO)
```
This branch tests whether the simulator is actually necessary. It does not
use a hand-authored bottleneck-to-knob rule.
### Search policy
Legal configurations are enumerated independently of telemetry. A generic
cost-aware acquisition rule ranks candidates from predicted improvement,
uncertainty, and measured H20 cost. The current production harness's
bottleneck scores, topology-first ordering, and hand-set relief constants are
not consumed by either branch. The validator may enforce legality,
full-config no-repeat, failure accounting, and resource caps only.
## Hypotheses
| ID | Hypothesis | Direct test | Failure meaning |
|---|---|---|---|
| H0 | Existing artifacts can express a common, direct-measurement state without heuristic labels. | Engine/simulator extractor coverage and invariants. | Route is not currently implementable. |
| H1 | Simulator errors are predictable from engine/simulator state discrepancy at measured anchors. | Task-held-out pairwise inversion correction and new-inversion rate. | Telemetry is diagnostic but cannot correct the surface. |
| H2 | Telemetry alone predicts useful config transitions beyond outcome-only history. | Telemetry-only versus real-outcome-only sequential replay. | Direct telemetry-guided tuning has no independent value. |
| H3 | Residual correction changes actual tuning decisions and cost. | H20-hours to 95% oracle and regret AUC against the strongest safe baseline. | No system contribution even if H1/H2 prediction metrics improve. |
## Common-state contract
Only directly observed or exactly reconstructed quantities are admitted.
| Quantity | vLLM Layer-1 | Frontier | R0 status |
|---|---|---|---|
| Scheduled requests / batch size | Per scheduler step | Existing per-batch metric, disabled in P1 output | Common after CPU replay |
| Scheduled prefill/decode tokens | Per scheduler step | Existing per-batch metrics | Common after CPU replay |
| Scheduler/batch rate | Monotonic step timestamps | Batch count / simulated duration | Common after CPU replay |
| Waiting queue area | Time-weighted queue gauge | Sum of request waiting times | Common aggregate |
| Running request area | Time-weighted running gauge | Sum of E2E minus waiting time | Common aggregate, semantics audited |
| Preemption count | Per step | Per request | Common |
| KV usage/headroom | Exact blocks and ratio | Not in committed output | Engine-only until exact reconstruction exists |
| CUDA graph mode/padding | Exact per step | Not modeled | Engine-only omitted-mechanism signal |
| Request TTFT/TPOT/pass rate | Exact real outcomes | Exact simulated request metrics | Common outcome, not state |
Unavailable fields remain null. They cannot be imputed from a human
`prefill/decode/queueing` label.
Frontier already contains the required detailed batch and timestamped
stage-batch ledger output. P1 disabled it for artifact size. R0 replays the
same immutable fixtures with the existing output flags enabled; it does not
change the simulator model or calibration.
## Data separation
- Phase 6 / `chat_w20260311_1000`: development only.
- P1 / `chat_w20260312_1000`: development only.
- R1 / `chat_w20260313_1000`: new development surface.
- R2: trace windows not used for feature, model, threshold, candidate-space,
cutoff, or acquisition decisions.
- Splits are by complete workload/SLO task. Anchor- or pair-level random
splits are prohibited.
- Sequential-policy seeds measure algorithmic variability; they are not
counted as independent system tasks.
The two existing development tasks have an important limitation: the now-
available SLO-gated simulator reading already retains the real oracle at its
top rank/tie. They therefore cannot establish a positive end-to-end ranking
claim. They are used for plumbing, known false-feasible cases, and negative
evidence. R1 must be run as an unbiased complete surface, not selected after
observing simulator success or failure.
## Step-by-step roadmap
### R0.1 — Inventory and roadmap
Deliverables:
- this roadmap;
- rolling untracked `ONGOING.md`;
- exact engine/simulator field and artifact inventory.
Gate: every claimed input has an authoritative file path and provenance.
### R0.2 — Common-state plumbing
Deliverables:
- `runs/telemetry-residual/common_state.py`;
- synthetic correctness tests;
- one exact P1 Frontier replay with individual batch metrics and the full
stage-batch ledger enabled;
- paired engine/simulator state summary for the same fixture.
Gate:
- replay request count and SLO scorer exactly agree with the committed replay;
- batch/ledger outputs are non-empty;
- all counters are non-negative, ratios bounded, times monotonic;
- no GPU is visible to Frontier;
- output volume is practical before expanding to twelve replays.
### R0.3 — Development residual/headroom audit
Use all frozen P1 primary fixtures and corresponding engine intervals. Produce:
- common-state residuals per anchor;
- simulator-error labels and continuous SLO/goodput residuals;
- ordered source/target diagnostic that removes both config identities from
both roles in every training fold;
- oracle upper bound for cross-candidate correction;
- explicit comparison with simulator+outcome and telemetry-only features.
R0 is a feasibility gate, not headline evidence. Proceed to R1 only if:
1. state features are collected with the measured source anchor, vary across
cells, and are available before any target config is evaluated;
2. at least one known simulator error has a state discrepancy not exposed by
the matched external prefix outcome;
3. a prior-preserving model can correct development errors without introducing
a larger number of new errors under regularization sensitivity;
4. an oracle cross-candidate correction has at least 15% sequential tuning-cost
headroom under full startup/warm-up accounting.
### R0 result and decision
R0 completed without a data-validity red flag, but failed condition 3. The
decision is **STOP_BEFORE_R1**; no H20 job was launched for this route.
- All 12 detailed Frontier CPU replays exactly reproduced their committed SLO
scorers. Runtime was 23.943--54.786 seconds per replay, detailed artifacts
were 4.12--13.53 MB, CUDA visibility was empty, and there were zero failures.
- The paired surface contains 12 real/sim anchors, two known simulator
false-feasible anchors, and 120 legal cross-config ordered transitions. A
fold removes both the source and target TP/MNS identity from source and
target roles; the two offered-load anchors remain part of the same task.
- Raw Frontier feasibility is 83.33% on the repeated transition view. The
structurally correct hybrid model uses
`r_target = r_source + delta_r`; the direct model uses
`y_target = y_source + delta_y` and never reads simulator fields.
- Direct telemetry is not robust relative to real-outcome-only: its accuracy
delta over L2 `{0.1,1,10,100}` is `{-0.83,+1.67,0,-4.17}` percentage points,
and its best absolute accuracy is 54.17%, below the raw simulator's 83.33%.
- Hybrid telemetry raises classification accuracy over the corresponding
simulator+outcome transition regression by 1.67--4.17 percentage points,
but worsens pass-rate RMSE by 0.141--0.201 and MAE by 0.084--0.125. Its full
correction reaches only 46.67--53.33% absolute accuracy.
- Across 24 nonzero `(L2, raw-simulator-prior weight)` combinations, no model
both corrects an existing simulator error without more new errors and avoids
worsening RMSE/MAE. Whenever a correction fixes at least one error, it
corrupts at least 11 previously correct transitions.
- A perfect correction could skip the frozen simulator rank-2 real final and
save 0.043469 H20-hours: 15.45% of the prospective online `k=2` cost, or
14.40% when the prior failed launch is charged. On this development task the
simulator top-1 already is the real oracle with zero regret, so headroom
versus the observed-safe top-1 baseline is 0%.
The result does not prove that engine telemetry is useless. It shows that the
current one-task anchor-transition evidence cannot support either a safe
simulator-residual tuner or a simulator-free telemetry tuner. A larger model
or an R1 run would add capacity/data after a failed gate and is therefore not
authorized under this roadmap.
### R1 — New development surface
Status: **NOT LAUNCHED; CLOSED BY R0**.
Frozen starting setup:
- host: dash0, eight NVIDIA H20 GPUs;
- cells run solo; no co-location for SLO verdicts;
- patched vLLM 0.24.1.dev3, Qwen3-30B-A3B BF16;
- trace: `chat_w20260313_1000`;
- output tokens: exactly 128;
- SLO: stepped TTFT 2/4/6 seconds, TPOT 50 ms, pass rate at least 0.95;
- config surface: TP `{1,2,4}` × MNS `{8,16,32,64}`;
- hard campaign cap: 4 H20-hours.
The load ladder, repetitions, randomized order, exact commands, expected wall
time, and artifact paths are frozen only after R0. A resolved echo is required
before launch.
R1 passes only if a frozen sequential replay shows at least 15% E2E H20-hour
headroom over the strongest safe baseline with final regret at most 5%. R1 is
development evidence and cannot be reported as the held-out result.
### R2 — Held-out sequential tuning
Status: **NOT LAUNCHED; CLOSED BY R0**.
Required baselines:
1. random search;
2. real-outcome-only Bayesian/sequential search;
3. Frontier ranking plus real top-k final;
4. simulator plus real-outcome residual;
5. telemetry-only transition tuner;
6. simulator plus telemetry residual tuner;
7. complete real surface as oracle, not as a cost competitor.
Primary metric: end-to-end H20-hours to first reach 95% of the real full-surface
SLO-goodput oracle. Secondary metrics are cost-normalized regret AUC, final
regret at fixed budgets, oracle false-prune, wall time, and per-task regressions.
The route is successful only if the winning telemetry method reduces the
primary cost by at least 20% versus the strongest safe baseline and ends within
5% regret on every headline task. If hybrid beats telemetry-only by at least
10%, simulator residual correction is the primary method. If telemetry-only
is within 5% or better, the simulator dependency is removed. If neither clears
the contribution bar, the route is closed and telemetry remains a diagnostic
facility only.
## Cost discipline
- R0 simulator work is CPU-only and must set empty CUDA visibility.
- R1 cannot exceed 4 H20-hours.
- R2 receives no budget until R1 passes.
- Startup, warm-up, burn-in, failed launches, real probes, continuation, and
final validation are charged. Benchmark-only annotation repeats are
reported separately and cannot disappear from campaign accounting.
## Final R0 sanity block
| Data | n | Min | Max | Distinct | Checked invariant |
|---|---:|---:|---:|---:|---|
| Phase 6 cells | 12 | TP1/MNS8 | TP4/MNS64 | 12 | Surface not identical; solo SLO tier authoritative |
| Phase 6 Layer-1 primary steps | 37 streams | 343 | 12,103 | 37 | Contiguous; zero drops |
| P1 primary anchors | 12 | infeasible | feasible | 2 labels | 7 feasible / 5 infeasible |
| P1 Frontier runtime | 12 | 24.093 s | 54.575 s | 12 | CPU-only; zero failures |
| Detailed Frontier replay runtime | 12 | 23.943 s | 54.786 s | 12 | Exact committed scorers; CUDA hidden |
| Detailed artifact bytes | 12 | 4,123,724 | 13,527,776 | 12 | Non-negative; practical CPU replay size |
| Cross-config transitions | 120 | real pass 0.1067 | real pass 1.0 | 6 outcomes | Both endpoint config identities held out |
| State residual vectors | 12 | 16 fields | 16 fields | 12 vectors | Finite; no missing common field |
| R0 E2E cost values | 4 | 0.237914 | 0.301935 H20-h | 4 | Non-negative; `k=1/2`, online/conservative |
Checked invariants: non-negative counts and costs; pass rates in `[0,1]`;
simulator results not all identical; exact request count/hash agreement; Layer-1
step continuity and zero drops; no co-resident SLO measurements; no calibration
or evaluation split reuse for a future headline claim. No current red flag
invalidates R0 plumbing. The R0 tuning gate itself failed because safe
prior-preserving correction was absent.

View File

@@ -0,0 +1,391 @@
# AITuner tuning核心挑战、统一成本口径与研究路线
日期2026-07-15Asia/Singapore
状态:**问题定义与历史成本审计完成;新的 tuner 贡献尚未建立。**
## 结论先行
我们不应该把 tuning 定义成“根据当前 telemetry 判断哪个 cap 满了,再调对应 knob”。这个定义同时遗漏了 knob interaction、反事实识别、实验成本和跨任务失配。更准确的问题是
> 给定模型、engine version、hardware、workload、SLO 和一个声明好的合法配置空间tuner 如何用最少的真实 GPU 成本,依次选择可能包含多个 knob 的 intervention找到 SLO-goodput regret 不超过 `epsilon` 的配置?
AITuner 可以形成的系统贡献应当是:
> **一个 intervention-calibrated、action-conditioned、cost-aware 的 tuner它从真实 engine trajectory 和已测 intervention 中学习联合 config action 的反事实收益分布,并以 cost-to-oracle 而非规则命中率作为目标。Harness 只负责实验语义、合法性、配对、记账和可复现性,不负责用人工 bottleneck rule 决定 action。**
现有结果支持这个问题值得做,但不支持宣称它已经解决:
- 在真实 `TP x MNS` surface 上one-knob-at-a-time 会停在比 oracle 低 **25.6%** 的 coordinate-wise local optimum。
- 在 action-aware pilot 中,增加 MBBT 在“几乎从未独占打满 MBBT cap”的情况下仍把 source goodput 提高 **48.0%--77.1%**;因此 `cap -> knob` 不是完整模型。
- 同一 dash0 任务上,当前 guided harness 到 5% empirical regret 只比纯 LLM 少 **5.85%** H20-hours到 2% regret 则少 **61.09%**。这说明必须比较完整 cost--regret curve不能只比较最终最好值。
- Frontier 的 decision-bearing throughput top-1 在 12-cell surface 上有 **30.46%** real regret。Simulator 本身的边际 GPU cost 是 0但通过 real-final 恢复 oracle 需要 tie-expanded 4 个真实 cell**0.7828 reconstructed H20-hours**
## 1. Tuning 问题和成功标准
固定 task context
```text
T = {model, engine build, hardware, workload/trace, SLO, legal config space C}
```
每个完整配置 `c in C` 的目标为:
```text
f_T(c) = max request_rate_per_gpu
subject to request SLO pass rate >= target
```
有限空间 oracle 为:
```text
f*_T = max_{c in C} f_T(c)
regret(c) = 1 - f_T(c) / f*_T
```
顺序 tuner 在第 `t` 步基于历史 `D_t` 选择一个完整 config intervention
```text
a_t = c_t -> c_{t+1}
```
成功不是“最后找到一个不错的值”,而是同时满足:
1. `regret(best_t) <= epsilon`
2. 达到该点之前的 all-in H20-hours 最小;
3. launch、correctness、SLO 和失败率约束不退化;
4. 结论在 held-out task 上成立,而不是在用于设计规则的 task 上成立。
### 1.1 GPU cost 的统一定义
未来实验的 task-marginal cost 应定义为:
```text
C_task = sum_j allocated_GPU_count_j
* (GPU_idle_or_release_time_j - allocation_start_time_j)
```
它包括 method 实际触发的 startup、warm-up、prefix/full replay、confirmation、failure、cleanup如果 LLM 思考期间 GPU 仍被占用也计入。Simulator/模型的一次性 onboarding 成本单独报告:
```text
C_e2e(N tasks) = C_profile_or_training / N + C_task
```
另外报告 CPU-hours、LLM API latency/cost但不把它们伪装成 GPU-hours。构建 benchmark oracle 的 exhaustive annotation cost 是公共评测成本,单独报告,不计入任何方法;同时可给一个将其等量加回所有方法的 conservative view。
历史记录没有 allocation start/release timestamp。本次只能从每个 `engine.log` 的首末时间戳重建:
```text
C_engine_lower_bound = parallel_size * engine_log_span / 3600
```
因此下面所有历史 H20-hour 数字都是 **engine-lifetime lower bound**,不是 all-in cost。尤其 simulator 的一次性 H20 operator profiling 成本没有记录,不能称为完全免费。
### 1.2 两种 oracle 必须分开
- **Exact finite-surface oracle**:声明好的 12-cell `TP x MNS` 空间全部真实测量oracle 是 `TP2/MNS32 = 3.2833 req/s/GPU`
- **Broader empirical reference**dash0 两个 sequential run 中观察到的最好值 `3.35 req/s/GPU`。它包含 surface 外的 MBBT/chunk/GMU action但只是 best observed不是全局 oracle。
不能把 empirical best 写成 global oracle也不能让每个方法使用不同的 oracle 定义。
## 2. 现有方案的 cost-to-oracle 审计
可复算输入和完整结果在:
- `runs/tuning-cost/manifest.json`
- `runs/tuning-cost/analyze.py`
- `runs/tuning-cost/metrics.json`
### 2.1 严格同任务对照:纯 LLM vs 当前 guided harness
两组均为 dash0、Qwen3-30B-A3B、community-vLLM 0.20.0、8xH20 可见、`chat_w20260311_1000`、input 0--8k、output 128、replay scale 0.1、TTFT 2/4/6s、TPOT 50ms、pass rate 0.95。除 tuner method 和服务端口外,固定 task spec 相同。
Reference 是两组中 best observed `3.35 req/s/GPU`
| Method | 到 <=5% regret | 到 <=2% regret | 到 <=1% regret | 完整 run 成本 | 最终 best |
|---|---:|---:|---:|---:|---:|
| Pure LLM, no harness | 0.2847 H20htrial 2regret 2.736% | 1.1458trial 6regret 1.493% | 1.3719trial 7regret 0% | 2.2825 | 3.35 |
| Guided harness v2 | 0.2681 H20htrial 2regret 2.736% | 0.4458trial 3regret 1.990% | 未达到 | 0.6231 | 3.30regret 1.493% |
直接结论:
- 5% endpointguided 比 pure LLM 少 **5.85%**,不是 material contribution。
- 2% endpointguided 比 pure LLM 少 **61.09%**,有明显 headroom signal但只有一个 task不能外推。
- Pure LLM 在 trial 7 已找到 best observed之后又花了 `2.2825 - 1.3719 = 0.9106 H20h` 而没有改进,说明 trustworthy stopping 本身就是成本来源。
- Pure LLM 的 trial 3 使用当前 binary 不支持的 `--expert-parallel-size` 并在 launch 前失败。当前 harness 的 legality/version contract 有实际价值,但它仍不是性能 action-ranking 贡献。
### 2.2 Simulator零边际 GPU cost 不等于零 tuning cost
Frontier fidelity suite 在 CPU 上执行 184 个 simulation耗时 **2.055 CPU-hours**simulation 本身为 0 marginal H20-hours。其对应的 exact dash1 12-cell real surface annotation lower bound 为 **3.5953 H20-hours**
Decision-bearing `frozen-calibrated/throughput-proxy`
| Policy | Real cells evaluated | Real-final H20h lower bound | Selected real regret |
|---|---:|---:|---:|
| Simulator-only top-1 | 0 | 0 | **30.46%**,选 TP1/MNS64 |
| Throughput top-1 + real final | 1 | 0.1353 | **30.46%** |
| Throughput top-2 + real final | 2 | 0.2672 | **30.46%** |
| Throughput nominal top-3 + real final | tie-expanded 4 | 0.7828 | 0%,找到 TP2/MNS32 |
Post-hoc `SLO-gated` reading 把 `{TP2/MNS32, TP2/MNS64}` 放在 top tie bucket测两个 cell 需 **0.5156 H20h** 并能找到 oracle。但它不是 preregistered decision-bearing policy而且 anchor verdict 中有 21 个 false-feasible、7 个 false-infeasible只能作为诊断上界不能反写成 prospective simulator 结果。
Pure LLM/harness 数据来自 dash0simulator exact surface 来自 dash1。模型、engine、trace、GPU type 匹配,但 host 和 campaign 不同。因此两块内部可以直接比较,跨块只能做 development-level 指示paper 结论必须在同 host、同 task execution protocol 下重跑。
### 2.3 我们要达到的成本目标
在当前 reconstructed lower-bound 口径下,一个有意义的单任务 development bar 是:
| Endpoint | 当前最强同任务 baseline | 20% reduction bar | 兼顾 post-hoc sim+real 的 30% bar | 暂定目标 |
|---|---:|---:|---:|---:|
| <=5% empirical regret | guided 0.2681 | 0.2144 | 0.3609 | **<=0.2144 H20h** |
| <=2% empirical regret | guided 0.4458 | 0.3567 | 0.3609 | **<=0.3567 H20h** |
这两个数字不是 paper result只用于检查 proposed method 是否有足够 headroom
- 5% endpoint 已经由 baseline + TP2 两个完整 trial 达到。任何必须先跑 source 再跑 target 的 telemetry tuner 都不能靠减少 trial count 获得 20% 优势;它必须能够 one-shot warm-start、跳过 baseline或安全地缩短其中一次测量。
- 2% endpoint 有更合理的结构性空间:从一个 source 直接选择 joint `TP2 + MBBT/chunk` target可能跳过当前中间 trial如果仍按当前三次完整 trial 顺序执行,就不会达到 bar。
Paper-facing gate 不使用这些跨 campaign 绝对数,而使用 prospective same-host all-in cost在每个 held-out task 上 regret <=5%,相对最强 safe outcome-only/current harness 至少省 20%,相对 frozen simulator+real 至少省 30%,并报告 task-level paired confidence interval。
## 3. 四个最核心的 tuning challenge
### Challenge 1响应面是联合、条件化且 regime-dependent 的
#### 问题本质
一般情况下:
```text
f(c) != base + sum_k effect_k(c_k)
```
一个 knob 的 effect 是当前完整 context 的函数:
```text
Delta_x(c, workload, engine state)
```
它可能随 topology、另一个 runtime knob、load、SLO 或 engine version 改变大小甚至改变符号。因此不能先分别求每个 knob 的最优值再 merge也不能固定一个低质量 context 去判断另一个 knob。
#### 已有真实证据
在 C1 12-cell real surface
- `MNS 8 -> 32` 在 TP1/TP2/TP4 下分别提升约 **8.7% / 44.3% / 90.3%**
- 从同一 `TP1/MNS8` 起点,先 tune MNS 再 TP 会停在 `TP4/MNS16 = 2.4417`;该点沿任一单维都没有 strictly improving move但 joint/global surface oracle `TP2/MNS32 = 3.2833`**34.5% relative to the local point**,即 local point 对 oracle 有 **25.6% regret**
- C3 中 `MBT 256 -> 384` 的 effect 根据 topology/MNS 从 0 到约 -9.2%`MNS 64 -> 128` 从 0 到约 +10.1%。
- Action-aware Regime A 中 MBBT 几乎从不作为 exclusive cap但 MBBT action 仍把 source goodput 提高 48.0%--77.1%。它通过 chunk size、prefill packing 和 scarce MNS slot residency 的联合变化获得收益。
这直接否定两类通用策略OAT/coordinate greedy以及 `which cap is full -> tune that knob`
#### Tuner 必须具备的能力
- Action 的基本单位是完整 `config delta`,允许 sparse joint action而不是孤立 knob/value。
- 对 topology/runtime family 使用 crossed anchors 或信息增益设计,主动测 interaction不是默认所有 interaction 都强。
- 能从数据判断 task 是 topology-dominant、runtime-interaction-dominant 还是 flat/noisy并据此分配实验而不是把固定 search order 写进规则。
### Challenge 2当前状态是 observational signaltuning 需要 counterfactual identification
#### 问题本质
一次 telemetry trace 只能告诉我们:
```text
P(engine trajectory | current config, workload)
```
Tuning 真正需要的是:
```text
P(Delta SLO-goodput, failure, cost
| source trajectory, proposed full-config action)
```
Queue、KV、padding、split prefill 等状态既可能是原因,也可能是 workload/config 的结果。看见某种状态,不等于知道哪个 action 能修复它。一个 action 也可能同时改变多条机制;例如 MBBT 同时改变总 token budget、per-request chunk 和 multi-request packing现有 telemetry 的解释是 mechanism-consistent不是已完成的 causal decomposition。
#### 已有真实证据
- 5/10 秒 telemetry 确实太短300 秒 phase-aware experiment 中MNS action 的 queue/padding 机制直到 replay 75%--100% 才稳定出现。
- 但 external TTFT outcome 在 25% 已完美区分该 action 是否修复 SLO。Telemetry 解释了 why却没有比 outcome 更早或更可靠地指导 tuning。
- 3.125 req/s/GPU 的 source 无法在 timeout 内 drain另一组 source 已达 offered ceiling 的 99.1%--100%,数学上不可能通过 10% improvement gate。没有 exposure/headroom 和 censoring control模型学到的不是 action response。
- Same-config repeats 与 matched intervention 的波动不可忽略;只比较两个未经配对的 run 会混入 arrival/order/warm-state noise。
#### Tuner 必须具备的能力
- 训练样本必须是 exact-workload paired intervention`(source trajectory, action) -> target delta`,保留失败和 censoring。
- 使用 phase-binned continuous trajectory而不是人工 bottleneck label 或 threshold rule。
- 输出 response distribution 和 uncertainty证据不足时 abstain而不是强行给 diagnosis。
- Telemetry 的价值必须通过同 cutoff、同 model capacity 的 outcome-only ablation 证明。若不能降低 end-to-end H20-hoursinstrumentation 只保留为 debugging/解释工具。
### Challenge 3这是异构成本下的 sequential experimental design不是静态 ranking
#### 问题本质
每个 trial 的成本不同TP4 是 TP1 的四倍 GPU multiplierstartup/warm-up 可能主导短 probe失败也有成本同时 tuner 不知道 oracle只能在 exploitation、information gain 和 cost 之间权衡。选对 top-1 的 accuracy 不能代表 tuning 效果。
必须回答三个连续问题:
1. 下一次测哪个联合 action
2. 测多久,何时 continuation/confirmation
3. 什么证据允许停止,并声称 best 已在 `epsilon` 内?
#### 已有真实证据
- Pure LLM 达到 best observed 后仍浪费 0.9106 reconstructed H20h。
- Simulator top-1 虽然 0 marginal GPUh却因 rank error 损失 30.46%real-final 的 k 增大又迅速增加 H20h。
- 5% endpoint 上两个方法都只需两个 trialselection-count headroom 很小2% endpoint 才暴露 action quality 和 stopping 的巨大差异。
- Prefix 不是天然便宜:如果 startup、warm-up 和稳定状态形成占主要成本,缩短 replay window 未必带来等比例 H20h reduction。
#### Tuner 必须具备的能力
- Acquisition 直接优化 expected regret reduction / predicted H20 cost并把 failure probability 纳入约束。
- 在 run 前做与 tuning policy 分离的 workload admissibility check避免 outcome ceiling、无法 drain、无请求或 measurement cap。
- 使用 uncertainty-aware continuation 和 stopstop criterion 针对声明的 candidate set 中“仍存在 >epsilon improvement 的概率”,而不是连续几次没提升。
- 主结果报告 H20-hours-to-5%/2%/1%、fixed-budget regret 和 cost-normalized regret AUC不 metric shopping。
### Challenge 4任何 mechanism model 都有 fidelity 和 transfer boundary
#### 问题本质
Simulator、learned surrogate、LLM prior 都是近似。Workload、SLO、model、hardware、engine version 改变后operator cost、scheduler state transition、合法 flag 和 response surface 都可能变化。模型在 calibration task 上解释得好,不表示能在 held-out task 上排序正确。
#### 已有真实证据
- Frontier throughput reading 在完全匹配的 12-cell task 上仍把 real oracle 排错top-1 regret 30.46%。这说明预测绝对 throughput 还不够,局部 rank fidelity 才是 tuning 关键。
- Post-hoc SLO reading 的 top bucket 正确,但有大量 anchor feasibility error也没有 prospective policy status。
- Pure LLM 提出了当前 community-vLLM binary 不支持的 flagengine/API version knowledge 本身会漂移。
- 已有 cross-version experiment 中 vLLM 0.20 的强配置在 0.24 上出现大幅退化,说明 response prior 不能无条件迁移。
#### Tuner 必须具备的能力
- Simulator 只能作为 prior mean 或 candidate prior真实 outcome 是 authoritative update。
- 学习 simulator residual`sim prediction + source state + action` 映射到 real response而不是用 telemetry 重新实现另一个无校准 simulator。
- 对 task-level OOD 显式提高 uncertainty/abstaintrain/test 按完整 task 分割,不能按 request、anchor 或同一 surface cell 随机分割。
- 分开报告 cold-start profile/training cost 与 per-task marginal cost并在 N=1/10/100 等 amortization horizon 下展示。
## 4. 对应的系统设计
### 4.1 Harness从 rule-based tuner 收缩成 experimental control plane
Harness 保留以下确定性职责:
- engine-version-aware config schema、合法性和资源约束
- 完整 config/action canonicalization禁止隐式 merge 和重复试验;
- exact trace/request/arrival/length hash配对、随机化和 counter-rotation
- engine trajectory、external outcome、failure/censoring 的统一时间轴;
- all-in GPU cost ledger、oracle annotation 分账、budget enforcement
- data sanity、coverage、SLO/correctness 和 stop-proof audit。
Harness **不**包含 `queue > N -> increase MNS``cap full -> tune knob` 或人工 diagnosis-to-action mapping。这里的规则是实验语义和安全 invariant不是性能决策 heuristic。
### 4.2 Action-conditioned response model
每条学习记录为:
```text
x = {source full config,
workload/SLO context,
source external outcome,
phase-binned engine trajectory}
a = normalized full-config delta
y = {Delta SLO-goodput, target feasibility/failure, measured H20 cost}
```
学习:
```text
p_theta(y | x, a, optional simulator prediction)
```
第一版应使用适合小数据且有 uncertainty 的 action-conditioned Gaussian-process/bootstrapped surrogatekernel/feature ablation包括
1. config + external outcome
2. 同样输入 + telemetry trajectory
3. simulator + config + outcome
4. 同样输入 + telemetry residual features。
Telemetry 保留 continuous phase distributionsqueue/running residency、MNS/token slack、prefill/decode composition、partial/split prefill、step duration、KV、graph/padding。模型学习它们与 action 的 interaction不先压成 bottleneck label。
### 4.3 Cost-aware policy
在合法的 single/joint candidate set 上选择:
```text
a* = argmax_a
expected constrained improvement(a)
/ expected all-in H20 cost(a)
```
探索项来自 posterior uncertainty/information gainlaunch/SLO failure 有显式 penalty。Simulator 可提供 prior mean但 simulator 与 real discrepancy 会被 posterior residual 更新。一次 target measurement 后更新 response model并重新计算下一步 action 或停止概率。
LLM 在这个 tuning core 中不是 telemetry classifier。它最多作为可移除的 candidate/prior source提出 schema 内的 sparse joint actions 或检索 engine mechanism每个 proposal 都由同一个 response model、cost acquisition 和 real validator 评分。只有 `with LLM` 相对 `same tuner without LLM` 在 held-out tasks 上继续降低 cost-to-oracle才能讨论 LLM 必要性。
### 4.4 Stop 条件
对一个预先声明的有限 candidate set满足以下条件才 stop
```text
P(exists c: f(c) > best_observed / (1 - epsilon) | D_t) < alpha
```
并且 best config 通过独立 confirmation、SLO/correctness gateremaining candidate 的 cost-aware value of information 低于阈值。停止原因、posterior coverage 和未测区域必须写入 audit。
## 5. 下一阶段如何证明,而不是再次构造 heuristic
### R0已有数据 retrospective premise check
- 用 C1/C3 response surfaces 检查 joint model 是否能避免 OAT trap。
- 用 action-aware paired records 比较 outcome-only 与 +telemetry 的 action-delta calibration。
- 用 SimFid surface 比较 direct model 与 simulator-residual model 的 rank/regret。
- 所有 feature、kernel、candidate encoding 在 held-out task 结果之前冻结。
R0 只能筛选 model family不能作为 paper result因为现有 tasks 已参与路线设计。
### R1prospective same-host cost-to-oracle pilot
- dash0 8xH20固定 engine build/modelserialized placement禁止共置干扰。
- 至少一个未参与 feature/threshold 选择的新 trace window选择非 ceiling、可 drain 的 offered load。
- 声明一个可穷举的小 surface至少包含 topology/runtime crossed actions而不是只有一个 MNS ladder。
- Oracle annotation 与 tuner online actions 分开记账method 只能看到当时可用的数据。
- 运行 random/search、OAT、纯 LLM、当前 guided harness、frozen simulator+real、outcome-only response、+telemetry response、sim-residual +telemetry。
- 比较完整 H20 cost-to-regret curve而不是 action classification accuracy。
Pilot opening gate
1. telemetry model 相对相同 response model 去掉 telemetry确实改变至少一个正确的 prospective action ranking
2. 最终 regret <=5%,无 false-safe accept
3. all-in H20-hours 相对 strongest safe outcome-only 至少下降 20%
4. 如果使用 simulator需相对 frozen simulator+real 至少下降 30%
5. instrumentation overhead <=1%,所有成本和失败均计入。
若 1--5 任一失败,就不能把 telemetry/harness 写成 tuning contribution保留其 debugging/measurement 价值即可。
### R2task-held-out replication
至少 3 个 workload window x 2 个 SLO regime按完整 task 做 leave-one-task-out 或固定 train/test split。报告每个 task 的 regret、安全和成本以及 task-level paired bootstrap CI。只有 R2 通过,才能把单 task 的 61.09% lower-bound saving 升级为项目贡献。
## 6. 当前能说与不能说的贡献
当前能说:
- 我们有真实反例证明 OAT 和 cap-to-knob mapping 不是通用 tuning strategy。
- Harness 的 legality、exact replay、failure/cost accounting 有必要的实验基础设施价值。
- 当前 guided sequence 在一个严格同任务比较中显著减少了达到 2% empirical regret 的 reconstructed engine cost。
- Simulator 的边际计算便宜,但 rank error 会转化成显著 real regret 或更多 real-final 成本。
当前不能说:
- telemetry 已经对 end-to-end tuning 提供独立增益;现有 direct pilot 对此为 negative。
- 当前 harness 的 heuristic action ranking 是系统贡献5% endpoint 只省 5.85%。
- LLM 是必要组件;尚无同 policy 的 with/without LLM held-out ablation。
- simulator 总 tuning cost 是 0profile GPU cost 未审计real verification 不能忽略。
- 3.35 是 global oracle或 dash0 与 dash1 数字是完全 controlled comparison。
## Data sanity
- Dash0 sequential numeric scoresn=9min/max `1.1042/3.35`distinct=7两组 config outcome 不全相同。
- Exact surface scoresn=12min/max `1.2833/3.2833`distinct=812 cells 完整且与 simulator metrics 中的 real scores 一致。
- Reconstructed trial/cell attempts 包括 4 个无 engine timestamp 的失败n=32min/max `0/0.49778 H20h`distinct=26所有可重建成本均非负。
- Sequential regret observationsn=16min/max `0/0.34328`distinct=6全部在 `[0,1]`
- Checked invariantsdash0 fixed task contexts 相同(除 method/porttrial counts 与 manifest 相符engine log timestamps monotonicsurface cell 唯一且 MBT=8192simulator 无失败且 predictions 不全相同scores/results 不全相同cost 非负regret bounded。
- Measurement limitationprimary 12-cell campaign 的 4 个 TP4 pre-ready failure 没有 engine timestamp随后由 companion campaign 完整重跑;其失败成本在 engine-lifetime reconstruction 中为 0。因此 `3.5953 H20h` 是 completed annotation lower bound不能作为 all-in annotation cost。这个缺口已显式保留没有在其上建立 total-cost claim。

View File

@@ -182,6 +182,59 @@ def rsync_pull(config: FleetConfig, host: HostSpec, remote_path: str, local_path
run_local(argv, cwd=config.project_root, capture_output=True, check=True)
def scp_push(config: FleetConfig, host: HostSpec) -> None:
ensure_remote_dir(config, host, host.sync_remote_path)
local_src = str(config.sync.local_path.resolve()) + "/."
remote_dst = f"{host.ssh_alias}:{host.sync_remote_path.rstrip('/')}/"
argv = [
"scp",
"-o",
"BatchMode=yes",
"-o",
f"ConnectTimeout={config.ssh_timeout_sec}",
"-r",
"-p",
local_src,
remote_dst,
]
run_local(argv, cwd=config.project_root, capture_output=True, check=True)
def scp_pull(config: FleetConfig, host: HostSpec, remote_path: str, local_path: Path) -> None:
remote_src = remote_path
if remote_path.endswith("/"):
ensure_dir(local_path)
remote_src = remote_path.rstrip("/") + "/."
else:
ensure_dir(local_path.parent)
argv = [
"scp",
"-o",
"BatchMode=yes",
"-o",
f"ConnectTimeout={config.ssh_timeout_sec}",
"-r",
"-p",
f"{host.ssh_alias}:{remote_src}",
str(local_path),
]
run_local(argv, cwd=config.project_root, capture_output=True, check=True)
def sync_push(config: FleetConfig, host: HostSpec) -> None:
if config.sync.mode == "rsync":
rsync_push(config, host)
else:
scp_push(config, host)
def sync_pull(config: FleetConfig, host: HostSpec, remote_path: str, local_path: Path) -> None:
if config.sync.mode == "rsync":
rsync_pull(config, host, remote_path, local_path)
else:
scp_pull(config, host, remote_path, local_path)
def ensure_remote_dir(config: FleetConfig, host: HostSpec, remote_path: str) -> None:
run_ssh(config, host, f"mkdir -p {shlex.quote(remote_path)}", capture_output=True, check=True)
@@ -206,8 +259,10 @@ def load_config(path: Path) -> FleetConfig:
local_path=relative_to_root(project_root, sync_raw.get("local_path"), project_root),
exclude=[str(item) for item in sync_raw.get("exclude", [])],
)
if sync.mode != "rsync":
if sync.mode not in {"rsync", "scp"}:
raise FleetError(f"unsupported sync.mode: {sync.mode}")
if sync.mode == "scp" and sync.exclude:
raise FleetError("sync.exclude is not supported for sync.mode=scp")
scheduler_raw = raw.get("scheduler", {})
scheduler = SchedulerSpec(
@@ -639,7 +694,7 @@ def harvest_run(config: FleetConfig, manifest: dict[str, Any]) -> dict[str, Any]
local_base = ensure_dir(config.artifacts_dir / refreshed["run_id"])
remote_run_dir = refreshed["remote_run_dir"].rstrip("/")
rsync_pull(config, host, f"{remote_run_dir}/", local_base / "remote_run")
sync_pull(config, host, f"{remote_run_dir}/", local_base / "remote_run")
for artifact in refreshed.get("artifacts", []):
artifact_remote = f"{refreshed['remote_sync_path'].rstrip('/')}/{artifact}"
@@ -652,7 +707,7 @@ def harvest_run(config: FleetConfig, manifest: dict[str, Any]) -> dict[str, Any]
check=False,
)
if check.returncode == 0:
rsync_pull(config, host, artifact_remote, target)
sync_pull(config, host, artifact_remote, target)
refreshed["harvested_at"] = utc_now()
write_run_manifest(config, refreshed)
@@ -866,7 +921,7 @@ def dispatch_jobs(
)
continue
if host.name not in synced_hosts:
rsync_push(config, host)
sync_push(config, host)
synced_hosts.add(host.name)
manifest = launch_job(config, host, job, gpu_ids)
manifests.append(manifest)

View File

@@ -0,0 +1,63 @@
#!/usr/bin/env python3
"""Add explicit MBBT/config provenance to the accepted Phase-6 replay client."""
from __future__ import annotations
import argparse
import json
import sys
from pathlib import Path
PHASE6 = Path(__file__).resolve().parents[1] / "opprof-phase6"
sys.path.insert(0, str(PHASE6))
import opprof_phase6_client as base # noqa: E402
def parser() -> argparse.ArgumentParser:
result = argparse.ArgumentParser()
result.add_argument("command", choices=("warmup", "run-anchor"))
result.add_argument("--study", required=True)
result.add_argument("--cell", required=True)
result.add_argument("--anchor", type=float, required=True)
result.add_argument("--tp", type=int, required=True)
result.add_argument("--mns", type=int, required=True)
result.add_argument("--mbbt", type=int, required=True)
result.add_argument("--base-url", required=True)
result.add_argument("--result-dir", required=True)
result.add_argument("--disable-slo-early-stop", action="store_true")
return result
def main() -> None:
args = parser().parse_args()
result = base.run_replay(args, warmup=args.command == "warmup")
result.update(
{
"schema": "action-aware-pilot-result-v0",
"config_id": args.cell,
"mbbt": args.mbbt,
}
)
base.atomic_json(Path(args.result_dir) / "result.json", result)
print(
json.dumps(
{
key: result[key]
for key in (
"config_id",
"mns",
"mbbt",
"kind",
"pass_rate",
"feasible",
)
},
sort_keys=True,
)
)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,697 @@
#!/usr/bin/env python3
"""Audit source-only constraint signals against crossed real interventions."""
from __future__ import annotations
import argparse
import hashlib
import json
import math
import os
import statistics
import sys
from pathlib import Path
from typing import Any, Iterable, Mapping
HERE = Path(__file__).resolve().parent
COMMON_STATE = HERE.parent / "telemetry-residual"
sys.path.insert(0, str(COMMON_STATE))
from common_state import summarize_engine # noqa: E402
SCHEMA = "action-aware-constraint-pilot-audit-v0"
def sha256_file(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for chunk in iter(lambda: source.read(1 << 20), b""):
digest.update(chunk)
return digest.hexdigest()
def atomic_json(path: Path, payload: Any) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
temporary = path.with_suffix(path.suffix + ".tmp")
temporary.write_text(
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
)
os.replace(temporary, path)
def numeric(values: Iterable[float]) -> dict[str, Any]:
finite = [float(value) for value in values]
if not finite:
raise ValueError("numeric summary requires values")
if any(not math.isfinite(value) for value in finite):
raise ValueError("numeric summary received non-finite values")
return {
"n": len(finite),
"min": min(finite),
"max": max(finite),
"distinct_n": len(set(finite)),
}
def distribution(values: Iterable[float]) -> dict[str, Any]:
finite = [float(value) for value in values]
summary = numeric(finite)
return {
**summary,
"mean": statistics.fmean(finite),
"p50": quantile(finite, 0.50),
"p95": quantile(finite, 0.95),
"p99": quantile(finite, 0.99),
}
def quantile(values: Iterable[float], probability: float) -> float:
ordered = sorted(float(value) for value in values)
if not ordered:
raise ValueError("quantile requires values")
position = probability * (len(ordered) - 1)
lower = math.floor(position)
upper = math.ceil(position)
if lower == upper:
return ordered[lower]
weight = position - lower
return ordered[lower] * (1.0 - weight) + ordered[upper] * weight
def load_jsonl(path: Path) -> list[dict[str, Any]]:
records = []
with path.open(encoding="utf-8") as source:
for line_number, line in enumerate(source, 1):
try:
records.append(json.loads(line))
except json.JSONDecodeError as error:
raise ValueError(f"{path}:{line_number}: invalid JSON") from error
return records
def binding_summary(
records: list[Mapping[str, Any]], *, mns: int, mbbt: int
) -> dict[str, Any]:
if not records:
raise ValueError("binding summary requires scheduler records")
counts = {
"mns_exclusive": 0,
"mbbt_exclusive": 0,
"both": 0,
"waiting_unresolved": 0,
"waiting": 0,
}
running_utilization = []
token_utilization = []
kv_usage = []
preemptions = 0
for record in records:
waiting = int(record["queues"]["waiting"]) + int(
record["queues"]["deferred"]
)
running = int(record["queues"]["running"])
scheduled_tokens = int(record["prefill_tokens"]) + int(
record["decode_tokens"]
)
if running > mns:
raise ValueError("running requests exceed configured MNS")
if scheduled_tokens > mbbt:
raise ValueError("scheduled tokens exceed configured MBBT")
mns_hit = waiting > 0 and running == mns
mbbt_hit = waiting > 0 and scheduled_tokens == mbbt
if waiting > 0:
counts["waiting"] += 1
if mns_hit and mbbt_hit:
counts["both"] += 1
elif mns_hit:
counts["mns_exclusive"] += 1
elif mbbt_hit:
counts["mbbt_exclusive"] += 1
else:
counts["waiting_unresolved"] += 1
running_utilization.append(running / mns)
token_utilization.append(scheduled_tokens / mbbt)
kv_usage.append(float(record["kv"]["usage"]))
preemptions += int(record["preemptions"])
count = len(records)
return {
"records": count,
**{f"{name}_count": value for name, value in counts.items()},
**{f"{name}_fraction": value / count for name, value in counts.items()},
"running_utilization_mean": statistics.fmean(running_utilization),
"running_utilization_max": max(running_utilization),
"token_utilization_mean": statistics.fmean(token_utilization),
"token_utilization_max": max(token_utilization),
"kv_usage_mean": statistics.fmean(kv_usage),
"kv_usage_max": max(kv_usage),
"preemptions": preemptions,
}
def telemetry_coverage(
records: list[Mapping[str, Any]], *, start_ns: int, end_ns: int
) -> tuple[dict[str, float], bool]:
if not records:
raise ValueError("telemetry coverage requires records")
submit_gaps = [
(int(right["submit_mono_ns"]) - int(left["submit_mono_ns"])) / 1e9
for left, right in zip(records, records[1:], strict=False)
]
uncovered_gaps = [
max(
0,
int(right["submit_mono_ns"]) - int(left["complete_mono_ns"]),
)
/ 1e9
for left, right in zip(records, records[1:], strict=False)
]
coverage = {
"start_gap_s": (int(records[0]["submit_mono_ns"]) - start_ns) / 1e9,
"end_gap_s": (end_ns - int(records[-1]["submit_mono_ns"])) / 1e9,
"max_internal_submit_gap_s": max(submit_gaps, default=0.0),
"max_uncovered_gap_s": max(uncovered_gaps, default=0.0),
}
covered = (
0.0 <= coverage["start_gap_s"] <= 1.0
and 0.0 <= coverage["end_gap_s"] <= 1.0
and 0.0 <= coverage["max_uncovered_gap_s"] <= 1.0
)
return coverage, covered
def mechanism_summary(records: list[Mapping[str, Any]]) -> dict[str, Any]:
executed = [record for record in records if bool(record["model_executed"])]
if not executed:
raise ValueError("mechanism summary requires executed steps")
prefill = [record for record in executed if int(record["prefill_tokens"]) > 0]
decode_only = [
record for record in executed if int(record["prefill_tokens"]) == 0
]
if not prefill or not decode_only:
raise ValueError("mechanism summary requires prefill and decode-only steps")
def durations_ms(selected: list[Mapping[str, Any]]) -> list[float]:
values = [
(int(record["complete_mono_ns"]) - int(record["submit_mono_ns"]))
/ 1e6
for record in selected
]
if any(value < 0.0 for value in values):
raise ValueError("engine step duration must be non-negative")
return values
chunk_keys = ("first", "middle", "final", "unsplit", "tokens")
chunks = {
key: sum(int(record["chunked_prefill"][key]) for record in executed)
for key in chunk_keys
}
prefill_tokens = [int(record["prefill_tokens"]) for record in prefill]
prefill_requests = sum(int(record["prefill_requests"]) for record in prefill)
prefix_queries = sum(
int(record["prefix"]["local"]["queries"]) for record in executed
)
prefix_hits = sum(
int(record["prefix"]["local"]["hits"]) for record in executed
)
invariants = {
"nonnegative_counts": all(
value >= 0
for value in (
*chunks.values(),
prefill_requests,
prefix_queries,
prefix_hits,
)
),
"chunk_tokens_match_prefill_tokens": chunks["tokens"]
== sum(prefill_tokens),
"prefix_hits_bounded": 0 <= prefix_hits <= prefix_queries,
}
return {
"executed_steps": len(executed),
"step_duration_ms": distribution(durations_ms(executed)),
"prefill_steps": len(prefill),
"prefill_step_duration_ms": distribution(durations_ms(prefill)),
"decode_only_steps": len(decode_only),
"decode_only_step_duration_ms": distribution(durations_ms(decode_only)),
"prefill": {
"requests": prefill_requests,
"requests_per_step": prefill_requests / len(prefill),
"tokens": sum(prefill_tokens),
"tokens_per_step": distribution(prefill_tokens),
"chunks": chunks,
},
"prefix": {
"queries": prefix_queries,
"hits": prefix_hits,
"hit_rate": prefix_hits / prefix_queries if prefix_queries else 0.0,
},
"sanity": {"invariants": invariants},
}
def request_summary(path: Path, expected_count: int) -> dict[str, Any]:
rows = load_jsonl(path)
if len(rows) != expected_count:
raise ValueError(f"request row count mismatch: {path}")
ttft = [float(row["ttft_ms"]) for row in rows if row["ttft_ms"] is not None]
tpot = [float(row["tpot_ms"]) for row in rows if row["tpot_ms"] is not None]
if not ttft or not tpot:
raise ValueError(f"missing request latency values: {path}")
return {
"ttft_ms": {f"p{int(p * 100)}": quantile(ttft, p) for p in (0.5, 0.95, 0.99)},
"tpot_ms": {f"p{int(p * 100)}": quantile(tpot, p) for p in (0.5, 0.95, 0.99)},
}
def load_stream(session_root: Path) -> tuple[list[dict[str, Any]], dict[str, Any]]:
streams = sorted((session_root / "opprof").glob("*.jsonl"))
sidecars = sorted((session_root / "opprof").glob("*.jsonl.footer.json"))
if len(streams) != 1 or len(sidecars) != 1:
raise ValueError(f"expected one OpProf stream and sidecar: {session_root}")
decoded = load_jsonl(streams[0])
records = [row for row in decoded if "step_index" in row]
footers = [row for row in decoded if row.get("record_type") == "footer"]
sidecar = json.loads(sidecars[0].read_text(encoding="utf-8"))
indexes = [int(row["step_index"]) for row in records]
invariants = {
"one_footer_last": len(footers) == 1 and decoded[-1] is footers[0],
"sidecar_final": sidecar.get("final") is True,
"zero_drops": sidecar.get("dropped_records") == 0,
"written_matches_records": sidecar.get("written_records") == len(records),
"contiguous_step_indexes": indexes == list(range(len(indexes))),
"monotonic_timestamps": all(
int(right["submit_mono_ns"]) >= int(left["submit_mono_ns"])
for left, right in zip(records, records[1:], strict=False)
),
}
return records, {
"stream": str(streams[0]),
"stream_sha256": sha256_file(streams[0]),
"records": len(records),
"invariants": invariants,
}
def analyze_run(
*,
run_root: Path,
config: Mapping[str, Any],
repetition: int,
expected: Mapping[str, Any],
stream_records: list[Mapping[str, Any]],
duration_s: float,
phase_fractions: list[float],
) -> dict[str, Any]:
result_root = run_root / "sessions" / str(config["id"]) / f"rep{repetition}"
result_path = result_root / "result.json"
result = json.loads(result_path.read_text(encoding="utf-8"))
selection = result["selection"]
invariants = {
"result_schema": result.get("schema") == "action-aware-pilot-result-v0",
"config_id": result.get("config_id") == config["id"],
"tp": int(result.get("tp", -1)) == 4,
"mns": int(result.get("mns", -1)) == int(config["mns"]),
"mbbt": int(result.get("mbbt", -1)) == int(config["mbbt"]),
"uncensored": not bool(result.get("early_stopped", True)),
"slo_early_stop_disabled": result.get("slo_early_stop_disabled") is True,
"selection_count": int(selection["count"]) == int(expected["selected_count"]),
"request_accounting": int(result["observed_count"])
== int(expected["selected_count"]),
"request_hash": selection["request_id_order_sha256"]
== expected["request_id_order_sha256"],
"arrival_hash": selection["arrival_order_sha256"]
== expected["arrival_order_sha256"],
"length_hash": selection["raw_length_order_sha256"]
== expected["input_length_order_sha256"],
}
start_ns = int(result["interval"]["start_mono_ns"])
arrival_end_ns = start_ns + round(duration_s * 1e9)
full_records = [
record
for record in stream_records
if start_ns <= int(record["submit_mono_ns"]) <= arrival_end_ns
]
if not full_records:
raise ValueError(f"no telemetry records in measured window: {result_path}")
coverage, invariants["telemetry_coverage"] = telemetry_coverage(
full_records, start_ns=start_ns, end_ns=arrival_end_ns
)
binding = binding_summary(
full_records, mns=int(config["mns"]), mbbt=int(config["mbbt"])
)
mechanism = mechanism_summary(full_records)
invariants["mechanism_summary"] = all(
mechanism["sanity"]["invariants"].values()
)
phases = {}
for fraction in phase_fractions:
phase_end = start_ns + round(duration_s * fraction * 1e9)
phase_records = [
record
for record in full_records
if int(record["submit_mono_ns"]) <= phase_end
]
phases[f"{fraction:.2f}"] = binding_summary(
phase_records, mns=int(config["mns"]), mbbt=int(config["mbbt"])
)
state = summarize_engine(
full_records,
start_ns=start_ns,
end_ns=arrival_end_ns,
request_count=int(result["observed_count"]),
)
latency = request_summary(
result_root / "requests.jsonl", int(result["observed_count"])
)
return {
"config_id": config["id"],
"mns": int(config["mns"]),
"mbbt": int(config["mbbt"]),
"repetition": repetition,
"result_path": str(result_path),
"result_sha256": sha256_file(result_path),
"selection": {
"count": int(selection["count"]),
"request_id_order_sha256": selection["request_id_order_sha256"],
"arrival_order_sha256": selection["arrival_order_sha256"],
"raw_length_order_sha256": selection["raw_length_order_sha256"],
},
"outcome": {
"pass_rate": float(result["pass_rate"]),
"feasible": bool(result["feasible"]),
"slo_pass_count": int(result["slo_pass_count"]),
"slo_goodput_req_s": int(result["slo_pass_count"]) / duration_s,
"elapsed_s": float(result["interval"]["elapsed_s"]),
**latency,
},
"binding": binding,
"mechanism": mechanism,
"phases": phases,
"state": state,
"coverage": coverage,
"invariants": invariants,
}
def median(values: Iterable[float]) -> float:
return float(statistics.median(float(value) for value in values))
def evaluate_decisions(
runs: list[Mapping[str, Any]], manifest: Mapping[str, Any]
) -> dict[str, Any]:
by_key = {
(str(run["config_id"]), int(run["repetition"])): run for run in runs
}
repetitions = sorted(int(key) for key in manifest["repetitions"])
regime_results = {}
all_predictions = []
crossed_pass = True
binding_pass = True
material_ambiguity = False
for regime_name, regime in manifest["regimes"].items():
rows = []
source_runs = []
for repetition in repetitions:
source = by_key[(str(regime["source"]), repetition)]
mns_target = by_key[(str(regime["actions"]["mns"]), repetition)]
mbbt_target = by_key[(str(regime["actions"]["mbbt"]), repetition)]
source_runs.append(source)
source_goodput = float(source["outcome"]["slo_goodput_req_s"])
mns_goodput = float(mns_target["outcome"]["slo_goodput_req_s"])
mbbt_goodput = float(mbbt_target["outcome"]["slo_goodput_req_s"])
observed = (
"mns"
if mns_goodput > mbbt_goodput
else "mbbt"
if mbbt_goodput > mns_goodput
else "tie"
)
mns_score = float(source["binding"]["mns_exclusive_fraction"])
mbbt_score = float(source["binding"]["mbbt_exclusive_fraction"])
predicted = (
"mns"
if mns_score > mbbt_score
else "mbbt"
if mbbt_score > mns_score
else "tie"
)
phase_predictions = {}
for phase, summary in source["phases"].items():
left = float(summary["mns_exclusive_fraction"])
right = float(summary["mbbt_exclusive_fraction"])
phase_predictions[phase] = (
"mns" if left > right else "mbbt" if right > left else "tie"
)
margin = (
abs(mns_goodput - mbbt_goodput) / source_goodput
if source_goodput > 0
else None
)
row = {
"repetition": repetition,
"source_goodput_req_s": source_goodput,
"mns_target_goodput_req_s": mns_goodput,
"mbbt_target_goodput_req_s": mbbt_goodput,
"observed_winner": observed,
"predicted_winner": predicted,
"prediction_correct": predicted == observed,
"relative_winner_margin_over_source": margin,
"mns_exclusive_fraction": mns_score,
"mbbt_exclusive_fraction": mbbt_score,
"phase_predictions": phase_predictions,
"phase_stable": all(value == predicted for value in phase_predictions.values()),
}
rows.append(row)
all_predictions.append(row)
expected_winner = "mns" if regime_name == "A" else "mbbt"
minimum_margin = float(manifest["gates"]["minimum_relative_winner_margin"])
regime_crossed = all(
row["observed_winner"] == expected_winner
and row["relative_winner_margin_over_source"] is not None
and row["relative_winner_margin_over_source"] >= minimum_margin
for row in rows
)
crossed_pass &= regime_crossed
winning_key = f"{expected_winner}_exclusive_fraction"
losing_key = (
"mbbt_exclusive_fraction" if expected_winner == "mns" else "mns_exclusive_fraction"
)
winning_median = median(row[winning_key] for row in rows)
losing_median = median(row[losing_key] for row in rows)
ratio_pass = winning_median >= float(
manifest["gates"]["minimum_exclusive_ratio"]
) * losing_median
regime_binding = (
all(row["prediction_correct"] and row["phase_stable"] for row in rows)
and winning_median
>= float(manifest["gates"]["minimum_exclusive_fraction"])
and ratio_pass
)
binding_pass &= regime_binding
ambiguity_median = median(
float(run["binding"]["both_fraction"])
+ float(run["binding"]["waiting_unresolved_fraction"])
for run in source_runs
)
score_gap_median = median(
abs(
float(run["binding"]["mns_exclusive_fraction"])
- float(run["binding"]["mbbt_exclusive_fraction"])
)
for run in source_runs
)
kv_max_median = median(
float(run["binding"]["kv_usage_max"]) for run in source_runs
)
any_preemption = any(
int(run["binding"]["preemptions"]) > 0 for run in source_runs
)
regime_material = (
ambiguity_median >= score_gap_median
or kv_max_median >= float(manifest["gates"]["material_kv_usage"])
or any_preemption
)
material_ambiguity |= regime_material
regime_results[regime_name] = {
"source": regime["source"],
"actions": regime["actions"],
"expected_winner": expected_winner,
"crossed_response_pass": regime_crossed,
"binding_pass": regime_binding,
"winning_exclusive_median": winning_median,
"losing_exclusive_median": losing_median,
"exclusive_ratio_pass": ratio_pass,
"ambiguity_median": ambiguity_median,
"exclusive_gap_median": score_gap_median,
"kv_usage_max_median": kv_max_median,
"any_preemption": any_preemption,
"material_ambiguity": regime_material,
"repetitions": rows,
}
if not crossed_pass:
decision = "STOP_WORKLOAD_NOT_CROSSED"
elif not binding_pass:
decision = "STOP_BINDING_NOT_PREDICTIVE"
elif material_ambiguity:
decision = "OPEN_EXACT_ATTRIBUTION_ABLATION"
else:
decision = "STOP_NO_NEW_INSTRUMENTATION_NEEDED"
correct = sum(int(row["prediction_correct"]) for row in all_predictions)
return {
"decision": decision,
"crossed_response_pass": crossed_pass,
"binding_pass": binding_pass,
"material_ambiguity": material_ambiguity,
"regimes": regime_results,
"baselines": {
"always_mns_correct": sum(
int(row["observed_winner"] == "mns") for row in all_predictions
),
"always_mbbt_correct": sum(
int(row["observed_winner"] == "mbbt") for row in all_predictions
),
"binding_correct": correct,
"decision_count": len(all_predictions),
},
}
def analyze(run_root: Path, manifest_path: Path) -> dict[str, Any]:
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
if manifest.get("schema") not in {
"action-aware-constraint-pilot-manifest-v0",
"action-aware-constraint-pilot-manifest-v1",
}:
raise ValueError("unexpected manifest schema")
duration_s = float(manifest["engine"]["duration_s"])
phase_fractions = [float(value) for value in manifest["gates"]["phase_fractions"]]
runs = []
stream_audits = []
for config in manifest["configs"]:
session_root = run_root / "sessions" / str(config["id"])
stream_records, stream_audit = load_stream(session_root)
stream_audit["config_id"] = config["id"]
stream_audits.append(stream_audit)
for repetition in sorted(int(key) for key in manifest["repetitions"]):
runs.append(
analyze_run(
run_root=run_root,
config=config,
repetition=repetition,
expected=manifest["repetitions"][str(repetition)]["selection"],
stream_records=stream_records,
duration_s=duration_s,
phase_fractions=phase_fractions,
)
)
invariants = {
"fifteen_runs": len(runs) == 15,
"five_streams": len(stream_audits) == 5,
"all_run_invariants": all(
all(bool(value) for value in run["invariants"].values()) for run in runs
),
"all_stream_invariants": all(
all(bool(value) for value in stream["invariants"].values())
for stream in stream_audits
),
"nonnegative_counters": all(
all(
float(run["binding"][key]) >= 0
for key in (
"mns_exclusive_count",
"mbbt_exclusive_count",
"both_count",
"waiting_unresolved_count",
"preemptions",
)
)
for run in runs
),
"ratios_bounded": all(
all(
0.0 <= float(run["binding"][key]) <= 1.0
for key in (
"mns_exclusive_fraction",
"mbbt_exclusive_fraction",
"both_fraction",
"waiting_unresolved_fraction",
"kv_usage_mean",
"kv_usage_max",
)
)
for run in runs
),
"per_config_results_not_all_identical": len(
{float(run["outcome"]["pass_rate"]) for run in runs}
)
> 1,
}
red_flags = [name for name, passed in invariants.items() if not passed]
decisions = (
evaluate_decisions(runs, manifest)
if not red_flags
else {
"decision": "STOP_DATA_INVALID",
"crossed_response_pass": False,
"binding_pass": False,
"material_ambiguity": False,
"regimes": {},
"baselines": {},
}
)
payload = {
"schema": SCHEMA,
"decision": decisions["decision"],
"manifest": str(manifest_path),
"manifest_sha256": sha256_file(manifest_path),
"run_root": str(run_root),
"runs": runs,
"streams": stream_audits,
"decision_audit": decisions,
"sanity": {
"runs": len(runs),
"pass_rate": numeric(run["outcome"]["pass_rate"] for run in runs),
"slo_goodput_req_s": numeric(
run["outcome"]["slo_goodput_req_s"] for run in runs
),
"telemetry_records_per_run": numeric(
run["binding"]["records"] for run in runs
),
"mns_values": numeric(run["mns"] for run in runs),
"mbbt_values": numeric(run["mbbt"] for run in runs),
"invariants": invariants,
"red_flags": red_flags,
},
}
return payload
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--run-root", type=Path, required=True)
parser.add_argument("--manifest", type=Path, required=True)
parser.add_argument("--output", type=Path, required=True)
args = parser.parse_args()
payload = analyze(args.run_root, args.manifest)
atomic_json(args.output, payload)
print(
json.dumps(
{
"decision": payload["decision"],
"sanity": payload["sanity"],
"decision_audit": payload["decision_audit"],
},
indent=2,
sort_keys=True,
)
)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,227 @@
{
"budget": {
"expected_h20_hours": [
6.0,
7.2
],
"expected_wall_minutes": [
90,
110
],
"global_hard_cap_h20_hours": 8.0,
"hard_cap_h20_hours": 7.614013100465138,
"prior_attempt_artifact": "/home/admin/cpfs/wjh/action-aware-constraint-v0-20260714/operational-stop-v0.json",
"prior_attempt_h20_hours": 0.38598689953486126,
"safety_h20_hours": 0.25,
"session_estimate_h20_hours": 1.35
},
"burnin": {
"anchor": 0.18919793755240089,
"arrival_order_sha256": "6c0ac4cb9a30ef501eeeacc8e6cc631c345e976db5ccf530ea5a1ec706d62a24",
"input_length_order_sha256": "7939cc20e1a00d1031d27d71508789f38decbbbb6ea59a1df18b2ec342fd2ef8",
"offered_req_s": 8.5,
"offered_req_s_per_gpu": 2.125,
"request_id_order_sha256": "84f4809acbc8acd3b1d14dfa357134a1dc0b9287341624b33f598dafeef54dc7",
"selected_count": 510,
"study": "/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/studies/burnin-tp4.json",
"study_sha256": "5d6c2098042909a863efd3112818fbee9bafe96f22898ac98b66846dbe1fef0f"
},
"configs": [
{
"id": "b_base",
"mbbt": 2048,
"mns": 64,
"repetition_order": [
1,
2,
3
]
},
{
"id": "a_base",
"mbbt": 8192,
"mns": 16,
"repetition_order": [
2,
3,
1
]
},
{
"id": "shared",
"mbbt": 8192,
"mns": 64,
"repetition_order": [
3,
1,
2
]
},
{
"id": "b_mns",
"mbbt": 2048,
"mns": 128,
"repetition_order": [
1,
3,
2
]
},
{
"id": "a_mbbt",
"mbbt": 16384,
"mns": 16,
"repetition_order": [
2,
1,
3
]
}
],
"engine": {
"burnin_max_elapsed_s": 90.0,
"client_timeout_s": 450.0,
"disable_slo_early_stop": true,
"duration_s": 300.0,
"tp": 4
},
"gates": {
"material_kv_usage": 0.9,
"minimum_exclusive_fraction": 0.1,
"minimum_exclusive_ratio": 5.0,
"minimum_relative_winner_margin": 0.1,
"phase_fractions": [
0.25,
0.5,
0.75,
1.0
]
},
"regimes": {
"A": {
"actions": {
"mbbt": "a_mbbt",
"mns": "shared"
},
"source": "a_base"
},
"B": {
"actions": {
"mbbt": "shared",
"mns": "b_mns"
},
"source": "b_base"
}
},
"repetitions": {
"1": {
"merged_trace": {
"bytes": 337429767,
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep1.jsonl",
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
"rows": 9420,
"sha256": "68983266aa0e66aa589562f7c08edbd966f9ba4405e20c105adb43777d2dfbf5",
"source_sha256": [
"b242d1d9086df3accab57b4c92445d5edd581e12f47e12cea227aa63964c6930",
"d23b549f7b69af3647308677bbf76f818a3c226a1c98f9a9f93f09ceee46be87"
],
"sources": [
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low1.jsonl",
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high1.jsonl"
]
},
"selection": {
"anchor": 0.48686986110831465,
"arrival_order_sha256": "c2ad99986ce558da5901a9c5ec0a00bd69f198c981d8779235f2773a5c87f1c0",
"input_length_order_sha256": "9442bfebdc3fab5062dc1f4d688dc28c02afe3fd806c56dd8159f0ac7e6d0b94",
"offered_req_s": 8.5,
"offered_req_s_per_gpu": 2.125,
"request_id_order_sha256": "0bb61dbc9c26875e991d0d4f984134910d37463e5063f86ee960cf4f8aafb771",
"selected_count": 2550,
"target_count": 2550,
"target_req_s_per_gpu": 2.125
},
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep1-tp4.json",
"study_sha256": "ecfff96e33d458eb1e3b9a6d24386f00cc6f1b19ff926e2ec6320b3f671a7ae3"
},
"2": {
"merged_trace": {
"bytes": 337509330,
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep2.jsonl",
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
"rows": 9457,
"sha256": "f38e8938f6a481fc6725b71b21aa04ff7eaf79783cdfd6e41aa2f074156f00c2",
"source_sha256": [
"4cbb0baac082bd54af562ce2f39104c5c23b4671672da365a67b1e8c146adf9f",
"bb0bcd2564a88000f435f12feb21c7c902eafc9ea5fe916adfe9d1eae47f3f9a"
],
"sources": [
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low2.jsonl",
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high2.jsonl"
]
},
"selection": {
"anchor": 0.4825698948735577,
"arrival_order_sha256": "b9fc12cf3f86bc8a79bee65296e65aa2b8bf2aeca46b2887094c669adcbb9a00",
"input_length_order_sha256": "d8d4bd6fc8ba852a45605b673b6b3e4f33b58f459e69f2a032d226ee175b074e",
"offered_req_s": 8.5,
"offered_req_s_per_gpu": 2.125,
"request_id_order_sha256": "56a0616b6b54abafd37875c7cb25f8639afef2706ccc55dfbe568f45859ea382",
"selected_count": 2550,
"target_count": 2550,
"target_req_s_per_gpu": 2.125
},
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep2-tp4.json",
"study_sha256": "d92a576db031db24bb58f354ea725d7f7567cb76699d387117ac5a6c9317bbb9"
},
"3": {
"merged_trace": {
"bytes": 337450256,
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep3.jsonl",
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
"rows": 9431,
"sha256": "3094084b0bb20cc02eecf465091a5c919b4e5b112f704cdc36a563d1efdcee46",
"source_sha256": [
"1f7ececb142f9a363d2d1ca25eb7b8488b2cc319a51b55faa384f2a3d51f2142",
"6f326234791e1cff4ff866bface0d097d0d6e3844eebb1c97653d8e9c35e9397"
],
"sources": [
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low3.jsonl",
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high3.jsonl"
]
},
"selection": {
"anchor": 0.48664343020532463,
"arrival_order_sha256": "efce7339e22d3618cb4d55e6b55bfddb2c563c18faba2a992d5829c13e3f55e9",
"input_length_order_sha256": "0792b05fff6729fbd92ab2bb4cb6d31bea7799e232ad42772936bc06efbafb54",
"offered_req_s": 8.5,
"offered_req_s_per_gpu": 2.125,
"request_id_order_sha256": "2a2fabe2c4cf176aeb7e0d32fb8e7dbb1f27429a2e7a0cd18d7d186f23096f19",
"selected_count": 2550,
"target_count": 2550,
"target_req_s_per_gpu": 2.125
},
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep3-tp4.json",
"study_sha256": "fb8ffe256dace32f4ca8a8d49b662d98c3b69b94ecc8fa826e43068b238884ab"
}
},
"sanity": {
"invariants": {
"all_repetition_orders_are_permutations": true,
"five_unique_configs": true,
"same_load_all_repetitions": true,
"shared_endpoint_reused_by_both_regimes": true,
"three_disjoint_repetitions": true
},
"red_flags": []
},
"schema": "action-aware-constraint-pilot-manifest-v1",
"source": {
"base_manifest": "/home/gahow/phd/aituner/runs/intervention-response-v2/pilot-manifest-v3.json",
"base_manifest_sha256": "273db1181dcc9d6b64439650d0642ebe553b12e6aa9adebfbe3758a7977e5611",
"source_trace": "/home/admin/cpfs/wjh/aituner/aituner/trace_windows/traces/chat_w20260312_1000.jsonl",
"source_trace_sha256": "875ba869775deb78086477919f03b322da14e2673c7d070e26528c4190912757",
"window_id": "chat_w20260312_1000"
},
"status": "PASS"
}

View File

@@ -0,0 +1,227 @@
{
"budget": {
"expected_h20_hours": [
6.0,
7.2
],
"expected_wall_minutes": [
90,
110
],
"global_hard_cap_h20_hours": 8.0,
"hard_cap_h20_hours": 7.295602157380846,
"prior_attempt_artifact": "/home/admin/cpfs/wjh/aituner/aituner-action-aware-20260714/runs/action-aware-v0/prior-attempts-v2.json",
"prior_attempt_h20_hours": 0.7043978426191542,
"safety_h20_hours": 0.25,
"session_estimate_h20_hours": 1.35
},
"burnin": {
"anchor": 0.18919793755240089,
"arrival_order_sha256": "6c0ac4cb9a30ef501eeeacc8e6cc631c345e976db5ccf530ea5a1ec706d62a24",
"input_length_order_sha256": "7939cc20e1a00d1031d27d71508789f38decbbbb6ea59a1df18b2ec342fd2ef8",
"offered_req_s": 8.5,
"offered_req_s_per_gpu": 2.125,
"request_id_order_sha256": "84f4809acbc8acd3b1d14dfa357134a1dc0b9287341624b33f598dafeef54dc7",
"selected_count": 510,
"study": "/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/studies/burnin-tp4.json",
"study_sha256": "5d6c2098042909a863efd3112818fbee9bafe96f22898ac98b66846dbe1fef0f"
},
"configs": [
{
"id": "b_base",
"mbbt": 2048,
"mns": 64,
"repetition_order": [
1,
2,
3
]
},
{
"id": "a_base",
"mbbt": 8192,
"mns": 16,
"repetition_order": [
2,
3,
1
]
},
{
"id": "shared",
"mbbt": 8192,
"mns": 64,
"repetition_order": [
3,
1,
2
]
},
{
"id": "b_mns",
"mbbt": 2048,
"mns": 128,
"repetition_order": [
1,
3,
2
]
},
{
"id": "a_mbbt",
"mbbt": 16384,
"mns": 16,
"repetition_order": [
2,
1,
3
]
}
],
"engine": {
"burnin_max_elapsed_s": 90.0,
"client_timeout_s": 450.0,
"disable_slo_early_stop": true,
"duration_s": 300.0,
"tp": 4
},
"gates": {
"material_kv_usage": 0.9,
"minimum_exclusive_fraction": 0.1,
"minimum_exclusive_ratio": 5.0,
"minimum_relative_winner_margin": 0.1,
"phase_fractions": [
0.25,
0.5,
0.75,
1.0
]
},
"regimes": {
"A": {
"actions": {
"mbbt": "a_mbbt",
"mns": "shared"
},
"source": "a_base"
},
"B": {
"actions": {
"mbbt": "shared",
"mns": "b_mns"
},
"source": "b_base"
}
},
"repetitions": {
"1": {
"merged_trace": {
"bytes": 337429767,
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep1.jsonl",
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
"rows": 9420,
"sha256": "68983266aa0e66aa589562f7c08edbd966f9ba4405e20c105adb43777d2dfbf5",
"source_sha256": [
"b242d1d9086df3accab57b4c92445d5edd581e12f47e12cea227aa63964c6930",
"d23b549f7b69af3647308677bbf76f818a3c226a1c98f9a9f93f09ceee46be87"
],
"sources": [
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low1.jsonl",
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high1.jsonl"
]
},
"selection": {
"anchor": 0.48686986110831465,
"arrival_order_sha256": "c2ad99986ce558da5901a9c5ec0a00bd69f198c981d8779235f2773a5c87f1c0",
"input_length_order_sha256": "9442bfebdc3fab5062dc1f4d688dc28c02afe3fd806c56dd8159f0ac7e6d0b94",
"offered_req_s": 8.5,
"offered_req_s_per_gpu": 2.125,
"request_id_order_sha256": "0bb61dbc9c26875e991d0d4f984134910d37463e5063f86ee960cf4f8aafb771",
"selected_count": 2550,
"target_count": 2550,
"target_req_s_per_gpu": 2.125
},
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep1-tp4.json",
"study_sha256": "ecfff96e33d458eb1e3b9a6d24386f00cc6f1b19ff926e2ec6320b3f671a7ae3"
},
"2": {
"merged_trace": {
"bytes": 337509330,
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep2.jsonl",
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
"rows": 9457,
"sha256": "f38e8938f6a481fc6725b71b21aa04ff7eaf79783cdfd6e41aa2f074156f00c2",
"source_sha256": [
"4cbb0baac082bd54af562ce2f39104c5c23b4671672da365a67b1e8c146adf9f",
"bb0bcd2564a88000f435f12feb21c7c902eafc9ea5fe916adfe9d1eae47f3f9a"
],
"sources": [
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low2.jsonl",
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high2.jsonl"
]
},
"selection": {
"anchor": 0.4825698948735577,
"arrival_order_sha256": "b9fc12cf3f86bc8a79bee65296e65aa2b8bf2aeca46b2887094c669adcbb9a00",
"input_length_order_sha256": "d8d4bd6fc8ba852a45605b673b6b3e4f33b58f459e69f2a032d226ee175b074e",
"offered_req_s": 8.5,
"offered_req_s_per_gpu": 2.125,
"request_id_order_sha256": "56a0616b6b54abafd37875c7cb25f8639afef2706ccc55dfbe568f45859ea382",
"selected_count": 2550,
"target_count": 2550,
"target_req_s_per_gpu": 2.125
},
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep2-tp4.json",
"study_sha256": "d92a576db031db24bb58f354ea725d7f7567cb76699d387117ac5a6c9317bbb9"
},
"3": {
"merged_trace": {
"bytes": 337450256,
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep3.jsonl",
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
"rows": 9431,
"sha256": "3094084b0bb20cc02eecf465091a5c919b4e5b112f704cdc36a563d1efdcee46",
"source_sha256": [
"1f7ececb142f9a363d2d1ca25eb7b8488b2cc319a51b55faa384f2a3d51f2142",
"6f326234791e1cff4ff866bface0d097d0d6e3844eebb1c97653d8e9c35e9397"
],
"sources": [
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low3.jsonl",
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high3.jsonl"
]
},
"selection": {
"anchor": 0.48664343020532463,
"arrival_order_sha256": "efce7339e22d3618cb4d55e6b55bfddb2c563c18faba2a992d5829c13e3f55e9",
"input_length_order_sha256": "0792b05fff6729fbd92ab2bb4cb6d31bea7799e232ad42772936bc06efbafb54",
"offered_req_s": 8.5,
"offered_req_s_per_gpu": 2.125,
"request_id_order_sha256": "2a2fabe2c4cf176aeb7e0d32fb8e7dbb1f27429a2e7a0cd18d7d186f23096f19",
"selected_count": 2550,
"target_count": 2550,
"target_req_s_per_gpu": 2.125
},
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep3-tp4.json",
"study_sha256": "fb8ffe256dace32f4ca8a8d49b662d98c3b69b94ecc8fa826e43068b238884ab"
}
},
"sanity": {
"invariants": {
"all_repetition_orders_are_permutations": true,
"five_unique_configs": true,
"same_load_all_repetitions": true,
"shared_endpoint_reused_by_both_regimes": true,
"three_disjoint_repetitions": true
},
"red_flags": []
},
"schema": "action-aware-constraint-pilot-manifest-v1",
"source": {
"base_manifest": "/home/gahow/phd/aituner/runs/intervention-response-v2/pilot-manifest-v3.json",
"base_manifest_sha256": "273db1181dcc9d6b64439650d0642ebe553b12e6aa9adebfbe3758a7977e5611",
"source_trace": "/home/admin/cpfs/wjh/aituner/aituner/trace_windows/traces/chat_w20260312_1000.jsonl",
"source_trace_sha256": "875ba869775deb78086477919f03b322da14e2673c7d070e26528c4190912757",
"window_id": "chat_w20260312_1000"
},
"status": "PASS"
}

View File

@@ -0,0 +1,223 @@
{
"budget": {
"expected_h20_hours": [
6.0,
7.2
],
"expected_wall_minutes": [
90,
110
],
"hard_cap_h20_hours": 8.0,
"safety_h20_hours": 0.25,
"session_estimate_h20_hours": 1.35
},
"burnin": {
"anchor": 0.18919793755240089,
"arrival_order_sha256": "6c0ac4cb9a30ef501eeeacc8e6cc631c345e976db5ccf530ea5a1ec706d62a24",
"input_length_order_sha256": "7939cc20e1a00d1031d27d71508789f38decbbbb6ea59a1df18b2ec342fd2ef8",
"offered_req_s": 8.5,
"offered_req_s_per_gpu": 2.125,
"request_id_order_sha256": "84f4809acbc8acd3b1d14dfa357134a1dc0b9287341624b33f598dafeef54dc7",
"selected_count": 510,
"study": "/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/studies/burnin-tp4.json",
"study_sha256": "5d6c2098042909a863efd3112818fbee9bafe96f22898ac98b66846dbe1fef0f"
},
"configs": [
{
"id": "b_base",
"mbbt": 256,
"mns": 64,
"repetition_order": [
1,
2,
3
]
},
{
"id": "a_base",
"mbbt": 8192,
"mns": 16,
"repetition_order": [
2,
3,
1
]
},
{
"id": "shared",
"mbbt": 8192,
"mns": 64,
"repetition_order": [
3,
1,
2
]
},
{
"id": "b_mns",
"mbbt": 256,
"mns": 128,
"repetition_order": [
1,
3,
2
]
},
{
"id": "a_mbbt",
"mbbt": 16384,
"mns": 16,
"repetition_order": [
2,
1,
3
]
}
],
"engine": {
"client_timeout_s": 450.0,
"disable_slo_early_stop": true,
"duration_s": 300.0,
"tp": 4
},
"gates": {
"material_kv_usage": 0.9,
"minimum_exclusive_fraction": 0.1,
"minimum_exclusive_ratio": 5.0,
"minimum_relative_winner_margin": 0.1,
"phase_fractions": [
0.25,
0.5,
0.75,
1.0
]
},
"regimes": {
"A": {
"actions": {
"mbbt": "a_mbbt",
"mns": "shared"
},
"source": "a_base"
},
"B": {
"actions": {
"mbbt": "shared",
"mns": "b_mns"
},
"source": "b_base"
}
},
"repetitions": {
"1": {
"merged_trace": {
"bytes": 337429767,
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep1.jsonl",
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
"rows": 9420,
"sha256": "68983266aa0e66aa589562f7c08edbd966f9ba4405e20c105adb43777d2dfbf5",
"source_sha256": [
"b242d1d9086df3accab57b4c92445d5edd581e12f47e12cea227aa63964c6930",
"d23b549f7b69af3647308677bbf76f818a3c226a1c98f9a9f93f09ceee46be87"
],
"sources": [
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low1.jsonl",
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high1.jsonl"
]
},
"selection": {
"anchor": 0.48686986110831465,
"arrival_order_sha256": "c2ad99986ce558da5901a9c5ec0a00bd69f198c981d8779235f2773a5c87f1c0",
"input_length_order_sha256": "9442bfebdc3fab5062dc1f4d688dc28c02afe3fd806c56dd8159f0ac7e6d0b94",
"offered_req_s": 8.5,
"offered_req_s_per_gpu": 2.125,
"request_id_order_sha256": "0bb61dbc9c26875e991d0d4f984134910d37463e5063f86ee960cf4f8aafb771",
"selected_count": 2550,
"target_count": 2550,
"target_req_s_per_gpu": 2.125
},
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep1-tp4.json",
"study_sha256": "ecfff96e33d458eb1e3b9a6d24386f00cc6f1b19ff926e2ec6320b3f671a7ae3"
},
"2": {
"merged_trace": {
"bytes": 337509330,
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep2.jsonl",
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
"rows": 9457,
"sha256": "f38e8938f6a481fc6725b71b21aa04ff7eaf79783cdfd6e41aa2f074156f00c2",
"source_sha256": [
"4cbb0baac082bd54af562ce2f39104c5c23b4671672da365a67b1e8c146adf9f",
"bb0bcd2564a88000f435f12feb21c7c902eafc9ea5fe916adfe9d1eae47f3f9a"
],
"sources": [
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low2.jsonl",
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high2.jsonl"
]
},
"selection": {
"anchor": 0.4825698948735577,
"arrival_order_sha256": "b9fc12cf3f86bc8a79bee65296e65aa2b8bf2aeca46b2887094c669adcbb9a00",
"input_length_order_sha256": "d8d4bd6fc8ba852a45605b673b6b3e4f33b58f459e69f2a032d226ee175b074e",
"offered_req_s": 8.5,
"offered_req_s_per_gpu": 2.125,
"request_id_order_sha256": "56a0616b6b54abafd37875c7cb25f8639afef2706ccc55dfbe568f45859ea382",
"selected_count": 2550,
"target_count": 2550,
"target_req_s_per_gpu": 2.125
},
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep2-tp4.json",
"study_sha256": "d92a576db031db24bb58f354ea725d7f7567cb76699d387117ac5a6c9317bbb9"
},
"3": {
"merged_trace": {
"bytes": 337450256,
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep3.jsonl",
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
"rows": 9431,
"sha256": "3094084b0bb20cc02eecf465091a5c919b4e5b112f704cdc36a563d1efdcee46",
"source_sha256": [
"1f7ececb142f9a363d2d1ca25eb7b8488b2cc319a51b55faa384f2a3d51f2142",
"6f326234791e1cff4ff866bface0d097d0d6e3844eebb1c97653d8e9c35e9397"
],
"sources": [
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low3.jsonl",
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high3.jsonl"
]
},
"selection": {
"anchor": 0.48664343020532463,
"arrival_order_sha256": "efce7339e22d3618cb4d55e6b55bfddb2c563c18faba2a992d5829c13e3f55e9",
"input_length_order_sha256": "0792b05fff6729fbd92ab2bb4cb6d31bea7799e232ad42772936bc06efbafb54",
"offered_req_s": 8.5,
"offered_req_s_per_gpu": 2.125,
"request_id_order_sha256": "2a2fabe2c4cf176aeb7e0d32fb8e7dbb1f27429a2e7a0cd18d7d186f23096f19",
"selected_count": 2550,
"target_count": 2550,
"target_req_s_per_gpu": 2.125
},
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep3-tp4.json",
"study_sha256": "fb8ffe256dace32f4ca8a8d49b662d98c3b69b94ecc8fa826e43068b238884ab"
}
},
"sanity": {
"invariants": {
"all_repetition_orders_are_permutations": true,
"five_unique_configs": true,
"same_load_all_repetitions": true,
"shared_endpoint_reused_by_both_regimes": true,
"three_disjoint_repetitions": true
},
"red_flags": []
},
"schema": "action-aware-constraint-pilot-manifest-v0",
"source": {
"base_manifest": "/home/gahow/phd/aituner/runs/intervention-response-v2/pilot-manifest-v3.json",
"base_manifest_sha256": "273db1181dcc9d6b64439650d0642ebe553b12e6aa9adebfbe3758a7977e5611",
"source_trace": "/home/admin/cpfs/wjh/aituner/aituner/trace_windows/traces/chat_w20260312_1000.jsonl",
"source_trace_sha256": "875ba869775deb78086477919f03b322da14e2673c7d070e26528c4190912757",
"window_id": "chat_w20260312_1000"
},
"status": "PASS"
}

View File

@@ -0,0 +1,638 @@
#!/usr/bin/env python3
"""Serialized controller for the crossed-constraint action-aware pilot."""
from __future__ import annotations
import argparse
import json
import os
import shlex
import signal
import subprocess
import sys
import time
from pathlib import Path
from typing import Any, Mapping
HERE = Path(__file__).resolve().parent
PHASE6 = HERE.parent / "opprof-phase6"
sys.path.insert(0, str(PHASE6))
import opprof_phase6_controller as base # noqa: E402
SCHEMA = "action-aware-constraint-pilot-state-v0"
def atomic_json(path: Path, payload: Any) -> None:
base.atomic_json(path, payload)
def wait_all_idle(timeout_s: float = 30.0) -> None:
deadline = time.monotonic() + timeout_s
last_error: Exception | None = None
while time.monotonic() < deadline:
try:
base.assert_all_idle()
return
except RuntimeError as error:
last_error = error
time.sleep(1.0)
raise last_error or RuntimeError("GPU idle timeout")
def configure(args: argparse.Namespace, manifest: Mapping[str, Any]) -> None:
base.WORKDIR = args.run_root.parent
base.RUN_ROOT = args.run_root
base.STATE = args.run_root / "controller-state.json"
base.SOURCE = args.vllm_source
base.VENV = args.venv
base.AITUNER = args.aituner_root
base.MODEL = args.model
base.CLIENT = args.client
base.GPU_LIMIT = float(manifest["budget"]["hard_cap_h20_hours"])
base.MARKER = "action-aware-constraint-pilot-v0"
def validate_inputs(args: argparse.Namespace, manifest: Mapping[str, Any]) -> None:
if manifest.get("schema") not in {
"action-aware-constraint-pilot-manifest-v0",
"action-aware-constraint-pilot-manifest-v1",
}:
raise RuntimeError("unexpected action-aware manifest schema")
if manifest.get("status") != "PASS":
raise RuntimeError("action-aware manifest did not pass preflight")
red_flags = manifest.get("sanity", {}).get("red_flags", [])
if red_flags:
raise RuntimeError(f"manifest red flags: {red_flags}")
required = {
"manifest": args.manifest,
"aituner_root": args.aituner_root,
"vllm_source": args.vllm_source,
"venv_python": args.venv / "bin/python",
"venv_vllm": args.venv / "bin/vllm",
"model": args.model,
"client": args.client,
"burnin_study": Path(manifest["burnin"]["study"]),
}
for repetition, item in manifest["repetitions"].items():
required[f"rep{repetition}_study"] = Path(item["study"])
required[f"rep{repetition}_trace"] = Path(item["merged_trace"]["path"])
missing = {name: str(path) for name, path in required.items() if not path.exists()}
if missing:
raise RuntimeError(f"action-aware input paths missing: {missing}")
def config_map(manifest: Mapping[str, Any]) -> dict[str, dict[str, Any]]:
return {str(item["id"]): dict(item) for item in manifest["configs"]}
def server_command(
config: Mapping[str, Any], *, gpus: tuple[int, ...], port: int
) -> list[str]:
return [
"taskset",
"-c",
base.cpu_mask(gpus),
str(base.VENV / "bin/vllm"),
"serve",
str(base.MODEL),
"--host",
"127.0.0.1",
"--port",
str(port),
"--served-model-name",
"qwen3-30b-a3b-community",
"--max-num-batched-tokens",
str(config["mbbt"]),
"--max-num-seqs",
str(config["mns"]),
"--tensor-parallel-size",
"4",
"--shutdown-timeout",
"120",
]
def client_command(
entry: Mapping[str, Any],
config: Mapping[str, Any],
*,
study: str,
anchor: float,
output: Path,
warmup: bool,
) -> list[str]:
command = [
"taskset",
"-c",
base.cpu_mask(entry["gpus"]),
str(base.VENV / "bin/python"),
str(base.CLIENT),
"warmup" if warmup else "run-anchor",
"--study",
study,
"--cell",
str(config["id"]),
"--anchor",
str(anchor),
"--tp",
"4",
"--mns",
str(config["mns"]),
"--mbbt",
str(config["mbbt"]),
"--base-url",
f"http://127.0.0.1:{entry['port']}",
"--result-dir",
str(output),
"--disable-slo-early-stop",
]
return command
def remaining_projection(
manifest: Mapping[str, Any], *, completed_sessions: int
) -> float:
remaining = len(manifest["configs"]) - completed_sessions
return (
remaining * float(manifest["budget"]["session_estimate_h20_hours"])
+ float(manifest["budget"]["safety_h20_hours"])
)
def dry_run_plan(
args: argparse.Namespace, manifest: Mapping[str, Any]
) -> dict[str, Any]:
sessions = []
for index, config in enumerate(manifest["configs"]):
entry = {"gpus": (0, 1, 2, 3), "port": 9050 + index}
session_root = args.run_root / "sessions" / str(config["id"])
first_repetition = str(config["repetition_order"][0])
first = manifest["repetitions"][first_repetition]
commands = {
"server": server_command(config, gpus=entry["gpus"], port=entry["port"]),
"warmup": client_command(
entry,
config,
study=first["study"],
anchor=float(first["selection"]["anchor"]),
output=session_root / "warmup",
warmup=True,
),
"burnin": client_command(
entry,
config,
study=manifest["burnin"]["study"],
anchor=float(manifest["burnin"]["anchor"]),
output=session_root / "burnin",
warmup=False,
),
}
for repetition in config["repetition_order"]:
item = manifest["repetitions"][str(repetition)]
commands[f"rep{repetition}"] = client_command(
entry,
config,
study=item["study"],
anchor=float(item["selection"]["anchor"]),
output=session_root / f"rep{repetition}",
warmup=False,
)
sessions.append(
{
"config": config["id"],
"mns": config["mns"],
"mbbt": config["mbbt"],
"port": entry["port"],
"repetition_order": config["repetition_order"],
"commands": {
role: shlex.join(command) for role, command in commands.items()
},
}
)
return {
"schema": "action-aware-constraint-pilot-dry-run-v0",
"status": "PASS",
"manifest": str(args.manifest),
"run_root": str(args.run_root),
"projected_h20_hours": remaining_projection(
manifest, completed_sessions=0
),
"hard_cap_h20_hours": manifest["budget"]["hard_cap_h20_hours"],
"sessions": sessions,
}
def load_state(path: Path, hard_cap: float) -> dict[str, Any]:
if path.exists():
return json.loads(path.read_text(encoding="utf-8"))
return {
"schema": SCHEMA,
"status": "initialized",
"hard_cap_h20_hours": hard_cap,
"gpu_hours_total": 0.0,
"completed_sessions": 0,
"sessions": {},
"failures": [],
"started_at": time.time(),
}
def append_echo(run_root: Path, line: str) -> None:
run_root.mkdir(parents=True, exist_ok=True)
with (run_root / "launch-echo.log").open("a", encoding="utf-8") as target:
target.write(line + "\n")
print(line, flush=True)
def start_server(
*,
args: argparse.Namespace,
config: Mapping[str, Any],
index: int,
) -> dict[str, Any]:
gpus = (0, 1, 2, 3)
session_root = args.run_root / "sessions" / str(config["id"])
session_root.mkdir(parents=True, exist_ok=True)
port = 9050 + index
command = server_command(config, gpus=gpus, port=port)
with (session_root / "commands.log").open("a", encoding="utf-8") as log:
log.write(f"SERVER {shlex.join(command)}\n")
server_log = (session_root / "server.log").open("ab", buffering=0)
environment = os.environ.copy()
environment.update(
{
"CUDA_VISIBLE_DEVICES": "0,1,2,3",
"VLLM_OPPROF_DIR": str(session_root / "opprof"),
"OPPROF_PHASE6_MARKER": base.MARKER,
"AITUNER_ROOT": str(base.AITUNER),
"HF_HUB_OFFLINE": "1",
"TRANSFORMERS_OFFLINE": "1",
"PYTHONUNBUFFERED": "1",
}
)
server = subprocess.Popen(
command,
cwd=base.SOURCE,
env=environment,
stdout=server_log,
stderr=subprocess.STDOUT,
start_new_session=True,
)
base.OWNED_PGIDS.add(server.pid)
return {
"cell": str(config["id"]),
"gpus": gpus,
"port": port,
"dir": session_root,
"server": server,
"server_handle": server_log,
"spawned_at": time.time(),
"results": [],
}
def validate_result(
result: Mapping[str, Any],
*,
config: Mapping[str, Any],
selection: Mapping[str, Any],
role: str,
warmup: bool,
) -> None:
if result.get("schema") != "action-aware-pilot-result-v0":
raise RuntimeError(f"unexpected result schema: {role}")
if result.get("config_id") != config["id"]:
raise RuntimeError(f"config id mismatch: {role}")
if int(result["tp"]) != 4:
raise RuntimeError(f"TP mismatch: {role}")
if int(result["mns"]) != int(config["mns"]):
raise RuntimeError(f"MNS mismatch: {role}")
if int(result["mbbt"]) != int(config["mbbt"]):
raise RuntimeError(f"MBBT mismatch: {role}")
if result.get("slo_early_stop_disabled") is not True:
raise RuntimeError(f"SLO early stop was not disabled: {role}")
if warmup:
if result["kind"] != "warmup" or int(result["selection"]["count"]) != 16:
raise RuntimeError(f"invalid warmup: {role}")
return
if bool(result["early_stopped"]):
raise RuntimeError(f"uncensored run early-stopped: {role}")
if int(result["selection"]["count"]) != int(selection["selected_count"]):
raise RuntimeError(f"selection count mismatch: {role}")
if int(result["observed_count"]) != int(selection["selected_count"]):
raise RuntimeError(f"request accounting mismatch: {role}")
for result_key, selection_key in (
("request_id_order_sha256", "request_id_order_sha256"),
("arrival_order_sha256", "arrival_order_sha256"),
("raw_length_order_sha256", "input_length_order_sha256"),
):
if result["selection"][result_key] != selection[selection_key]:
raise RuntimeError(f"selection hash mismatch {result_key}: {role}")
def burnin_gate(
result: Mapping[str, Any],
*,
expected_count: int,
maximum_elapsed_s: float,
) -> dict[str, Any]:
if result.get("kind") != "anchor":
raise RuntimeError("burnin gate received a non-anchor result")
if int(result["selection"]["count"]) != expected_count:
raise RuntimeError("burnin gate received the wrong request set")
elapsed_s = float(result["interval"]["elapsed_s"])
summary = {
"elapsed_s": elapsed_s,
"pass_rate": float(result["pass_rate"]),
"feasible": bool(result["feasible"]),
}
if elapsed_s > maximum_elapsed_s:
raise RuntimeError(
f"burnin throughput gate failed: {elapsed_s:.3f}s > "
f"{maximum_elapsed_s:.3f}s"
)
return summary
def run_client(
*,
entry: dict[str, Any],
config: Mapping[str, Any],
role: str,
study: str,
selection: Mapping[str, Any],
output: Path,
state: Mapping[str, Any],
timeout_s: float,
warmup: bool = False,
) -> dict[str, Any]:
command = client_command(
entry,
config,
study=study,
anchor=float(selection["anchor"]),
output=output,
warmup=warmup,
)
with (entry["dir"] / "commands.log").open("a", encoding="utf-8") as log:
log.write(f"CLIENT role={role} {shlex.join(command)}\n")
handle = (output.parent / f"{output.name}.log").open("ab", buffering=0)
environment = os.environ.copy()
environment.update({"AITUNER_ROOT": str(base.AITUNER), "PYTHONUNBUFFERED": "1"})
process = subprocess.Popen(
command,
cwd=base.WORKDIR,
env=environment,
stdout=handle,
stderr=subprocess.STDOUT,
start_new_session=True,
)
deadline = time.monotonic() + timeout_s
try:
while process.poll() is None:
if time.monotonic() > deadline:
raise TimeoutError(f"client timeout: {config['id']} {role}")
if entry["server"].poll() is not None:
raise RuntimeError(f"server exited during {config['id']} {role}")
base.assert_no_other_compute()
if state["gpu_hours_total"] + base.live_gpu_hours([entry]) >= base.GPU_LIMIT:
raise RuntimeError("action-aware pilot H20-hour hard cap reached")
time.sleep(1.0)
except Exception:
try:
os.killpg(process.pid, signal.SIGTERM)
except ProcessLookupError:
pass
try:
process.wait(timeout=10.0)
except subprocess.TimeoutExpired:
try:
os.killpg(process.pid, signal.SIGKILL)
except ProcessLookupError:
pass
process.wait(timeout=10.0)
raise
finally:
handle.close()
if process.returncode:
raise RuntimeError(
f"client failed: config={config['id']} role={role} rc={process.returncode}"
)
result = json.loads((output / "result.json").read_text(encoding="utf-8"))
validate_result(
result,
config=config,
selection=selection,
role=role,
warmup=warmup,
)
entry["results"].append(
{"anchor": float(selection["anchor"]), "dir": str(output), "kind": result["kind"]}
)
return result
def execute_session(
*,
args: argparse.Namespace,
manifest: Mapping[str, Any],
config: Mapping[str, Any],
index: int,
state: dict[str, Any],
state_path: Path,
) -> None:
name = str(config["id"])
if state["sessions"].get(name, {}).get("status") == "complete":
return
projection = remaining_projection(
manifest, completed_sessions=int(state["completed_sessions"])
)
if float(state["gpu_hours_total"]) + projection > base.GPU_LIMIT:
raise RuntimeError(f"projected cost exceeds cap before {name}")
load_values = {
float(item["selection"]["offered_req_s_per_gpu"])
for item in manifest["repetitions"].values()
}
load_text = (
f"{next(iter(load_values)):.6g}"
if len(load_values) == 1
else ",".join(f"{value:.6g}" for value in sorted(load_values))
)
echo = (
f"ACTION_AWARE_SESSION_ECHO host=dash0 config={name} tp=4 "
f"mns={config['mns']} mbbt={config['mbbt']} gpus=0-3 "
f"workload={manifest['source']['window_id']} load_per_gpu={load_text} "
f"duration_s={manifest['engine']['duration_s']} "
f"repetitions={','.join(map(str, config['repetition_order']))} "
f"source={args.manifest} output={args.run_root / 'sessions' / name} "
f"spent_h20h={state['gpu_hours_total']:.6f} "
f"remaining_projection_h20h={projection:.3f} cap_h20h={base.GPU_LIMIT:.1f}"
)
append_echo(args.run_root, echo)
wait_all_idle()
session_state = {
"status": "starting",
"mns": int(config["mns"]),
"mbbt": int(config["mbbt"]),
"repetition_order": list(config["repetition_order"]),
"started_at": time.time(),
"runs": [],
}
state["status"] = "running"
state["sessions"][name] = session_state
atomic_json(state_path, state)
entry = start_server(args=args, config=config, index=index)
failure: Exception | None = None
try:
base.wait_ready(entry)
first = manifest["repetitions"][str(config["repetition_order"][0])]
session_state["status"] = "warmup"
atomic_json(state_path, state)
run_client(
entry=entry,
config=config,
role="warmup",
study=first["study"],
selection=first["selection"],
output=entry["dir"] / "warmup",
state=state,
timeout_s=180.0,
warmup=True,
)
session_state["status"] = "burnin"
atomic_json(state_path, state)
burnin = manifest["burnin"]
burnin_result = run_client(
entry=entry,
config=config,
role="burnin",
study=burnin["study"],
selection=burnin,
output=entry["dir"] / "burnin",
state=state,
timeout_s=float(manifest["engine"]["client_timeout_s"]),
)
session_state["burnin"] = burnin_gate(
burnin_result,
expected_count=int(burnin["selected_count"]),
maximum_elapsed_s=float(manifest["engine"]["burnin_max_elapsed_s"]),
)
atomic_json(state_path, state)
session_state["status"] = "measured"
atomic_json(state_path, state)
for repetition in config["repetition_order"]:
item = manifest["repetitions"][str(repetition)]
role = f"rep{repetition}"
result = run_client(
entry=entry,
config=config,
role=role,
study=item["study"],
selection=item["selection"],
output=entry["dir"] / role,
state=state,
timeout_s=float(manifest["engine"]["client_timeout_s"]),
)
session_state["runs"].append(
{
"repetition": int(repetition),
"pass_rate": result["pass_rate"],
"feasible": result["feasible"],
"slo_pass_count": result["slo_pass_count"],
"elapsed_s": result["interval"]["elapsed_s"],
}
)
atomic_json(state_path, state)
session_state["status"] = "stopping"
atomic_json(state_path, state)
except Exception as error: # noqa: BLE001
failure = error
finally:
try:
base.stop_entry(entry)
except Exception as error: # noqa: BLE001
failure = failure or error
time.sleep(2.0)
try:
wait_all_idle()
except Exception as error: # noqa: BLE001
failure = failure or error
session_hours = base.live_gpu_hours([entry])
state["gpu_hours_total"] += session_hours
session_state["gpu_hours"] = session_hours
if failure is not None:
session_state["status"] = "failed"
session_state["failure"] = repr(failure)
state["status"] = "failed"
state["failures"].append({"session": name, "failure": repr(failure)})
atomic_json(state_path, state)
raise failure
validation = base.validate_cell(entry)
session_state["validation"] = validation
session_state["status"] = "complete"
session_state["completed_at"] = time.time()
state["completed_sessions"] += 1
atomic_json(state_path, state)
def parser() -> argparse.ArgumentParser:
result = argparse.ArgumentParser()
result.add_argument("--manifest", type=Path, required=True)
result.add_argument("--run-root", type=Path, required=True)
result.add_argument("--aituner-root", type=Path, required=True)
result.add_argument("--vllm-source", type=Path, required=True)
result.add_argument("--venv", type=Path, required=True)
result.add_argument("--model", type=Path, required=True)
result.add_argument("--client", type=Path, required=True)
result.add_argument("--dry-run", action="store_true")
return result
def main() -> None:
args = parser().parse_args()
manifest = json.loads(args.manifest.read_text(encoding="utf-8"))
validate_inputs(args, manifest)
configure(args, manifest)
if args.dry_run:
print(json.dumps(dry_run_plan(args, manifest), indent=2, sort_keys=True))
return
args.run_root.mkdir(parents=True, exist_ok=True)
copied_manifest = args.run_root / "pilot-manifest.json"
if not copied_manifest.exists():
atomic_json(copied_manifest, manifest)
state_path = args.run_root / "controller-state.json"
state = load_state(state_path, base.GPU_LIMIT)
state["status"] = "running"
atomic_json(state_path, state)
for index, config in enumerate(manifest["configs"]):
execute_session(
args=args,
manifest=manifest,
config=config,
index=index,
state=state,
state_path=state_path,
)
state["status"] = "complete"
state["completed_at"] = time.time()
atomic_json(state_path, state)
wait_all_idle()
print(
json.dumps(
{
"status": state["status"],
"completed_sessions": state["completed_sessions"],
"gpu_hours_total": state["gpu_hours_total"],
},
sort_keys=True,
)
)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,195 @@
#!/usr/bin/env python3
"""Freeze the crossed-constraint action-aware development pilot."""
from __future__ import annotations
import argparse
import hashlib
import json
import os
from pathlib import Path
from typing import Any
SCHEMA_V0 = "action-aware-constraint-pilot-manifest-v0"
SCHEMA_V1 = "action-aware-constraint-pilot-manifest-v1"
def configs(token_source_mbbt: int) -> tuple[dict[str, Any], ...]:
return (
{
"id": "b_base",
"mns": 64,
"mbbt": token_source_mbbt,
"repetition_order": [1, 2, 3],
},
{"id": "a_base", "mns": 16, "mbbt": 8192, "repetition_order": [2, 3, 1]},
{"id": "shared", "mns": 64, "mbbt": 8192, "repetition_order": [3, 1, 2]},
{
"id": "b_mns",
"mns": 128,
"mbbt": token_source_mbbt,
"repetition_order": [1, 3, 2],
},
{"id": "a_mbbt", "mns": 16, "mbbt": 16384, "repetition_order": [2, 1, 3]},
)
def sha256_file(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for chunk in iter(lambda: source.read(1 << 20), b""):
digest.update(chunk)
return digest.hexdigest()
def atomic_json(path: Path, payload: Any) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
temporary = path.with_suffix(path.suffix + ".tmp")
temporary.write_text(
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
)
os.replace(temporary, path)
def build(
base_path: Path,
*,
token_source_mbbt: int = 256,
prior_attempt_h20_hours: float = 0.0,
prior_attempt_artifact: str | None = None,
) -> dict[str, Any]:
if token_source_mbbt <= 0:
raise ValueError("token source MBBT must be positive")
if prior_attempt_h20_hours < 0.0 or prior_attempt_h20_hours >= 8.0:
raise ValueError("prior attempt cost must be in [0, 8)")
base = json.loads(base_path.read_text(encoding="utf-8"))
if base.get("schema") != "intervention-response-phase-aware-pilot-manifest-v3":
raise ValueError("unexpected base manifest schema")
if base.get("status") != "PASS":
raise ValueError("base manifest did not pass its preflight")
if sorted(int(key) for key in base["repetitions"]) != [1, 2, 3]:
raise ValueError("base manifest must contain exactly three repetitions")
repetitions = {}
selection_hashes = []
for repetition in (1, 2, 3):
source = base["repetitions"][str(repetition)]
selection = dict(source["selections"]["mid"])
selection_hashes.append(selection["request_id_order_sha256"])
repetitions[str(repetition)] = {
"study": source["study"],
"study_sha256": source["study_sha256"],
"selection": selection,
"merged_trace": source["merged_trace"],
}
frozen_configs = configs(token_source_mbbt)
config_ids = [str(config["id"]) for config in frozen_configs]
schema = (
SCHEMA_V0
if token_source_mbbt == 256 and prior_attempt_h20_hours == 0.0
else SCHEMA_V1
)
payload = {
"schema": schema,
"status": "PASS",
"source": {
"base_manifest": str(base_path.resolve()),
"base_manifest_sha256": sha256_file(base_path),
"window_id": base["source"]["window_id"],
"source_trace": base["source"]["source_trace"],
"source_trace_sha256": base["source"]["source_trace_sha256"],
},
"engine": {
"tp": 4,
"duration_s": 300.0,
"disable_slo_early_stop": True,
"client_timeout_s": 450.0,
"burnin_max_elapsed_s": 90.0,
},
"burnin": base["burnin"],
"repetitions": repetitions,
"configs": [dict(config) for config in frozen_configs],
"regimes": {
"A": {
"source": "a_base",
"actions": {"mns": "shared", "mbbt": "a_mbbt"},
},
"B": {
"source": "b_base",
"actions": {"mns": "b_mns", "mbbt": "shared"},
},
},
"budget": {
"global_hard_cap_h20_hours": 8.0,
"hard_cap_h20_hours": 8.0 - prior_attempt_h20_hours,
"prior_attempt_h20_hours": prior_attempt_h20_hours,
"prior_attempt_artifact": prior_attempt_artifact,
"session_estimate_h20_hours": 1.35,
"safety_h20_hours": 0.25,
"expected_h20_hours": [6.0, 7.2],
"expected_wall_minutes": [90, 110],
},
"gates": {
"minimum_relative_winner_margin": 0.10,
"minimum_exclusive_fraction": 0.10,
"minimum_exclusive_ratio": 5.0,
"phase_fractions": [0.25, 0.50, 0.75, 1.0],
"material_kv_usage": 0.90,
},
"sanity": {
"invariants": {
"five_unique_configs": len(config_ids) == len(set(config_ids)) == 5,
"three_disjoint_repetitions": len(set(selection_hashes)) == 3,
"same_load_all_repetitions": len(
{
float(item["selection"]["offered_req_s_per_gpu"])
for item in repetitions.values()
}
)
== 1,
"all_repetition_orders_are_permutations": all(
sorted(config["repetition_order"]) == [1, 2, 3]
for config in frozen_configs
),
}
},
}
payload["sanity"]["invariants"]["shared_endpoint_reused_by_both_regimes"] = (
payload["regimes"]["A"]["actions"]["mns"]
== payload["regimes"]["B"]["actions"]["mbbt"]
== "shared"
)
payload["sanity"]["red_flags"] = [
name
for name, passed in payload["sanity"]["invariants"].items()
if not passed
]
if payload["sanity"]["red_flags"]:
payload["status"] = "FAIL"
return payload
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--base-manifest", type=Path, required=True)
parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--token-source-mbbt", type=int, default=256)
parser.add_argument("--prior-attempt-h20-hours", type=float, default=0.0)
parser.add_argument("--prior-attempt-artifact")
args = parser.parse_args()
payload = build(
args.base_manifest,
token_source_mbbt=args.token_source_mbbt,
prior_attempt_h20_hours=args.prior_attempt_h20_hours,
prior_attempt_artifact=args.prior_attempt_artifact,
)
atomic_json(args.output, payload)
print(json.dumps(payload["sanity"], sort_keys=True))
if payload["status"] != "PASS":
raise SystemExit("manifest preflight failed")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,24 @@
{
"global_hard_cap_h20_hours": 8.0,
"invariants": {
"all_gpus_idle_after_each_stop": true,
"no_completed_measured_runs": true,
"no_prior_runtime_data_reused": true
},
"prior_attempt_h20_hours": 0.7043978426191542,
"schema": "action-aware-prior-attempts-v2",
"stops": [
{
"artifact": "/home/admin/cpfs/wjh/action-aware-constraint-v0-20260714/operational-stop-v0.json",
"h20_hours": 0.38598689953486126,
"reason": "MBBT256 burn-in remained throughput-backlogged",
"stage": "burnin"
},
{
"artifact": "/home/admin/cpfs/wjh/action-aware-constraint-v1-20260714/operational-stop-v1.json",
"h20_hours": 0.31841094308429296,
"reason": "controller passed the warmup result to the burn-in gate",
"stage": "first measured run in flight; zero measured results completed"
}
]
}

View File

@@ -0,0 +1,292 @@
#!/usr/bin/env python3
from __future__ import annotations
import copy
import importlib.util
from pathlib import Path
from types import SimpleNamespace
HERE = Path(__file__).resolve().parent
ROOT = HERE.parents[1]
def load(name: str, filename: str):
spec = importlib.util.spec_from_file_location(name, HERE / filename)
module = importlib.util.module_from_spec(spec)
assert spec.loader is not None
spec.loader.exec_module(module)
return module
def record(*, waiting: int, running: int, tokens: int) -> dict:
return {
"queues": {"waiting": waiting, "deferred": 0, "running": running},
"prefill_tokens": tokens,
"decode_tokens": 0,
"kv": {"usage": 0.5},
"preemptions": 0,
}
def fake_run(
config: str,
repetition: int,
*,
goodput: float,
mns_score: float = 0.0,
mbbt_score: float = 0.0,
ambiguous: float = 0.0,
) -> dict:
binding = {
"mns_exclusive_fraction": mns_score,
"mbbt_exclusive_fraction": mbbt_score,
"both_fraction": ambiguous,
"waiting_unresolved_fraction": 0.0,
"kv_usage_max": 0.5,
"preemptions": 0,
}
phases = {
phase: {
"mns_exclusive_fraction": mns_score,
"mbbt_exclusive_fraction": mbbt_score,
}
for phase in ("0.25", "0.50", "0.75", "1.00")
}
return {
"config_id": config,
"repetition": repetition,
"outcome": {"slo_goodput_req_s": goodput},
"binding": binding,
"phases": phases,
}
def main() -> None:
analysis = load("action_aware_analysis", "analyze_pilot.py")
summary = analysis.binding_summary(
[
record(waiting=1, running=16, tokens=8),
record(waiting=1, running=8, tokens=32),
record(waiting=1, running=16, tokens=32),
record(waiting=1, running=8, tokens=8),
record(waiting=0, running=8, tokens=8),
],
mns=16,
mbbt=32,
)
assert summary["mns_exclusive_count"] == 1
assert summary["mbbt_exclusive_count"] == 1
assert summary["both_count"] == 1
assert summary["waiting_unresolved_count"] == 1
assert summary["waiting_count"] == 4
# A per-step stream may have a submit gap above one second when the
# preceding model execution itself spans that interval. Such a gap is
# covered telemetry, not a dropped-record interval.
asynchronous = [
{"submit_mono_ns": 0, "complete_mono_ns": 1_200_000_000},
{"submit_mono_ns": 1_100_000_000, "complete_mono_ns": 1_300_000_000},
]
coverage, covered = analysis.telemetry_coverage(
asynchronous, start_ns=0, end_ns=1_100_000_000
)
assert coverage["max_internal_submit_gap_s"] == 1.1
assert coverage["max_uncovered_gap_s"] == 0.0
assert covered
missing = copy.deepcopy(asynchronous)
missing[0]["complete_mono_ns"] = 0
assert not analysis.telemetry_coverage(
missing, start_ns=0, end_ns=1_100_000_000
)[1]
mechanism = analysis.mechanism_summary(
[
{
"model_executed": True,
"submit_mono_ns": 0,
"complete_mono_ns": 2_000_000,
"prefill_tokens": 8,
"prefill_requests": 2,
"chunked_prefill": {
"first": 1,
"middle": 0,
"final": 0,
"unsplit": 1,
"tokens": 8,
},
"prefix": {"local": {"queries": 10, "hits": 2}},
},
{
"model_executed": True,
"submit_mono_ns": 2_000_000,
"complete_mono_ns": 3_000_000,
"prefill_tokens": 0,
"prefill_requests": 0,
"chunked_prefill": {
"first": 0,
"middle": 0,
"final": 0,
"unsplit": 0,
"tokens": 0,
},
"prefix": {"local": {"queries": 0, "hits": 0}},
},
]
)
assert mechanism["prefill"]["requests_per_step"] == 2.0
assert mechanism["prefill"]["chunks"]["first"] == 1
assert mechanism["prefix"]["hit_rate"] == 0.2
assert all(mechanism["sanity"]["invariants"].values())
manifest = {
"repetitions": {str(index): {} for index in (1, 2, 3)},
"regimes": {
"A": {
"source": "a_base",
"actions": {"mns": "shared", "mbbt": "a_mbbt"},
},
"B": {
"source": "b_base",
"actions": {"mns": "b_mns", "mbbt": "shared"},
},
},
"gates": {
"minimum_relative_winner_margin": 0.10,
"minimum_exclusive_fraction": 0.10,
"minimum_exclusive_ratio": 5.0,
"material_kv_usage": 0.90,
},
}
runs = []
for repetition in (1, 2, 3):
runs.extend(
[
fake_run(
"a_base",
repetition,
goodput=1.0,
mns_score=0.8,
mbbt_score=0.01,
),
fake_run(
"b_base",
repetition,
goodput=1.0,
mns_score=0.01,
mbbt_score=0.7,
),
fake_run("shared", repetition, goodput=3.0),
fake_run("a_mbbt", repetition, goodput=1.5),
fake_run("b_mns", repetition, goodput=1.2),
]
)
result = analysis.evaluate_decisions(runs, manifest)
assert result["decision"] == "STOP_NO_NEW_INSTRUMENTATION_NEEDED"
assert result["baselines"] == {
"always_mns_correct": 3,
"always_mbbt_correct": 3,
"binding_correct": 6,
"decision_count": 6,
}
ambiguous = copy.deepcopy(runs)
for run in ambiguous:
if run["config_id"] == "b_base":
run["binding"]["both_fraction"] = 0.8
assert (
analysis.evaluate_decisions(ambiguous, manifest)["decision"]
== "OPEN_EXACT_ATTRIBUTION_ABLATION"
)
wrong = copy.deepcopy(runs)
for run in wrong:
if run["config_id"] == "b_base":
run["binding"]["mns_exclusive_fraction"] = 0.8
run["binding"]["mbbt_exclusive_fraction"] = 0.01
for phase in run["phases"].values():
phase["mns_exclusive_fraction"] = 0.8
phase["mbbt_exclusive_fraction"] = 0.01
assert (
analysis.evaluate_decisions(wrong, manifest)["decision"]
== "STOP_BINDING_NOT_PREDICTIVE"
)
prepare = load("action_aware_prepare", "prepare_pilot.py")
frozen = prepare.build(
ROOT / "runs/intervention-response-v2/pilot-manifest-v3.json"
)
assert frozen["status"] == "PASS"
assert frozen["sanity"]["red_flags"] == []
assert [config["id"] for config in frozen["configs"]] == [
"b_base",
"a_base",
"shared",
"b_mns",
"a_mbbt",
]
controller = load("action_aware_controller", "pilot_controller.py")
args = SimpleNamespace(
manifest=Path("/tmp/manifest.json"),
run_root=Path("/tmp/action-aware"),
aituner_root=Path("/tmp/aituner"),
vllm_source=Path("/tmp/vllm"),
venv=Path("/tmp/venv"),
model=Path("/tmp/model"),
client=Path("/tmp/client.py"),
)
controller.configure(args, frozen)
plan = controller.dry_run_plan(args, frozen)
assert plan["status"] == "PASS"
assert len(plan["sessions"]) == 5
assert plan["projected_h20_hours"] == 7.0
assert "--max-num-batched-tokens 256" in plan["sessions"][0]["commands"]["server"]
revised = prepare.build(
ROOT / "runs/intervention-response-v2/pilot-manifest-v3.json",
token_source_mbbt=2048,
prior_attempt_h20_hours=0.38598689953486126,
prior_attempt_artifact="/tmp/operational-stop-v0.json",
)
assert revised["schema"] == "action-aware-constraint-pilot-manifest-v1"
assert revised["configs"][0]["mbbt"] == 2048
assert revised["configs"][3]["mbbt"] == 2048
assert revised["budget"]["hard_cap_h20_hours"] < 8.0
controller.configure(args, revised)
revised_plan = controller.dry_run_plan(args, revised)
assert revised_plan["projected_h20_hours"] < revised_plan["hard_cap_h20_hours"]
assert (
"--max-num-batched-tokens 2048"
in revised_plan["sessions"][0]["commands"]["server"]
)
accepted_burnin = {
"kind": "anchor",
"selection": {"count": 510},
"interval": {"elapsed_s": 61.25},
"pass_rate": 0.5,
"feasible": False,
}
assert controller.burnin_gate(
accepted_burnin, expected_count=510, maximum_elapsed_s=90.0
)["elapsed_s"] == 61.25
warmup = copy.deepcopy(accepted_burnin)
warmup["kind"] = "warmup"
try:
controller.burnin_gate(warmup, expected_count=510, maximum_elapsed_s=90.0)
except RuntimeError as error:
assert "non-anchor" in str(error)
else:
raise AssertionError("warmup incorrectly passed the burnin gate")
slow = copy.deepcopy(accepted_burnin)
slow["interval"]["elapsed_s"] = 91.0
try:
controller.burnin_gate(slow, expected_count=510, maximum_elapsed_s=90.0)
except RuntimeError as error:
assert "throughput gate failed" in str(error)
else:
raise AssertionError("slow burnin incorrectly passed the throughput gate")
print("action-aware constraint pilot: PASS")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,325 @@
#!/usr/bin/env python3
"""Audit held-out action/measurement choices against the exact 2x2 surface."""
from __future__ import annotations
import argparse
import hashlib
import json
import math
import os
import statistics
from pathlib import Path
from typing import Any, Mapping
SCHEMA = "active-intervention-prospective-audit-v0"
def sha256_file(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for chunk in iter(lambda: source.read(1 << 20), b""):
digest.update(chunk)
return digest.hexdigest()
def atomic_json(path: Path, payload: Any) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
temporary = path.with_suffix(path.suffix + ".tmp")
temporary.write_text(
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
)
os.replace(temporary, path)
def numeric(values: list[float]) -> dict[str, Any]:
finite = [float(value) for value in values]
if not finite or any(not math.isfinite(value) for value in finite):
raise ValueError("numeric summary requires finite values")
return {
"n": len(finite),
"min": min(finite),
"max": max(finite),
"distinct_n": len(set(finite)),
}
def load_surface(
manifest: Mapping[str, Any], run_root: Path
) -> tuple[dict[str, Any], list[dict[str, Any]]]:
rows = []
aggregate = {}
duration_s = float(manifest["engine"]["duration_s"])
tp = int(manifest["engine"]["tp"])
for config in manifest["configs"]:
config_id = str(config["id"])
values = []
for repetition in sorted(int(key) for key in manifest["repetitions"]):
expected = manifest["repetitions"][str(repetition)]["selection"]
result_path = (
run_root / "sessions" / config_id / f"rep{repetition}" / "result.json"
)
result = json.loads(result_path.read_text(encoding="utf-8"))
if result["selection"]["request_id_order_sha256"] != expected[
"request_id_order_sha256"
]:
raise ValueError(f"request hash mismatch: {config_id} rep{repetition}")
offered_total = float(expected["offered_req_s_per_gpu"]) * tp
normalized = float(result["slo_pass_count"]) / duration_s / offered_total
values.append(normalized)
rows.append(
{
"config_id": config_id,
"mns": int(config["mns"]),
"mbbt": int(config["mbbt"]),
"repetition": repetition,
"normalized_slo_goodput": normalized,
"slo_goodput_req_s": float(result["slo_pass_count"]) / duration_s,
"pass_rate": float(result["pass_rate"]),
"elapsed_s": float(result["interval"]["elapsed_s"]),
"result": str(result_path),
"result_sha256": sha256_file(result_path),
}
)
aggregate[config_id] = {
"normalized_slo_goodput_values": values,
"median_normalized_slo_goodput": float(statistics.median(values)),
"sanity": numeric(values),
}
return aggregate, rows
def source_cost_estimate(
*,
source_session: Mapping[str, Any],
source_rows: list[Mapping[str, Any]],
cutoff_s: float,
tp: int,
) -> dict[str, float]:
actual_h20_hours = float(source_session["gpu_hours"])
measured_replay_h20_hours = (
tp * sum(float(row["elapsed_s"]) for row in source_rows) / 3600.0
)
fixed_h20_hours = max(0.0, actual_h20_hours - measured_replay_h20_hours)
prefix_replay_h20_hours = tp * len(source_rows) * cutoff_s / 3600.0
return {
"actual_full_session_h20_hours": actual_h20_hours,
"fixed_startup_warmup_burnin_cleanup_h20_hours": fixed_h20_hours,
"prefix_replay_h20_hours_lower_bound": prefix_replay_h20_hours,
"counterfactual_all_in_h20_hours_lower_bound": fixed_h20_hours
+ prefix_replay_h20_hours,
}
def replay_policy(
*,
mode: str,
manifest: Mapping[str, Any],
decision: Mapping[str, Any],
surface: Mapping[str, Any],
session_costs: Mapping[str, float],
source_cost: Mapping[str, float],
) -> dict[str, Any]:
acceptable_regret = float(manifest["gates"]["acceptable_regret"])
source_id = str(manifest["source_config_id"])
oracle = max(
float(item["median_normalized_slo_goodput"]) for item in surface.values()
)
cumulative = float(source_cost["counterfactual_all_in_h20_hours_lower_bound"])
source_score = float(surface[source_id]["median_normalized_slo_goodput"])
source_regret = 1.0 - source_score / oracle if oracle > 0 else 0.0
points = [
{
"action_id": "noop",
"config_id": source_id,
"score": source_score,
"regret": source_regret,
"cumulative_h20_hours_lower_bound": cumulative,
}
]
hit = points[0] if source_regret <= acceptable_regret + 1e-12 else None
seen = {source_id}
for action_id in decision["decisions"][mode]["intervention_order"]:
config_id = str(manifest["actions"][action_id])
if config_id in seen:
continue
seen.add(config_id)
cumulative += float(session_costs[config_id])
score = float(surface[config_id]["median_normalized_slo_goodput"])
regret = 1.0 - score / oracle if oracle > 0 else 0.0
point = {
"action_id": action_id,
"config_id": config_id,
"score": score,
"regret": regret,
"cumulative_h20_hours_lower_bound": cumulative,
}
points.append(point)
if hit is None and regret <= acceptable_regret + 1e-12:
hit = point
return {
"mode": mode,
"measurement_cutoff_s": float(
decision["decisions"][mode]["selected_cutoff_s"]
),
"selected_action": decision["decisions"][mode]["selected_action"],
"decision_kind": decision["decisions"][mode]["decision_kind"],
"intervention_order": decision["decisions"][mode]["intervention_order"],
"source_cost": dict(source_cost),
"oracle_normalized_slo_goodput": oracle,
"cost_to_acceptable": hit,
"reached_acceptable": hit is not None,
"points": points,
}
def build_audit(
*, manifest_path: Path, decision_path: Path, run_root: Path
) -> dict[str, Any]:
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
decision = json.loads(decision_path.read_text(encoding="utf-8"))
state_path = run_root / "controller-state.json"
state = json.loads(state_path.read_text(encoding="utf-8"))
if manifest.get("schema") != "active-intervention-prospective-manifest-v0":
raise ValueError("unexpected prospective manifest schema")
if decision.get("schema") != "active-intervention-prospective-decision-v0":
raise ValueError("unexpected prospective decision schema")
if decision["manifest_sha256"] != sha256_file(manifest_path):
raise ValueError("decision does not match prospective manifest")
surface, rows = load_surface(manifest, run_root)
source_id = str(manifest["source_config_id"])
sessions = state["sessions"]
session_costs = {
config_id: float(sessions[config_id]["gpu_hours"])
for config_id in surface
}
source_rows = [row for row in rows if row["config_id"] == source_id]
policies = {}
for mode in ("outcome_only", "telemetry"):
cost = source_cost_estimate(
source_session=sessions[source_id],
source_rows=source_rows,
cutoff_s=float(decision["decisions"][mode]["selected_cutoff_s"]),
tp=int(manifest["engine"]["tp"]),
)
policies[mode] = replay_policy(
mode=mode,
manifest=manifest,
decision=decision,
surface=surface,
session_costs=session_costs,
source_cost=cost,
)
outcome_hit = policies["outcome_only"]["cost_to_acceptable"]
telemetry_hit = policies["telemetry"]["cost_to_acceptable"]
if outcome_hit is None or telemetry_hit is None:
reduction = None
else:
outcome_cost = float(outcome_hit["cumulative_h20_hours_lower_bound"])
telemetry_cost = float(telemetry_hit["cumulative_h20_hours_lower_bound"])
reduction = 1.0 - telemetry_cost / outcome_cost if outcome_cost > 0 else 0.0
confirmation_trigger = bool(
reduction is not None
and reduction
>= float(manifest["gates"]["confirmation_trigger_gpu_cost_reduction"])
and policies["telemetry"]["reached_acceptable"]
)
contribution_gate = bool(
reduction is not None
and reduction >= float(manifest["gates"]["contribution_gpu_cost_reduction"])
and policies["telemetry"]["reached_acceptable"]
)
status = (
"TRIGGER_ACTUAL_EARLY_STOP_CONFIRMATION"
if confirmation_trigger
else "STOP_NO_PROSPECTIVE_GPU_COST_SIGNAL"
)
normalized_values = [float(row["normalized_slo_goodput"]) for row in rows]
costs = list(session_costs.values())
invariants = {
"controller_complete": state.get("status") == "complete",
"four_sessions_complete": len(sessions) == 4
and all(item.get("status") == "complete" for item in sessions.values()),
"twelve_surface_outcomes": len(rows) == 12,
"nonnegative_goodput": all(value >= 0.0 for value in normalized_values),
"normalized_goodput_bounded": all(value <= 1.0 + 1e-12 for value in normalized_values),
"surface_not_all_identical": len(set(normalized_values)) > 1,
"nonnegative_session_costs": all(value >= 0.0 for value in costs),
"policy_replay_reaches_oracle_surface": all(
policy["reached_acceptable"] for policy in policies.values()
),
}
red_flags = [name for name, passed in invariants.items() if not passed]
if red_flags:
status = "STOP_SANITY"
return {
"schema": SCHEMA,
"status": status,
"claim_boundary": (
"Prospective exact-surface replay. Prefix source costs reconstruct the "
"measured fixed overhead plus selected replay seconds; actual early-stop "
"confirmation is required before claiming GPU-cost reduction."
),
"manifest": str(manifest_path),
"manifest_sha256": sha256_file(manifest_path),
"decision": str(decision_path),
"decision_sha256": sha256_file(decision_path),
"controller_state": str(state_path),
"controller_state_sha256": sha256_file(state_path),
"surface": surface,
"rows": rows,
"session_costs_h20_hours": session_costs,
"annotation_campaign_h20_hours": float(state["gpu_hours_total"]),
"policies": policies,
"comparison": {
"telemetry_gpu_cost_reduction_fraction": reduction,
"confirmation_trigger": confirmation_trigger,
"contribution_gate": contribution_gate,
"confirmation_trigger_threshold": manifest["gates"][
"confirmation_trigger_gpu_cost_reduction"
],
"contribution_threshold": manifest["gates"][
"contribution_gpu_cost_reduction"
],
"action_changed": policies["outcome_only"]["selected_action"]
!= policies["telemetry"]["selected_action"],
"measurement_changed": policies["outcome_only"]["measurement_cutoff_s"]
!= policies["telemetry"]["measurement_cutoff_s"],
},
"sanity": {
"invariants": invariants,
"red_flags": red_flags,
"normalized_slo_goodput": numeric(normalized_values),
"session_h20_hours": numeric(costs),
},
}
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--manifest", type=Path, required=True)
parser.add_argument("--decision", type=Path, required=True)
parser.add_argument("--run-root", type=Path, required=True)
parser.add_argument("--output", type=Path, required=True)
args = parser.parse_args()
audit = build_audit(
manifest_path=args.manifest,
decision_path=args.decision,
run_root=args.run_root,
)
atomic_json(args.output, audit)
print(
json.dumps(
{
"status": audit["status"],
"comparison": audit["comparison"],
"sanity": audit["sanity"],
},
sort_keys=True,
)
)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,324 @@
#!/usr/bin/env python3
"""Extract paired source/action examples from the accepted action-aware run."""
from __future__ import annotations
import argparse
import hashlib
import json
import math
import os
import sys
from pathlib import Path
from statistics import fmean
from typing import Any, Mapping
PHASES = ("0.25", "0.50", "0.75", "1.00")
HERE = Path(__file__).resolve().parent
COMMON_STATE = HERE.parent / "telemetry-residual"
sys.path.insert(0, str(COMMON_STATE))
from common_state import summarize_engine # noqa: E402
def sha256_file(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for chunk in iter(lambda: source.read(1 << 20), b""):
digest.update(chunk)
return digest.hexdigest()
def atomic_json(path: Path, payload: Any) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
temporary = path.with_suffix(path.suffix + ".tmp")
temporary.write_text(
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
)
os.replace(temporary, path)
def load_jsonl(path: Path) -> list[dict[str, Any]]:
records = []
with path.open(encoding="utf-8") as source:
for line_number, line in enumerate(source, 1):
if not line.strip():
continue
try:
records.append(json.loads(line))
except json.JSONDecodeError as error:
raise ValueError(f"{path}:{line_number}: invalid JSON") from error
if not records:
raise ValueError(f"{path}: no request records")
return records
def prefix_outcome(
requests: list[Mapping[str, Any]], *, cutoff_s: float, offered_total: float
) -> dict[str, float]:
admitted = [request for request in requests if float(request["arrival_s"]) <= cutoff_s]
completed = [
request
for request in requests
if request.get("completed_elapsed_s") is not None
and float(request["completed_elapsed_s"]) <= cutoff_s
]
if not admitted:
raise ValueError("prefix has no admitted requests")
admitted_ids = {str(request["request_id"]) for request in admitted}
if any(str(request["request_id"]) not in admitted_ids for request in completed):
raise ValueError("prefix completion precedes admission")
passed = sum(bool(request["slo_pass"]) for request in completed)
ttft = [float(request["ttft_ms"]) for request in completed]
tpot = [float(request["tpot_ms"]) for request in completed]
total = len(requests)
return {
"normalized_slo_goodput": passed / cutoff_s / offered_total,
"admitted_fraction": len(admitted) / total,
"completed_over_admitted": len(completed) / len(admitted),
"completed_pass_rate": passed / max(1, len(completed)),
"completed_fail_fraction_of_total": (len(completed) - passed) / total,
"outstanding_over_admitted": (len(admitted) - len(completed)) / len(admitted),
"ttft_max_over_slo_max": max(ttft, default=0.0) / 6000.0,
"ttft_mean_over_slo_max": fmean(ttft) / 6000.0 if ttft else 0.0,
"tpot_max_over_slo": max(tpot, default=0.0) / 50.0,
"tpot_mean_over_slo": fmean(tpot) / 50.0 if tpot else 0.0,
"admitted_input_tokens_mean_over_limit": fmean(
float(request["raw_input_tokens"]) for request in admitted
)
/ 8192.0,
}
def telemetry_record(state: Mapping[str, Any]) -> dict[str, float]:
common = state["common"]
engine = state["engine_only"]
executed_steps = int(state["sanity"]["executed_steps"])
if executed_steps <= 0:
raise ValueError("telemetry phase contains no executed engine steps")
return {
"scheduler_steps_per_s": float(common["scheduler_steps_per_s"]),
"batch_size_mean": float(common["batch_size"]["mean"]),
"batch_size_cv": float(common["batch_size"]["cv"]),
"batch_tokens_mean": float(common["batch_tokens"]["mean"]),
"batch_tokens_cv": float(common["batch_tokens"]["cv"]),
"decode_batch_size_mean": float(common["decode_batch_size"]["mean"]),
"decode_batch_size_cv": float(common["decode_batch_size"]["cv"]),
"prefill_token_fraction": float(common["prefill_token_fraction"]),
"queue_waiting_mean": float(common["queue_waiting_mean"]),
"queue_running_mean": float(common["queue_running_mean"]),
"preemptions_per_step": float(common["preemptions"]) / executed_steps,
"kv_usage_mean": float(engine["kv_usage_mean"]),
"kv_usage_max": float(engine["kv_usage_max"]),
"kv_usage_end_minus_start": float(engine["kv_usage_end_minus_start"]),
"graph_none_share": float(engine["graph_none_share"]),
"graph_full_share": float(engine["graph_full_share"]),
"graph_padding_fraction": float(engine["graph_padding_fraction"]),
}
def load_stream(path: Path, *, expected_sha256: str) -> list[dict[str, Any]]:
if sha256_file(path) != expected_sha256:
raise ValueError(f"engine stream hash mismatch: {path}")
decoded = load_jsonl(path)
records = [row for row in decoded if "step_index" in row]
if not records:
raise ValueError(f"engine stream has no Layer-1 records: {path}")
return records
def build_dataset(
*, audit_path: Path, manifest_path: Path, run_root: Path
) -> dict[str, Any]:
audit = json.loads(audit_path.read_text(encoding="utf-8"))
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
if audit.get("schema") != "action-aware-constraint-pilot-audit-v0":
raise ValueError("unexpected action-aware audit schema")
if audit["sanity"]["red_flags"]:
raise ValueError(f"action-aware audit red flags: {audit['sanity']['red_flags']}")
configs = {str(item["id"]): item for item in manifest["configs"]}
runs = {
(str(run["config_id"]), int(run["repetition"])): run
for run in audit["runs"]
}
source_ids = {str(regime["source"]) for regime in manifest["regimes"].values()}
stream_entries = {
str(item["config_id"]): item
for item in audit["streams"]
if str(item["config_id"]) in source_ids
}
if set(stream_entries) != source_ids:
raise ValueError("audit is missing a source config engine stream")
streams = {
config_id: load_stream(
Path(item["stream"]), expected_sha256=str(item["stream_sha256"])
)
for config_id, item in stream_entries.items()
}
examples = []
request_hashes = []
for regime_name, regime in sorted(manifest["regimes"].items()):
source_id = str(regime["source"])
for repetition in sorted(int(value) for value in manifest["repetitions"]):
source_run = runs[(source_id, repetition)]
source_config = configs[source_id]
request_path = run_root / "sessions" / source_id / f"rep{repetition}" / "requests.jsonl"
requests = load_jsonl(request_path)
request_hashes.append(sha256_file(request_path))
offered_rate_per_gpu = float(
manifest["repetitions"][str(repetition)]["selection"][
"offered_req_s_per_gpu"
]
)
offered_total = offered_rate_per_gpu * int(manifest["engine"]["tp"])
source_goodput = float(source_run["outcome"]["slo_goodput_req_s"])
source_normalized = min(1.0, source_goodput / offered_total)
decision_id = f"{regime_name}-rep{repetition}"
for phase in PHASES:
cutoff_s = float(manifest["engine"]["duration_s"]) * float(phase)
outcome = prefix_outcome(
requests, cutoff_s=cutoff_s, offered_total=offered_total
)
admitted_count = sum(
float(request["arrival_s"]) <= cutoff_s for request in requests
)
start_ns = int(source_run["state"]["interval"]["start_ns"])
phase_state = summarize_engine(
streams[source_id],
start_ns=start_ns,
end_ns=start_ns + round(cutoff_s * 1e9),
request_count=admitted_count,
)
if not all(phase_state["sanity"]["invariants"].values()):
raise ValueError(
f"engine state invariant failed: {decision_id} phase {phase}"
)
telemetry = telemetry_record(phase_state)
actions = {"noop": source_id, **regime["actions"]}
for action_name, target_id in sorted(actions.items()):
target_run = runs[(str(target_id), repetition)]
target_config = configs[str(target_id)]
target_goodput = float(target_run["outcome"]["slo_goodput_req_s"])
normalized = target_goodput / offered_total
if not 0.0 <= normalized <= 1.0 + 1e-12:
raise ValueError("target normalized goodput is outside [0, 1]")
examples.append(
{
"phase": phase,
"cutoff_s": cutoff_s,
"decision_id": decision_id,
"regime": regime_name,
"repetition": repetition,
"source": {
"config_id": source_id,
"mns": int(source_config["mns"]),
"mbbt": int(source_config["mbbt"]),
"offered_rate_per_gpu": offered_rate_per_gpu,
"outcome": outcome,
"telemetry": telemetry,
},
"action": {
"id": action_name,
"target_config_id": str(target_id),
"target_mns": int(target_config["mns"]),
"target_mbbt": int(target_config["mbbt"]),
},
"target_slo_goodput_req_s": target_goodput,
"target_normalized_goodput": min(1.0, normalized),
"source_normalized_goodput": source_normalized,
"target_delta_normalized_goodput": min(1.0, normalized)
- source_normalized,
}
)
invariants = {
"expected_examples": len(examples) == len(PHASES) * 2 * 3 * 3,
"four_phases": sorted({example["phase"] for example in examples})
== sorted(PHASES),
"six_decisions": len({example["decision_id"] for example in examples}) == 6,
"three_actions_per_decision_phase": all(
sum(
item["decision_id"] == decision
and item["phase"] == phase
for item in examples
)
== 3
for decision in {item["decision_id"] for item in examples}
for phase in PHASES
),
"targets_not_all_identical": len(
{example["target_normalized_goodput"] for example in examples}
)
> 1,
"bounded_prefix_ratios": all(
0.0 <= float(value) <= 1.0
for example in examples
for key, value in example["source"]["outcome"].items()
if key
in {
"admitted_fraction",
"completed_over_admitted",
"completed_pass_rate",
"completed_fail_fraction_of_total",
"outstanding_over_admitted",
}
),
"direct_telemetry_without_binding_labels": all(
not any(token in key for token in ("exclusive", "unresolved", "both"))
for example in examples
for key in example["source"]["telemetry"]
),
"treatment_effects_bounded": all(
-1.0 <= float(example["target_delta_normalized_goodput"]) <= 1.0
for example in examples
),
}
red_flags = [name for name, passed in invariants.items() if not passed]
if red_flags:
raise RuntimeError(f"training dataset sanity failed: {red_flags}")
return {
"schema": "active-intervention-training-v0",
"status": "VALID",
"provenance": {
"audit": str(audit_path),
"audit_sha256": sha256_file(audit_path),
"manifest": str(manifest_path),
"manifest_sha256": sha256_file(manifest_path),
"run_root": str(run_root),
"source_request_sha256": sorted(set(request_hashes)),
"source_stream_sha256": sorted(
str(item["stream_sha256"]) for item in stream_entries.values()
),
},
"examples": examples,
"sanity": {"invariants": invariants, "red_flags": red_flags},
}
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--audit", type=Path, required=True)
parser.add_argument("--manifest", type=Path, required=True)
parser.add_argument("--run-root", type=Path, required=True)
parser.add_argument("--output", type=Path, required=True)
args = parser.parse_args()
dataset = build_dataset(
audit_path=args.audit,
manifest_path=args.manifest,
run_root=args.run_root,
)
atomic_json(args.output, dataset)
print(
json.dumps(
{
"status": dataset["status"],
"examples": len(dataset["examples"]),
"sanity": dataset["sanity"],
},
sort_keys=True,
)
)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,287 @@
#!/usr/bin/env python3
"""Small-data action-response model for the active intervention pilot.
The model predicts the paired normalized SLO-goodput treatment effect from a
source measurement and a full MNS/MBBT action. Telemetry features are direct,
continuous engine measurements; there is no diagnosis-to-action rule here.
"""
from __future__ import annotations
import math
from dataclasses import dataclass
from typing import Any, Iterable, Mapping, Sequence
import numpy as np
PREFIX_FEATURES = (
"normalized_slo_goodput",
"admitted_fraction",
"completed_over_admitted",
"completed_pass_rate",
"completed_fail_fraction_of_total",
"outstanding_over_admitted",
"ttft_max_over_slo_max",
"ttft_mean_over_slo_max",
"tpot_max_over_slo",
"tpot_mean_over_slo",
"admitted_input_tokens_mean_over_limit",
)
TELEMETRY_FEATURES = (
"scheduler_steps_per_s",
"batch_size_mean",
"batch_size_cv",
"batch_tokens_mean",
"batch_tokens_cv",
"decode_batch_size_mean",
"decode_batch_size_cv",
"prefill_token_fraction",
"queue_waiting_mean",
"queue_running_mean",
"preemptions_per_step",
"kv_usage_mean",
"kv_usage_max",
"kv_usage_end_minus_start",
"graph_none_share",
"graph_full_share",
"graph_padding_fraction",
)
def _finite(value: Any, name: str) -> float:
result = float(value)
if not math.isfinite(result):
raise ValueError(f"{name} must be finite")
return result
def feature_vector(
example: Mapping[str, Any], *, include_telemetry: bool
) -> tuple[list[str], np.ndarray]:
source = example["source"]
action = example["action"]
source_log_mns = math.log2(_finite(source["mns"], "source MNS"))
source_log_mbbt = math.log2(_finite(source["mbbt"], "source MBBT"))
target_log_mns = math.log2(_finite(action["target_mns"], "target MNS"))
target_log_mbbt = math.log2(_finite(action["target_mbbt"], "target MBBT"))
delta_mns = target_log_mns - source_log_mns
delta_mbbt = target_log_mbbt - source_log_mbbt
names = [
"source_log2_mns",
"source_log2_mbbt",
"target_log2_mns",
"target_log2_mbbt",
"delta_log2_mns",
"delta_log2_mbbt",
"delta_product",
"offered_rate_per_gpu",
]
values = [
source_log_mns,
source_log_mbbt,
target_log_mns,
target_log_mbbt,
delta_mns,
delta_mbbt,
delta_mns * delta_mbbt,
_finite(source["offered_rate_per_gpu"], "offered rate"),
]
for name in PREFIX_FEATURES:
names.append(f"outcome.{name}")
values.append(_finite(source["outcome"][name], name))
if include_telemetry:
for name in TELEMETRY_FEATURES:
value = _finite(source["telemetry"][name], name)
names.extend(
(
f"telemetry.{name}",
f"telemetry.{name}*delta_mns",
f"telemetry.{name}*delta_mbbt",
)
)
values.extend((value, value * delta_mns, value * delta_mbbt))
vector = np.asarray(values, dtype=np.float64)
if not np.all(np.isfinite(vector)):
raise ValueError("feature vector contains a non-finite value")
return names, vector
@dataclass(frozen=True)
class RidgeModel:
feature_names: tuple[str, ...]
mean: np.ndarray
scale: np.ndarray
weights: np.ndarray
intercept: float
regularization: float
def predict(self, values: np.ndarray) -> float:
if values.shape != self.mean.shape:
raise ValueError("ridge prediction feature shape mismatch")
normalized = (values - self.mean) / self.scale
return float(self.intercept + normalized @ self.weights)
def to_json(self) -> dict[str, Any]:
return {
"feature_names": list(self.feature_names),
"mean": self.mean.tolist(),
"scale": self.scale.tolist(),
"weights": self.weights.tolist(),
"intercept": self.intercept,
"regularization": self.regularization,
}
@classmethod
def from_json(cls, payload: Mapping[str, Any]) -> "RidgeModel":
return cls(
feature_names=tuple(str(value) for value in payload["feature_names"]),
mean=np.asarray(payload["mean"], dtype=np.float64),
scale=np.asarray(payload["scale"], dtype=np.float64),
weights=np.asarray(payload["weights"], dtype=np.float64),
intercept=float(payload["intercept"]),
regularization=float(payload["regularization"]),
)
def fit_ridge(
examples: Sequence[Mapping[str, Any]],
*,
include_telemetry: bool,
regularization: float,
) -> RidgeModel:
if not examples:
raise ValueError("ridge fit requires examples")
if regularization <= 0:
raise ValueError("ridge regularization must be positive")
encoded = [
feature_vector(example, include_telemetry=include_telemetry)
for example in examples
]
names = encoded[0][0]
if any(item[0] != names for item in encoded):
raise ValueError("feature names changed across examples")
x = np.stack([item[1] for item in encoded])
y = np.asarray(
[
_finite(example["target_delta_normalized_goodput"], "target effect")
for example in examples
],
dtype=np.float64,
)
mean = x.mean(axis=0)
scale = x.std(axis=0)
scale[scale < 1e-12] = 1.0
normalized = (x - mean) / scale
intercept = float(y.mean())
centered = y - intercept
system = normalized.T @ normalized + regularization * np.eye(x.shape[1])
weights = np.linalg.solve(system, normalized.T @ centered)
return RidgeModel(
feature_names=tuple(names),
mean=mean,
scale=scale,
weights=weights,
intercept=intercept,
regularization=regularization,
)
def fit_jackknife_ensemble(
examples: Sequence[Mapping[str, Any]],
*,
include_telemetry: bool,
regularization: float,
group_key: str = "decision_id",
) -> list[RidgeModel]:
groups = sorted({str(example[group_key]) for example in examples})
if len(groups) < 3:
raise ValueError("jackknife ensemble requires at least three groups")
models = []
for held_out in groups:
training = [
example for example in examples if str(example[group_key]) != held_out
]
models.append(
fit_ridge(
training,
include_telemetry=include_telemetry,
regularization=regularization,
)
)
return models
def ensemble_predict(
models: Sequence[RidgeModel],
example: Mapping[str, Any],
*,
include_telemetry: bool,
) -> dict[str, float]:
if not models:
raise ValueError("ensemble prediction requires models")
source = example["source"]
action = example["action"]
if (
int(action["target_mns"]) == int(source["mns"])
and int(action["target_mbbt"]) == int(source["mbbt"])
):
return {"mean": 0.0, "std": 0.0, "min": 0.0, "max": 0.0, "distinct_n": 1}
names, values = feature_vector(example, include_telemetry=include_telemetry)
if any(model.feature_names != tuple(names) for model in models):
raise ValueError("ensemble feature schema mismatch")
raw = np.asarray([model.predict(values) for model in models], dtype=np.float64)
clipped = np.clip(raw, -1.0, 1.0)
return {
"mean": float(clipped.mean()),
"std": float(clipped.std(ddof=0)),
"min": float(clipped.min()),
"max": float(clipped.max()),
"distinct_n": len(set(float(value) for value in clipped)),
}
def select_action(
models: Sequence[RidgeModel],
candidates: Sequence[Mapping[str, Any]],
*,
include_telemetry: bool,
confidence_z: float = 1.0,
minimum_margin: float = 0.02,
) -> dict[str, Any]:
if len(candidates) < 2:
raise ValueError("action selection requires at least two candidates")
rows = []
for example in candidates:
prediction = ensemble_predict(
models, example, include_telemetry=include_telemetry
)
rows.append(
{
"action_id": str(example["action"]["id"]),
"prediction": prediction,
"lower": prediction["mean"] - confidence_z * prediction["std"],
"upper": prediction["mean"] + confidence_z * prediction["std"],
}
)
rows.sort(key=lambda row: (-row["prediction"]["mean"], row["action_id"]))
best, second = rows[:2]
margin = float(best["prediction"]["mean"] - second["prediction"]["mean"])
confident = bool(
margin >= minimum_margin and best["lower"] > second["upper"]
)
return {
"selected_action": best["action_id"],
"confident": confident,
"predicted_margin": margin,
"candidates": rows,
}
def models_to_json(models: Iterable[RidgeModel]) -> list[dict[str, Any]]:
return [model.to_json() for model in models]
def models_from_json(payload: Iterable[Mapping[str, Any]]) -> list[RidgeModel]:
return [RidgeModel.from_json(item) for item in payload]

View File

@@ -0,0 +1,363 @@
#!/usr/bin/env python3
"""Freeze the unseen-trace 2x2 active intervention development surface."""
from __future__ import annotations
import argparse
import hashlib
import json
import math
import os
import sys
from pathlib import Path
from typing import Any
AITUNER_ROOT = Path(os.environ.get("AITUNER_ROOT", Path(__file__).resolve().parents[2]))
sys.path.insert(0, str(AITUNER_ROOT / "src"))
from aituner.spec import load_study_spec # noqa: E402
from aituner.trace import load_trace_requests, select_requests_for_threshold # noqa: E402
SCHEMA = "active-intervention-prospective-manifest-v0"
TP = 4
REPETITIONS = (1, 2, 3)
DURATION_S = 300.0
REPLAY_TIME_SCALE = 0.5
OFFERED_RATE_PER_GPU = 2.75
TARGET_COUNT = round(OFFERED_RATE_PER_GPU * DURATION_S * TP)
WINDOW_ID = "chat_w20260313_1000"
ENGINE_VERSION = "0.24.1.dev3+g668cfb7e2"
def atomic_json(path: Path, payload: Any) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
temporary = path.with_suffix(path.suffix + ".tmp")
temporary.write_text(
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
)
os.replace(temporary, path)
def sha256_file(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for chunk in iter(lambda: source.read(1 << 20), b""):
digest.update(chunk)
return digest.hexdigest()
def order_hash(values: list[str]) -> str:
return hashlib.sha256("\n".join(values).encode()).hexdigest()
def configs() -> list[dict[str, Any]]:
return [
{
"id": "source_mns32_mbbt4096",
"mns": 32,
"mbbt": 4096,
"repetition_order": [1, 2, 3],
},
{
"id": "mns64_mbbt4096",
"mns": 64,
"mbbt": 4096,
"repetition_order": [2, 3, 1],
},
{
"id": "mns32_mbbt8192",
"mns": 32,
"mbbt": 8192,
"repetition_order": [3, 1, 2],
},
{
"id": "joint_mns64_mbbt8192",
"mns": 64,
"mbbt": 8192,
"repetition_order": [1, 3, 2],
},
]
def partition_trace(source: Path, output_root: Path) -> dict[str, Any]:
source_sha = sha256_file(source)
output_root.mkdir(parents=True, exist_ok=True)
paths = {rep: output_root / f"rep{rep}.jsonl" for rep in REPETITIONS}
temporary = {rep: path.with_suffix(".jsonl.tmp") for rep, path in paths.items()}
handles = {rep: temporary[rep].open("w", encoding="utf-8") for rep in REPETITIONS}
counts = {rep: 0 for rep in REPETITIONS}
id_digests = {rep: hashlib.sha256() for rep in REPETITIONS}
total = 0
try:
with source.open(encoding="utf-8") as input_file:
for line_number, line in enumerate(input_file, start=1):
if not line.strip():
continue
row = json.loads(line)
original_id = str(row.get("request_id") or row.get("id") or line_number)
digest = hashlib.sha256(
f"{source_sha}:{line_number}:{original_id}".encode()
).hexdigest()
repetition = int(digest[:16], 16) % len(REPETITIONS) + 1
row["request_id"] = f"active-r{repetition}-{digest}"
handles[repetition].write(json.dumps(row, ensure_ascii=False) + "\n")
counts[repetition] += 1
total += 1
id_digests[repetition].update(row["request_id"].encode() + b"\n")
finally:
for handle in handles.values():
handle.close()
for repetition in REPETITIONS:
os.replace(temporary[repetition], paths[repetition])
partitions = {
str(rep): {
"path": str(paths[rep]),
"rows": counts[rep],
"bytes": paths[rep].stat().st_size,
"sha256": sha256_file(paths[rep]),
"request_id_order_sha256": id_digests[rep].hexdigest(),
}
for rep in REPETITIONS
}
return {
"source": str(source),
"source_sha256": source_sha,
"source_rows": total,
"partition_rule": "sha256(source_sha:line_number:original_id) modulo 3",
"partitions": partitions,
}
def materialize_study(
base_study: Path,
target: Path,
*,
repetition: int,
trace_path: Path,
windows_path: Path,
) -> None:
payload = json.loads(base_study.read_text(encoding="utf-8"))
payload["study_id"] = f"active-intervention-trace13-rep{repetition}"
payload["hardware"]["host_candidates"] = ["dash0"]
payload["engine"]["engine_version"] = ENGINE_VERSION
trace = payload["trace"]
trace.update(
{
"windows_path": str(windows_path),
"window_id": WINDOW_ID,
"trace_file_override": str(trace_path),
"completion_tokens_override": 128,
"replay_time_scale": REPLAY_TIME_SCALE,
"early_stop_max_lag_s": None,
"early_stop_max_elapsed_s": 360.0,
"restart_engine_after_early_stop": False,
"adaptive_stop": {"enabled": False},
}
)
atomic_json(target, payload)
def attainable_anchor(requests: list[Any], target_count: int) -> tuple[float, list[Any]]:
ordered = sorted(float(request.sampling_u) for request in requests)
if target_count <= 0 or target_count > len(ordered):
raise ValueError(
f"target count {target_count} is outside available range 1..{len(ordered)}"
)
candidates = []
for index in sorted({target_count - 1, min(target_count, len(ordered) - 1)}):
anchor = ordered[index]
selected = select_requests_for_threshold(requests, threshold=anchor)
candidates.append((abs(len(selected) - target_count), len(selected), anchor, selected))
_error, _count, anchor, selected = min(
candidates, key=lambda item: (item[0], item[1], item[2])
)
return anchor, selected
def selection_record(selected: list[Any]) -> dict[str, Any]:
return {
"anchor": max(float(request.sampling_u) for request in selected),
"selected_count": len(selected),
"target_count": TARGET_COUNT,
"offered_req_s": len(selected) / DURATION_S,
"offered_req_s_per_gpu": len(selected) / DURATION_S / TP,
"request_id_order_sha256": order_hash([request.row_id for request in selected]),
"arrival_order_sha256": order_hash(
[f"{request.arrival_s:.12f}" for request in selected]
),
"input_length_order_sha256": order_hash(
[str(request.prompt_tokens_hint) for request in selected]
),
}
def build(
*,
base_study: Path,
base_action_manifest: Path,
source_trace: Path,
windows_path: Path,
private_root: Path,
policy_path: Path,
) -> dict[str, Any]:
base_manifest = json.loads(base_action_manifest.read_text(encoding="utf-8"))
if base_manifest.get("status") != "PASS":
raise ValueError("base action-aware manifest did not pass")
policy = json.loads(policy_path.read_text(encoding="utf-8"))
if policy.get("schema") != "active-intervention-policy-v0":
raise ValueError("unexpected frozen policy schema")
if policy.get("sanity", {}).get("red_flags"):
raise ValueError("frozen policy contains red flags")
partition = partition_trace(source_trace, private_root / "traces")
repetitions = {}
selected_sets: list[set[str]] = []
for repetition in REPETITIONS:
trace_path = Path(partition["partitions"][str(repetition)]["path"])
study_path = private_root / "studies" / f"rep{repetition}-tp4.json"
materialize_study(
base_study,
study_path,
repetition=repetition,
trace_path=trace_path,
windows_path=windows_path,
)
study = load_study_spec(study_path)
window, requests = load_trace_requests(study, study_spec_path=study_path)
duration_s = float(window.window_end - window.window_start)
if not math.isclose(duration_s, DURATION_S, abs_tol=1e-9):
raise ValueError(f"rep{repetition}: duration {duration_s} != {DURATION_S}")
_anchor, selected = attainable_anchor(requests, TARGET_COUNT)
record = selection_record(selected)
selected_sets.append({request.row_id for request in selected})
repetitions[str(repetition)] = {
"study": str(study_path),
"study_sha256": sha256_file(study_path),
"trace": partition["partitions"][str(repetition)],
"available_filtered_requests": len(requests),
"selection": record,
}
frozen_configs = configs()
config_ids = {str(config["id"]) for config in frozen_configs}
invariants = {
"three_nonempty_trace_partitions": all(
int(item["rows"]) > 0 for item in partition["partitions"].values()
),
"partition_rows_conserved": sum(
int(item["rows"]) for item in partition["partitions"].values()
)
== int(partition["source_rows"]),
"selected_sets_disjoint": all(
not selected_sets[left] & selected_sets[right]
for left in range(len(selected_sets))
for right in range(left + 1, len(selected_sets))
),
"target_count_attained": all(
abs(int(item["selection"]["selected_count"]) - TARGET_COUNT) <= 1
for item in repetitions.values()
),
"four_unique_configs": len(config_ids) == 4,
"two_by_two_surface": {
(int(config["mns"]), int(config["mbbt"]))
for config in frozen_configs
}
== {(32, 4096), (64, 4096), (32, 8192), (64, 8192)},
"repetition_orders_are_permutations": all(
sorted(config["repetition_order"]) == list(REPETITIONS)
for config in frozen_configs
),
}
red_flags = [name for name, passed in invariants.items() if not passed]
return {
"schema": SCHEMA,
"status": "PASS" if not red_flags else "STOP",
"source": {
"window_id": WINDOW_ID,
"source_trace": str(source_trace),
"source_trace_sha256": partition["source_sha256"],
"windows_path": str(windows_path),
"base_study": str(base_study),
"base_study_sha256": sha256_file(base_study),
"base_action_manifest": str(base_action_manifest),
"base_action_manifest_sha256": sha256_file(base_action_manifest),
},
"policy": {
"path": str(policy_path),
"sha256": sha256_file(policy_path),
"status": policy["status"],
"training": policy["training"],
"measurement_policy": policy["measurement_policy"],
"launch_reason": (
"bounded unseen-trace joint-action test after a negative narrow "
"retrospective replay"
),
},
"engine": {
"tp": TP,
"duration_s": DURATION_S,
"client_timeout_s": 450.0,
"burnin_max_elapsed_s": 90.0,
"disable_slo_early_stop": True,
},
"burnin": base_manifest["burnin"],
"private": {"trace_partition": partition},
"repetitions": repetitions,
"configs": frozen_configs,
"source_config_id": "source_mns32_mbbt4096",
"actions": {
"noop": "source_mns32_mbbt4096",
"mns": "mns64_mbbt4096",
"mbbt": "mns32_mbbt8192",
"joint": "joint_mns64_mbbt8192",
},
"checkpoints": {
"fractions": [0.25, 0.50, 0.75, 1.0],
"seconds": [75.0, 150.0, 225.0, 300.0],
},
"gates": {
"acceptable_regret": 0.02,
"source_ceiling_normalized_goodput": 0.98,
"confirmation_trigger_gpu_cost_reduction": 0.10,
"contribution_gpu_cost_reduction": 0.20,
"maximum_task_regret": 0.05,
},
"budget": {
"hard_cap_h20_hours": 6.0,
"session_estimate_h20_hours": 1.3,
"safety_h20_hours": 0.3,
"expected_h20_hours": [4.6, 5.5],
"expected_wall_minutes": [75, 100],
},
"sanity": {"invariants": invariants, "red_flags": red_flags},
}
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--base-study", type=Path, required=True)
parser.add_argument("--base-action-manifest", type=Path, required=True)
parser.add_argument("--source-trace", type=Path, required=True)
parser.add_argument("--windows-path", type=Path, required=True)
parser.add_argument("--private-root", type=Path, required=True)
parser.add_argument("--policy", type=Path, required=True)
parser.add_argument("--output", type=Path, required=True)
args = parser.parse_args()
payload = build(
base_study=args.base_study,
base_action_manifest=args.base_action_manifest,
source_trace=args.source_trace,
windows_path=args.windows_path,
private_root=args.private_root,
policy_path=args.policy,
)
atomic_json(args.output, payload)
print(json.dumps({"status": payload["status"], "sanity": payload["sanity"]}))
if payload["status"] != "PASS":
raise SystemExit("prospective manifest preflight failed")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,198 @@
#!/usr/bin/env python3
"""Run source first, select the next intervention, then annotate the 2x2 surface."""
from __future__ import annotations
import argparse
import json
import sys
from pathlib import Path
from typing import Any, Mapping
HERE = Path(__file__).resolve().parent
ACTION_DIR = HERE.parent / "action-aware-v0"
sys.path.insert(0, str(ACTION_DIR))
sys.path.insert(0, str(HERE))
import pilot_controller as action_controller # noqa: E402
import prospective_decision # noqa: E402
SCHEMA = "active-intervention-prospective-state-v0"
def validate_inputs(args: argparse.Namespace, manifest: Mapping[str, Any]) -> None:
if manifest.get("schema") != "active-intervention-prospective-manifest-v0":
raise RuntimeError("unexpected active intervention manifest schema")
if manifest.get("status") != "PASS" or manifest["sanity"]["red_flags"]:
raise RuntimeError("active intervention manifest did not pass preflight")
required = {
"manifest": args.manifest,
"policy": args.policy,
"aituner_root": args.aituner_root,
"vllm_source": args.vllm_source,
"venv_python": args.venv / "bin/python",
"venv_vllm": args.venv / "bin/vllm",
"model": args.model,
"client": args.client,
"burnin_study": Path(manifest["burnin"]["study"]),
}
for repetition, item in manifest["repetitions"].items():
required[f"rep{repetition}_study"] = Path(item["study"])
required[f"rep{repetition}_trace"] = Path(item["trace"]["path"])
missing = {name: str(path) for name, path in required.items() if not path.exists()}
if missing:
raise RuntimeError(f"active intervention input paths missing: {missing}")
if prospective_decision.sha256_file(args.policy) != manifest["policy"]["sha256"]:
raise RuntimeError("active intervention policy hash mismatch")
def dry_run(args: argparse.Namespace, manifest: Mapping[str, Any]) -> dict[str, Any]:
plan = action_controller.dry_run_plan(args, manifest)
return {
"schema": "active-intervention-prospective-dry-run-v0",
"status": "PASS",
"manifest": str(args.manifest),
"policy": str(args.policy),
"source_first": manifest["source_config_id"],
"post_source_order": "selected by telemetry policy; all remaining cells then annotated",
"candidate_actions": manifest["actions"],
"projected_h20_hours": plan["projected_h20_hours"],
"hard_cap_h20_hours": plan["hard_cap_h20_hours"],
"sessions": plan["sessions"],
}
def load_or_build_decision(
*, args: argparse.Namespace, run_root: Path
) -> dict[str, Any]:
path = run_root / "active-decision.json"
if path.exists():
decision = json.loads(path.read_text(encoding="utf-8"))
if decision.get("manifest_sha256") != prospective_decision.sha256_file(
args.manifest
):
raise RuntimeError("existing active decision has a different manifest")
if decision.get("policy_sha256") != prospective_decision.sha256_file(args.policy):
raise RuntimeError("existing active decision has a different policy")
return decision
decision = prospective_decision.build_decision(
manifest_path=args.manifest,
policy_path=args.policy,
run_root=run_root,
)
prospective_decision.atomic_json(path, decision)
return decision
def parser() -> argparse.ArgumentParser:
result = argparse.ArgumentParser()
result.add_argument("--manifest", type=Path, required=True)
result.add_argument("--policy", type=Path, required=True)
result.add_argument("--run-root", type=Path, required=True)
result.add_argument("--aituner-root", type=Path, required=True)
result.add_argument("--vllm-source", type=Path, required=True)
result.add_argument("--venv", type=Path, required=True)
result.add_argument("--model", type=Path, required=True)
result.add_argument("--client", type=Path, required=True)
result.add_argument("--dry-run", action="store_true")
return result
def main() -> None:
args = parser().parse_args()
manifest = json.loads(args.manifest.read_text(encoding="utf-8"))
validate_inputs(args, manifest)
action_controller.configure(args, manifest)
action_controller.base.MARKER = "active-intervention-prospective-v0"
if args.dry_run:
print(json.dumps(dry_run(args, manifest), indent=2, sort_keys=True))
return
args.run_root.mkdir(parents=True, exist_ok=True)
copied_manifest = args.run_root / "prospective-manifest.json"
if not copied_manifest.exists():
action_controller.atomic_json(copied_manifest, manifest)
state_path = args.run_root / "controller-state.json"
state = action_controller.load_state(
state_path, float(manifest["budget"]["hard_cap_h20_hours"])
)
state["schema"] = SCHEMA
state["status"] = "running"
action_controller.atomic_json(state_path, state)
configs = {str(item["id"]): dict(item) for item in manifest["configs"]}
config_indexes = {
str(item["id"]): index for index, item in enumerate(manifest["configs"])
}
source_id = str(manifest["source_config_id"])
action_controller.execute_session(
args=args,
manifest=manifest,
config=configs[source_id],
index=config_indexes[source_id],
state=state,
state_path=state_path,
)
decision = load_or_build_decision(args=args, run_root=args.run_root)
state["active_decision"] = {
"path": str(args.run_root / "active-decision.json"),
"status": decision["status"],
"outcome_only": {
key: decision["decisions"]["outcome_only"][key]
for key in ("selected_cutoff_s", "decision_kind", "selected_action")
},
"telemetry": {
key: decision["decisions"]["telemetry"][key]
for key in ("selected_cutoff_s", "decision_kind", "selected_action")
},
}
action_controller.atomic_json(state_path, state)
if decision["status"] != "SELECTED":
state["status"] = decision["status"].lower()
state["completed_at"] = action_controller.time.time()
action_controller.atomic_json(state_path, state)
action_controller.wait_all_idle()
print(json.dumps({"status": state["status"], "decision": decision["status"]}))
return
action_order = decision["decisions"]["telemetry"]["intervention_order"]
execution_order = [source_id]
for action_id in action_order:
target_id = str(manifest["actions"][action_id])
if target_id not in execution_order:
execution_order.append(target_id)
for config_id in configs:
if config_id not in execution_order:
execution_order.append(config_id)
state["execution_order"] = execution_order
action_controller.atomic_json(state_path, state)
for config_id in execution_order[1:]:
action_controller.execute_session(
args=args,
manifest=manifest,
config=configs[config_id],
index=config_indexes[config_id],
state=state,
state_path=state_path,
)
state["status"] = "complete"
state["completed_at"] = action_controller.time.time()
action_controller.atomic_json(state_path, state)
action_controller.wait_all_idle()
print(
json.dumps(
{
"status": state["status"],
"completed_sessions": state["completed_sessions"],
"gpu_hours_total": state["gpu_hours_total"],
"execution_order": execution_order,
},
sort_keys=True,
)
)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,441 @@
#!/usr/bin/env python3
"""Choose measurement horizon and next intervention from a completed source run."""
from __future__ import annotations
import argparse
import hashlib
import importlib.util
import json
import math
import os
import statistics
import sys
from pathlib import Path
from typing import Any, Mapping, Sequence
import numpy as np
HERE = Path(__file__).resolve().parent
COMMON_STATE = HERE.parent / "telemetry-residual"
sys.path.insert(0, str(COMMON_STATE))
from common_state import summarize_engine # noqa: E402
SCHEMA = "active-intervention-prospective-decision-v0"
def load_module(name: str, path: Path):
spec = importlib.util.spec_from_file_location(name, path)
module = importlib.util.module_from_spec(spec)
assert spec.loader is not None
sys.modules[spec.name] = module
spec.loader.exec_module(module)
return module
MODEL = load_module("active_intervention_prospective_model", HERE / "model.py")
EXTRACT = load_module(
"active_intervention_prospective_extract", HERE / "extract_training.py"
)
def sha256_file(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for chunk in iter(lambda: source.read(1 << 20), b""):
digest.update(chunk)
return digest.hexdigest()
def atomic_json(path: Path, payload: Any) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
temporary = path.with_suffix(path.suffix + ".tmp")
temporary.write_text(
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
)
os.replace(temporary, path)
def numeric(values: Sequence[float]) -> dict[str, Any]:
finite = [float(value) for value in values]
if not finite or any(not math.isfinite(value) for value in finite):
raise ValueError("numeric summary requires finite values")
return {
"n": len(finite),
"min": min(finite),
"max": max(finite),
"distinct_n": len(set(finite)),
}
def load_engine_records(source_root: Path) -> tuple[list[dict[str, Any]], Path]:
streams = sorted((source_root / "opprof").glob("*.jsonl"))
if len(streams) != 1:
raise ValueError(f"expected one source engine stream, found {len(streams)}")
records = [
row for row in EXTRACT.load_jsonl(streams[0]) if "step_index" in row
]
if not records:
raise ValueError("source engine stream has no Layer-1 records")
return records, streams[0]
def candidate_example(
*,
source_config: Mapping[str, Any],
target_config: Mapping[str, Any],
action_id: str,
offered_rate_per_gpu: float,
outcome: Mapping[str, float],
telemetry: Mapping[str, float],
) -> dict[str, Any]:
return {
"source": {
"mns": int(source_config["mns"]),
"mbbt": int(source_config["mbbt"]),
"offered_rate_per_gpu": float(offered_rate_per_gpu),
"outcome": dict(outcome),
"telemetry": dict(telemetry),
},
"action": {
"id": action_id,
"target_mns": int(target_config["mns"]),
"target_mbbt": int(target_config["mbbt"]),
},
}
def aggregate_checkpoint(
*,
models: Sequence[Any],
examples_by_action: Mapping[str, Sequence[Mapping[str, Any]]],
include_telemetry: bool,
confidence_z: float,
minimum_margin: float,
) -> dict[str, Any]:
rows = []
for action_id, examples in sorted(examples_by_action.items()):
raw = []
for example in examples:
source = example["source"]
action = example["action"]
noop = (
int(source["mns"]) == int(action["target_mns"])
and int(source["mbbt"]) == int(action["target_mbbt"])
)
if noop:
raw.extend(0.0 for _model in models)
continue
names, values = MODEL.feature_vector(
example, include_telemetry=include_telemetry
)
if any(model.feature_names != tuple(names) for model in models):
raise ValueError("prospective feature schema does not match frozen model")
raw.extend(model.predict(values) for model in models)
clipped = np.clip(np.asarray(raw, dtype=np.float64), -1.0, 1.0)
prediction = {
"mean": float(clipped.mean()),
"std": float(clipped.std(ddof=0)),
"min": float(clipped.min()),
"max": float(clipped.max()),
"distinct_n": len(set(float(value) for value in clipped)),
"sample_n": int(clipped.size),
}
rows.append(
{
"action_id": action_id,
"prediction": prediction,
"lower": prediction["mean"] - confidence_z * prediction["std"],
"upper": prediction["mean"] + confidence_z * prediction["std"],
}
)
rows.sort(key=lambda row: (-row["prediction"]["mean"], row["action_id"]))
best, second = rows[:2]
margin = float(best["prediction"]["mean"] - second["prediction"]["mean"])
confident = bool(
margin >= minimum_margin and best["lower"] > second["upper"]
)
return {
"selected_action": best["action_id"],
"confident": confident,
"predicted_margin": margin,
"candidates": rows,
}
def apply_measurement_and_acquisition(checkpoints: list[dict[str, Any]]) -> dict[str, Any]:
selected = checkpoints[-1]
stop_reason = "full_measurement_fallback"
for previous, current in zip(checkpoints, checkpoints[1:], strict=False):
if (
previous["confident"]
and current["confident"]
and previous["selected_action"] == current["selected_action"]
):
selected = current
stop_reason = "two_consecutive_confident_checkpoints"
break
candidates = selected["candidates"]
mean_best = candidates[0]
non_noop = [row for row in candidates if row["action_id"] != "noop"]
if selected["confident"]:
chosen = mean_best
decision_kind = "exploit"
else:
positive_ucb = [row for row in non_noop if float(row["upper"]) > 0.0]
if positive_ucb:
chosen = max(
positive_ucb,
key=lambda row: (float(row["upper"]), row["action_id"]),
)
decision_kind = "diagnostic_ucb"
else:
chosen = next(row for row in candidates if row["action_id"] == "noop")
decision_kind = "abstain_no_positive_ucb"
remaining = [row for row in candidates if row["action_id"] != chosen["action_id"]]
remaining.sort(key=lambda row: (-float(row["upper"]), row["action_id"]))
order = [chosen["action_id"], *(row["action_id"] for row in remaining)]
return {
"selected_phase": selected["phase"],
"selected_cutoff_s": selected["cutoff_s"],
"measurement_stop_reason": stop_reason,
"decision_kind": decision_kind,
"selected_action": chosen["action_id"],
"intervention_order": order,
"selected_checkpoint": selected,
"checkpoints": checkpoints,
}
def build_decision(
*, manifest_path: Path, policy_path: Path, run_root: Path
) -> dict[str, Any]:
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
policy = json.loads(policy_path.read_text(encoding="utf-8"))
if manifest.get("schema") != "active-intervention-prospective-manifest-v0":
raise ValueError("unexpected prospective manifest schema")
if policy.get("schema") != "active-intervention-policy-v0":
raise ValueError("unexpected frozen policy schema")
if sha256_file(policy_path) != manifest["policy"]["sha256"]:
raise ValueError("frozen policy hash changed after manifest preparation")
configs = {str(item["id"]): item for item in manifest["configs"]}
source_id = str(manifest["source_config_id"])
source_config = configs[source_id]
source_root = run_root / "sessions" / source_id
engine_records, stream_path = load_engine_records(source_root)
phases = [f"{fraction:.2f}" for fraction in manifest["checkpoints"]["fractions"]]
confidence_z = float(policy["measurement_policy"]["confidence_z"])
minimum_margin = float(policy["measurement_policy"]["minimum_margin"])
examples: dict[str, dict[str, dict[str, Mapping[str, Any]]]] = {}
source_measurements: dict[str, dict[str, Any]] = {}
source_normalized = []
telemetry_values = []
for repetition in sorted(int(key) for key in manifest["repetitions"]):
item = manifest["repetitions"][str(repetition)]
result_root = source_root / f"rep{repetition}"
result = json.loads((result_root / "result.json").read_text(encoding="utf-8"))
if result["selection"]["request_id_order_sha256"] != item["selection"][
"request_id_order_sha256"
]:
raise ValueError(f"source request hash mismatch: rep{repetition}")
requests = EXTRACT.load_jsonl(result_root / "requests.jsonl")
offered_rate = float(item["selection"]["offered_req_s_per_gpu"])
offered_total = offered_rate * int(manifest["engine"]["tp"])
source_normalized.append(
float(result["slo_pass_count"])
/ float(manifest["engine"]["duration_s"])
/ offered_total
)
start_ns = int(result["interval"]["start_mono_ns"])
examples[str(repetition)] = {}
source_measurements[str(repetition)] = {
"result": str(result_root / "result.json"),
"result_sha256": sha256_file(result_root / "result.json"),
"request_sha256": sha256_file(result_root / "requests.jsonl"),
"phases": {},
}
for phase, cutoff_s in zip(
phases, manifest["checkpoints"]["seconds"], strict=True
):
outcome = EXTRACT.prefix_outcome(
requests, cutoff_s=float(cutoff_s), offered_total=offered_total
)
admitted_count = sum(
float(request["arrival_s"]) <= float(cutoff_s)
for request in requests
)
state = summarize_engine(
engine_records,
start_ns=start_ns,
end_ns=start_ns + round(float(cutoff_s) * 1e9),
request_count=admitted_count,
)
if not all(state["sanity"]["invariants"].values()):
raise ValueError(
f"source engine state invariant failed: rep{repetition} {phase}"
)
telemetry = EXTRACT.telemetry_record(state)
telemetry_values.extend(float(value) for value in telemetry.values())
source_measurements[str(repetition)]["phases"][phase] = {
"cutoff_s": float(cutoff_s),
"outcome": outcome,
"telemetry": telemetry,
"engine_sanity": state["sanity"],
}
examples[str(repetition)][phase] = {
action_id: candidate_example(
source_config=source_config,
target_config=configs[str(target_id)],
action_id=action_id,
offered_rate_per_gpu=offered_rate,
outcome=outcome,
telemetry=telemetry,
)
for action_id, target_id in manifest["actions"].items()
}
decisions = {}
for mode, include_telemetry in (("outcome_only", False), ("telemetry", True)):
checkpoints = []
for phase, cutoff_s in zip(
phases, manifest["checkpoints"]["seconds"], strict=True
):
models = MODEL.models_from_json(policy["phases"][phase][mode]["models"])
examples_by_action = {
action_id: [
examples[str(repetition)][phase][action_id]
for repetition in sorted(int(key) for key in manifest["repetitions"])
]
for action_id in manifest["actions"]
}
checkpoint = aggregate_checkpoint(
models=models,
examples_by_action=examples_by_action,
include_telemetry=include_telemetry,
confidence_z=confidence_z,
minimum_margin=minimum_margin,
)
checkpoints.append(
{"phase": phase, "cutoff_s": float(cutoff_s), **checkpoint}
)
decisions[mode] = apply_measurement_and_acquisition(checkpoints)
ceiling = float(manifest["gates"]["source_ceiling_normalized_goodput"])
source_median = float(statistics.median(source_normalized))
status = "STOP_SOURCE_CEILING" if source_median >= ceiling else "SELECTED"
phase_admission_monotonic = all(
all(
left <= right + 1e-12
for left, right in zip(values, values[1:], strict=False)
)
for repetition in source_measurements.values()
for values in (
[
float(repetition["phases"][phase]["outcome"]["admitted_fraction"])
for phase in phases
],
)
)
telemetry_ratio_keys = {
"prefill_token_fraction",
"kv_usage_mean",
"kv_usage_max",
"graph_none_share",
"graph_full_share",
"graph_padding_fraction",
}
telemetry_records = [
measurement["telemetry"]
for repetition in source_measurements.values()
for measurement in repetition["phases"].values()
]
invariants = {
"three_source_repetitions": len(source_normalized) == 3,
"source_goodput_nonnegative": all(value >= 0.0 for value in source_normalized),
"source_goodput_bounded": all(
value <= 1.0 + 1e-12 for value in source_normalized
),
"four_actions": set(manifest["actions"]) == {"noop", "mns", "mbbt", "joint"},
"four_checkpoints": len(phases) == 4,
"finite_telemetry": all(math.isfinite(value) for value in telemetry_values),
"nonnegative_telemetry": all(
float(value) >= 0.0
for record in telemetry_records
for key, value in record.items()
if key != "kv_usage_end_minus_start"
),
"telemetry_ratios_bounded": all(
0.0 <= float(record[key]) <= 1.0 + 1e-12
for record in telemetry_records
for key in telemetry_ratio_keys
),
"telemetry_not_all_identical": len(set(telemetry_values)) > 1,
"phase_admission_monotonic": phase_admission_monotonic,
"orders_are_permutations": all(
set(decisions[mode]["intervention_order"]) == set(manifest["actions"])
for mode in decisions
),
}
red_flags = [name for name, passed in invariants.items() if not passed]
if red_flags:
status = "STOP_SANITY"
return {
"schema": SCHEMA,
"status": status,
"manifest": str(manifest_path),
"manifest_sha256": sha256_file(manifest_path),
"policy": str(policy_path),
"policy_sha256": sha256_file(policy_path),
"source_stream": str(stream_path),
"source_stream_sha256": sha256_file(stream_path),
"source_measurements": source_measurements,
"source_normalized_goodput": {
"values": source_normalized,
"median": source_median,
**numeric(source_normalized),
},
"decisions": decisions,
"sanity": {
"invariants": invariants,
"red_flags": red_flags,
"telemetry_values": numeric(telemetry_values),
},
}
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--manifest", type=Path, required=True)
parser.add_argument("--policy", type=Path, required=True)
parser.add_argument("--run-root", type=Path, required=True)
parser.add_argument("--output", type=Path, required=True)
args = parser.parse_args()
decision = build_decision(
manifest_path=args.manifest, policy_path=args.policy, run_root=args.run_root
)
atomic_json(args.output, decision)
print(
json.dumps(
{
"status": decision["status"],
"source_normalized_goodput": decision["source_normalized_goodput"],
"outcome_only": {
key: decision["decisions"]["outcome_only"][key]
for key in ("selected_cutoff_s", "decision_kind", "selected_action")
},
"telemetry": {
key: decision["decisions"]["telemetry"][key]
for key in ("selected_cutoff_s", "decision_kind", "selected_action")
},
},
sort_keys=True,
)
)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,91 @@
#!/usr/bin/env python3
from __future__ import annotations
import importlib.util
import sys
from pathlib import Path
HERE = Path(__file__).resolve().parent
def load_model():
spec = importlib.util.spec_from_file_location(
"active_intervention_model", HERE / "model.py"
)
module = importlib.util.module_from_spec(spec)
assert spec.loader is not None
sys.modules[spec.name] = module
spec.loader.exec_module(module)
return module
def example(model, decision: str, action: str, pressure: float, target: float):
outcome = {
name: 0.5 for name in model.PREFIX_FEATURES
}
telemetry = {name: 0.0 for name in model.TELEMETRY_FEATURES}
telemetry["queue_waiting_mean"] = pressure
telemetry["batch_size_mean"] = pressure
return {
"decision_id": decision,
"source": {
"mns": 16,
"mbbt": 8192,
"offered_rate_per_gpu": 2.0,
"outcome": outcome,
"telemetry": telemetry,
},
"action": {
"id": action,
"target_mns": 64 if action == "mns" else 16,
"target_mbbt": 8192 if action == "mns" else 16384,
},
"target_normalized_goodput": target,
"target_delta_normalized_goodput": target - 0.5,
}
def main() -> None:
model = load_model()
examples = []
for index, pressure in enumerate((0.2, 0.5, 0.8), 1):
examples.extend(
(
example(model, f"d{index}", "mns", pressure, 0.5 + pressure / 2),
example(model, f"d{index}", "mbbt", pressure, 0.6 - pressure / 4),
)
)
fitted = model.fit_ridge(
examples, include_telemetry=True, regularization=1.0
)
encoded = fitted.to_json()
restored = model.RidgeModel.from_json(encoded)
names, values = model.feature_vector(examples[-2], include_telemetry=True)
assert tuple(names) == restored.feature_names
assert abs(fitted.predict(values) - restored.predict(values)) < 1e-12
ensemble = model.fit_jackknife_ensemble(
examples, include_telemetry=True, regularization=1.0
)
decision = model.select_action(
ensemble, examples[-2:], include_telemetry=True, minimum_margin=0.0
)
assert decision["selected_action"] == "mns"
assert all(-1.0 <= row["prediction"]["mean"] <= 1.0 for row in decision["candidates"])
noop = example(model, "noop", "noop", 0.8, 0.5)
noop["action"]["target_mbbt"] = 8192
prediction = model.ensemble_predict(
ensemble, noop, include_telemetry=True
)
assert prediction == {
"mean": 0.0,
"std": 0.0,
"min": 0.0,
"max": 0.0,
"distinct_n": 1,
}
print("active intervention model: PASS")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,199 @@
#!/usr/bin/env python3
from __future__ import annotations
import importlib.util
import json
import sys
import tempfile
from pathlib import Path
HERE = Path(__file__).resolve().parent
def load(name: str, path: Path):
spec = importlib.util.spec_from_file_location(name, path)
module = importlib.util.module_from_spec(spec)
assert spec.loader is not None
sys.modules[spec.name] = module
spec.loader.exec_module(module)
return module
def write_json(path: Path, payload) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload) + "\n", encoding="utf-8")
def write_jsonl(path: Path, rows) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(
"".join(json.dumps(row) + "\n" for row in rows), encoding="utf-8"
)
def engine_record(index: int, timestamp_ns: int) -> dict:
alternate = index % 2
return {
"step_index": index,
"submit_mono_ns": timestamp_ns,
"model_executed": True,
"scheduled_requests": 1 + alternate,
"decode_batch_size": alternate,
"prefill_tokens": 8 + alternate,
"decode_tokens": alternate,
"preemptions": 0,
"queues": {"waiting": alternate, "running": 1 + alternate},
"kv": {"usage": 0.1 + 0.01 * alternate},
"cudagraph": {
"runtime_mode": "FULL" if alternate else "NONE",
"bucket_tokens": 16,
"padding_tokens": alternate,
},
"dropped_records_before": 0,
}
def main() -> None:
extractor = load("active_intervention_extract_test", HERE / "extract_training.py")
trainer = load("active_intervention_train_test", HERE / "train_policy.py")
with tempfile.TemporaryDirectory() as temporary:
root = Path(temporary)
run_root = root / "runs"
configs = [
{"id": "a_base", "mns": 16, "mbbt": 8192},
{"id": "a_mns", "mns": 64, "mbbt": 8192},
{"id": "a_mbbt", "mns": 16, "mbbt": 16384},
{"id": "b_base", "mns": 64, "mbbt": 2048},
{"id": "b_mns", "mns": 128, "mbbt": 2048},
{"id": "b_mbbt", "mns": 64, "mbbt": 8192},
]
manifest = {
"engine": {"duration_s": 300.0, "tp": 4},
"configs": configs,
"repetitions": {
str(rep): {"selection": {"offered_req_s_per_gpu": 0.01}}
for rep in (1, 2, 3)
},
"regimes": {
"A": {
"source": "a_base",
"actions": {"mns": "a_mns", "mbbt": "a_mbbt"},
},
"B": {
"source": "b_base",
"actions": {"mns": "b_mns", "mbbt": "b_mbbt"},
},
},
}
manifest_path = root / "manifest.json"
write_json(manifest_path, manifest)
streams = []
source_starts: dict[tuple[str, int], int] = {}
for source_index, source_id in enumerate(("a_base", "b_base")):
rows = []
index = 0
for repetition in (1, 2, 3):
start_ns = int((source_index * 2000 + repetition * 400) * 1e9)
source_starts[(source_id, repetition)] = start_ns
for second in (1, 30, 76, 105, 151, 180, 226, 255):
rows.append(engine_record(index, start_ns + int(second * 1e9)))
index += 1
stream_path = root / f"{source_id}-stream.jsonl"
write_jsonl(stream_path, rows)
streams.append(
{
"config_id": source_id,
"stream": str(stream_path),
"stream_sha256": extractor.sha256_file(stream_path),
}
)
request_rows = [
{
"request_id": f"r{index}",
"arrival_s": arrival,
"completed_elapsed_s": arrival + 10,
"slo_pass": index != 3,
"ttft_ms": 1000 + index * 100,
"tpot_ms": 20 + index,
"raw_input_tokens": 1000 + index * 100,
}
for index, arrival in enumerate((5.0, 80.0, 155.0, 230.0), 1)
]
for source_id in ("a_base", "b_base"):
for repetition in (1, 2, 3):
write_jsonl(
run_root
/ "sessions"
/ source_id
/ f"rep{repetition}"
/ "requests.jsonl",
request_rows,
)
goodput = {
"a_base": 0.020,
"a_mns": 0.036,
"a_mbbt": 0.028,
"b_base": 0.032,
"b_mns": 0.030,
"b_mbbt": 0.038,
}
runs = []
for config in configs:
for repetition in (1, 2, 3):
item = {
"config_id": config["id"],
"repetition": repetition,
"outcome": {
"slo_goodput_req_s": goodput[config["id"]]
+ repetition * 0.0001
},
}
if config["id"] in ("a_base", "b_base"):
start_ns = source_starts[(config["id"], repetition)]
item["state"] = {
"interval": {
"start_ns": start_ns,
"end_ns": start_ns + int(300 * 1e9),
}
}
runs.append(item)
audit = {
"schema": "action-aware-constraint-pilot-audit-v0",
"sanity": {"red_flags": []},
"streams": streams,
"runs": runs,
}
audit_path = root / "audit.json"
write_json(audit_path, audit)
dataset = extractor.build_dataset(
audit_path=audit_path, manifest_path=manifest_path, run_root=run_root
)
assert dataset["status"] == "VALID"
assert len(dataset["examples"]) == 72
assert not dataset["sanity"]["red_flags"]
assert all(
"exclusive" not in feature
for example in dataset["examples"]
for feature in example["source"]["telemetry"]
)
dataset_path = root / "dataset.json"
write_json(dataset_path, dataset)
policy = trainer.build_policy(dataset_path)
assert policy["status"] in {
"RETROSPECTIVE_GPU_COST_SIGNAL",
"NO_RETROSPECTIVE_GPU_COST_SIGNAL",
}
assert policy["training"]["acceptable_regret"] == 0.02
assert policy["sequential_replay"]["outcome_only"]["decision_n"] == 6
assert policy["sequential_replay"]["telemetry"]["decision_n"] == 6
assert not policy["sanity"]["red_flags"]
print("active intervention pipeline: PASS")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,190 @@
#!/usr/bin/env python3
from __future__ import annotations
import importlib.util
import json
import sys
import tempfile
from pathlib import Path
HERE = Path(__file__).resolve().parent
def load(name: str, path: Path):
spec = importlib.util.spec_from_file_location(name, path)
module = importlib.util.module_from_spec(spec)
assert spec.loader is not None
sys.modules[spec.name] = module
spec.loader.exec_module(module)
return module
def write_json(path: Path, payload) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(json.dumps(payload) + "\n", encoding="utf-8")
def main() -> None:
prepare = load("active_intervention_prepare_test", HERE / "prepare_prospective.py")
decision_module = load(
"active_intervention_decision_test", HERE / "prospective_decision.py"
)
analyzer = load("active_intervention_audit_test", HERE / "analyze_prospective.py")
with tempfile.TemporaryDirectory() as temporary:
root = Path(temporary)
source = root / "source.jsonl"
source.write_text(
"".join(
json.dumps(
{
"request_id": f"request-{index}",
"timestamp": float(index),
"sampling_u": index / 100.0,
}
)
+ "\n"
for index in range(60)
),
encoding="utf-8",
)
partition = prepare.partition_trace(source, root / "partitions")
assert sum(item["rows"] for item in partition["partitions"].values()) == 60
ids = []
for item in partition["partitions"].values():
assert item["rows"] > 0
ids.extend(
json.loads(line)["request_id"]
for line in Path(item["path"]).read_text(encoding="utf-8").splitlines()
)
assert len(ids) == len(set(ids)) == 60
checkpoints = [
{
"phase": "0.25",
"cutoff_s": 75.0,
"selected_action": "joint",
"confident": True,
"candidates": [
{"action_id": "joint", "upper": 0.5, "prediction": {"mean": 0.4}},
{"action_id": "mns", "upper": 0.2, "prediction": {"mean": 0.1}},
{"action_id": "mbbt", "upper": 0.1, "prediction": {"mean": 0.05}},
{"action_id": "noop", "upper": 0.0, "prediction": {"mean": 0.0}},
],
},
{
"phase": "0.50",
"cutoff_s": 150.0,
"selected_action": "joint",
"confident": True,
"candidates": [
{"action_id": "joint", "upper": 0.45, "prediction": {"mean": 0.4}},
{"action_id": "mns", "upper": 0.2, "prediction": {"mean": 0.1}},
{"action_id": "mbbt", "upper": 0.1, "prediction": {"mean": 0.05}},
{"action_id": "noop", "upper": 0.0, "prediction": {"mean": 0.0}},
],
},
]
selected = decision_module.apply_measurement_and_acquisition(checkpoints)
assert selected["selected_cutoff_s"] == 150.0
assert selected["selected_action"] == "joint"
configs = prepare.configs()
repetitions = {
str(rep): {
"selection": {
"offered_req_s_per_gpu": 0.25,
"request_id_order_sha256": f"hash-{rep}",
}
}
for rep in (1, 2, 3)
}
manifest = {
"schema": "active-intervention-prospective-manifest-v0",
"engine": {"duration_s": 300.0, "tp": 4},
"repetitions": repetitions,
"configs": configs,
"source_config_id": "source_mns32_mbbt4096",
"actions": {
"noop": "source_mns32_mbbt4096",
"mns": "mns64_mbbt4096",
"mbbt": "mns32_mbbt8192",
"joint": "joint_mns64_mbbt8192",
},
"gates": {
"acceptable_regret": 0.02,
"confirmation_trigger_gpu_cost_reduction": 0.10,
"contribution_gpu_cost_reduction": 0.20,
},
}
manifest_path = root / "manifest.json"
write_json(manifest_path, manifest)
run_root = root / "run"
scores = {
"source_mns32_mbbt4096": 0.5,
"mns64_mbbt4096": 0.8,
"mns32_mbbt8192": 0.7,
"joint_mns64_mbbt8192": 1.0,
}
sessions = {}
for config in configs:
config_id = config["id"]
sessions[config_id] = {"status": "complete", "gpu_hours": 1.2}
for repetition in (1, 2, 3):
result = {
"selection": {
"request_id_order_sha256": f"hash-{repetition}"
},
"slo_pass_count": round(scores[config_id] * 300),
"pass_rate": scores[config_id],
"interval": {"elapsed_s": 300.0},
}
write_json(
run_root
/ "sessions"
/ config_id
/ f"rep{repetition}"
/ "result.json",
result,
)
state = {
"status": "complete",
"gpu_hours_total": 4.8,
"sessions": sessions,
}
write_json(run_root / "controller-state.json", state)
mode_base = {
"selected_cutoff_s": 300.0,
"selected_action": "mns",
"decision_kind": "exploit",
"intervention_order": ["mns", "mbbt", "joint", "noop"],
}
mode_telemetry = {
"selected_cutoff_s": 150.0,
"selected_action": "joint",
"decision_kind": "exploit",
"intervention_order": ["joint", "mns", "mbbt", "noop"],
}
decision = {
"schema": "active-intervention-prospective-decision-v0",
"manifest_sha256": analyzer.sha256_file(manifest_path),
"decisions": {
"outcome_only": mode_base,
"telemetry": mode_telemetry,
},
}
decision_path = root / "decision.json"
write_json(decision_path, decision)
audit = analyzer.build_audit(
manifest_path=manifest_path,
decision_path=decision_path,
run_root=run_root,
)
assert audit["status"] == "TRIGGER_ACTUAL_EARLY_STOP_CONFIRMATION"
assert audit["comparison"]["telemetry_gpu_cost_reduction_fraction"] > 0.10
assert not audit["sanity"]["red_flags"]
print("active intervention prospective pipeline: PASS")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,151 @@
{
"schema": "active-intervention-trace13-result-summary-v0",
"status": "STOP_NO_PROSPECTIVE_GPU_COST_SIGNAL",
"provenance": {
"aituner_commit": "39b767e384fc49da53b99ad06a3e2ca1b6ac37d6",
"vllm_commit": "4b253fd8619764b6971a7f2e3a3aa7545f6ace05",
"manifest_sha256": "3bd25ae0ca040729a6351635f14447b3c789d4d86fb0fe4d65940735ad225a78",
"policy_sha256": "4f096d3a5f8c38771e956dfd576dd6cd5d5691286ab258028dbddbe07b20078c",
"decision_sha256": "86b7089151c480e18b5ae6ed65c4e4a3e11159dd0c16d5851312d4d9196d5ca2",
"controller_state_sha256": "d10d3cbe5ce20eacc1392f52e79d16ae23b5e15301123d5e85e48d53ae676cda",
"audit_sha256": "bd95b45e5d4cb93b5ad2f722b7dc96553d64eb8a5d5aae3a82f29c2d015fe3f6",
"remote_root": "/home/admin/cpfs/wjh/active-intervention-prospective-20260715"
},
"cost": {
"annotation_campaign_h20_hours": 5.0379046784506905,
"hard_cap_h20_hours": 6.0,
"outcome_only_cost_to_acceptable_h20_hours_lower_bound": 2.4284364508172893,
"telemetry_cost_to_acceptable_h20_hours_lower_bound": 2.4284364508172893,
"telemetry_gpu_cost_reduction_fraction": 0.0,
"session_h20_hours": {
"source_mns32_mbbt4096": 1.3566088432735868,
"mns64_mbbt4096": 1.256969277858734,
"mns32_mbbt8192": 1.254111782974667,
"joint_mns64_mbbt8192": 1.1702147743437026
}
},
"policy_comparison": {
"outcome_only": {
"measurement_cutoff_s": 300.0,
"selected_action": "joint",
"intervention_order": ["joint", "mns", "mbbt", "noop"]
},
"telemetry": {
"measurement_cutoff_s": 300.0,
"selected_action": "joint",
"intervention_order": ["joint", "mns", "mbbt", "noop"]
},
"action_changed": false,
"measurement_changed": false,
"confirmation_trigger": false,
"contribution_gate": false
},
"surface": {
"source_mns32_mbbt4096": {
"normalized_slo_goodput": [
0.40090909090909094,
0.3978787878787879,
0.42060606060606065
],
"median": 0.40090909090909094
},
"mns64_mbbt4096": {
"normalized_slo_goodput": [1.0, 0.9996969696969698, 1.0],
"median": 1.0
},
"mns32_mbbt8192": {
"normalized_slo_goodput": [
0.44393939393939397,
0.41515151515151516,
0.4260606060606061
],
"median": 0.42606060606060603
},
"joint_mns64_mbbt8192": {
"normalized_slo_goodput": [1.0, 1.0, 1.0],
"median": 1.0
}
},
"selected_checkpoint_prediction": {
"actual_median_effect": {
"noop": 0.0,
"mns": 0.5990909090909091,
"mbbt": 0.02515151515151509,
"joint": 0.5990909090909091
},
"outcome_only_predicted_effect": {
"noop": 0.0,
"mns": 0.2886250281729182,
"mbbt": 0.17933598309437812,
"joint": 0.3205015384324615
},
"telemetry_predicted_effect": {
"noop": 0.0,
"mns": 0.26117798146236215,
"mbbt": 0.09686132563074483,
"joint": 0.35190199346536294
},
"actual_joint_minus_mns": 0.0,
"outcome_only_joint_minus_mns": 0.0318765102595433,
"telemetry_joint_minus_mns": 0.09072401200300079
},
"engine_mechanism": {
"source_mns32_mbbt4096": {
"scheduler_records": 41086,
"waiting_fraction": 0.9312174463320839,
"mns_exclusive_fraction": 0.8536484447256973,
"mbbt_exclusive_fraction": 0.01114734946210388,
"both_fraction": 0.06642165214428272,
"running_utilization_mean": 0.9738878510928297,
"token_utilization_mean": 0.15694342925509905,
"kv_usage_mean": 0.027507593814715858,
"preemptions": 0
},
"mns64_mbbt4096": {
"scheduler_records": 37001,
"waiting_fraction": 0.053809356503878275,
"mns_exclusive_fraction": 0.0,
"mbbt_exclusive_fraction": 0.053809356503878275,
"running_utilization_mean": 0.5410364753655307,
"token_utilization_mean": 0.17425695631536986,
"kv_usage_mean": 0.030549112146415616,
"preemptions": 0
},
"mns32_mbbt8192": {
"scheduler_records": 41348,
"waiting_fraction": 0.9119425365192996,
"mns_exclusive_fraction": 0.9108542130211861,
"mbbt_exclusive_fraction": 0.0003627744993711909,
"running_utilization_mean": 0.9652567838831383,
"token_utilization_mean": 0.0779606414231039,
"kv_usage_mean": 0.027355900366122385,
"preemptions": 0
},
"joint_mns64_mbbt8192": {
"scheduler_records": 40416,
"waiting_fraction": 0.0088826207442597,
"mns_exclusive_fraction": 0.0,
"mbbt_exclusive_fraction": 0.0088826207442597,
"running_utilization_mean": 0.49403392220902614,
"token_utilization_mean": 0.07978070546782215,
"kv_usage_mean": 0.028003770774946098,
"preemptions": 0
}
},
"sanity": {
"surface_outcomes": {"n": 12, "min": 0.3978787878787879, "max": 1.0, "distinct_n": 8},
"session_h20_hours": {"n": 4, "min": 1.1702147743437026, "max": 1.3566088432735868, "distinct_n": 4},
"scheduler_records": {"n": 4, "min": 37001, "max": 41348, "distinct_n": 4},
"invariants": {
"controller_complete": true,
"four_sessions_complete": true,
"twelve_surface_outcomes": true,
"ratios_bounded": true,
"nonnegative_counts_and_costs": true,
"surface_not_all_identical": true,
"request_hashes_match": true,
"no_censored_runs": true
},
"red_flags": []
}
}

View File

@@ -0,0 +1,519 @@
#!/usr/bin/env python3
"""Train and audit outcome-only versus telemetry action-response policies."""
from __future__ import annotations
import argparse
import hashlib
import importlib.util
import json
import math
import os
import sys
from collections import defaultdict
from pathlib import Path
from typing import Any, Mapping, Sequence
HERE = Path(__file__).resolve().parent
def _load_model():
spec = importlib.util.spec_from_file_location(
"active_intervention_model", HERE / "model.py"
)
module = importlib.util.module_from_spec(spec)
assert spec.loader is not None
sys.modules[spec.name] = module
spec.loader.exec_module(module)
return module
MODEL = _load_model()
REGULARIZATION = 10.0
MINIMUM_MARGIN = 0.02
CONFIDENCE_Z = 1.0
ACCEPTABLE_REGRET = 0.02
def sha256_file(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for chunk in iter(lambda: source.read(1 << 20), b""):
digest.update(chunk)
return digest.hexdigest()
def atomic_json(path: Path, payload: Any) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
temporary = path.with_suffix(path.suffix + ".tmp")
temporary.write_text(
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
)
os.replace(temporary, path)
def grouped(
examples: Sequence[Mapping[str, Any]], key: str
) -> dict[str, list[Mapping[str, Any]]]:
result: dict[str, list[Mapping[str, Any]]] = defaultdict(list)
for example in examples:
result[str(example[key])].append(example)
return dict(result)
def evaluate_grouped_cv(
examples: Sequence[Mapping[str, Any]],
*,
include_telemetry: bool,
holdout_key: str,
) -> dict[str, Any]:
holdouts = grouped(examples, holdout_key)
decision_rows = []
for held_out, test_examples in sorted(holdouts.items()):
training = [example for example in examples if str(example[holdout_key]) != held_out]
if len({str(example["decision_id"]) for example in training}) < 2:
continue
model = MODEL.fit_ridge(
training,
include_telemetry=include_telemetry,
regularization=REGULARIZATION,
)
for decision_id, candidates in sorted(grouped(test_examples, "decision_id").items()):
predictions = []
for candidate in candidates:
source = candidate["source"]
action = candidate["action"]
noop = (
int(action["target_mns"]) == int(source["mns"])
and int(action["target_mbbt"]) == int(source["mbbt"])
)
if noop:
prediction = 0.0
else:
names, vector = MODEL.feature_vector(
candidate, include_telemetry=include_telemetry
)
if tuple(names) != model.feature_names:
raise ValueError("cross-validation feature schema mismatch")
prediction = max(-1.0, min(1.0, model.predict(vector)))
predictions.append(
{
"action_id": str(candidate["action"]["id"]),
"prediction": prediction,
"real": float(candidate["target_normalized_goodput"]),
}
)
predictions.sort(key=lambda row: (-row["prediction"], row["action_id"]))
selected = predictions[0]
oracle = max(row["real"] for row in predictions)
regret = 1.0 - selected["real"] / oracle if oracle > 0 else 0.0
best_actions = {
row["action_id"] for row in predictions if math.isclose(row["real"], oracle)
}
acceptable_actions = {
row["action_id"]
for row in predictions
if oracle <= 0
or 1.0 - float(row["real"]) / oracle <= ACCEPTABLE_REGRET + 1e-12
}
decision_rows.append(
{
"holdout": held_out,
"decision_id": decision_id,
"selected_action": selected["action_id"],
"best_actions": sorted(best_actions),
"acceptable_actions": sorted(acceptable_actions),
"correct": regret <= ACCEPTABLE_REGRET + 1e-12,
"selected_real": selected["real"],
"oracle_real": oracle,
"regret": regret,
"predictions": predictions,
}
)
if not decision_rows:
return {"status": "INSUFFICIENT_GROUPS", "decisions": []}
regrets = [float(row["regret"]) for row in decision_rows]
return {
"status": "VALID",
"holdout_key": holdout_key,
"acceptable_regret": ACCEPTABLE_REGRET,
"decision_n": len(decision_rows),
"correct_n": sum(bool(row["correct"]) for row in decision_rows),
"accuracy": sum(bool(row["correct"]) for row in decision_rows) / len(decision_rows),
"mean_regret": sum(regrets) / len(regrets),
"max_regret": max(regrets),
"decisions": decision_rows,
}
def paired_delta(outcome: Mapping[str, Any], telemetry: Mapping[str, Any]) -> dict[str, Any]:
if outcome.get("status") != "VALID" or telemetry.get("status") != "VALID":
return {"status": "INSUFFICIENT_GROUPS"}
outcome_by_id = {row["decision_id"]: row for row in outcome["decisions"]}
telemetry_by_id = {row["decision_id"]: row for row in telemetry["decisions"]}
common = sorted(set(outcome_by_id) & set(telemetry_by_id))
rows = []
for decision_id in common:
before = outcome_by_id[decision_id]
after = telemetry_by_id[decision_id]
rows.append(
{
"decision_id": decision_id,
"outcome_action": before["selected_action"],
"telemetry_action": after["selected_action"],
"action_changed": before["selected_action"] != after["selected_action"],
"regret_delta": float(after["regret"]) - float(before["regret"]),
"telemetry_corrected": (not before["correct"]) and bool(after["correct"]),
"telemetry_harmed": bool(before["correct"]) and (not after["correct"]),
}
)
return {
"status": "VALID",
"decision_n": len(rows),
"action_changed_n": sum(row["action_changed"] for row in rows),
"corrected_n": sum(row["telemetry_corrected"] for row in rows),
"harmed_n": sum(row["telemetry_harmed"] for row in rows),
"mean_regret_delta": (
sum(float(row["regret_delta"]) for row in rows) / len(rows) if rows else 0.0
),
"rows": rows,
}
def evaluate_sequential_measurement_cv(
examples: Sequence[Mapping[str, Any]],
*,
include_telemetry: bool,
holdout_key: str,
) -> dict[str, Any]:
"""Replay a two-consecutive-confident-checkpoint measurement policy."""
phases = sorted({str(example["phase"]) for example in examples}, key=float)
holdouts = grouped(examples, holdout_key)
rows = []
full_duration_s = max(float(example["cutoff_s"]) for example in examples)
for held_out, test_examples in sorted(holdouts.items()):
training = [
example for example in examples if str(example[holdout_key]) != held_out
]
if len({str(example["decision_id"]) for example in training}) < 3:
continue
phase_models = {}
for phase in phases:
phase_training = [
example for example in training if str(example["phase"]) == phase
]
phase_models[phase] = MODEL.fit_jackknife_ensemble(
phase_training,
include_telemetry=include_telemetry,
regularization=REGULARIZATION,
)
for decision_id, decision_examples in sorted(
grouped(test_examples, "decision_id").items()
):
checkpoints = []
by_phase = grouped(decision_examples, "phase")
for phase in phases:
candidates = by_phase[phase]
decision = MODEL.select_action(
phase_models[phase],
candidates,
include_telemetry=include_telemetry,
confidence_z=CONFIDENCE_Z,
minimum_margin=MINIMUM_MARGIN,
)
checkpoints.append(
{
"phase": phase,
"cutoff_s": float(candidates[0]["cutoff_s"]),
**decision,
}
)
selected_checkpoint = checkpoints[-1]
stop_reason = "full_measurement_fallback"
for previous, current in zip(checkpoints, checkpoints[1:], strict=False):
if (
previous["confident"]
and current["confident"]
and previous["selected_action"] == current["selected_action"]
):
selected_checkpoint = current
stop_reason = "two_consecutive_confident_checkpoints"
break
candidates = by_phase[str(selected_checkpoint["phase"])]
real_by_action = {
str(candidate["action"]["id"]): float(
candidate["target_normalized_goodput"]
)
for candidate in candidates
}
target_by_action = {
str(candidate["action"]["id"]): str(
candidate["action"]["target_config_id"]
)
for candidate in candidates
}
selected_action = str(selected_checkpoint["selected_action"])
oracle = max(real_by_action.values())
selected_real = real_by_action[selected_action]
regret = 1.0 - selected_real / oracle if oracle > 0 else 0.0
source_tp = 4
target_s = 0.0 if selected_action == "noop" else full_duration_s
replay_gpu_seconds = source_tp * (
float(selected_checkpoint["cutoff_s"]) + target_s
)
rows.append(
{
"holdout": held_out,
"decision_id": decision_id,
"selected_phase": str(selected_checkpoint["phase"]),
"selected_cutoff_s": float(selected_checkpoint["cutoff_s"]),
"stop_reason": stop_reason,
"selected_action": selected_action,
"selected_target_config_id": target_by_action[selected_action],
"selected_real": selected_real,
"oracle_real": oracle,
"regret": regret,
"acceptable": regret <= ACCEPTABLE_REGRET + 1e-12,
"replay_gpu_seconds_lower_bound": replay_gpu_seconds,
"checkpoints": checkpoints,
}
)
if not rows:
return {"status": "INSUFFICIENT_GROUPS", "decisions": []}
regrets = [float(row["regret"]) for row in rows]
cutoffs = [float(row["selected_cutoff_s"]) for row in rows]
costs = [float(row["replay_gpu_seconds_lower_bound"]) for row in rows]
return {
"status": "VALID",
"holdout_key": holdout_key,
"measurement_rule": "earliest two consecutive confident checkpoints; otherwise full",
"acceptable_regret": ACCEPTABLE_REGRET,
"decision_n": len(rows),
"acceptable_n": sum(bool(row["acceptable"]) for row in rows),
"mean_regret": sum(regrets) / len(regrets),
"max_regret": max(regrets),
"mean_cutoff_s": sum(cutoffs) / len(cutoffs),
"total_replay_gpu_seconds_lower_bound": sum(costs),
"total_replay_h20_hours_lower_bound": sum(costs) / 3600.0,
"decisions": rows,
}
def paired_sequential_delta(
outcome: Mapping[str, Any], telemetry: Mapping[str, Any]
) -> dict[str, Any]:
if outcome.get("status") != "VALID" or telemetry.get("status") != "VALID":
return {"status": "INSUFFICIENT_GROUPS"}
before_by_id = {row["decision_id"]: row for row in outcome["decisions"]}
after_by_id = {row["decision_id"]: row for row in telemetry["decisions"]}
rows = []
for decision_id in sorted(set(before_by_id) & set(after_by_id)):
before = before_by_id[decision_id]
after = after_by_id[decision_id]
rows.append(
{
"decision_id": decision_id,
"outcome_action": before["selected_action"],
"telemetry_action": after["selected_action"],
"outcome_cutoff_s": before["selected_cutoff_s"],
"telemetry_cutoff_s": after["selected_cutoff_s"],
"outcome_regret": before["regret"],
"telemetry_regret": after["regret"],
"regret_delta": float(after["regret"]) - float(before["regret"]),
"gpu_seconds_delta": float(
after["replay_gpu_seconds_lower_bound"]
)
- float(before["replay_gpu_seconds_lower_bound"]),
"telemetry_corrected": (not before["acceptable"])
and bool(after["acceptable"]),
"telemetry_harmed": bool(before["acceptable"])
and (not after["acceptable"]),
}
)
outcome_cost = float(outcome["total_replay_gpu_seconds_lower_bound"])
telemetry_cost = float(telemetry["total_replay_gpu_seconds_lower_bound"])
return {
"status": "VALID",
"decision_n": len(rows),
"corrected_n": sum(row["telemetry_corrected"] for row in rows),
"harmed_n": sum(row["telemetry_harmed"] for row in rows),
"outcome_replay_gpu_seconds_lower_bound": outcome_cost,
"telemetry_replay_gpu_seconds_lower_bound": telemetry_cost,
"gpu_cost_reduction_fraction": (
1.0 - telemetry_cost / outcome_cost if outcome_cost > 0 else 0.0
),
"rows": rows,
}
def build_policy(dataset_path: Path) -> dict[str, Any]:
dataset = json.loads(dataset_path.read_text(encoding="utf-8"))
if dataset.get("status") != "VALID" or dataset["sanity"]["red_flags"]:
raise ValueError("training dataset is not valid")
examples = dataset["examples"]
phases = sorted({str(example["phase"]) for example in examples}, key=float)
phase_results = {}
incremental_candidates = []
for phase in phases:
selected = [example for example in examples if str(example["phase"]) == phase]
outcome_cv = evaluate_grouped_cv(
selected, include_telemetry=False, holdout_key="repetition"
)
telemetry_cv = evaluate_grouped_cv(
selected, include_telemetry=True, holdout_key="repetition"
)
outcome_regime = evaluate_grouped_cv(
selected, include_telemetry=False, holdout_key="regime"
)
telemetry_regime = evaluate_grouped_cv(
selected, include_telemetry=True, holdout_key="regime"
)
delta = paired_delta(outcome_cv, telemetry_cv)
outcome_models = MODEL.fit_jackknife_ensemble(
selected,
include_telemetry=False,
regularization=REGULARIZATION,
)
telemetry_models = MODEL.fit_jackknife_ensemble(
selected,
include_telemetry=True,
regularization=REGULARIZATION,
)
incremental = bool(
delta.get("status") == "VALID"
and int(delta["corrected_n"]) >= 1
and int(delta["harmed_n"]) == 0
and float(delta["mean_regret_delta"]) < -1e-12
and float(telemetry_cv["max_regret"]) <= 0.05
and telemetry_regime.get("status") == "VALID"
and float(telemetry_regime["mean_regret"])
<= float(outcome_regime["mean_regret"]) + 1e-12
and float(telemetry_regime["max_regret"]) <= 0.05
)
if incremental:
incremental_candidates.append(phase)
phase_results[phase] = {
"cutoff_s": float(selected[0]["cutoff_s"]),
"outcome_only": {
"leave_repetition_out": outcome_cv,
"leave_regime_out": outcome_regime,
"models": MODEL.models_to_json(outcome_models),
},
"telemetry": {
"leave_repetition_out": telemetry_cv,
"leave_regime_out": telemetry_regime,
"models": MODEL.models_to_json(telemetry_models),
},
"paired_incremental": delta,
"incremental_gate": incremental,
}
outcome_sequential = evaluate_sequential_measurement_cv(
examples, include_telemetry=False, holdout_key="repetition"
)
telemetry_sequential = evaluate_sequential_measurement_cv(
examples, include_telemetry=True, holdout_key="repetition"
)
sequential_delta = paired_sequential_delta(
outcome_sequential, telemetry_sequential
)
retrospective_cost_gate = bool(
sequential_delta.get("status") == "VALID"
and int(sequential_delta["harmed_n"]) == 0
and int(telemetry_sequential["acceptable_n"])
>= int(outcome_sequential["acceptable_n"])
and float(telemetry_sequential["max_regret"]) <= 0.05
and float(sequential_delta["gpu_cost_reduction_fraction"]) >= 0.10
)
status = (
"RETROSPECTIVE_GPU_COST_SIGNAL"
if retrospective_cost_gate
else "NO_RETROSPECTIVE_GPU_COST_SIGNAL"
)
target_values = [float(example["target_normalized_goodput"]) for example in examples]
effect_values = [
float(example["target_delta_normalized_goodput"]) for example in examples
]
invariants = {
"four_phases": len(phases) == 4,
"targets_bounded": all(0.0 <= value <= 1.0 for value in target_values),
"targets_not_all_identical": len(set(target_values)) > 1,
"effects_bounded": all(-1.0 <= value <= 1.0 for value in effect_values),
"effects_not_all_identical": len(set(effect_values)) > 1,
"models_present_every_phase": all(
phase_results[phase][mode]["models"]
for phase in phases
for mode in ("outcome_only", "telemetry")
),
}
red_flags = [name for name, passed in invariants.items() if not passed]
if red_flags:
raise RuntimeError(f"policy sanity failed: {red_flags}")
return {
"schema": "active-intervention-policy-v0",
"status": status,
"training": {
"dataset": str(dataset_path),
"dataset_sha256": sha256_file(dataset_path),
"examples": len(examples),
"decisions": len({example["decision_id"] for example in examples}),
"regularization": REGULARIZATION,
"confidence_z": CONFIDENCE_Z,
"minimum_margin": MINIMUM_MARGIN,
"acceptable_regret": ACCEPTABLE_REGRET,
},
"measurement_policy": {
"rule": "earliest two consecutive confident checkpoints; otherwise full",
"checkpoints": [phase_results[phase]["cutoff_s"] for phase in phases],
"confidence_z": CONFIDENCE_Z,
"minimum_margin": MINIMUM_MARGIN,
},
"sequential_replay": {
"outcome_only": outcome_sequential,
"telemetry": telemetry_sequential,
"paired_delta": sequential_delta,
"retrospective_gpu_cost_gate": retrospective_cost_gate,
"minimum_cost_reduction_fraction": 0.10,
},
"phases": phase_results,
"sanity": {
"invariants": invariants,
"red_flags": red_flags,
"target_normalized_goodput": {
"n": len(target_values),
"min": min(target_values),
"max": max(target_values),
"distinct_n": len(set(target_values)),
},
"target_delta_normalized_goodput": {
"n": len(effect_values),
"min": min(effect_values),
"max": max(effect_values),
"distinct_n": len(set(effect_values)),
},
},
}
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--dataset", type=Path, required=True)
parser.add_argument("--output", type=Path, required=True)
args = parser.parse_args()
policy = build_policy(args.dataset)
atomic_json(args.output, policy)
print(
json.dumps(
{
"status": policy["status"],
"measurement_policy": policy["measurement_policy"],
"sanity": policy["sanity"],
},
sort_keys=True,
)
)
if __name__ == "__main__":
main()

View File

@@ -37,6 +37,41 @@ def selection_for(
return manifest["cells"][cell]["targets"][level]["selections"][role]
def campaign_gpu_accounting(
primary_state_path: Path, prior_state_paths: tuple[Path, ...] = ()
) -> dict[str, Any]:
attempts = []
for role, path in (
[("prior_failure", path) for path in prior_state_paths]
+ [("primary", primary_state_path)]
):
state = json.loads(path.read_text(encoding="utf-8"))
gpu_hours = float(state["gpu_hours_total"])
attempts.append(
{
"role": role,
"path": str(path.resolve()),
"sha256": sha256_file(path),
"status": state["status"],
"h20_hours": gpu_hours,
}
)
total = sum(attempt["h20_hours"] for attempt in attempts)
primary = json.loads(primary_state_path.read_text(encoding="utf-8"))
hard_cap = float(primary["hard_cap_h20_hours"])
return {
"attempts": attempts,
"aggregate_h20_hours": total,
"hard_cap_h20_hours": hard_cap,
"invariants": {
"costs_nonnegative": all(
attempt["h20_hours"] >= 0.0 for attempt in attempts
),
"aggregate_below_cap": 0.0 <= total < hard_cap,
},
}
def build_pilot_examples(
manifest: dict[str, Any], run_root: Path, cutoff_s: float
) -> tuple[list[PrefixExample], list[dict[str, Any]], list[str]]:
@@ -125,11 +160,13 @@ def analyze(
manifest_path: Path,
model_path: Path,
run_root: Path,
prior_state_paths: tuple[Path, ...] = (),
) -> dict[str, Any]:
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
models = json.loads(model_path.read_text(encoding="utf-8"))
state_path = run_root / "controller-state.json"
state = json.loads(state_path.read_text(encoding="utf-8"))
gpu_accounting = campaign_gpu_accounting(state_path, prior_state_paths)
cutoff_s = float(models["cutoff_s"])
threshold = float(models["accept_probability"])
examples, details, red_flags = build_pilot_examples(manifest, run_root, cutoff_s)
@@ -171,7 +208,7 @@ def analyze(
detail["actual_timestamped_outcomes"] == 0 for detail in details
):
red_flags.append("no_exact_request_timestamps")
if float(state["gpu_hours_total"]) >= float(state["hard_cap_h20_hours"]):
if not all(gpu_accounting["invariants"].values()):
red_flags.append("hard_cap_exceeded")
outcome_errors = outcome_policy["false_accept"] + outcome_policy["false_reject"]
@@ -232,8 +269,8 @@ def analyze(
"opens_expanded_p2": pilot_pass,
},
"gpu": {
"actual_h20_hours": state["gpu_hours_total"],
"hard_cap_h20_hours": state["hard_cap_h20_hours"],
"primary_attempt_h20_hours": state["gpu_hours_total"],
**gpu_accounting,
},
"sanity": {
"red_flags": red_flags,
@@ -277,9 +314,15 @@ def main() -> None:
parser.add_argument("--manifest", type=Path, required=True)
parser.add_argument("--frozen-models", type=Path, required=True)
parser.add_argument("--run-root", type=Path, required=True)
parser.add_argument("--prior-state", type=Path, action="append", default=[])
parser.add_argument("--output", type=Path, required=True)
args = parser.parse_args()
result = analyze(args.manifest, args.frozen_models, args.run_root)
result = analyze(
args.manifest,
args.frozen_models,
args.run_root,
tuple(args.prior_state),
)
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(result, indent=2, sort_keys=True) + "\n")
print(json.dumps({

View File

@@ -0,0 +1,429 @@
#!/usr/bin/env python3
"""Replay the P1 simulator shortlist under full and prefix policies."""
from __future__ import annotations
import argparse
import json
import math
import subprocess
from pathlib import Path
from typing import Any
from analyze_prefixes import numeric, sha256_file
AITUNER_ROOT = Path(__file__).resolve().parents[2]
FROZEN_K = 2
CUTOFF_S = 5.0
THRESHOLD = 0.95
def git_capture(*arguments: str) -> str:
return subprocess.run(
["git", "-C", str(AITUNER_ROOT), *arguments],
check=True,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
).stdout
def setup_costs(state: dict[str, Any]) -> dict[str, float]:
result = {}
for cell, payload in state["cells"].items():
tp = int(payload["tp"])
annotation_intervals = sum(
float(run["elapsed_s"]) * tp / 3600.0
for run in payload["runs"]
if run["role"] not in {"low1", "high1"}
)
primary_intervals = sum(
float(run["elapsed_s"]) * tp / 3600.0
for run in payload["runs"]
if run["role"] in {"low1", "high1"}
)
setup = float(payload["gpu_hours"]) - annotation_intervals - primary_intervals
if setup < -1e-12:
raise ValueError(f"negative inferred setup cost: {cell}={setup}")
result[cell] = max(0.0, setup)
return result
def build_candidates(
manifest: dict[str, Any],
state: dict[str, Any],
strong: dict[str, Any],
) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
baseline_probability = strong["headline"]["sim_plus_outcome"]["probability"]
instrument_probability = strong["headline"][
"sim_plus_outcome_plus_instrumentation"
]["probability"]
setup = setup_costs(state)
anchors = []
for detail, baseline_p, instrument_p in zip(
strong["pilot_examples"], baseline_probability, instrument_probability
):
cell = str(detail["cell"])
level = str(detail["level"])
role = f"{level}1"
selection = manifest["cells"][cell]["targets"][level]["selections"][role]
run = next(
item for item in state["cells"][cell]["runs"] if item["role"] == role
)
tp = int(state["cells"][cell]["tp"])
full_cost = float(run["elapsed_s"]) * tp / 3600.0
prefix_cost = min(CUTOFF_S, float(run["elapsed_s"])) * tp / 3600.0
anchors.append(
{
"cell": cell,
"level": level,
"role": role,
"tp": tp,
"real_feasible": bool(detail["adjudicated_feasible"]),
"real_goodput_req_s_per_gpu": float(
selection["offered_req_s_per_gpu"]
),
"sim_feasible": bool(detail["sim_slo_feasible"]),
"sim_pass_rate": float(detail["sim_slo_pass_rate"]),
"sim_throughput_req_s_per_gpu": float(
detail["sim_completed_throughput_per_gpu"]
),
"baseline_probability": float(baseline_p),
"instrument_probability": float(instrument_p),
"setup_h20_hours": setup[cell],
"full_trial_h20_hours": full_cost,
"prefix_h20_hours": prefix_cost,
}
)
candidates = []
for cell in sorted(manifest["cells"]):
feasible = [
anchor for anchor in anchors if anchor["cell"] == cell and anchor["sim_feasible"]
]
if not feasible:
continue
candidates.append(
max(feasible, key=lambda anchor: anchor["sim_throughput_req_s_per_gpu"])
)
candidates.sort(
key=lambda anchor: (
-anchor["sim_throughput_req_s_per_gpu"],
anchor["cell"],
)
)
return anchors, candidates
def expanded_top_k(candidates: list[dict[str, Any]], k: int) -> list[dict[str, Any]]:
if not candidates or k <= 0:
return []
boundary = candidates[min(k, len(candidates)) - 1][
"sim_throughput_req_s_per_gpu"
]
return [
candidate
for candidate in candidates
if candidate["sim_throughput_req_s_per_gpu"] >= boundary - 1e-12
]
def selected_result(
evaluated: list[dict[str, Any]], feasible_key: str
) -> tuple[str | None, float | None]:
feasible = [candidate for candidate in evaluated if candidate[feasible_key]]
if not feasible:
return None, None
best = max(feasible, key=lambda candidate: candidate["real_goodput_req_s_per_gpu"])
return str(best["cell"]), float(best["real_goodput_req_s_per_gpu"])
def replay(
shortlist: list[dict[str, Any]],
*,
probability_key: str | None,
oracle_goodput: float,
common_failure_h20_hours: float,
) -> dict[str, Any]:
evaluated = []
online_cost = 0.0
early_accept = 0
early_reject = 0
false_accept = 0
false_reject = 0
for candidate in shortlist:
current = dict(candidate)
online_cost += current["setup_h20_hours"]
if probability_key is None:
predicted_feasible = current["real_feasible"]
online_cost += current["full_trial_h20_hours"]
action = "full"
else:
probability = float(current[probability_key])
if probability >= THRESHOLD:
predicted_feasible = True
early_accept += 1
online_cost += current["prefix_h20_hours"]
action = "early_accept"
false_accept += int(not current["real_feasible"])
elif probability <= 1.0 - THRESHOLD:
predicted_feasible = False
early_reject += 1
online_cost += current["prefix_h20_hours"]
action = "early_reject"
false_reject += int(current["real_feasible"])
else:
predicted_feasible = current["real_feasible"]
online_cost += current["full_trial_h20_hours"]
action = "continue_full"
current["policy_feasible"] = predicted_feasible
current["action"] = action
evaluated.append(current)
selected_cell, selected_goodput = selected_result(evaluated, "policy_feasible")
regret = (
1.0 - selected_goodput / oracle_goodput
if selected_goodput is not None and oracle_goodput > 0
else None
)
return {
"selected_cell": selected_cell,
"selected_real_goodput_req_s_per_gpu": selected_goodput,
"real_regret": regret,
"online_h20_hours": online_cost,
"conservative_h20_hours_with_prior_failure": (
online_cost + common_failure_h20_hours
),
"early_accept": early_accept,
"early_reject": early_reject,
"false_accept": false_accept,
"false_reject": false_reject,
"evaluated": [
{
"cell": item["cell"],
"level": item["level"],
"action": item["action"],
"real_feasible": item["real_feasible"],
"policy_feasible": item["policy_feasible"],
}
for item in evaluated
],
}
def analyze(
manifest_path: Path,
state_path: Path,
prior_state_path: Path,
strong_path: Path,
) -> dict[str, Any]:
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
state = json.loads(state_path.read_text(encoding="utf-8"))
prior = json.loads(prior_state_path.read_text(encoding="utf-8"))
strong = json.loads(strong_path.read_text(encoding="utf-8"))
anchors, candidates = build_candidates(manifest, state, strong)
oracle_anchor = max(
(anchor for anchor in anchors if anchor["real_feasible"]),
key=lambda anchor: anchor["real_goodput_req_s_per_gpu"],
)
oracle_goodput = float(oracle_anchor["real_goodput_req_s_per_gpu"])
common_failure = float(prior["gpu_hours_total"])
by_k = {}
for k in (1, 2, 3, 6):
shortlist = expanded_top_k(candidates, k)
full = replay(
shortlist,
probability_key=None,
oracle_goodput=oracle_goodput,
common_failure_h20_hours=common_failure,
)
baseline = replay(
shortlist,
probability_key="baseline_probability",
oracle_goodput=oracle_goodput,
common_failure_h20_hours=common_failure,
)
instrument = replay(
shortlist,
probability_key="instrument_probability",
oracle_goodput=oracle_goodput,
common_failure_h20_hours=common_failure,
)
for result in (baseline, instrument):
result["online_cost_reduction_vs_full"] = (
1.0 - result["online_h20_hours"] / full["online_h20_hours"]
)
result["conservative_cost_reduction_vs_full"] = 1.0 - (
result["conservative_h20_hours_with_prior_failure"]
/ full["conservative_h20_hours_with_prior_failure"]
)
by_k[str(k)] = {
"actual_shortlist_size": len(shortlist),
"shortlist": [candidate["cell"] for candidate in shortlist],
"sim_top_k_plus_real_final": full,
"sim_plus_outcome": baseline,
"sim_plus_outcome_plus_instrumentation": instrument,
}
frozen = by_k[str(FROZEN_K)]
full = frozen["sim_top_k_plus_real_final"]
baseline = frozen["sim_plus_outcome"]
instrument = frozen["sim_plus_outcome_plus_instrumentation"]
baseline_safe = baseline["false_accept"] == 0 and baseline["false_reject"] == 0
instrument_safe = (
instrument["false_accept"] == 0 and instrument["false_reject"] == 0
)
incremental_reduction = (
1.0 - instrument["online_h20_hours"] / baseline["online_h20_hours"]
if baseline_safe and instrument_safe and baseline["online_h20_hours"] > 0
else None
)
contribution_gate = {
"frozen_k": FROZEN_K,
"instrument_safe": instrument_safe,
"outcome_baseline_safe": baseline_safe,
"instrument_regret_at_most_5pct": (
instrument["real_regret"] is not None
and instrument["real_regret"] <= 0.05
),
"instrument_cost_reduction_vs_full_at_least_30pct": (
instrument["online_cost_reduction_vs_full"] >= 0.30
),
"instrument_cost_reduction_vs_outcome_at_least_20pct": (
incremental_reduction is not None and incremental_reduction >= 0.20
),
"incremental_reduction_vs_outcome": incremental_reduction,
}
contribution_gate["passes"] = all(
contribution_gate[key]
for key in (
"instrument_safe",
"outcome_baseline_safe",
"instrument_regret_at_most_5pct",
"instrument_cost_reduction_vs_full_at_least_30pct",
"instrument_cost_reduction_vs_outcome_at_least_20pct",
)
)
red_flags = []
if state["status"] != "complete" or int(state["completed_cells"]) != 6:
red_flags.append("pilot_incomplete")
if strong["status"] != "PASS" or strong["sanity"]["red_flags"]:
red_flags.append("strong_input_invalid")
if len(anchors) != 12 or len(candidates) != 6:
red_flags.append("unexpected_surface_size")
probabilities = [
value
for anchor in anchors
for value in (anchor["baseline_probability"], anchor["instrument_probability"])
]
costs = [
value
for anchor in anchors
for value in (
anchor["setup_h20_hours"],
anchor["full_trial_h20_hours"],
anchor["prefix_h20_hours"],
)
]
if not all(0.0 <= value <= 1.0 for value in probabilities):
red_flags.append("probability_out_of_range")
if not all(value >= 0.0 and math.isfinite(value) for value in costs):
red_flags.append("invalid_cost")
return {
"schema": "fidelity-pilot-e2e-v1",
"status": "PASS" if not red_flags else "STOP",
"scope": "held-out P1 replay; gate diagnostic, not paper-facing evidence",
"ranking": [
{
"rank": rank,
"cell": candidate["cell"],
"level": candidate["level"],
"sim_throughput_req_s_per_gpu": candidate[
"sim_throughput_req_s_per_gpu"
],
"real_feasible": candidate["real_feasible"],
"real_goodput_req_s_per_gpu": candidate[
"real_goodput_req_s_per_gpu"
],
}
for rank, candidate in enumerate(candidates, start=1)
],
"real_oracle": {
"cell": oracle_anchor["cell"],
"level": oracle_anchor["level"],
"goodput_req_s_per_gpu": oracle_goodput,
},
"by_k": by_k,
"contribution_gate": contribution_gate,
"analysis": {
"script": str(Path(__file__).resolve()),
"script_sha256": sha256_file(Path(__file__).resolve()),
"aituner_git_head": git_capture("rev-parse", "HEAD").strip(),
"aituner_git_status_short": git_capture("status", "--short"),
},
"provenance": {
"manifest": str(manifest_path.resolve()),
"manifest_sha256": sha256_file(manifest_path),
"controller_state": str(state_path.resolve()),
"controller_state_sha256": sha256_file(state_path),
"prior_state": str(prior_state_path.resolve()),
"prior_state_sha256": sha256_file(prior_state_path),
"strong_metrics": str(strong_path.resolve()),
"strong_metrics_sha256": sha256_file(strong_path),
},
"sanity": {
"red_flags": red_flags,
"anchors": numeric([1 for _ in anchors]),
"candidates": numeric([1 for _ in candidates]),
"probabilities": numeric(probabilities),
"costs_h20_hours": numeric(costs),
"invariants": {
"anchors_12": len(anchors) == 12,
"candidates_6": len(candidates) == 6,
"probabilities_bounded": all(
0.0 <= value <= 1.0 for value in probabilities
),
"costs_nonnegative": all(value >= 0.0 for value in costs),
"per_config_not_all_identical": len(
{candidate["sim_throughput_req_s_per_gpu"] for candidate in candidates}
)
> 1,
"tie_expansion_applied": True,
},
},
}
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--manifest", type=Path, required=True)
parser.add_argument("--controller-state", type=Path, required=True)
parser.add_argument("--prior-state", type=Path, required=True)
parser.add_argument("--strong-metrics", type=Path, required=True)
parser.add_argument("--output", type=Path, required=True)
args = parser.parse_args()
result = analyze(
args.manifest,
args.controller_state,
args.prior_state,
args.strong_metrics,
)
args.output.write_text(json.dumps(result, indent=2, sort_keys=True) + "\n")
print(
json.dumps(
{
"status": result["status"],
"red_flags": result["sanity"]["red_flags"],
"contribution_gate": result["contribution_gate"],
},
sort_keys=True,
)
)
if result["status"] != "PASS":
raise RuntimeError(result["sanity"]["red_flags"])
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,298 @@
#!/usr/bin/env python3
"""Audit telemetry against a simulator-aware outcome calibration baseline.
This is a retrospective headroom check. It strengthens the earlier
outcome-only baseline by giving both nested models the same per-anchor
Frontier throughput and SLO predictions. The only additional inputs to the
larger model are real engine Layer-1 features.
"""
from __future__ import annotations
import argparse
import hashlib
import json
import math
from pathlib import Path
from typing import Any
import numpy as np
from analyze_existing import (
DEFAULT_REGULARIZATION,
REGULARIZATION_SENSITIVITY,
_classification_metrics,
_fit_logistic,
_group_bootstrap_delta,
_mcnemar_exact_p,
_sigmoid,
)
from analyze_prefixes import (
INSTRUMENTATION_FEATURES,
OUTCOME_FEATURES,
PrefixExample,
build_examples,
numeric,
policy_metrics,
sha256_file,
)
SIMULATOR_FEATURES = (
"log_sim_completed_throughput_per_gpu",
"sim_slo_pass_rate",
"sim_slo_feasible",
)
def load_simulator_features(raw_root: Path) -> tuple[dict[tuple[str, float], tuple[float, ...]], str]:
features: dict[tuple[str, float], tuple[float, ...]] = {}
digest = hashlib.sha256()
paths = sorted(raw_root.glob("*/trial-0001/run_manifest.json"))
for manifest_path in paths:
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
run = manifest["run"]
if run["mode"] != "frozen-calibrated":
continue
scorer_path = manifest_path.parent / "scorer_output.json"
scorer = json.loads(scorer_path.read_text(encoding="utf-8"))
key = (str(run["cell_id"]), float(run["sampling_u"]))
if key in features:
raise ValueError(f"duplicate frozen simulator run: {key}")
throughput = float(scorer["throughput_requests_per_second_per_gpu"])
pass_rate = float(scorer["slo"]["pass_rate"])
if throughput <= 0 or not 0.0 <= pass_rate <= 1.0:
raise ValueError(f"invalid simulator output: {key}")
features[key] = (
math.log(throughput),
pass_rate,
float(bool(scorer["slo"]["feasible"])),
)
for path in (manifest_path, scorer_path):
digest.update(str(path.relative_to(raw_root)).encode())
digest.update(path.read_bytes())
return features, digest.hexdigest()
def simulator_row(
example: PrefixExample,
features: dict[tuple[str, float], tuple[float, ...]],
) -> tuple[float, ...]:
matches = [
values
for (cell, anchor), values in features.items()
if cell == example.cell
and math.isclose(anchor, example.anchor, rel_tol=0.0, abs_tol=1e-12)
]
if len(matches) != 1:
raise ValueError(
f"expected one simulator match for {example.cell}/{example.anchor}: {len(matches)}"
)
return matches[0]
def grouped_predictions(
examples: list[PrefixExample],
simulator: dict[tuple[str, float], tuple[float, ...]],
*,
instrumentation_aware: bool,
regularization: float,
) -> tuple[np.ndarray, np.ndarray, list[str]]:
probabilities: list[float] = []
labels: list[int] = []
groups: list[str] = []
for held_out in sorted({example.cell for example in examples}):
train = [example for example in examples if example.cell != held_out]
test = [example for example in examples if example.cell == held_out]
def row(example: PrefixExample) -> np.ndarray:
values = example.outcome + simulator_row(example, simulator)
if instrumentation_aware:
values += example.instrumentation
return np.asarray((1.0, *values), dtype=np.float64)
x_train = np.stack([row(example) for example in train])
x_test = np.stack([row(example) for example in test])
y_train = np.asarray([example.feasible for example in train], dtype=np.float64)
mean = x_train[:, 1:].mean(axis=0)
standard_deviation = x_train[:, 1:].std(axis=0)
standard_deviation[standard_deviation < 1e-8] = 1.0
x_train[:, 1:] = (x_train[:, 1:] - mean) / standard_deviation
x_test[:, 1:] = (x_test[:, 1:] - mean) / standard_deviation
weights = _fit_logistic(x_train, y_train, regularization)
probabilities.extend(_sigmoid(x_test @ weights).tolist())
labels.extend(example.feasible for example in test)
groups.extend(held_out for _ in test)
return (
np.asarray(labels, dtype=np.int64),
np.asarray(probabilities, dtype=np.float64),
groups,
)
def analyze(
phase6_path: Path,
phase6_raw_root: Path,
simulator_raw_root: Path,
simulator_metrics_path: Path,
) -> dict[str, Any]:
phase6 = json.loads(phase6_path.read_text(encoding="utf-8"))
examples = build_examples(phase6, phase6_raw_root, 5.0)
simulator, simulator_raw_sha256 = load_simulator_features(simulator_raw_root)
red_flags = []
try:
matched = [simulator_row(example, simulator) for example in examples]
except ValueError as error:
matched = []
red_flags.append(str(error))
sensitivity = {}
if matched:
for regularization in REGULARIZATION_SENSITIVITY:
labels, baseline_probability, groups = grouped_predictions(
examples,
simulator,
instrumentation_aware=False,
regularization=regularization,
)
instrument_labels, instrument_probability, instrument_groups = grouped_predictions(
examples,
simulator,
instrumentation_aware=True,
regularization=regularization,
)
if not np.array_equal(labels, instrument_labels) or groups != instrument_groups:
raise AssertionError("nested baseline folds differ")
baseline_correct = (baseline_probability >= 0.5) == labels
instrument_correct = (instrument_probability >= 0.5) == labels
paired = {
"both_correct": int(np.sum(baseline_correct & instrument_correct)),
"sim_outcome_only_correct": int(
np.sum(baseline_correct & ~instrument_correct)
),
"instrumentation_only_correct": int(
np.sum(~baseline_correct & instrument_correct)
),
"both_wrong": int(np.sum(~baseline_correct & ~instrument_correct)),
}
paired["mcnemar_exact_two_sided_p"] = _mcnemar_exact_p(
paired["sim_outcome_only_correct"],
paired["instrumentation_only_correct"],
)
sensitivity[str(regularization)] = {
"sim_plus_outcome": {
"classification": _classification_metrics(labels, baseline_probability),
"policy_0p95": policy_metrics(
examples, labels, baseline_probability, 0.95
),
},
"sim_plus_outcome_plus_instrumentation": {
"classification": _classification_metrics(labels, instrument_probability),
"policy_0p95": policy_metrics(
examples, labels, instrument_probability, 0.95
),
},
"paired_correctness": paired,
"group_bootstrap": _group_bootstrap_delta(
labels,
baseline_probability,
instrument_probability,
groups,
),
}
headline = sensitivity.get(str(DEFAULT_REGULARIZATION))
simulator_pass_rates = [row[1] for row in matched]
labels = [example.feasible for example in examples]
if len(examples) != 37:
red_flags.append("examples_not_37")
if len(simulator) != 92:
red_flags.append("frozen_simulator_runs_not_92")
if len(set(labels)) != 2:
red_flags.append("single_label")
if matched and not all(0.0 <= value <= 1.0 for value in simulator_pass_rates):
red_flags.append("simulator_pass_rate_out_of_range")
return {
"schema": "fidelity-strong-baseline-v1",
"status": "PASS" if not red_flags else "STOP",
"scope": "retrospective one-task headroom audit; not contribution evidence",
"comparison": (
"same 5-second prefix, folds, logistic family, regularization, and frozen "
"Frontier outputs; the only nested difference is real Layer-1 engine state"
),
"features": {
"shared_outcome": list(OUTCOME_FEATURES),
"shared_simulator": list(SIMULATOR_FEATURES),
"instrumentation_only": list(INSTRUMENTATION_FEATURES),
},
"headline_regularization": DEFAULT_REGULARIZATION,
"headline": headline,
"regularization_sensitivity": sensitivity,
"provenance": {
"phase6_metrics": str(phase6_path.resolve()),
"phase6_metrics_sha256": sha256_file(phase6_path),
"phase6_raw_root": str(phase6_raw_root.resolve()),
"simulator_metrics": str(simulator_metrics_path.resolve()),
"simulator_metrics_sha256": sha256_file(simulator_metrics_path),
"simulator_raw_root": str(simulator_raw_root.resolve()),
"frozen_simulator_manifest_scorer_set_sha256": simulator_raw_sha256,
},
"decision": {
"contribution_established": False,
"prospective_requirement": (
"repeat sim+outcome versus sim+outcome+instrumentation on complete held-out tasks"
),
},
"sanity": {
"red_flags": red_flags,
"examples": numeric([1 for _ in examples]),
"labels": {
**numeric(labels),
"positive": sum(labels),
"negative": len(labels) - sum(labels),
},
"matched_simulator_pass_rate": numeric(simulator_pass_rates),
"frozen_simulator_runs": len(simulator),
"invariants": {
"all_examples_matched_once": len(matched) == len(examples),
"same_nested_folds": True,
"simulator_ratios_bounded": all(
0.0 <= value <= 1.0 for value in simulator_pass_rates
),
"labels_not_identical": len(set(labels)) == 2,
"per_config_results_not_all_identical": len(set(simulator_pass_rates)) > 1,
},
},
}
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--phase6-metrics", type=Path, required=True)
parser.add_argument("--phase6-raw-root", type=Path, required=True)
parser.add_argument("--simulator-raw-root", type=Path, required=True)
parser.add_argument("--simulator-metrics", type=Path, required=True)
parser.add_argument("--output", type=Path, required=True)
args = parser.parse_args()
result = analyze(
args.phase6_metrics,
args.phase6_raw_root,
args.simulator_raw_root,
args.simulator_metrics,
)
args.output.write_text(json.dumps(result, indent=2, sort_keys=True) + "\n")
print(
json.dumps(
{
"status": result["status"],
"output": str(args.output),
"red_flags": result["sanity"]["red_flags"],
},
sort_keys=True,
)
)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,507 @@
#!/usr/bin/env python3
"""Exploratory P1 audit against the strengthened simulator-aware baseline.
P1 was already running when the strong baseline was added, so this script is
not paper-facing prospective evidence. It trains only on the historical
Phase-6 task and evaluates the exact P1 primary probes. Both nested models
receive identical Frontier predictions; engine telemetry is the sole feature
difference.
"""
from __future__ import annotations
import argparse
import json
import math
import subprocess
from pathlib import Path
from typing import Any
import numpy as np
from analyze_existing import (
DEFAULT_REGULARIZATION,
REGULARIZATION_SENSITIVITY,
_classification_metrics,
_fit_logistic,
_mcnemar_exact_p,
_sigmoid,
)
from analyze_pilot import build_pilot_examples, campaign_gpu_accounting
from analyze_prefixes import (
INSTRUMENTATION_FEATURES,
OUTCOME_FEATURES,
PrefixExample,
build_examples,
numeric,
policy_metrics,
sha256_file,
)
from analyze_strong_baseline import (
SIMULATOR_FEATURES,
load_simulator_features,
simulator_row,
)
AITUNER_ROOT = Path(__file__).resolve().parents[2]
def git_capture(*arguments: str) -> str:
return subprocess.run(
["git", "-C", str(AITUNER_ROOT), *arguments],
check=True,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
).stdout
def load_pilot_simulator(
path: Path,
) -> tuple[dict[tuple[str, str], tuple[float, ...]], list[str]]:
payload = json.loads(path.read_text(encoding="utf-8"))
red_flags = []
if payload.get("status") != "PASS":
red_flags.append("pilot_simulator_not_pass")
features: dict[tuple[str, str], tuple[float, ...]] = {}
for item in payload.get("results", []):
key = (str(item["cell"]), str(item["role"]))
if key in features:
red_flags.append(f"duplicate_pilot_simulator_{key[0]}_{key[1]}")
continue
scorer = item["scorer"]
throughput = float(scorer["throughput_requests_per_second_per_gpu"])
pass_rate = float(scorer["slo"]["pass_rate"])
if throughput <= 0:
red_flags.append(f"nonpositive_pilot_simulator_throughput_{key[0]}_{key[1]}")
if not 0.0 <= pass_rate <= 1.0:
red_flags.append(f"pilot_simulator_ratio_out_of_range_{key[0]}_{key[1]}")
features[key] = (
math.log(throughput),
pass_rate,
float(bool(scorer["slo"]["feasible"])),
)
if len(features) != 12:
red_flags.append("pilot_simulator_entries_not_12")
return features, red_flags
def fit_model(
examples: list[PrefixExample],
simulator: list[tuple[float, ...]],
*,
instrumentation_aware: bool,
regularization: float,
) -> dict[str, Any]:
rows = []
for example, simulator_features in zip(examples, simulator):
values = example.outcome + simulator_features
if instrumentation_aware:
values += example.instrumentation
rows.append((1.0, *values))
matrix = np.asarray(rows, dtype=np.float64)
labels = np.asarray([example.feasible for example in examples], dtype=np.float64)
mean = matrix[:, 1:].mean(axis=0)
standard_deviation = matrix[:, 1:].std(axis=0)
standard_deviation[standard_deviation < 1e-8] = 1.0
standardized = matrix.copy()
standardized[:, 1:] = (standardized[:, 1:] - mean) / standard_deviation
weights = _fit_logistic(standardized, labels, regularization)
return {
"instrumentation_aware": instrumentation_aware,
"regularization": regularization,
"feature_mean": mean,
"feature_standard_deviation": standard_deviation,
"weights": weights,
}
def predict_model(
model: dict[str, Any],
examples: list[PrefixExample],
simulator: list[tuple[float, ...]],
) -> np.ndarray:
rows = []
for example, simulator_features in zip(examples, simulator):
values = example.outcome + simulator_features
if model["instrumentation_aware"]:
values += example.instrumentation
rows.append((1.0, *values))
matrix = np.asarray(rows, dtype=np.float64)
matrix[:, 1:] = (
matrix[:, 1:] - model["feature_mean"]
) / model["feature_standard_deviation"]
return _sigmoid(matrix @ model["weights"])
def covariate_shift(
training_examples: list[PrefixExample],
training_simulator: list[tuple[float, ...]],
pilot_examples: list[PrefixExample],
pilot_simulator: list[tuple[float, ...]],
*,
instrumentation_aware: bool,
) -> dict[str, Any]:
def matrix(
examples: list[PrefixExample], simulator: list[tuple[float, ...]]
) -> np.ndarray:
rows = []
for example, simulator_features in zip(examples, simulator):
values = example.outcome + simulator_features
if instrumentation_aware:
values += example.instrumentation
rows.append(values)
return np.asarray(rows, dtype=np.float64)
training = matrix(training_examples, training_simulator)
pilot = matrix(pilot_examples, pilot_simulator)
mean = training.mean(axis=0)
standard_deviation = training.std(axis=0)
standard_deviation[standard_deviation < 1e-8] = 1.0
absolute_z = np.abs((pilot - mean) / standard_deviation)
names = [*OUTCOME_FEATURES, *SIMULATOR_FEATURES]
if instrumentation_aware:
names.extend(INSTRUMENTATION_FEATURES)
return {
"values": numeric(absolute_z.ravel().tolist()),
"count_gt_3": int(np.sum(absolute_z > 3.0)),
"count_gt_5": int(np.sum(absolute_z > 5.0)),
"total_feature_values": int(absolute_z.size),
"per_feature_max_abs_z": {
name: float(value) for name, value in zip(names, absolute_z.max(axis=0))
},
}
def comparison(
training_examples: list[PrefixExample],
training_simulator: list[tuple[float, ...]],
pilot_examples: list[PrefixExample],
pilot_simulator: list[tuple[float, ...]],
regularization: float,
) -> dict[str, Any]:
labels = np.asarray([example.feasible for example in pilot_examples], dtype=np.int64)
baseline_model = fit_model(
training_examples,
training_simulator,
instrumentation_aware=False,
regularization=regularization,
)
instrument_model = fit_model(
training_examples,
training_simulator,
instrumentation_aware=True,
regularization=regularization,
)
baseline_probability = predict_model(
baseline_model, pilot_examples, pilot_simulator
)
instrument_probability = predict_model(
instrument_model, pilot_examples, pilot_simulator
)
baseline_correct = (baseline_probability >= 0.5) == labels
instrument_correct = (instrument_probability >= 0.5) == labels
paired = {
"both_correct": int(np.sum(baseline_correct & instrument_correct)),
"sim_outcome_only_correct": int(
np.sum(baseline_correct & ~instrument_correct)
),
"instrumentation_only_correct": int(
np.sum(~baseline_correct & instrument_correct)
),
"both_wrong": int(np.sum(~baseline_correct & ~instrument_correct)),
}
paired["mcnemar_exact_two_sided_p"] = _mcnemar_exact_p(
paired["sim_outcome_only_correct"], paired["instrumentation_only_correct"]
)
return {
"sim_plus_outcome": {
"classification": _classification_metrics(labels, baseline_probability),
"policy_0p95": policy_metrics(
pilot_examples, labels, baseline_probability, 0.95
),
"probability": baseline_probability.tolist(),
},
"sim_plus_outcome_plus_instrumentation": {
"classification": _classification_metrics(labels, instrument_probability),
"policy_0p95": policy_metrics(
pilot_examples, labels, instrument_probability, 0.95
),
"probability": instrument_probability.tolist(),
},
"paired_correctness": paired,
}
def analyze(
phase6_path: Path,
phase6_raw_root: Path,
training_simulator_root: Path,
pilot_manifest_path: Path,
pilot_run_root: Path,
pilot_simulator_path: Path,
prior_state_paths: tuple[Path, ...] = (),
) -> dict[str, Any]:
phase6 = json.loads(phase6_path.read_text(encoding="utf-8"))
pilot_manifest = json.loads(pilot_manifest_path.read_text(encoding="utf-8"))
pilot_state_path = pilot_run_root / "controller-state.json"
pilot_state = json.loads(pilot_state_path.read_text(encoding="utf-8"))
gpu_accounting = campaign_gpu_accounting(
pilot_state_path, prior_state_paths
)
training_examples = build_examples(phase6, phase6_raw_root, 5.0)
training_simulator_map, training_simulator_sha256 = load_simulator_features(
training_simulator_root
)
training_simulator = [
simulator_row(example, training_simulator_map)
for example in training_examples
]
pilot_examples, pilot_details, red_flags = build_pilot_examples(
pilot_manifest, pilot_run_root, 5.0
)
pilot_simulator_map, simulator_red_flags = load_pilot_simulator(
pilot_simulator_path
)
red_flags.extend(simulator_red_flags)
pilot_simulator = []
for example, detail in zip(pilot_examples, pilot_details):
role = f"{detail['level']}1"
key = (example.cell, role)
if key not in pilot_simulator_map:
red_flags.append(f"missing_pilot_simulator_{example.cell}_{role}")
pilot_simulator.append((0.0, 0.0, 0.0))
else:
pilot_simulator.append(pilot_simulator_map[key])
sensitivity = {}
if not red_flags:
for regularization in REGULARIZATION_SENSITIVITY:
sensitivity[str(regularization)] = comparison(
training_examples,
training_simulator,
pilot_examples,
pilot_simulator,
regularization,
)
headline = sensitivity.get(str(DEFAULT_REGULARIZATION))
labels = [example.feasible for example in pilot_examples]
simulator_pass_rates = [row[1] for row in pilot_simulator]
simulator_labels = [int(row[2]) for row in pilot_simulator]
if len(training_examples) != 37:
red_flags.append("training_examples_not_37")
if len(pilot_examples) != 12:
red_flags.append("pilot_examples_not_12")
if len(set(labels)) != 2:
red_flags.append("pilot_single_label")
if len(set(simulator_pass_rates)) <= 1:
red_flags.append("pilot_simulator_results_identical")
if pilot_state.get("status") != "complete" or int(
pilot_state.get("completed_cells", 0)
) != 6:
red_flags.append("pilot_campaign_incomplete")
if any(detail["actual_timestamped_outcomes"] == 0 for detail in pilot_details):
red_flags.append("pilot_no_exact_request_timestamps")
all_cell_validations = all(
cell.get("validation") is not None
and all(cell["validation"]["invariants"].values())
for cell in pilot_state.get("cells", {}).values()
)
if not all_cell_validations:
red_flags.append("pilot_cell_validation_failed")
if not all(gpu_accounting["invariants"].values()):
red_flags.append("pilot_hard_cap_exceeded")
covariate_diagnostics = {
"sim_plus_outcome": covariate_shift(
training_examples,
training_simulator,
pilot_examples,
pilot_simulator,
instrumentation_aware=False,
),
"sim_plus_outcome_plus_instrumentation": covariate_shift(
training_examples,
training_simulator,
pilot_examples,
pilot_simulator,
instrumentation_aware=True,
),
}
if headline is None:
decision = {
"strong_incremental_gate": False,
"reason": "analysis red flag prevented nested comparison",
}
else:
baseline_policy = headline["sim_plus_outcome"]["policy_0p95"]
instrument_policy = headline[
"sim_plus_outcome_plus_instrumentation"
]["policy_0p95"]
baseline_errors = baseline_policy["false_accept"] + baseline_policy["false_reject"]
instrument_errors = (
instrument_policy["false_accept"] + instrument_policy["false_reject"]
)
baseline_reduction = baseline_policy["valid_cost_reduction_fraction"]
instrument_reduction = instrument_policy["valid_cost_reduction_fraction"]
reduction_delta = (
instrument_reduction - baseline_reduction
if baseline_reduction is not None and instrument_reduction is not None
else None
)
per_lambda_safe_and_better = []
for item in sensitivity.values():
baseline = item["sim_plus_outcome"]["policy_0p95"]
instrument = item["sim_plus_outcome_plus_instrumentation"]["policy_0p95"]
base_errors = baseline["false_accept"] + baseline["false_reject"]
inst_errors = instrument["false_accept"] + instrument["false_reject"]
base_reduction = baseline["valid_cost_reduction_fraction"]
inst_reduction = instrument["valid_cost_reduction_fraction"]
per_lambda_safe_and_better.append(
inst_errors == 0
and inst_errors <= base_errors
and base_reduction is not None
and inst_reduction is not None
and inst_reduction > base_reduction
)
decision = {
"strong_incremental_gate": bool(
not red_flags
and instrument_errors == 0
and instrument_errors <= baseline_errors
and reduction_delta is not None
and reduction_delta >= 0.15
),
"regularization_robust": all(per_lambda_safe_and_better),
"valid_cost_reduction_fraction_delta": reduction_delta,
"scope": "exploratory task; may choose P2 design but cannot establish contribution",
}
return {
"schema": "fidelity-strong-pilot-v1",
"status": "PASS" if not red_flags else "STOP",
"scope": (
"post-amendment exploratory P1 audit; strong model was not frozen before "
"partial P1 outcomes, so this is not prospective contribution evidence"
),
"features": {
"shared_outcome": list(OUTCOME_FEATURES),
"shared_simulator": list(SIMULATOR_FEATURES),
"instrumentation_only": list(INSTRUMENTATION_FEATURES),
},
"headline_regularization": DEFAULT_REGULARIZATION,
"headline": headline,
"regularization_sensitivity": sensitivity,
"simulator_only": {
"classification": _classification_metrics(
np.asarray(labels, dtype=np.int64),
np.asarray(simulator_labels, dtype=np.float64),
)
if labels
else None,
"predicted_feasible": simulator_labels,
},
"pilot_examples": [
{
**detail,
"sim_completed_throughput_per_gpu": math.exp(simulator[0]),
"sim_slo_pass_rate": simulator[1],
"sim_slo_feasible": bool(simulator[2]),
}
for detail, simulator in zip(pilot_details, pilot_simulator)
],
"covariate_shift_diagnostic": covariate_diagnostics,
"decision": decision,
"gpu": {
"primary_attempt_h20_hours": pilot_state["gpu_hours_total"],
**gpu_accounting,
},
"analysis": {
"script": str(Path(__file__).resolve()),
"script_sha256": sha256_file(Path(__file__).resolve()),
"aituner_git_head": git_capture("rev-parse", "HEAD").strip(),
"aituner_git_status_short": git_capture("status", "--short"),
},
"provenance": {
"phase6_metrics": str(phase6_path.resolve()),
"phase6_metrics_sha256": sha256_file(phase6_path),
"phase6_raw_root": str(phase6_raw_root.resolve()),
"training_simulator_root": str(training_simulator_root.resolve()),
"training_simulator_manifest_scorer_set_sha256": training_simulator_sha256,
"pilot_manifest": str(pilot_manifest_path.resolve()),
"pilot_manifest_sha256": sha256_file(pilot_manifest_path),
"pilot_run_root": str(pilot_run_root.resolve()),
"pilot_controller_state": str(pilot_state_path.resolve()),
"pilot_controller_state_sha256": sha256_file(pilot_state_path),
"pilot_simulator": str(pilot_simulator_path.resolve()),
"pilot_simulator_sha256": sha256_file(pilot_simulator_path),
},
"sanity": {
"red_flags": red_flags,
"training_examples": numeric([1 for _ in training_examples]),
"pilot_labels": {
**numeric(labels),
"positive": sum(labels),
"negative": len(labels) - sum(labels),
},
"pilot_simulator_pass_rate": numeric(simulator_pass_rates),
"invariants": {
"training_examples_37": len(training_examples) == 37,
"pilot_examples_12": len(pilot_examples) == 12,
"pilot_cells_6": len({example.cell for example in pilot_examples}) == 6,
"pilot_both_labels": len(set(labels)) == 2,
"simulator_ratios_bounded": all(
0.0 <= value <= 1.0 for value in simulator_pass_rates
),
"per_config_not_all_identical": len(set(simulator_pass_rates)) > 1,
"all_prefixes_exact_monotonic": all(
example.completion_time_source in {"exact_monotonic", "none_completed"}
for example in pilot_examples
),
"all_cell_validations": all_cell_validations,
"gpu_cost_nonnegative_below_cap": (
all(gpu_accounting["invariants"].values())
),
},
},
}
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--phase6-metrics", type=Path, required=True)
parser.add_argument("--phase6-raw-root", type=Path, required=True)
parser.add_argument("--training-simulator-root", type=Path, required=True)
parser.add_argument("--pilot-manifest", type=Path, required=True)
parser.add_argument("--pilot-run-root", type=Path, required=True)
parser.add_argument("--pilot-simulator", type=Path, required=True)
parser.add_argument("--prior-state", type=Path, action="append", default=[])
parser.add_argument("--output", type=Path, required=True)
args = parser.parse_args()
result = analyze(
args.phase6_metrics,
args.phase6_raw_root,
args.training_simulator_root,
args.pilot_manifest,
args.pilot_run_root,
args.pilot_simulator,
tuple(args.prior_state),
)
args.output.write_text(json.dumps(result, indent=2, sort_keys=True) + "\n")
print(
json.dumps(
{
"status": result["status"],
"red_flags": result["sanity"]["red_flags"],
"decision": result["decision"],
},
sort_keys=True,
)
)
if result["status"] != "PASS":
raise RuntimeError(result["sanity"]["red_flags"])
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,334 @@
#!/usr/bin/env python3
"""Prepare exact Frontier fixtures for the P1 primary low/high probes.
Prompt-bearing band traces remain under ``--private-root``. The emitted
fixtures and public manifest contain token IDs, block IDs, hashes, and
aggregate metadata, but no prompt text.
"""
from __future__ import annotations
import argparse
import hashlib
import importlib.util
import json
import math
import subprocess
import sys
from pathlib import Path
from typing import Any
from transformers import AutoTokenizer
HERE = Path(__file__).resolve().parent
AITUNER_ROOT = HERE.parents[1]
sys.path.insert(0, str(HERE))
import prepare_pilot as pilot # noqa: E402
PRIMARY_ROLES = ("low1", "high1")
def load_module(path: Path):
module_root = str(path.parent.resolve())
if module_root not in sys.path:
sys.path.insert(0, module_root)
spec = importlib.util.spec_from_file_location("simfid_s2rb_prepare", path)
if spec is None or spec.loader is None:
raise ImportError(path)
module = importlib.util.module_from_spec(spec)
sys.modules[spec.name] = module
spec.loader.exec_module(module)
return module
def order_hash(values: list[str]) -> str:
return hashlib.sha256("\n".join(values).encode()).hexdigest()
def sha256_file(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for chunk in iter(lambda: source.read(1 << 20), b""):
digest.update(chunk)
return digest.hexdigest()
def git_capture(root: Path, *arguments: str) -> str:
return subprocess.run(
["git", "-C", str(root), *arguments],
check=True,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
).stdout
def raw_rows(path: Path) -> dict[int, dict[str, Any]]:
result = {}
with path.open(encoding="utf-8") as source:
for index, line in enumerate(source):
if line.strip():
result[index] = json.loads(line)
return result
def selected_hashes(
selected: list[Any], rows: dict[int, dict[str, Any]]
) -> dict[str, str]:
identifiers = []
arrivals = []
lengths = []
for item in selected:
row = rows[item.row_index]
identifiers.append(str(row.get("request_id") or row.get("id") or item.row_index))
arrivals.append(f"{float(item.timestamp) * 0.1:.12f}")
lengths.append(str(int(item.input_length)))
return {
"request_id_order_sha256": order_hash(identifiers),
"arrival_order_sha256": order_hash(arrivals),
"input_length_order_sha256": order_hash(lengths),
}
def kv_blocks(raw_root: Path, cell: str) -> int:
stream = next((raw_root / cell / "opprof").glob("*.jsonl"))
with stream.open(encoding="utf-8") as source:
for line in source:
record = json.loads(line)
if "step_index" in record:
return int(record["kv"]["total_blocks"])
raise ValueError(f"no Layer-1 record for {cell}")
def source_window(windows_path: Path, window_id: str) -> tuple[dict[str, Any], Path]:
return pilot.resolve_source_trace(windows_path, window_id)
def prepare(args: argparse.Namespace) -> dict[str, Any]:
simulator = load_module(args.replayserve_root / "tools/simfid_s2rb_prepare.py")
manifest = json.loads(args.pilot_manifest.read_text(encoding="utf-8"))
window, trace = source_window(args.source_windows, args.source_window_id)
if args.band_root is not None:
role_paths = {
role: (args.band_root / f"{role}.jsonl").resolve()
for role in PRIMARY_ROLES
}
band_stats = {
role: manifest["private"]["band_stats"][role]
for role in PRIMARY_ROLES
}
for role, path in role_paths.items():
if sha256_file(path) != band_stats[role]["sha256"]:
raise ValueError(f"pre-materialized band hash mismatch: {role}")
private_windows = None
else:
private_windows, all_band_stats = pilot.materialize_bands(
trace, window, args.private_root
)
private_payload = json.loads(private_windows.read_text(encoding="utf-8"))
role_paths = {
item["fidelity_pilot_role"]: (
private_windows.parent / item["trace_file"]
).resolve()
for item in private_payload["windows"]
}
band_stats = {
role: all_band_stats[role]
for role in PRIMARY_ROLES
}
tokenizer = AutoTokenizer.from_pretrained(
args.tokenizer, local_files_only=True, use_fast=True
)
fixture_root = args.output / "fixtures"
config_root = args.output / "configs"
fixture_root.mkdir(parents=True, exist_ok=True)
config_root.mkdir(parents=True, exist_ok=True)
entries = []
red_flags = []
for role in PRIMARY_ROLES:
trace_path = role_paths[role]
retained, trace_stats = simulator.scan_trace(trace_path)
rows = raw_rows(trace_path)
primary_pool = [retained[(index * len(retained)) // 512] for index in range(512)]
selections: dict[str, list[Any]] = {}
selected_union: set[int] = set()
for cell, cell_manifest in sorted(manifest["cells"].items()):
level = "low" if role.startswith("low") else "high"
expected = cell_manifest["targets"][level]["selections"][role]
pool = retained if int(cell_manifest["tp"]) == 4 else primary_pool
selected = [item for item in pool if item.sampling_u <= float(expected["anchor"])]
selections[cell] = selected
selected_union.update(item.row_index for item in selected)
hashes = selected_hashes(selected, rows)
if len(selected) != int(expected["selected_count"]):
red_flags.append(f"selection_count_{cell}_{role}")
for key, value in hashes.items():
if value != expected[key]:
red_flags.append(f"selection_hash_{cell}_{role}_{key}")
token_gates, selected_records, block_stats = simulator.tokenize_and_hash(
trace=trace_path,
tokenizer=tokenizer,
retained=retained,
selected_union=selected_union,
)
if any(gate["status"] != "pass" for gate in token_gates.values()):
red_flags.append(f"token_gate_{role}")
for cell, selected in selections.items():
cell_manifest = manifest["cells"][cell]
level = "low" if role.startswith("low") else "high"
expected = cell_manifest["targets"][level]["selections"][role]
fixture_id = f"fidelity_p1_{cell}_{role}"
cell_record = {
"cell_id": cell,
"tensor_parallel_size": int(cell_manifest["tp"]),
"max_num_seqs": int(cell_manifest["mns"]),
"store_role": "companion" if int(cell_manifest["tp"]) == 4 else "primary",
"kv_capacity": {
"block_size_tokens": 16,
"num_blocks": kv_blocks(args.phase6_raw_root, cell),
},
}
probe = {
"probe_index": 0 if role == "low1" else 1,
"sampling_u": float(expected["anchor"]),
}
fixture = simulator.create_fixture(
fixture_root=fixture_root,
fixture_id=fixture_id,
cell=cell_record,
probe=probe,
row_indexes=[item.row_index for item in selected],
meta_by_index={item.row_index: item for item in retained},
selected_records=selected_records,
)
config_path = config_root / f"{fixture_id}.json"
config = simulator.build_config(
path=config_path,
cell=cell_record,
mode="frozen-calibrated",
fixture_ids=[fixture_id],
frontier_root=args.frontier_root,
cache_dir=args.cache_dir,
)
entries.append(
{
"cell": cell,
"role": role,
"level": level,
"anchor": expected["anchor"],
"selected_count": len(selected),
"fixture_id": fixture_id,
"fixture_manifest": str(
(fixture_root / fixture_id / "fixture_manifest.json").resolve()
),
"frontier_csv": fixture["frontier_csv"]["path"],
"sidecar": fixture["sidecar_jsonl"]["path"],
"config": str(config_path.resolve()),
"calibration_scale": config["calibration"]["a_tp"],
}
)
if block_stats["selected_union_records"] != len(selected_union):
red_flags.append(f"selected_union_{role}")
if trace_stats["retained_inclusive_0_8192"] < 512:
red_flags.append(f"retained_too_small_{role}")
selected_counts = [int(entry["selected_count"]) for entry in entries]
calibration = [float(entry["calibration_scale"]) for entry in entries]
result = {
"schema": "fidelity-p1-frontier-prepared-v1",
"status": "PASS" if not red_flags else "STOP",
"source": {
"pilot_manifest": str(args.pilot_manifest.resolve()),
"source_windows": str(args.source_windows.resolve()),
"source_window_id": args.source_window_id,
"source_trace": str(trace.resolve()),
"private_windows": (
str(private_windows.resolve()) if private_windows is not None else None
),
"pre_materialized_band_root": (
str(args.band_root.resolve()) if args.band_root is not None else None
),
"band_stats": band_stats,
},
"simulator": {
"replayserve_root": str(args.replayserve_root.resolve()),
"frontier_root": str(args.frontier_root.resolve()),
"tokenizer": str(args.tokenizer.resolve()),
"mode": "frozen-calibrated",
},
"generator": {
"script": str(Path(__file__).resolve()),
"script_sha256": sha256_file(Path(__file__).resolve()),
"aituner_git_head": git_capture(AITUNER_ROOT, "rev-parse", "HEAD").strip(),
"aituner_git_status_short": git_capture(AITUNER_ROOT, "status", "--short"),
},
"entries": entries,
"sanity": {
"red_flags": red_flags,
"n": len(entries),
"selected_count": {
"n": len(selected_counts),
"min": min(selected_counts),
"max": max(selected_counts),
"distinct_n": len(set(selected_counts)),
},
"calibration_scale": {
"n": len(calibration),
"min": min(calibration),
"max": max(calibration),
"distinct_n": len(set(calibration)),
},
"invariants": {
"entries_12": len(entries) == 12,
"roles_2": {entry["role"] for entry in entries} == set(PRIMARY_ROLES),
"cells_6": len({entry["cell"] for entry in entries}) == 6,
"selected_nonnegative": all(value > 0 for value in selected_counts),
"per_config_not_identical": len(set(selected_counts)) > 1,
},
},
}
args.public_manifest.parent.mkdir(parents=True, exist_ok=True)
args.public_manifest.write_text(json.dumps(result, indent=2, sort_keys=True) + "\n")
return result
def parser() -> argparse.ArgumentParser:
result = argparse.ArgumentParser()
result.add_argument("--pilot-manifest", type=Path, required=True)
result.add_argument("--source-windows", type=Path, required=True)
result.add_argument("--source-window-id", required=True)
result.add_argument("--private-root", type=Path, required=True)
result.add_argument("--band-root", type=Path)
result.add_argument("--output", type=Path, required=True)
result.add_argument("--public-manifest", type=Path, required=True)
result.add_argument("--phase6-raw-root", type=Path, required=True)
result.add_argument("--replayserve-root", type=Path, required=True)
result.add_argument("--frontier-root", type=Path, required=True)
result.add_argument("--cache-dir", type=Path, required=True)
result.add_argument("--tokenizer", type=Path, required=True)
return result
def main() -> None:
result = prepare(parser().parse_args())
print(
json.dumps(
{
"status": result["status"],
"entries": len(result["entries"]),
"red_flags": result["sanity"]["red_flags"],
},
sort_keys=True,
)
)
if result["status"] != "PASS":
raise RuntimeError(result["sanity"]["red_flags"])
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,303 @@
#!/usr/bin/env python3
"""Run and score a frozen Frontier probe manifest, CPU only."""
from __future__ import annotations
import argparse
import hashlib
import importlib.util
import json
import os
import subprocess
import sys
import time
from pathlib import Path
from typing import Any
def load_module(name: str, path: Path):
module_root = str(path.parent.resolve())
if module_root not in sys.path:
sys.path.insert(0, module_root)
spec = importlib.util.spec_from_file_location(name, path)
if spec is None or spec.loader is None:
raise ImportError(path)
module = importlib.util.module_from_spec(spec)
sys.modules[spec.name] = module
spec.loader.exec_module(module)
return module
def sha256_file(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as source:
for chunk in iter(lambda: source.read(1 << 20), b""):
digest.update(chunk)
return digest.hexdigest()
def atomic_json(path: Path, payload: Any) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
temporary = path.with_suffix(path.suffix + ".tmp")
temporary.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n")
os.replace(temporary, path)
def git_capture(root: Path, *arguments: str) -> str:
return subprocess.run(
["git", "-C", str(root), *arguments],
check=True,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
).stdout
def execute(args: argparse.Namespace) -> dict[str, Any]:
args.output = args.output.resolve()
prepared = json.loads(args.prepared_manifest.read_text(encoding="utf-8"))
if prepared["status"] != "PASS":
raise RuntimeError("prepared simulator manifest did not pass")
expected_runs = int(prepared.get("expected_runs", len(prepared["entries"])))
if expected_runs != len(prepared["entries"]):
raise RuntimeError(
f"prepared manifest expected {expected_runs} runs but contains "
f"{len(prepared['entries'])} entries"
)
driver = load_module(
"simfid_execution_driver",
args.replayserve_root
/ "runs/simfid_s2rb/results/execution_driver.py",
)
head = git_capture(args.frontier_root, "rev-parse", "HEAD").strip()
status_short = git_capture(args.frontier_root, "status", "--short")
aituner_root = Path(__file__).resolve().parents[2]
aituner_head = git_capture(aituner_root, "rev-parse", "HEAD").strip()
aituner_status_short = git_capture(aituner_root, "status", "--short")
results = []
failures = []
gpu_visibility_disabled = True
for sequence, entry in enumerate(prepared["entries"]):
run_root = args.output / f"{sequence:03d}_{entry['fixture_id']}"
scorer_path = run_root / "scorer_output.json"
if scorer_path.is_file() and args.resume:
scorer = json.loads(scorer_path.read_text(encoding="utf-8"))
results.append({**entry, "sequence": sequence, "scorer": scorer, "resumed": True})
continue
run_root.mkdir(parents=True, exist_ok=True)
config_path = Path(entry["config"])
config = json.loads(config_path.read_text(encoding="utf-8"))
fixture_manifest_path = Path(entry["fixture_manifest"])
fixture = json.loads(fixture_manifest_path.read_text(encoding="utf-8"))
trace_path = Path(entry["frontier_csv"])
sidecar_path = Path(entry["sidecar"])
metrics_root = run_root / "frontier_metrics"
run_id = f"fidelity_p1_frontier_{sequence:02d}_{entry['cell']}_{entry['role']}"
knobs = config["frontier"]["knobs"]
command = driver.build_command(
trace_path=trace_path,
metrics_root=metrics_root,
run_id=run_id,
knobs=knobs,
)
driver.audit_command(command, knobs)
row = {
"hook_path": config["calibration"]["hook_path"],
"applied_a_tp": config["calibration"]["a_tp"],
"sidecar_path": str(sidecar_path),
"request_count": int(fixture["request_count"]),
"tensor_parallel_size": int(fixture["tensor_parallel_size"]),
}
environment = driver.environment_for(row)
gpu_visibility_disabled = gpu_visibility_disabled and (
environment.get("CUDA_VISIBLE_DEVICES") == ""
and environment.get("NVIDIA_VISIBLE_DEVICES") == "void"
)
run_manifest = {
"schema": "fidelity-p1-frontier-run-v1",
"sequence": sequence,
"cell": entry["cell"],
"role": entry["role"],
"anchor": entry["anchor"],
"request_count": entry["selected_count"],
"frontier": {
"root": str(args.frontier_root.resolve()),
"git_head": head,
"git_status_short": status_short,
},
"runner": {
"script": str(Path(__file__).resolve()),
"script_sha256": sha256_file(Path(__file__).resolve()),
"aituner_git_head": aituner_head,
"aituner_git_status_short": aituner_status_short,
},
"inputs": {
"config": str(config_path),
"config_sha256": sha256_file(config_path),
"fixture_manifest": str(fixture_manifest_path),
"fixture_manifest_sha256": sha256_file(fixture_manifest_path),
"frontier_csv": str(trace_path),
"frontier_csv_sha256": sha256_file(trace_path),
"sidecar": str(sidecar_path),
"sidecar_sha256": sha256_file(sidecar_path),
},
"environment": {
key: environment[key]
for key in (
"PYTHONPATH",
"FRONTIER_EXECUTION_TIME_SCALE",
"CUDA_VISIBLE_DEVICES",
"NVIDIA_VISIBLE_DEVICES",
"FRONTIER_LOG_LEVEL",
)
},
"command": command,
"contains_prompt_text": False,
}
atomic_json(run_root / "run_manifest.json", run_manifest)
start = time.time()
with (run_root / "stdout.log").open("w", encoding="utf-8") as stdout, (
run_root / "stderr.log"
).open("w", encoding="utf-8") as stderr:
try:
process = subprocess.run(
command,
cwd=args.frontier_root,
env=environment,
stdout=stdout,
stderr=stderr,
timeout=args.timeout_s,
)
return_code = int(process.returncode)
except subprocess.TimeoutExpired:
return_code = 124
runtime = time.time() - start
if return_code != 0:
failure = {
"sequence": sequence,
"cell": entry["cell"],
"role": entry["role"],
"return_code": return_code,
"runtime_s": runtime,
}
failures.append(failure)
atomic_json(run_root / "failure.json", failure)
break
system_path, request_path = driver.find_metrics(run_root)
scorer = driver.score_trial(row, system_path, request_path)
scorer["runtime_s"] = runtime
atomic_json(scorer_path, scorer)
results.append({**entry, "sequence": sequence, "scorer": scorer, "resumed": False})
print(
json.dumps(
{
"sequence": sequence,
"cell": entry["cell"],
"role": entry["role"],
"runtime_s": runtime,
"sim_pass_rate": scorer["slo"]["pass_rate"],
"sim_feasible": scorer["slo"]["feasible"],
},
sort_keys=True,
),
flush=True,
)
pass_rates = [float(item["scorer"]["slo"]["pass_rate"]) for item in results]
throughputs = [
float(item["scorer"]["throughput_requests_per_second_per_gpu"])
for item in results
]
runtimes = [float(item["scorer"]["runtime_s"]) for item in results]
red_flags = []
if failures:
red_flags.append("frontier_run_failure")
if len(results) != expected_runs:
red_flags.append("runs_not_expected")
if any(not 0.0 <= value <= 1.0 for value in pass_rates):
red_flags.append("pass_rate_out_of_range")
if any(value <= 0 for value in throughputs):
red_flags.append("nonpositive_throughput")
result = {
"schema": "fidelity-p1-frontier-result-v1",
"status": "PASS" if not red_flags else "STOP",
"prepared_manifest": str(args.prepared_manifest.resolve()),
"prepared_manifest_sha256": sha256_file(args.prepared_manifest),
"frontier": {
"root": str(args.frontier_root.resolve()),
"git_head": head,
"git_status_short": status_short,
},
"runner": {
"script": str(Path(__file__).resolve()),
"script_sha256": sha256_file(Path(__file__).resolve()),
"aituner_git_head": aituner_head,
"aituner_git_status_short": aituner_status_short,
},
"results": results,
"failures": failures,
"sanity": {
"red_flags": red_flags,
"n": len(results),
"pass_rate": {
"n": len(pass_rates),
"min": min(pass_rates) if pass_rates else None,
"max": max(pass_rates) if pass_rates else None,
"distinct_n": len(set(pass_rates)),
},
"throughput_per_gpu": {
"n": len(throughputs),
"min": min(throughputs) if throughputs else None,
"max": max(throughputs) if throughputs else None,
"distinct_n": len(set(throughputs)),
},
"runtime_s": {
"n": len(runtimes),
"min": min(runtimes) if runtimes else None,
"max": max(runtimes) if runtimes else None,
"distinct_n": len(set(runtimes)),
},
"invariants": {
"runs_expected": len(results) == expected_runs,
"expected_runs": expected_runs,
"zero_failures": not failures,
"ratios_bounded": all(0.0 <= value <= 1.0 for value in pass_rates),
"nonnegative_metrics": all(value > 0 for value in throughputs),
"per_config_not_identical": len(set(pass_rates)) > 1,
"gpu_visibility_disabled": gpu_visibility_disabled,
},
},
}
atomic_json(args.output / "metrics.json", result)
return result
def parser() -> argparse.ArgumentParser:
result = argparse.ArgumentParser()
result.add_argument("--prepared-manifest", type=Path, required=True)
result.add_argument("--output", type=Path, required=True)
result.add_argument("--replayserve-root", type=Path, required=True)
result.add_argument("--frontier-root", type=Path, required=True)
result.add_argument("--timeout-s", type=float, default=900.0)
result.add_argument("--resume", action="store_true")
return result
def main() -> None:
result = execute(parser().parse_args())
print(
json.dumps(
{
"status": result["status"],
"runs": len(result["results"]),
"red_flags": result["sanity"]["red_flags"],
},
sort_keys=True,
)
)
if result["status"] != "PASS":
raise RuntimeError(result["sanity"]["red_flags"])
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,477 @@
{
"comparison": "same 5-second prefix, folds, logistic family, regularization, and frozen Frontier outputs; the only nested difference is real Layer-1 engine state",
"decision": {
"contribution_established": false,
"prospective_requirement": "repeat sim+outcome versus sim+outcome+instrumentation on complete held-out tasks"
},
"features": {
"instrumentation_only": [
"model_steps_per_second",
"waiting_mean",
"waiting_max",
"waiting_nonzero_share",
"running_mean",
"running_max",
"decode_batch_mean",
"decode_batch_max",
"decode_batch_cv",
"kv_usage_mean",
"kv_usage_max",
"kv_usage_end_minus_start",
"graph_none_share",
"graph_full_share",
"padding_fraction",
"prefill_token_fraction",
"preemptions"
],
"shared_outcome": [
"log_offered_rate_per_gpu",
"log2_tp",
"log2_max_num_seqs",
"admitted_fraction",
"completed_over_admitted",
"completed_pass_rate",
"completed_fail_fraction_of_total",
"outstanding_over_admitted",
"ttft_max_over_slo_max",
"ttft_mean_over_slo_max",
"tpot_max_over_slo",
"tpot_mean_over_slo",
"admitted_input_tokens_mean_over_limit"
],
"shared_simulator": [
"log_sim_completed_throughput_per_gpu",
"sim_slo_pass_rate",
"sim_slo_feasible"
]
},
"headline": {
"group_bootstrap": {
"accuracy_delta_instrumentation_minus_outcome": {
"ci95": [
0.0,
0.18181818181818188
],
"point": 0.08108108108108103
},
"brier_delta_instrumentation_minus_outcome": {
"ci95": [
-0.04292727744470806,
0.019924730979981074
],
"point": -0.010145365131402809
},
"replicates": 10000,
"seed": 20260714,
"semantics": "group bootstrap over cells; diagnostic confidence interval"
},
"paired_correctness": {
"both_correct": 30,
"both_wrong": 4,
"instrumentation_only_correct": 3,
"mcnemar_exact_two_sided_p": 0.25,
"sim_outcome_only_correct": 0
},
"sim_plus_outcome": {
"classification": {
"accuracy": 0.8108108108108109,
"balanced_accuracy": 0.7242063492063493,
"brier": 0.1058226346682949,
"confusion": {
"false_negative": 3,
"false_positive": 4,
"true_negative": 5,
"true_positive": 25
},
"log_loss": 0.3011048455679668
},
"policy_0p95": {
"abstain_continue_full": 17,
"correctly_saved_h20_hours": 0.5429431818208333,
"decision_coverage": 0.5405405405405406,
"early_accept": 16,
"early_reject": 4,
"false_accept": 0,
"false_accept_examples": [],
"false_reject": 0,
"false_reject_examples": [],
"full_trial_h20_hours": 1.0669595034675,
"invalidly_saved_h20_hours": 0.0,
"remaining_h20_hours_at_cutoff": 0.957237281245278,
"saved_h20_hours_if_decisions_used": 0.5429431818208333,
"threshold": 0.95,
"valid_cost_reduction_fraction": 0.5088695307144538,
"valid_zero_error_policy": true
}
},
"sim_plus_outcome_plus_instrumentation": {
"classification": {
"accuracy": 0.8918918918918919,
"balanced_accuracy": 0.8154761904761905,
"brier": 0.0956772695368921,
"confusion": {
"false_negative": 1,
"false_positive": 3,
"true_negative": 6,
"true_positive": 27
},
"log_loss": 0.288823031828762
},
"policy_0p95": {
"abstain_continue_full": 12,
"correctly_saved_h20_hours": 0.7360063646722222,
"decision_coverage": 0.6756756756756757,
"early_accept": 20,
"early_reject": 5,
"false_accept": 0,
"false_accept_examples": [],
"false_reject": 0,
"false_reject_examples": [],
"full_trial_h20_hours": 1.0669595034675,
"invalidly_saved_h20_hours": 0.0,
"remaining_h20_hours_at_cutoff": 0.957237281245278,
"saved_h20_hours_if_decisions_used": 0.7360063646722222,
"threshold": 0.95,
"valid_cost_reduction_fraction": 0.6898165884274738,
"valid_zero_error_policy": true
}
}
},
"headline_regularization": 1.0,
"provenance": {
"frozen_simulator_manifest_scorer_set_sha256": "833842d96ecaa0b059ef99852621752f7989e63d100118b6025425fb119b7a55",
"phase6_metrics": "/home/gahow/phd/aituner/runs/opprof-phase6/phase6/metrics.json",
"phase6_metrics_sha256": "290ba7fcb8727291166de7e4d47afdc84e230052495c81dd087db0ace9f93a16",
"phase6_raw_root": "/home/gahow/phd/aituner/runs/opprof-phase6/phase6/solo-authoritative/cells",
"simulator_metrics": "/home/gahow/phd/replayserve/runs/simfid_s2rb/results/metrics.json",
"simulator_metrics_sha256": "55edb37d5692e979ab6f6dc6c65913a9db0aa0a836c350e4c05d9c38eee78206",
"simulator_raw_root": "/home/gahow/phd/replayserve/runs/simfid_s2rb/results/raw"
},
"regularization_sensitivity": {
"0.1": {
"group_bootstrap": {
"accuracy_delta_instrumentation_minus_outcome": {
"ci95": [
-0.17500000000000004,
0.0
],
"point": -0.08108108108108103
},
"brier_delta_instrumentation_minus_outcome": {
"ci95": [
-0.026383192545085435,
0.0607951286646285
],
"point": 0.019228316404518567
},
"replicates": 10000,
"seed": 20260714,
"semantics": "group bootstrap over cells; diagnostic confidence interval"
},
"paired_correctness": {
"both_correct": 30,
"both_wrong": 4,
"instrumentation_only_correct": 0,
"mcnemar_exact_two_sided_p": 0.25,
"sim_outcome_only_correct": 3
},
"sim_plus_outcome": {
"classification": {
"accuracy": 0.8918918918918919,
"balanced_accuracy": 0.8154761904761905,
"brier": 0.10990776306815446,
"confusion": {
"false_negative": 1,
"false_positive": 3,
"true_negative": 6,
"true_positive": 27
},
"log_loss": 0.328357763455984
},
"policy_0p95": {
"abstain_continue_full": 12,
"correctly_saved_h20_hours": 0.7402314096841667,
"decision_coverage": 0.6756756756756757,
"early_accept": 20,
"early_reject": 5,
"false_accept": 0,
"false_accept_examples": [],
"false_reject": 0,
"false_reject_examples": [],
"full_trial_h20_hours": 1.0669595034675,
"invalidly_saved_h20_hours": 0.0,
"remaining_h20_hours_at_cutoff": 0.957237281245278,
"saved_h20_hours_if_decisions_used": 0.7402314096841667,
"threshold": 0.95,
"valid_cost_reduction_fraction": 0.6937764809990414,
"valid_zero_error_policy": true
}
},
"sim_plus_outcome_plus_instrumentation": {
"classification": {
"accuracy": 0.8108108108108109,
"balanced_accuracy": 0.7619047619047619,
"brier": 0.12913607947267303,
"confusion": {
"false_negative": 4,
"false_positive": 3,
"true_negative": 6,
"true_positive": 24
},
"log_loss": 0.4373556318820343
},
"policy_0p95": {
"abstain_continue_full": 9,
"correctly_saved_h20_hours": 0.7469523484622221,
"decision_coverage": 0.7567567567567568,
"early_accept": 22,
"early_reject": 6,
"false_accept": 2,
"false_accept_examples": [
{
"anchor": 0.49609375,
"cell": "tp2_mns8",
"label_feasible": false,
"probability_feasible": 0.9869795738005246,
"remaining_h20_hours": 0.010117910306111111
},
{
"anchor": 0.033717411016,
"cell": "tp4_mns16",
"label_feasible": false,
"probability_feasible": 0.9855364057197005,
"remaining_h20_hours": 0.023106262014444445
}
],
"false_reject": 0,
"false_reject_examples": [],
"full_trial_h20_hours": 1.0669595034675,
"invalidly_saved_h20_hours": 0.03322417232055556,
"remaining_h20_hours_at_cutoff": 0.957237281245278,
"saved_h20_hours_if_decisions_used": 0.7801765207827777,
"threshold": 0.95,
"valid_cost_reduction_fraction": null,
"valid_zero_error_policy": false
}
}
},
"1.0": {
"group_bootstrap": {
"accuracy_delta_instrumentation_minus_outcome": {
"ci95": [
0.0,
0.18181818181818188
],
"point": 0.08108108108108103
},
"brier_delta_instrumentation_minus_outcome": {
"ci95": [
-0.04292727744470806,
0.019924730979981074
],
"point": -0.010145365131402809
},
"replicates": 10000,
"seed": 20260714,
"semantics": "group bootstrap over cells; diagnostic confidence interval"
},
"paired_correctness": {
"both_correct": 30,
"both_wrong": 4,
"instrumentation_only_correct": 3,
"mcnemar_exact_two_sided_p": 0.25,
"sim_outcome_only_correct": 0
},
"sim_plus_outcome": {
"classification": {
"accuracy": 0.8108108108108109,
"balanced_accuracy": 0.7242063492063493,
"brier": 0.1058226346682949,
"confusion": {
"false_negative": 3,
"false_positive": 4,
"true_negative": 5,
"true_positive": 25
},
"log_loss": 0.3011048455679668
},
"policy_0p95": {
"abstain_continue_full": 17,
"correctly_saved_h20_hours": 0.5429431818208333,
"decision_coverage": 0.5405405405405406,
"early_accept": 16,
"early_reject": 4,
"false_accept": 0,
"false_accept_examples": [],
"false_reject": 0,
"false_reject_examples": [],
"full_trial_h20_hours": 1.0669595034675,
"invalidly_saved_h20_hours": 0.0,
"remaining_h20_hours_at_cutoff": 0.957237281245278,
"saved_h20_hours_if_decisions_used": 0.5429431818208333,
"threshold": 0.95,
"valid_cost_reduction_fraction": 0.5088695307144538,
"valid_zero_error_policy": true
}
},
"sim_plus_outcome_plus_instrumentation": {
"classification": {
"accuracy": 0.8918918918918919,
"balanced_accuracy": 0.8154761904761905,
"brier": 0.0956772695368921,
"confusion": {
"false_negative": 1,
"false_positive": 3,
"true_negative": 6,
"true_positive": 27
},
"log_loss": 0.288823031828762
},
"policy_0p95": {
"abstain_continue_full": 12,
"correctly_saved_h20_hours": 0.7360063646722222,
"decision_coverage": 0.6756756756756757,
"early_accept": 20,
"early_reject": 5,
"false_accept": 0,
"false_accept_examples": [],
"false_reject": 0,
"false_reject_examples": [],
"full_trial_h20_hours": 1.0669595034675,
"invalidly_saved_h20_hours": 0.0,
"remaining_h20_hours_at_cutoff": 0.957237281245278,
"saved_h20_hours_if_decisions_used": 0.7360063646722222,
"threshold": 0.95,
"valid_cost_reduction_fraction": 0.6898165884274738,
"valid_zero_error_policy": true
}
}
},
"10.0": {
"group_bootstrap": {
"accuracy_delta_instrumentation_minus_outcome": {
"ci95": [
-0.13333333333333341,
0.05555555555555558
],
"point": -0.027027027027027084
},
"brier_delta_instrumentation_minus_outcome": {
"ci95": [
-0.03091105649870874,
0.01684192005239855
],
"point": -0.007318433328714388
},
"replicates": 10000,
"seed": 20260714,
"semantics": "group bootstrap over cells; diagnostic confidence interval"
},
"paired_correctness": {
"both_correct": 30,
"both_wrong": 4,
"instrumentation_only_correct": 1,
"mcnemar_exact_two_sided_p": 1.0,
"sim_outcome_only_correct": 2
},
"sim_plus_outcome": {
"classification": {
"accuracy": 0.8648648648648649,
"balanced_accuracy": 0.7222222222222222,
"brier": 0.10613344425735322,
"confusion": {
"false_negative": 0,
"false_positive": 5,
"true_negative": 4,
"true_positive": 28
},
"log_loss": 0.3404203142465075
},
"policy_0p95": {
"abstain_continue_full": 32,
"correctly_saved_h20_hours": 0.21727432337249997,
"decision_coverage": 0.13513513513513514,
"early_accept": 5,
"early_reject": 0,
"false_accept": 0,
"false_accept_examples": [],
"false_reject": 0,
"false_reject_examples": [],
"full_trial_h20_hours": 1.0669595034675,
"invalidly_saved_h20_hours": 0.0,
"remaining_h20_hours_at_cutoff": 0.957237281245278,
"saved_h20_hours_if_decisions_used": 0.21727432337249997,
"threshold": 0.95,
"valid_cost_reduction_fraction": 0.20363877229302757,
"valid_zero_error_policy": true
}
},
"sim_plus_outcome_plus_instrumentation": {
"classification": {
"accuracy": 0.8378378378378378,
"balanced_accuracy": 0.7420634920634921,
"brier": 0.09881501092863883,
"confusion": {
"false_negative": 2,
"false_positive": 4,
"true_negative": 5,
"true_positive": 26
},
"log_loss": 0.312914193285738
},
"policy_0p95": {
"abstain_continue_full": 30,
"correctly_saved_h20_hours": 0.2384080185036111,
"decision_coverage": 0.1891891891891892,
"early_accept": 6,
"early_reject": 1,
"false_accept": 0,
"false_accept_examples": [],
"false_reject": 0,
"false_reject_examples": [],
"full_trial_h20_hours": 1.0669595034675,
"invalidly_saved_h20_hours": 0.0,
"remaining_h20_hours_at_cutoff": 0.957237281245278,
"saved_h20_hours_if_decisions_used": 0.2384080185036111,
"threshold": 0.95,
"valid_cost_reduction_fraction": 0.22344617366339725,
"valid_zero_error_policy": true
}
}
}
},
"sanity": {
"examples": {
"distinct_n": 1,
"max": 1.0,
"min": 1.0,
"n": 37
},
"frozen_simulator_runs": 92,
"invariants": {
"all_examples_matched_once": true,
"labels_not_identical": true,
"per_config_results_not_all_identical": true,
"same_nested_folds": true,
"simulator_ratios_bounded": true
},
"labels": {
"distinct_n": 2,
"max": 1.0,
"min": 0.0,
"n": 37,
"negative": 9,
"positive": 28
},
"matched_simulator_pass_rate": {
"distinct_n": 12,
"max": 1.0,
"min": 0.06884057971014493,
"n": 37
},
"red_flags": []
},
"schema": "fidelity-strong-baseline-v1",
"scope": "retrospective one-task headroom audit; not contribution evidence",
"status": "PASS"
}

View File

@@ -0,0 +1,50 @@
#!/usr/bin/env python3
from __future__ import annotations
from analyze_pilot_e2e import expanded_top_k, replay
def candidate(
cell: str,
sim_score: float,
real_feasible: bool,
probability: float,
) -> dict[str, object]:
return {
"cell": cell,
"level": "high",
"sim_throughput_req_s_per_gpu": sim_score,
"real_goodput_req_s_per_gpu": sim_score,
"real_feasible": real_feasible,
"setup_h20_hours": 0.1,
"full_trial_h20_hours": 0.05,
"prefix_h20_hours": 0.01,
"instrument_probability": probability,
}
def main() -> None:
candidates = [
candidate("a", 3.0, True, 0.5),
candidate("b", 2.0, False, 0.01),
candidate("c", 2.0, True, 0.99),
]
shortlist = expanded_top_k(candidates, 2)
assert [item["cell"] for item in shortlist] == ["a", "b", "c"]
result = replay(
shortlist,
probability_key="instrument_probability",
oracle_goodput=3.0,
common_failure_h20_hours=0.02,
)
assert result["selected_cell"] == "a"
assert result["false_accept"] == 0
assert result["false_reject"] == 0
assert result["early_accept"] == 1
assert result["early_reject"] == 1
assert result["online_h20_hours"] > 0
print("fidelity pilot e2e: PASS")
if __name__ == "__main__":
main()

View File

@@ -14,6 +14,7 @@ sys.path.insert(0, str(HERE))
import pilot_controller as controller # noqa: E402
import prepare_pilot as prepare # noqa: E402
from analyze_pilot import campaign_gpu_accounting # noqa: E402
@dataclass
@@ -69,6 +70,32 @@ def main() -> None:
assert row["fidelity_pilot_band"] == role
assert abs(float(row["sampling_u"]) - 0.5) < 1e-12
prior = root / "prior-state.json"
primary = root / "primary-state.json"
prior.write_text(
json.dumps(
{
"status": "failed",
"gpu_hours_total": 0.02,
"hard_cap_h20_hours": 3.5,
}
),
encoding="utf-8",
)
primary.write_text(
json.dumps(
{
"status": "complete",
"gpu_hours_total": 1.5,
"hard_cap_h20_hours": 3.5,
}
),
encoding="utf-8",
)
accounting = campaign_gpu_accounting(primary, (prior,))
assert math.isclose(accounting["aggregate_h20_hours"], 1.52)
assert all(accounting["invariants"].values())
assert len(controller.ORDER) == 6
assert set(controller.ORDER) == set(prepare.CELLS)
assert math.isclose(

View File

@@ -0,0 +1,37 @@
#!/usr/bin/env python3
from __future__ import annotations
import json
from pathlib import Path
from analyze_strong_baseline import analyze
ROOT = Path(__file__).resolve().parents[2]
REPLAYSERVE = ROOT.parent / "replayserve"
def main() -> None:
result = analyze(
ROOT / "runs/opprof-phase6/phase6/metrics.json",
ROOT / "runs/opprof-phase6/phase6/solo-authoritative/cells",
REPLAYSERVE / "runs/simfid_s2rb/results/raw",
REPLAYSERVE / "runs/simfid_s2rb/results/metrics.json",
)
assert result["status"] == "PASS", json.dumps(result["sanity"], indent=2)
assert result["sanity"]["frozen_simulator_runs"] == 92
assert result["sanity"]["labels"]["n"] == 37
headline = result["headline"]
assert headline["sim_plus_outcome"]["policy_0p95"]["false_accept"] == 0
assert headline["sim_plus_outcome"]["policy_0p95"]["false_reject"] == 0
assert (
headline["sim_plus_outcome_plus_instrumentation"]["policy_0p95"][
"false_accept"
]
== 0
)
print("fidelity strong baseline: PASS")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,100 @@
#!/usr/bin/env python3
from __future__ import annotations
import json
import tempfile
from pathlib import Path
import numpy as np
from analyze_prefixes import PrefixExample
from prepare_pilot_simulator import load_module as load_prepare_module
from run_pilot_simulator import load_module as load_run_module
from analyze_strong_pilot import (
covariate_shift,
fit_model,
load_pilot_simulator,
predict_model,
)
def example(index: int) -> PrefixExample:
label = int(index >= 4)
return PrefixExample(
cell=f"cell-{index // 2}",
anchor=float(index),
cutoff_s=5.0,
tp=1,
full_elapsed_s=10.0,
feasible=label,
primary_feasible=label,
outcome=tuple(float(index + offset) for offset in range(13)),
instrumentation=tuple(float(index * offset + 1) for offset in range(17)),
completion_time_source="exact_monotonic",
)
def main() -> None:
examples = [example(index) for index in range(8)]
simulator = [(float(index), index / 10.0, float(index >= 4)) for index in range(8)]
for instrumentation_aware in (False, True):
model = fit_model(
examples,
simulator,
instrumentation_aware=instrumentation_aware,
regularization=1.0,
)
probability = predict_model(model, examples, simulator)
assert probability.shape == (8,)
assert np.all((probability >= 0.0) & (probability <= 1.0))
shift = covariate_shift(
examples,
simulator,
examples,
simulator,
instrumentation_aware=instrumentation_aware,
)
assert shift["values"]["min"] >= 0.0
assert shift["count_gt_3"] == 0
payload = {
"status": "PASS",
"results": [
{
"cell": f"cell-{index // 2}",
"role": "low1" if index % 2 == 0 else "high1",
"scorer": {
"throughput_requests_per_second_per_gpu": 1.0 + index,
"slo": {
"pass_rate": index / 12.0,
"feasible": index % 2 == 0,
},
},
}
for index in range(12)
],
}
with tempfile.TemporaryDirectory() as temporary:
path = Path(temporary) / "metrics.json"
path.write_text(json.dumps(payload), encoding="utf-8")
features, red_flags = load_pilot_simulator(path)
assert len(features) == 12
assert red_flags == []
with tempfile.TemporaryDirectory() as temporary:
root = Path(temporary)
(root / "prepare_dependency.py").write_text("VALUE = 17\n", encoding="utf-8")
(root / "prepare_target.py").write_text(
"from prepare_dependency import VALUE\n", encoding="utf-8"
)
assert load_prepare_module(root / "prepare_target.py").VALUE == 17
(root / "run_dependency.py").write_text("VALUE = 23\n", encoding="utf-8")
(root / "run_target.py").write_text(
"from run_dependency import VALUE\n", encoding="utf-8"
)
assert load_run_module("run_target", root / "run_target.py").VALUE == 23
print("fidelity strong pilot: PASS")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,4 @@
__pycache__/
cache/
replay/
figure-prototype.svg

View File

@@ -0,0 +1,249 @@
From 1f8900a4ac64e45754b03d0aa7c1dddab65785cf Mon Sep 17 00:00:00 2001
From: Gahow Wang <gahow.wang@gmail.com>
Date: Thu, 23 Jul 2026 15:27:40 +0800
Subject: [PATCH] Experiment with structured attention prefill predictor
---
.../shared_prediction_model_manager.py | 16 +++-
.../sklearn_execution_time_predictor.py | 16 +++-
.../structured_attention_prefill.py | 79 +++++++++++++++++++
.../unit/test_structured_attention_prefill.py | 59 ++++++++++++++
4 files changed, 165 insertions(+), 5 deletions(-)
create mode 100644 frontier/execution_time_predictor/structured_attention_prefill.py
create mode 100644 tests/unit/test_structured_attention_prefill.py
diff --git a/frontier/execution_time_predictor/shared_prediction_model_manager.py b/frontier/execution_time_predictor/shared_prediction_model_manager.py
index 8a65a49..4a21165 100644
--- a/frontier/execution_time_predictor/shared_prediction_model_manager.py
+++ b/frontier/execution_time_predictor/shared_prediction_model_manager.py
@@ -19,6 +19,9 @@ from frontier.execution_time_predictor.attention_tp_policy import (
from frontier.execution_time_predictor.attention_dataset_contract import (
enforce_mixed_attention_input_contract,
)
+from frontier.execution_time_predictor.structured_attention_prefill import (
+ StructuredAttentionPrefillRegressor,
+)
from frontier.logger import init_logger
from frontier.moe_gating_runtime import (
DEFAULT_MOE_GATING_RUNTIME_CONTEXT,
@@ -1254,7 +1257,10 @@ class ExecutionTimePredictionModelManager:
raise ValueError(
"Missing required column 'prefill_chunk_size' in attention profiling data."
)
- standard_prefill_df = prefill_df[prefill_df["prefill_chunk_size"] > 0].copy()
+ standard_prefill_df = prefill_df[
+ (prefill_df["prefill_chunk_size"] > 0)
+ & (prefill_df["batch_size"] == 1)
+ ].copy()
prefill_model_signature = f"attn_prefill_{attention_signature}"
if prefill_model_signature not in trained_model_signatures:
@@ -1742,7 +1748,13 @@ class ExecutionTimePredictionModelManager:
# initialization to generate missing cache files.
# ============================================================
- estimator, grid_search_params = self._create_estimator_and_params(execution_time_predictor_config)
+ if model_name == "attn_prefill":
+ estimator = StructuredAttentionPrefillRegressor()
+ grid_search_params = {}
+ else:
+ estimator, grid_search_params = self._create_estimator_and_params(
+ execution_time_predictor_config
+ )
cv = min(execution_time_predictor_config.k_fold_cv_splits, len(df)) if len(df) >= 2 else 2
diff --git a/frontier/execution_time_predictor/sklearn_execution_time_predictor.py b/frontier/execution_time_predictor/sklearn_execution_time_predictor.py
index 27b62bf..b7f5350 100644
--- a/frontier/execution_time_predictor/sklearn_execution_time_predictor.py
+++ b/frontier/execution_time_predictor/sklearn_execution_time_predictor.py
@@ -45,6 +45,9 @@ from frontier.execution_time_predictor.attention_tp_policy import (
from frontier.execution_time_predictor.attention_dataset_contract import (
enforce_mixed_attention_input_contract,
)
+from frontier.execution_time_predictor.structured_attention_prefill import (
+ StructuredAttentionPrefillRegressor,
+)
from frontier.logger import init_logger
from frontier.moe_gating_runtime import get_moe_gating_base_model_name
from frontier.profiling.cpu_overhead.schema import (
@@ -2573,8 +2576,12 @@ class SklearnExecutionTimePredictor(BaseExecutionTimePredictor):
if cached_model:
return cached_model
- model = self._get_estimator()
- grid_search_params = self._get_grid_search_params()
+ if model_name == "attn_prefill":
+ model = StructuredAttentionPrefillRegressor()
+ grid_search_params = {}
+ else:
+ model = self._get_estimator()
+ grid_search_params = self._get_grid_search_params()
if len(df) < self._config.k_fold_cv_splits:
cv = 2
@@ -2869,7 +2876,10 @@ class SklearnExecutionTimePredictor(BaseExecutionTimePredictor):
raise ValueError(
"Missing required column 'prefill_chunk_size' in attention profiling data."
)
- standard_prefill_df = prefill_df[prefill_df["prefill_chunk_size"] > 0].copy()
+ standard_prefill_df = prefill_df[
+ (prefill_df["prefill_chunk_size"] > 0)
+ & (prefill_df["batch_size"] == 1)
+ ].copy()
if len(standard_prefill_df) == 0:
raise ValueError(
"No standard prefill rows (prefill_chunk_size > 0) found in eager attention profiling data."
diff --git a/frontier/execution_time_predictor/structured_attention_prefill.py b/frontier/execution_time_predictor/structured_attention_prefill.py
new file mode 100644
index 0000000..1829047
--- /dev/null
+++ b/frontier/execution_time_predictor/structured_attention_prefill.py
@@ -0,0 +1,79 @@
+"""Structured latency model for single-request chunked prefill attention."""
+
+from typing import Any
+
+import numpy as np
+from sklearn.base import BaseEstimator, RegressorMixin
+from sklearn.isotonic import IsotonicRegression
+from sklearn.linear_model import LinearRegression
+
+
+class StructuredAttentionPrefillRegressor(RegressorMixin, BaseEstimator):
+ """Model attention as a monotone base curve plus continuous KV growth.
+
+ Input columns retain the existing Frontier contract:
+ ``[kv_cache_size, prefill_chunk_size_squared]``.
+ """
+
+ def fit(self, X: Any, y: Any) -> "StructuredAttentionPrefillRegressor":
+ values = self._as_feature_array(X)
+ target = np.asarray(y, dtype=float)
+ kv_cache_size = values[:, 0]
+ prefill_chunk_size = np.sqrt(np.maximum(values[:, 1], 0.0))
+
+ base_mask = np.isclose(kv_cache_size, 0.0)
+ growth_mask = kv_cache_size > 0.0
+ if not np.any(base_mask) or not np.any(growth_mask):
+ raise ValueError(
+ "structured attn_prefill training requires both KV=0 base rows "
+ "and KV>0 growth rows"
+ )
+
+ base_q = prefill_chunk_size[base_mask]
+ base_y = target[base_mask]
+ unique_q = np.unique(base_q)
+ grouped_y = np.asarray(
+ [np.mean(base_y[np.isclose(base_q, q)]) for q in unique_q],
+ dtype=float,
+ )
+ self._base_model = IsotonicRegression(
+ increasing=True,
+ out_of_bounds="clip",
+ ).fit(unique_q, grouped_y)
+
+ growth_q = prefill_chunk_size[growth_mask]
+ growth_kv = kv_cache_size[growth_mask]
+ growth_base = self._base_model.predict(growth_q)
+ growth_features = np.column_stack(
+ (growth_kv, growth_q * growth_kv)
+ )
+ self._growth_model = LinearRegression(
+ fit_intercept=False,
+ positive=True,
+ ).fit(growth_features, target[growth_mask] - growth_base)
+
+ self.n_features_in_ = 2
+ self._frontier_base_q_min = float(unique_q.min())
+ self._frontier_base_q_max = float(unique_q.max())
+ self._frontier_growth_kv_max = float(growth_kv.max())
+ return self
+
+ def predict(self, X: Any) -> np.ndarray:
+ values = self._as_feature_array(X)
+ kv_cache_size = values[:, 0]
+ prefill_chunk_size = np.sqrt(np.maximum(values[:, 1], 0.0))
+ base = self._base_model.predict(prefill_chunk_size)
+ growth_features = np.column_stack(
+ (kv_cache_size, prefill_chunk_size * kv_cache_size)
+ )
+ return np.maximum(base + self._growth_model.predict(growth_features), 0.0)
+
+ @staticmethod
+ def _as_feature_array(X: Any) -> np.ndarray:
+ values = np.asarray(X, dtype=float)
+ if values.ndim != 2 or values.shape[1] != 2:
+ raise ValueError(
+ "structured attn_prefill expects exactly two features: "
+ "kv_cache_size and prefill_chunk_size_squared"
+ )
+ return values
diff --git a/tests/unit/test_structured_attention_prefill.py b/tests/unit/test_structured_attention_prefill.py
new file mode 100644
index 0000000..12c4247
--- /dev/null
+++ b/tests/unit/test_structured_attention_prefill.py
@@ -0,0 +1,59 @@
+import pickle
+import unittest
+
+import numpy as np
+
+from frontier.execution_time_predictor.structured_attention_prefill import (
+ StructuredAttentionPrefillRegressor,
+)
+
+
+class StructuredAttentionPrefillRegressorTest(unittest.TestCase):
+ def setUp(self) -> None:
+ q = np.asarray([64, 128, 256, 512, 1024, 2048, 4096, 8192], dtype=float)
+ base = 0.05 + 1e-4 * q + 4e-8 * q**2
+ context_q = np.asarray([2048, 4096, 8192] * 3, dtype=float)
+ context_kv = np.repeat([8192, 16384, 24576], 3).astype(float)
+ context_y = (
+ np.interp(context_q, q, base)
+ + 1.5e-5 * context_kv
+ + 3e-8 * context_q * context_kv
+ )
+ self.X = np.column_stack(
+ (
+ np.concatenate((np.zeros_like(q), context_kv)),
+ np.concatenate((q**2, context_q**2)),
+ )
+ )
+ self.y = np.concatenate((base, context_y))
+
+ def test_recovers_structured_curve(self) -> None:
+ model = StructuredAttentionPrefillRegressor().fit(self.X, self.y)
+ np.testing.assert_allclose(model.predict(self.X), self.y, rtol=1e-6)
+
+ def test_prediction_is_nonnegative_and_monotone(self) -> None:
+ model = StructuredAttentionPrefillRegressor().fit(self.X, self.y)
+ q = np.arange(1, 8193, dtype=float)
+ for kv in (0, 8192, 32768, 40912):
+ X = np.column_stack((np.full_like(q, kv), q**2))
+ prediction = model.predict(X)
+ self.assertTrue(np.all(prediction >= 0))
+ self.assertTrue(np.all(np.diff(prediction) >= -1e-12))
+
+ kv = np.arange(0, 40913, 64, dtype=float)
+ for q_value in (64, 2048, 8192):
+ X = np.column_stack((kv, np.full_like(kv, q_value**2)))
+ self.assertTrue(np.all(np.diff(model.predict(X)) >= -1e-12))
+
+ def test_pickle_round_trip(self) -> None:
+ model = StructuredAttentionPrefillRegressor().fit(self.X, self.y)
+ restored = pickle.loads(pickle.dumps(model))
+ np.testing.assert_allclose(restored.predict(self.X), self.y, rtol=1e-6)
+
+ def test_requires_base_and_growth_rows(self) -> None:
+ with self.assertRaisesRegex(ValueError, "KV=0 base rows"):
+ StructuredAttentionPrefillRegressor().fit(self.X[:8], self.y[:8])
+
+
+if __name__ == "__main__":
+ unittest.main()
--
2.43.0

View File

@@ -0,0 +1,256 @@
#!/usr/bin/env python3
"""Offline predictor ablation for EXP-ATTN-STRUCTURED.
This is deliberately profile-only: it decides whether the structured model is
good enough to justify the expensive 7-cell trace replay.
"""
from __future__ import annotations
import argparse
import csv
import json
import sys
from pathlib import Path
from typing import Any
import numpy as np
import pandas as pd
from sklearn.ensemble import RandomForestRegressor
ROOT = Path(__file__).resolve().parent
REPO = ROOT.parents[1]
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser()
parser.add_argument(
"--profile",
type=Path,
default=REPO
/ "runs/frontier-prefill-kvgrowth-fix-v0/profiles/"
"profile-v5-kvgrowth/attention.csv",
)
parser.add_argument(
"--frontier-checkout",
type=Path,
default=Path("/tmp/frontier-attn-structured-v0"),
)
parser.add_argument("--output-root", type=Path, default=ROOT / "results")
return parser.parse_args()
def normalize_bool(series: pd.Series) -> pd.Series:
return series.astype(str).str.strip().str.lower().isin(
{"1", "true", "t", "yes", "y"}
)
def load_profile(path: Path) -> pd.DataFrame:
df = pd.read_csv(path).drop_duplicates()
for column in ("is_prefill", "is_true_mixed_batch"):
df[column] = normalize_bool(df[column])
df = df[
(df["n_embd"] == 2048)
& (df["n_q_head"] == 32)
& (df["n_kv_head"] == 4)
& (df["block_size"] == 16)
& df["is_prefill"]
& ~df["is_true_mixed_batch"]
& (df["prefill_chunk_size"] > 0)
].copy()
df["prefill_chunk_size_squared"] = df["prefill_chunk_size"] ** 2
return df
def mape(actual: np.ndarray, predicted: np.ndarray) -> float:
return float(np.mean(np.abs((predicted - actual) / actual)) * 100)
def make_rf() -> RandomForestRegressor:
# Exact best parameters selected by the current profile-v5 GridSearchCV.
return RandomForestRegressor(
random_state=0,
n_estimators=250,
max_depth=8,
min_samples_split=2,
)
def features(df: pd.DataFrame) -> pd.DataFrame:
return df[["kv_cache_size", "prefill_chunk_size_squared"]]
def score_model(
name: str,
estimator: Any,
train: pd.DataFrame,
single: pd.DataFrame,
grid: pd.DataFrame,
) -> dict[str, Any]:
target = "time_stats.attn_prefill.median"
estimator.fit(features(train), train[target])
grid_prediction = estimator.predict(features(grid))
single_prediction = estimator.predict(features(single))
heldout_actual: list[float] = []
heldout_prediction: list[float] = []
for context in sorted(grid["kv_cache_size"].unique()):
test = grid[grid["kv_cache_size"] == context]
fold_train = train.drop(index=test.index, errors="ignore")
fold_model = (
make_rf()
if name.startswith("rf")
else estimator.__class__()
)
fold_model.fit(features(fold_train), fold_train[target])
heldout_actual.extend(test[target].astype(float))
heldout_prediction.extend(fold_model.predict(features(test)))
q = np.arange(1, 8193, dtype=float)
q_deltas: list[float] = []
prediction_min: list[float] = []
for context in (0, 8192, 16384, 24576, 32768, 40912):
X = pd.DataFrame(
{
"kv_cache_size": np.full_like(q, context),
"prefill_chunk_size_squared": q**2,
}
)
prediction = estimator.predict(X)
prediction_min.append(float(prediction.min()))
q_deltas.append(float(np.diff(prediction).min()))
kv = np.arange(0, 40913, 64, dtype=float)
kv_deltas: list[float] = []
for query in (64, 512, 2048, 4096, 8192):
X = pd.DataFrame(
{
"kv_cache_size": kv,
"prefill_chunk_size_squared": np.full_like(kv, query**2),
}
)
kv_deltas.append(float(np.diff(estimator.predict(X)).min()))
heldout_actual_array = np.asarray(heldout_actual)
heldout_prediction_array = np.asarray(heldout_prediction)
return {
"candidate": name,
"training_rows": len(train),
"grid_fit_mape_pct": mape(
grid[target].to_numpy(), np.asarray(grid_prediction)
),
"single_fit_mape_pct": mape(
single[target].to_numpy(), np.asarray(single_prediction)
),
"heldout_context_mape_pct": mape(
heldout_actual_array, heldout_prediction_array
),
"heldout_context_max_abs_error_pct": float(
np.max(
np.abs(
(heldout_prediction_array - heldout_actual_array)
/ heldout_actual_array
)
)
* 100
),
"prediction_min_ms": min(prediction_min),
"q_min_delta_ms": min(q_deltas),
"kv_min_delta_ms": min(kv_deltas),
"monotone_and_nonnegative": (
min(prediction_min) >= 0
and min(q_deltas) >= -1e-12
and min(kv_deltas) >= -1e-12
),
}
def main() -> None:
args = parse_args()
sys.path.insert(0, str(args.frontier_checkout))
from frontier.execution_time_predictor.structured_attention_prefill import (
StructuredAttentionPrefillRegressor,
)
df = load_profile(args.profile)
records: list[dict[str, Any]] = []
data_audit: dict[str, Any] = {}
for tp in (1, 2, 4):
tp_df = df[df["num_tensor_parallel_workers"] == tp].copy()
single = tp_df[tp_df["batch_size"] == 1].copy()
grid = single[
single["prefill_chunk_size"].isin((2048, 4096, 8192))
& (single["kv_cache_size"] > 0)
].copy()
duplicate_groups = (
tp_df.groupby(
["kv_cache_size", "prefill_chunk_size_squared"]
)
.size()
.gt(1)
.sum()
)
data_audit[f"tp{tp}"] = {
"standard_rows": len(tp_df),
"single_request_rows": len(single),
"target_grid_rows": len(grid),
"duplicate_feature_groups": int(duplicate_groups),
}
candidates = (
("rf_all", make_rf(), tp_df),
("rf_single", make_rf(), single),
(
"structured_single",
StructuredAttentionPrefillRegressor(),
single,
),
)
for name, model, train in candidates:
result = score_model(name, model, train, single, grid)
result["tp"] = tp
records.append(result)
structured = [r for r in records if r["candidate"] == "structured_single"]
checks = {
"heldout_context_mape_le_5pct": all(
r["heldout_context_mape_pct"] <= 5 for r in structured
),
"monotone_and_nonnegative": all(
r["monotone_and_nonnegative"] for r in structured
),
}
checks["profile_gate"] = all(checks.values())
payload = {
"schema": "frontier-attn-structured-ablation-v1",
"profile": str(args.profile.resolve()),
"frontier_checkout": str(args.frontier_checkout.resolve()),
"data_audit": data_audit,
"results": records,
"checks": checks,
}
args.output_root.mkdir(parents=True, exist_ok=True)
(args.output_root / "predictor-ablation.json").write_text(
json.dumps(payload, indent=2)
)
with (args.output_root / "predictor-ablation.csv").open(
"w", newline=""
) as stream:
writer = csv.DictWriter(stream, fieldnames=list(records[0]))
writer.writeheader()
writer.writerows(records)
print(json.dumps(checks, indent=2))
for row in records:
print(
f"TP{row['tp']} {row['candidate']:18s} "
f"grid={row['grid_fit_mape_pct']:.2f}% "
f"heldout={row['heldout_context_mape_pct']:.2f}% "
f"max={row['heldout_context_max_abs_error_pct']:.2f}% "
f"monotone={row['monotone_and_nonnegative']}"
)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,304 @@
#!/usr/bin/env python3
"""Trial-aware verdict for the seven structured-attention trace replays."""
from __future__ import annotations
import csv
import json
import math
from pathlib import Path
from typing import Any
ROOT = Path(__file__).resolve().parent
REPO = ROOT.parents[1]
S3_REAL = REPO / "runs/frontier-s3-real-v0"
V5 = REPO / "runs/frontier-prefill-kvgrowth-fix-v0"
CELLS = {
"tp1_rho0p00125": {
"real": "frontier-tp1-real-r0p00125-t*",
"old": V5 / "sim-replay-tp1/v5/tp1_rho0p00125",
},
"tp1_rho0p0025": {
"real": "frontier-tp1-real-r0p0025-t*",
"old": V5 / "sim-replay-tp1/v5/tp1_rho0p0025",
},
"tp2_rho0p0025": {
"real": "frontier-s3-real-full-r0p0025-tp2-t*",
"old": V5 / "sim-replay/tp2_rho0p0025",
},
"tp2_rho0p005": {
"real": "frontier-s3-real-full-r0p005-tp2-t*",
"old": V5 / "sim-replay/tp2_rho0p005",
},
"tp4_rho0p0025": {
"real": "frontier-s3-real-full-r0p0025-tp4-t*",
"old": V5 / "sim-replay/tp4_rho0p0025",
},
"tp4_rho0p005": {
"real": "frontier-s3-real-full-r0p005-tp4-t*",
"old": V5 / "sim-replay/tp4_rho0p005",
},
"tp4_rho0p01": {
"real": "frontier-s3-real-full-r0p01-tp4-t*",
"old": V5 / "sim-replay/tp4_rho0p01",
},
}
METRICS = {
"ttft": ("ttft_ms", "ttft"),
"tpot": ("tpot_ms", "tpot"),
"e2e": ("e2e_ms", "request_e2e_time"),
}
QUANTILES = {"mean": None, "p50": 0.5, "p90": 0.9, "p99": 0.99}
def percentile(values: list[float], quantile: float) -> float:
ordered = sorted(values)
position = (len(ordered) - 1) * quantile
lower, upper = math.floor(position), math.ceil(position)
if lower == upper:
return ordered[lower]
return (
ordered[lower] * (upper - position)
+ ordered[upper] * (position - lower)
)
def summarize(values: list[float]) -> dict[str, float]:
return {
name: (
sum(values) / len(values)
if quantile is None
else percentile(values, quantile)
)
for name, quantile in QUANTILES.items()
}
def load_real_trials(pattern: str) -> list[list[dict[str, Any]]]:
trials = []
for run_root in sorted((S3_REAL / "fleet-artifacts").glob(pattern)):
results = list(
run_root.glob(
"artifacts/outputs/full-real/*/*/trial-*/results/result.json"
)
)
if len(results) != 1:
raise ValueError(f"expected one result in {run_root}, got {results}")
trials.append(json.loads(results[0].read_text())["requests"])
if len(trials) != 2:
raise ValueError(f"expected two real trials for {pattern}, got {len(trials)}")
return trials
def load_sim(root: Path) -> list[dict[str, str]]:
matches = list((root / "metrics").rglob("request_metrics.csv"))
if len(matches) != 1:
raise ValueError(f"expected one request_metrics.csv below {root}: {matches}")
rows = list(csv.DictReader(matches[0].open()))
rows.sort(key=lambda row: int(float(row["Request Id"])))
return rows
def distribution_bias(
real_rows: list[dict[str, Any]],
sim_rows: list[dict[str, str]],
) -> dict[str, dict[str, float]]:
output: dict[str, dict[str, float]] = {}
for metric, (real_key, sim_key) in METRICS.items():
pairs = [
(float(real[real_key]), float(sim[sim_key]))
for real, sim in zip(real_rows, sim_rows)
if real.get("success")
]
real_summary = summarize([pair[0] for pair in pairs])
sim_summary = summarize([pair[1] for pair in pairs])
output[metric] = {
name: (sim_summary[name] - real_summary[name]) / real_summary[name]
for name in QUANTILES
}
return output
def paired_relative_error(
real_rows: list[dict[str, Any]],
sim_rows: list[dict[str, str]],
) -> dict[str, dict[str, float]]:
output: dict[str, dict[str, float]] = {}
for metric, (real_key, sim_key) in METRICS.items():
errors = [
(float(sim[sim_key]) - float(real[real_key])) / float(real[real_key])
for real, sim in zip(real_rows, sim_rows)
if real.get("success") and float(real[real_key]) != 0
]
output[metric] = summarize(errors)
return output
def aggregate_trial_bias(
trial_biases: list[dict[str, dict[str, float]]],
) -> dict[str, dict[str, dict[str, float]]]:
return {
metric: {
quantile: {
"mean": sum(values) / len(values),
"min": min(values),
"max": max(values),
}
for quantile in QUANTILES
for values in [
[trial[metric][quantile] for trial in trial_biases]
]
}
for metric in METRICS
}
def legacy_pooled_bias(
real_trials: list[list[dict[str, Any]]],
sim_rows: list[dict[str, str]],
) -> dict[str, dict[str, float]]:
output: dict[str, dict[str, float]] = {}
for metric, (real_key, sim_key) in METRICS.items():
real_values = [
float(row[real_key])
for trial in real_trials
for row in trial[: len(sim_rows)]
if row.get("success")
]
sim_values = [float(row[sim_key]) for row in sim_rows]
real_summary = summarize(real_values)
sim_summary = summarize(sim_values)
output[metric] = {
name: (sim_summary[name] - real_summary[name]) / real_summary[name]
for name in QUANTILES
}
return output
def waiting_p99(sim_rows: list[dict[str, str]]) -> float:
return percentile(
[float(row["request_waiting_time_total"]) for row in sim_rows], 0.99
)
def main() -> None:
results: dict[str, Any] = {}
flat_rows: list[dict[str, Any]] = []
for label, paths in CELLS.items():
real_trials = load_real_trials(paths["real"])
old_sim = load_sim(paths["old"])
new_sim = load_sim(ROOT / "replay" / label)
old_trial_bias = [
distribution_bias(trial, old_sim) for trial in real_trials
]
new_trial_bias = [
distribution_bias(trial, new_sim) for trial in real_trials
]
old_legacy = legacy_pooled_bias(real_trials, old_sim)
new_legacy = legacy_pooled_bias(real_trials, new_sim)
wait_p99 = waiting_p99(new_sim)
results[label] = {
"old": {
"trialwise_distribution_bias": old_trial_bias,
"trialwise_distribution_bias_summary": aggregate_trial_bias(
old_trial_bias
),
"legacy_pooled_distribution_bias": old_legacy,
},
"new": {
"trialwise_distribution_bias": new_trial_bias,
"trialwise_distribution_bias_summary": aggregate_trial_bias(
new_trial_bias
),
"paired_relative_error": [
paired_relative_error(trial, new_sim)
for trial in real_trials
],
"legacy_pooled_distribution_bias": new_legacy,
"waiting_p99_ms": wait_p99,
"validity": (
"PASS_SUBCRITICAL"
if wait_p99 < 1000
else "GATE_FAIL_DIAGNOSTIC"
),
},
}
for metric in METRICS:
for quantile in QUANTILES:
flat_rows.append(
{
"cell": label,
"metric": metric,
"quantile": quantile,
"old_bias": old_legacy[metric][quantile],
"new_bias": new_legacy[metric][quantile],
"abs_bias_delta_pp": 100
* (
abs(new_legacy[metric][quantile])
- abs(old_legacy[metric][quantile])
),
"validity": results[label]["new"]["validity"],
}
)
tp1_checks = []
for cell in ("tp1_rho0p00125", "tp1_rho0p0025"):
for quantile in ("mean", "p99"):
old = results[cell]["old"]["legacy_pooled_distribution_bias"]["ttft"][
quantile
]
new = results[cell]["new"]["legacy_pooled_distribution_bias"]["ttft"][
quantile
]
tp1_checks.append(abs(old) - abs(new) >= 0.05)
regressions = [
row
for row in flat_rows
if row["cell"].startswith(("tp2", "tp4"))
and row["metric"] in ("ttft", "e2e")
and row["abs_bias_delta_pp"] > 5
]
checks = {
"tp1_ttft_mean_p99_improve_ge_5pp": all(tp1_checks),
"tp2_tp4_ttft_e2e_no_abs_regression_gt_5pp": not regressions,
"regressions": regressions,
}
checks["trace_gate"] = (
checks["tp1_ttft_mean_p99_improve_ge_5pp"]
and checks["tp2_tp4_ttft_e2e_no_abs_regression_gt_5pp"]
)
payload = {
"schema": "frontier-attn-structured-trial-aware-verdict-v1",
"metric_note": (
"Primary values are per-real-trial distribution biases with request "
"alignment by index. legacy_pooled reproduces the old milestone "
"quantile convention only for direct comparison."
),
"cells": results,
"checks": checks,
}
output = ROOT / "results"
output.mkdir(parents=True, exist_ok=True)
(output / "trace-verdict.json").write_text(json.dumps(payload, indent=2))
with (output / "trace-verdict.csv").open("w", newline="") as stream:
writer = csv.DictWriter(stream, fieldnames=list(flat_rows[0]))
writer.writeheader()
writer.writerows(flat_rows)
print(json.dumps(checks, indent=2))
for label, result in results.items():
old = result["old"]["legacy_pooled_distribution_bias"]["ttft"]
new = result["new"]["legacy_pooled_distribution_bias"]["ttft"]
print(
f"{label}: TTFT mean {old['mean']:+.1%}->{new['mean']:+.1%}, "
f"p99 {old['p99']:+.1%}->{new['p99']:+.1%}, "
f"waiting_p99={result['new']['waiting_p99_ms']:.0f}ms "
f"{result['new']['validity']}"
)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,82 @@
# 实验 EXP-ATTN-STRUCTURED结构化 predictor 能否关闭大 KV 端的 RF 欠拟合
> **状态:** 已完成profile gate PASSglobal merge gate FAIL
>
> Parent campaign[`../frontier-simulator-gap-campaign-v0/README.md`](../frontier-simulator-gap-campaign-v0/README.md)
## Claim 与决策
- **Parent claim** profile-v5 已补齐 chunked-prefill KV-context 测量,但当前 RF 仍在 TP1/2/4 的新网格上产生约 12%--14% self-fit MAPE并在 TP1 真实 trace 中留下 13% 到 22% TTFT 偏差。
- **目的:** 检查该 residual 是否来自可工程修复的 predictor representation而不是 profile 数据或 serving path。
- **Competing hypotheses**
- H1standard prefill 模型错误混入 pure multi-request rows且 RF 对连续 attention scaling 作阶梯平滑;使用单请求数据和结构化 `base(q)+KV×(a+bq)` 可关闭残余。
- H2残余主要来自未建模的 serving-path 组件;替换 predictor 不会改善 7-cell trace fidelity。
- **事前预测:**
- H1held-out context MAPE ≤5%TP1 TTFT mean/p99 绝对偏差至少改善 5 pp。
- H2profile gate 失败,或 profile gate 通过但 trace TTFT 几乎不动。
- **判定规则:**
- profile gateTP1/2/4 held-out context MAPE 均 ≤5%q/KV 单调且预测非负。
- trace gate两个 TP1 cell 的 TTFT mean/p99 |bias| 各改善 ≥5 ppTP2/TP4 任一 TTFT/E2E quantile 不恶化 >5 pp。
- profile gate 失败即停止trace gate 失败则回退 patch不进入 EXP-2。
## Setup
- **自变量:**
- A现有 RFstandard prefill 全部非 true-mixed rows。
- B现有 RF但仅 `batch_size=1`
- C`batch_size=1` 的 structured predictor
- `base(q)`KV=0 profile 的单调分段线性插值;
- growth非负 least-squares `KV×(a+bq)`
- **控制变量:** attention/linear/MoE/collective profile、trace、prefix cache、scheduler、graph mode、KV blocks、MNS、全部 argv。
- **System context** Qwen3-30B-A3B BF16H20Frontier `deadc4a3`TP1/2/4MNS16chunk 8192prefix caching。
- **Workload 或 trace** 现有 7-cell 60-min production chat trace matrixreal 侧每 cell 两个 trial。
- **Baselines** `docs/assets/frontier-fidelity/full-matrix.csv` 的 sim-v5。
- **Metrics**
- profilegrid fit MAPE、leave-one-context MAPE/max error、q/KV monotonicity
- tracerequest-ID paired bias每个 real trial 单独计算后报告 mean 与 trial interval
- queue validitywaiting p99TP1 超过 1 s 的 cell 标为 diagnostic。
## 预期产物与 review
- **预期数据:** `results/predictor-ablation.{json,csv}``replay/<cell>/``results/paired-verdict.json`
- **Figure prototype** `figure-prototype.png`;左图为 q8k 随 KV 增长的 actual/RF/structured右图为 7-cell TTFT bias 的事前期望。
- **人工 review** 已按 campaign 顺序批准执行。
- **Review 意见:** 只改 standard single-request predictor不得改 mixed predictor 或任何 profile row。
## 复现信息
- **Code** Frontier base `deadc4a321f0baaa534c6ebd17f974123733cdc2`;实验 patch 将保存为 `frontier-structured-attn.patch` 并记录 SHA256。
- **Environment** 本地 CPU replayPython dependency roots 复用 `runs/frontier-collective-joint-v0/counterfactual/joint-r2/manifest.json`
- **产物路径:** 本目录。
- **已知 deviation** milestone 文档将 7-cell 口径称为“逐 request paired”但旧脚本实际 pool 两个 real trial 后比较 quantile本实验会修正分析口径不改旧结果文件。
## 预分析事实
- 现有训练代码使用 `["kv_cache_size", "prefill_chunk_size_squared"]` 与 RF grid search。
- runtime cache 注释明确 standard model 是 per-request多请求 prefill 在模型存在时走 `attn_prefill_mixed`
- profile-v5 的 standard 训练集每 TP 有 29 行,其中单请求 23 行;有 4 组相同 `(KV,q²)` feature 对应多个 pure-batch 标签。
- 初步 structured candidate 的 leave-one-context MAPETP1 0.84%、TP2 1.61%、TP4 3.01%max error 分别 2.04%、3.47%、5.49%。这些是实现前的临时计算,须由版本化脚本复现后才进入结果。
## 结果
- **观察事实:**
- structured held-out-context MAPE 为 TP1/2/4=`0.84%/1.60%/3.01%`
当前 RF 为 `44.40%/44.20%/43.54%`。单调/非负 gate 通过。
- TP1 两点 TTFT mean bias `13.5/17.7% → 6.3/9.5%`p99
`16.8/22.1% → 8.6/14.4%`
- TP2 两点 TTFT mean bias `11.2/14.1% → 4.5/7.1%`p99
`17.3/19.4% → 7.7/9.0%`
- TP4 三点 TTFT mean bias `+2.5/+2.9/0.1% → +7.8/+8.4/+6.0%`
三点均使绝对误差恶化 `5.3--5.8 pp`,触发预设回归 gate。
- validity 重新审计TP1 两点 waiting p99=`1.34/1.89 s`TP2
ρ=.005=`1.17 s`,均标为 `GATE_FAIL_DIAGNOSTIC`。其余四点通过。
- **异常:** TP4 ρ=.005 的 TPOT p99 从 `+31.2%` 变为 `+36.6%`
表明该 tail 对 prefill/mixed-decode 相位敏感,不是本 patch 能关闭的稳定
decode predictor 偏差。
- **含义:** H1 的 representation 机制得到支持,但“全局替换 RF 可直接提升
7-cell fidelity”被反驳。TP4 原先接近零的 mean TTFT 含有 predictor
欠拟合与其它正向 residual 的误差抵消;单独修正 attention 会揭开后者。
- **Claim update** structured predictor 是明确的工程候选,但必须与 TP4
residual 联合收敛后才可 merge当前 patch 只保留为 ablation。
- **下一步:** EXP-2 先重算 structured 分支的 TP2 chunk-level residual
仅 residual ≥10% 才运行 GPU serving-path 三臂 profile。

Binary file not shown.

After

Width:  |  Height:  |  Size: 119 KiB

View File

@@ -0,0 +1,525 @@
{
"cc_cache": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/cc-cache",
"cells": {
"tp1_mns16": {
"argv": [
"/usr/bin/python3",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/run_frontier_with_curves.py",
"--simulation_mode",
"online",
"--sys_arch",
"co-location",
"--cc_backend_config_type",
"vidur",
"--cluster_config_num_replicas",
"1",
"--cluster_scheduler_config_type",
"sticky_round_robin",
"--replica_config_model_name",
"qwen3-a3b-30b-moe",
"--replica_config_device",
"h20",
"--replica_config_network_device",
"h20_dgx",
"--replica_config_attn_tensor_parallel_size",
"1",
"--replica_config_attn_data_parallel_size",
"1",
"--replica_config_moe_tensor_parallel_size",
"1",
"--replica_config_moe_expert_parallel_size",
"1",
"--replica_config_num_pipeline_stages",
"1",
"--replica_scheduler_config_type",
"vllm_v1",
"--decode_cuda_graph_mode",
"piecewise",
"--vllm_v1_scheduler_config_batch_size_cap",
"16",
"--vllm_v1_scheduler_config_max_tokens_in_batch",
"8192",
"--vllm_v1_scheduler_config_long_prefill_token_threshold",
"0",
"--vllm_v1_scheduler_config_block_size",
"16",
"--vllm_v1_scheduler_config_num_blocks_mode",
"explicit",
"--vllm_v1_scheduler_config_gpu_memory_utilization",
"0.92",
"--vllm_v1_scheduler_config_non_kv_cache_overhead_bytes",
"0",
"--request_generator_config_type",
"trace_replay",
"--trace_request_generator_config_trace_file",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp1-frontier.csv",
"--trace_request_generator_config_max_tokens",
"40960",
"--metrics_config_output_dir",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/sim/tp1_mns16/metrics",
"--metrics_config_run_id",
"joint_tp1_mns16",
"--metrics_config_write_metrics",
"--metrics_config_store_request_metrics",
"--metrics_config_store_batch_metrics",
"--metrics_config_store_token_completion_metrics",
"--metrics_config_store_utilization_metrics",
"--no-metrics_config_store_plots",
"--no-metrics_config_enable_chrome_trace",
"--no-metrics_config_write_json_trace",
"--metrics_config_store_frontier_stage_batch_ledger",
"--no-random_forrest_execution_time_predictor_config_enable_dummy_mode",
"--random_forrest_execution_time_predictor_config_linear_op_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/linear_op.csv",
"--random_forrest_execution_time_predictor_config_atten_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/attention.csv",
"--random_forrest_execution_time_predictor_config_moe_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/moe.csv",
"--random_forrest_execution_time_predictor_config_linear_op_kernel_only_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/linear_op.csv",
"--random_forrest_execution_time_predictor_config_atten_kernel_only_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/attention.csv",
"--random_forrest_execution_time_predictor_config_moe_kernel_only_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/moe.csv",
"--random_forrest_execution_time_predictor_config_prediction_max_prefill_chunk_size",
"8192",
"--random_forrest_execution_time_predictor_config_prediction_max_batch_size",
"32",
"--random_forrest_execution_time_predictor_config_prediction_max_tokens_per_request",
"40960",
"--random_forrest_execution_time_predictor_config_no_cache",
"--random_forrest_execution_time_predictor_config_skip_cpu_overhead_modeling",
"--vllm_v1_scheduler_config_num_blocks",
"20128",
"--vllm_v1_scheduler_config_enable_chunked_prefill",
"--random_forrest_execution_time_predictor_config_num_training_job_threads",
"4",
"--cudagraph_capture_sizes",
"1",
"2",
"4",
"8",
"16",
"24",
"32",
"--vidur_cc_backend_config_all_reduce_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/measured-allreduce.csv",
"--vidur_cc_backend_config_cache_dir",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/cc-cache",
"--vidur_cc_backend_config_k_fold_cv_splits",
"6",
"--vidur_cc_backend_config_num_training_job_threads",
"1",
"--metrics_config_cache_dir",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/model-cache"
],
"log": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/logs/tp1_mns16.log",
"source_command": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp1_mns16/tp1/command.json",
"source_command_sha256": "a9815797b1601bf6f6cdf0269e84acb376a84945609e338868dc8347aab650e6",
"usage": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/usage/tp1_mns16.json"
},
"tp2_mns16": {
"argv": [
"/usr/bin/python3",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/run_frontier_with_curves.py",
"--simulation_mode",
"online",
"--sys_arch",
"co-location",
"--cc_backend_config_type",
"vidur",
"--cluster_config_num_replicas",
"1",
"--cluster_scheduler_config_type",
"sticky_round_robin",
"--replica_config_model_name",
"qwen3-a3b-30b-moe",
"--replica_config_device",
"h20",
"--replica_config_network_device",
"h20_dgx",
"--replica_config_attn_tensor_parallel_size",
"2",
"--replica_config_attn_data_parallel_size",
"1",
"--replica_config_moe_tensor_parallel_size",
"2",
"--replica_config_moe_expert_parallel_size",
"1",
"--replica_config_num_pipeline_stages",
"1",
"--replica_scheduler_config_type",
"vllm_v1",
"--decode_cuda_graph_mode",
"piecewise",
"--vllm_v1_scheduler_config_batch_size_cap",
"16",
"--vllm_v1_scheduler_config_max_tokens_in_batch",
"8192",
"--vllm_v1_scheduler_config_long_prefill_token_threshold",
"0",
"--vllm_v1_scheduler_config_block_size",
"16",
"--vllm_v1_scheduler_config_num_blocks_mode",
"explicit",
"--vllm_v1_scheduler_config_gpu_memory_utilization",
"0.92",
"--vllm_v1_scheduler_config_non_kv_cache_overhead_bytes",
"0",
"--request_generator_config_type",
"trace_replay",
"--trace_request_generator_config_trace_file",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp2-frontier.csv",
"--trace_request_generator_config_max_tokens",
"40960",
"--metrics_config_output_dir",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/sim/tp2_mns16/metrics",
"--metrics_config_run_id",
"joint_tp2_mns16",
"--metrics_config_write_metrics",
"--metrics_config_store_request_metrics",
"--metrics_config_store_batch_metrics",
"--metrics_config_store_token_completion_metrics",
"--metrics_config_store_utilization_metrics",
"--no-metrics_config_store_plots",
"--no-metrics_config_enable_chrome_trace",
"--no-metrics_config_write_json_trace",
"--metrics_config_store_frontier_stage_batch_ledger",
"--no-random_forrest_execution_time_predictor_config_enable_dummy_mode",
"--random_forrest_execution_time_predictor_config_linear_op_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/linear_op.csv",
"--random_forrest_execution_time_predictor_config_atten_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/attention.csv",
"--random_forrest_execution_time_predictor_config_moe_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/moe.csv",
"--random_forrest_execution_time_predictor_config_linear_op_kernel_only_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/linear_op.csv",
"--random_forrest_execution_time_predictor_config_atten_kernel_only_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/attention.csv",
"--random_forrest_execution_time_predictor_config_moe_kernel_only_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/moe.csv",
"--random_forrest_execution_time_predictor_config_prediction_max_prefill_chunk_size",
"8192",
"--random_forrest_execution_time_predictor_config_prediction_max_batch_size",
"32",
"--random_forrest_execution_time_predictor_config_prediction_max_tokens_per_request",
"40960",
"--random_forrest_execution_time_predictor_config_no_cache",
"--random_forrest_execution_time_predictor_config_skip_cpu_overhead_modeling",
"--vllm_v1_scheduler_config_num_blocks",
"76620",
"--vllm_v1_scheduler_config_enable_chunked_prefill",
"--random_forrest_execution_time_predictor_config_num_training_job_threads",
"4",
"--cudagraph_capture_sizes",
"1",
"2",
"4",
"8",
"16",
"24",
"32",
"--vidur_cc_backend_config_all_reduce_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/measured-allreduce.csv",
"--vidur_cc_backend_config_cache_dir",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/cc-cache",
"--vidur_cc_backend_config_k_fold_cv_splits",
"6",
"--vidur_cc_backend_config_num_training_job_threads",
"1",
"--metrics_config_cache_dir",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/model-cache"
],
"log": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/logs/tp2_mns16.log",
"source_command": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp2_mns16/tp2/command.json",
"source_command_sha256": "61788a8810be301c9dbc006624aa19b6a932bc44d341b836861087833cffc3df",
"usage": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/usage/tp2_mns16.json"
},
"tp4_mns16": {
"argv": [
"/usr/bin/python3",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/run_frontier_with_curves.py",
"--simulation_mode",
"online",
"--sys_arch",
"co-location",
"--cc_backend_config_type",
"vidur",
"--cluster_config_num_replicas",
"1",
"--cluster_scheduler_config_type",
"sticky_round_robin",
"--replica_config_model_name",
"qwen3-a3b-30b-moe",
"--replica_config_device",
"h20",
"--replica_config_network_device",
"h20_dgx",
"--replica_config_attn_tensor_parallel_size",
"4",
"--replica_config_attn_data_parallel_size",
"1",
"--replica_config_moe_tensor_parallel_size",
"4",
"--replica_config_moe_expert_parallel_size",
"1",
"--replica_config_num_pipeline_stages",
"1",
"--replica_scheduler_config_type",
"vllm_v1",
"--decode_cuda_graph_mode",
"piecewise",
"--vllm_v1_scheduler_config_batch_size_cap",
"16",
"--vllm_v1_scheduler_config_max_tokens_in_batch",
"8192",
"--vllm_v1_scheduler_config_long_prefill_token_threshold",
"0",
"--vllm_v1_scheduler_config_block_size",
"16",
"--vllm_v1_scheduler_config_num_blocks_mode",
"explicit",
"--vllm_v1_scheduler_config_gpu_memory_utilization",
"0.92",
"--vllm_v1_scheduler_config_non_kv_cache_overhead_bytes",
"0",
"--request_generator_config_type",
"trace_replay",
"--trace_request_generator_config_trace_file",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp4-frontier.csv",
"--trace_request_generator_config_max_tokens",
"40960",
"--metrics_config_output_dir",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/sim/tp4_mns16/metrics",
"--metrics_config_run_id",
"joint_tp4_mns16",
"--metrics_config_write_metrics",
"--metrics_config_store_request_metrics",
"--metrics_config_store_batch_metrics",
"--metrics_config_store_token_completion_metrics",
"--metrics_config_store_utilization_metrics",
"--no-metrics_config_store_plots",
"--no-metrics_config_enable_chrome_trace",
"--no-metrics_config_write_json_trace",
"--metrics_config_store_frontier_stage_batch_ledger",
"--no-random_forrest_execution_time_predictor_config_enable_dummy_mode",
"--random_forrest_execution_time_predictor_config_linear_op_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/linear_op.csv",
"--random_forrest_execution_time_predictor_config_atten_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/attention.csv",
"--random_forrest_execution_time_predictor_config_moe_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/moe.csv",
"--random_forrest_execution_time_predictor_config_linear_op_kernel_only_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/linear_op.csv",
"--random_forrest_execution_time_predictor_config_atten_kernel_only_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/attention.csv",
"--random_forrest_execution_time_predictor_config_moe_kernel_only_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/moe.csv",
"--random_forrest_execution_time_predictor_config_prediction_max_prefill_chunk_size",
"8192",
"--random_forrest_execution_time_predictor_config_prediction_max_batch_size",
"32",
"--random_forrest_execution_time_predictor_config_prediction_max_tokens_per_request",
"40960",
"--random_forrest_execution_time_predictor_config_no_cache",
"--random_forrest_execution_time_predictor_config_skip_cpu_overhead_modeling",
"--vllm_v1_scheduler_config_num_blocks",
"191882",
"--vllm_v1_scheduler_config_enable_chunked_prefill",
"--random_forrest_execution_time_predictor_config_num_training_job_threads",
"4",
"--cudagraph_capture_sizes",
"1",
"2",
"4",
"8",
"16",
"24",
"32",
"--vidur_cc_backend_config_all_reduce_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/measured-allreduce.csv",
"--vidur_cc_backend_config_cache_dir",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/cc-cache",
"--vidur_cc_backend_config_k_fold_cv_splits",
"6",
"--vidur_cc_backend_config_num_training_job_threads",
"1",
"--metrics_config_cache_dir",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/model-cache"
],
"log": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/logs/tp4_mns16.log",
"source_command": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp4_mns16/tp4/command.json",
"source_command_sha256": "9bbcf10446336ba5885193f391dd628cd18ff64d91a51ebcd463ffd24be95532",
"usage": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/usage/tp4_mns16.json"
},
"tp4_mns32": {
"argv": [
"/usr/bin/python3",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/run_frontier_with_curves.py",
"--simulation_mode",
"online",
"--sys_arch",
"co-location",
"--cc_backend_config_type",
"vidur",
"--cluster_config_num_replicas",
"1",
"--cluster_scheduler_config_type",
"sticky_round_robin",
"--replica_config_model_name",
"qwen3-a3b-30b-moe",
"--replica_config_device",
"h20",
"--replica_config_network_device",
"h20_dgx",
"--replica_config_attn_tensor_parallel_size",
"4",
"--replica_config_attn_data_parallel_size",
"1",
"--replica_config_moe_tensor_parallel_size",
"4",
"--replica_config_moe_expert_parallel_size",
"1",
"--replica_config_num_pipeline_stages",
"1",
"--replica_scheduler_config_type",
"vllm_v1",
"--decode_cuda_graph_mode",
"piecewise",
"--vllm_v1_scheduler_config_batch_size_cap",
"32",
"--vllm_v1_scheduler_config_max_tokens_in_batch",
"8192",
"--vllm_v1_scheduler_config_long_prefill_token_threshold",
"0",
"--vllm_v1_scheduler_config_block_size",
"16",
"--vllm_v1_scheduler_config_num_blocks_mode",
"explicit",
"--vllm_v1_scheduler_config_gpu_memory_utilization",
"0.92",
"--vllm_v1_scheduler_config_non_kv_cache_overhead_bytes",
"0",
"--request_generator_config_type",
"trace_replay",
"--trace_request_generator_config_trace_file",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp4-frontier.csv",
"--trace_request_generator_config_max_tokens",
"40960",
"--metrics_config_output_dir",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/sim/tp4_mns32/metrics",
"--metrics_config_run_id",
"joint_tp4_mns32",
"--metrics_config_write_metrics",
"--metrics_config_store_request_metrics",
"--metrics_config_store_batch_metrics",
"--metrics_config_store_token_completion_metrics",
"--metrics_config_store_utilization_metrics",
"--no-metrics_config_store_plots",
"--no-metrics_config_enable_chrome_trace",
"--no-metrics_config_write_json_trace",
"--metrics_config_store_frontier_stage_batch_ledger",
"--no-random_forrest_execution_time_predictor_config_enable_dummy_mode",
"--random_forrest_execution_time_predictor_config_linear_op_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/linear_op.csv",
"--random_forrest_execution_time_predictor_config_atten_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/attention.csv",
"--random_forrest_execution_time_predictor_config_moe_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/moe.csv",
"--random_forrest_execution_time_predictor_config_linear_op_kernel_only_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/linear_op.csv",
"--random_forrest_execution_time_predictor_config_atten_kernel_only_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/attention.csv",
"--random_forrest_execution_time_predictor_config_moe_kernel_only_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/moe.csv",
"--random_forrest_execution_time_predictor_config_prediction_max_prefill_chunk_size",
"8192",
"--random_forrest_execution_time_predictor_config_prediction_max_batch_size",
"64",
"--random_forrest_execution_time_predictor_config_prediction_max_tokens_per_request",
"40960",
"--random_forrest_execution_time_predictor_config_no_cache",
"--random_forrest_execution_time_predictor_config_skip_cpu_overhead_modeling",
"--vllm_v1_scheduler_config_num_blocks",
"191786",
"--vllm_v1_scheduler_config_enable_chunked_prefill",
"--random_forrest_execution_time_predictor_config_num_training_job_threads",
"4",
"--cudagraph_capture_sizes",
"1",
"2",
"4",
"8",
"16",
"24",
"32",
"40",
"48",
"56",
"64",
"--vidur_cc_backend_config_all_reduce_input_file",
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/measured-allreduce.csv",
"--vidur_cc_backend_config_cache_dir",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/cc-cache",
"--vidur_cc_backend_config_k_fold_cv_splits",
"6",
"--vidur_cc_backend_config_num_training_job_threads",
"1",
"--metrics_config_cache_dir",
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/model-cache"
],
"log": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/logs/tp4_mns32.log",
"source_command": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp4_mns32/tp4/command.json",
"source_command_sha256": "fbc7dee55590b415ed1cde8072de835ed155a0c20ba0eb305c3cb22aa8065a51",
"usage": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/usage/tp4_mns32.json"
}
},
"collective_curve": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/results/collective-curve.json",
"collective_curve_sha256": "f9543649d4ea78f08240bf1284ab74083aa5cf5671ed47e386047f1453300b36",
"collective_curve_variant": "drop_mean",
"frontier_checkout": "/tmp/frontier-attn-structured-v0",
"frontier_commit": "1f8900a4ac64e45754b03d0aa7c1dddab65785cf",
"mode": "joint",
"model_cache": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/model-cache",
"moe_curve": "/home/gahow/phd/aituner/runs/frontier-fused-moe-profile-v0/results/fused-moe-curve.json",
"moe_curve_sha256": "b94d65d9d581adefcc6c14ed4920cce6a1136f74f1014737dc3e4249bc8250d2",
"python": "/usr/bin/python3",
"python_dependency_roots": [
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/python-deps",
"/home/gahow/.cache/uv/archive-v0/-_kzErLcPO5nASZFX8b9k",
"/home/gahow/.cache/uv/archive-v0/FbaBs_QJ9QKEbQ9V_4aIR",
"/home/gahow/.cache/uv/archive-v0/fuHsGXD0Lv_UjFC8yI4-7",
"/home/gahow/.cache/uv/archive-v0/jFGdqQLpB1eopfm9VxT3j",
"/home/gahow/.cache/uv/archive-v0/YWW6ExSJuPVvv4-qYQTin",
"/home/gahow/.cache/uv/archive-v0/3_qxZ5Ll-EpVAGZfbksfe"
],
"traces": {
"1": {
"first_arrival_s": 0.0,
"last_arrival_s": 595.348837209302,
"requests": 129,
"source_request_metrics": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp1_mns16/tp1/metrics/qwen3_a3b_30b_moe/online_serving/qwen30_trace_tp1_mns16_tp1/request_metrics.csv",
"source_sha256": "0b82e09644a5884fcd10d894b68495daefdabb32b770146c2f9ece37b8469f4f",
"trace": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp1-frontier.csv",
"trace_sha256": "59dd8996ff879ef94330004104dfdf515b791bce4036576eccc93290e9206dad"
},
"2": {
"first_arrival_s": 0.0,
"last_arrival_s": 297.674418604651,
"requests": 129,
"source_request_metrics": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp2_mns16/tp2/metrics/qwen3_a3b_30b_moe/online_serving/qwen30_trace_tp2_mns16_tp2/request_metrics.csv",
"source_sha256": "33983081bb20dd5e2053e9e3d13def8732e958150c9b47a8609ba345123f2316",
"trace": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp2-frontier.csv",
"trace_sha256": "64fc077b38274a76a8279884ac4115836cd1157c95119c64fabac50d81124f69"
},
"4": {
"first_arrival_s": 0.0,
"last_arrival_s": 148.837209302326,
"requests": 129,
"source_request_metrics": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp4_mns16/tp4/metrics/qwen3_a3b_30b_moe/online_serving/qwen30_trace_tp4_mns16_tp4/request_metrics.csv",
"source_sha256": "b36cd383c07b546d2c1f2fac754d5dbb92efd6880316b4233a7aef9fa1115a36",
"trace": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp4-frontier.csv",
"trace_sha256": "adb3d6f3932a44c86c3d9e7cf1e57739594e8c48d19a26b5aa54b37dce0e0c19"
}
}
}

View File

@@ -0,0 +1,72 @@
#!/usr/bin/env python3
"""Schematic figure frozen before EXP-ATTN-STRUCTURED execution."""
from pathlib import Path
import matplotlib
matplotlib.use("Agg")
import matplotlib.pyplot as plt
import numpy as np
ROOT = Path(__file__).resolve().parent
SURFACE = "#fcfcfb"
INK = "#111111"
MUTED = "#77736c"
GRID = "#dedbd2"
RF = "#d95f02"
STRUCTURED = "#1b75bc"
fig, axes = plt.subplots(1, 2, figsize=(10.8, 4.2), dpi=160)
fig.patch.set_facecolor(SURFACE)
for ax in axes:
ax.set_facecolor(SURFACE)
ax.grid(axis="y", color=GRID, linewidth=0.8)
ax.set_axisbelow(True)
ax.spines[["top", "right"]].set_visible(False)
ax.tick_params(colors=MUTED, labelsize=8)
kv = np.array([8, 16, 24], dtype=float)
actual = np.array([11.97, 19.92, 27.84])
rf = np.array([9.46, 16.40, 24.52])
structured_expected = np.array([12.0, 19.9, 27.9])
axes[0].plot(kv, actual, "o-", color=INK, label="profile actual")
axes[0].plot(kv, rf, "s--", color=RF, label="current RF")
axes[0].plot(
kv,
structured_expected,
"^:",
color=STRUCTURED,
label="structured (expected)",
)
axes[0].set_xlabel("KV context (ktok)")
axes[0].set_ylabel("TP1 q8k attention time (ms)")
axes[0].set_title("(a) Continuous KV growth", loc="left", fontsize=10)
axes[0].legend(frameon=False, fontsize=8)
labels = ["TP1\n.00125", "TP1\n.0025", "TP2\n.0025", "TP2\n.005",
"TP4\n.0025", "TP4\n.005", "TP4\n.01"]
x = np.arange(len(labels))
v5_mean = np.array([-13.5, -17.7, -11.2, -14.1, 2.5, 2.9, -0.2])
expected = np.array([-5, -8, -9, -11, 3, 3, 0])
axes[1].axhspan(-15, 15, color=GRID, alpha=0.5)
axes[1].axhline(0, color=MUTED, linewidth=0.8)
axes[1].plot(x, v5_mean, "o-", color=RF, label="sim-v5 measured")
axes[1].plot(x, expected, "s--", color=STRUCTURED, label="H1 expected")
axes[1].set_xticks(x, labels)
axes[1].set_ylabel("TTFT mean bias (%)")
axes[1].set_title("(b) 7-cell trace gate", loc="left", fontsize=10)
axes[1].legend(frameon=False, fontsize=8)
fig.suptitle(
"MOCK / schematic — EXP-ATTN-STRUCTURED (not measured results)",
x=0.01,
ha="left",
color=RF,
fontsize=9,
)
fig.tight_layout(rect=(0, 0, 1, 0.95))
fig.savefig(ROOT / "figure-prototype.png", facecolor=SURFACE)
fig.savefig(ROOT / "figure-prototype.svg", facecolor=SURFACE)
print(ROOT / "figure-prototype.png")

View File

@@ -0,0 +1,10 @@
candidate,training_rows,grid_fit_mape_pct,single_fit_mape_pct,heldout_context_mape_pct,heldout_context_max_abs_error_pct,prediction_min_ms,q_min_delta_ms,kv_min_delta_ms,monotone_and_nonnegative,tp
rf_all,29,15.074275515235467,19.842994757611535,44.40139318281943,82.78279487156401,0.06029164119272453,0.0,-4.2841601371801374e-05,False,1
rf_single,23,11.834992526698676,22.893412972496023,34.45356428541224,62.49334437588834,0.059327708247725125,-0.00022153525203457564,-1.4336001873005433e-05,False,1
structured_single,23,0.8432511364168856,2.227110646811972,0.841926169535806,2.0408978739639134,0.05679146709541477,0.0,0.0016745062683911627,True,1
rf_all,29,13.812689689573157,17.879870012606048,44.20143320278334,82.0656368501208,0.06000113548192927,-0.004361070463210395,-0.009949388915300408,False,2
rf_single,23,10.738962684891058,20.818820413545826,34.84360402100271,66.67253880294443,0.059971319361210015,-0.003811210796127021,-0.009949388915300408,False,2
structured_single,23,1.518412258922364,6.226292654776418,1.6047958673086566,3.464671475193195,0.05767893331746252,0.0,0.0013929374121726124,True,2
rf_all,29,14.117015331381916,15.49366794487052,43.53650868837558,81.33508178007524,0.05866772018640992,-0.0009967416035880083,-5.5955198407176e-05,False,4
rf_single,23,12.07517666989496,17.672692652721008,34.14562332866605,62.31478818862995,0.058213693721655094,-0.0012244979345549661,-8.259841203689389e-05,False,4
structured_single,23,3.088740104976349,5.969109476280061,3.0103379904473164,5.490598706238697,0.05747733327249683,0.0,0.0012048051417407057,True,4
1 candidate training_rows grid_fit_mape_pct single_fit_mape_pct heldout_context_mape_pct heldout_context_max_abs_error_pct prediction_min_ms q_min_delta_ms kv_min_delta_ms monotone_and_nonnegative tp
2 rf_all 29 15.074275515235467 19.842994757611535 44.40139318281943 82.78279487156401 0.06029164119272453 0.0 -4.2841601371801374e-05 False 1
3 rf_single 23 11.834992526698676 22.893412972496023 34.45356428541224 62.49334437588834 0.059327708247725125 -0.00022153525203457564 -1.4336001873005433e-05 False 1
4 structured_single 23 0.8432511364168856 2.227110646811972 0.841926169535806 2.0408978739639134 0.05679146709541477 0.0 0.0016745062683911627 True 1
5 rf_all 29 13.812689689573157 17.879870012606048 44.20143320278334 82.0656368501208 0.06000113548192927 -0.004361070463210395 -0.009949388915300408 False 2
6 rf_single 23 10.738962684891058 20.818820413545826 34.84360402100271 66.67253880294443 0.059971319361210015 -0.003811210796127021 -0.009949388915300408 False 2
7 structured_single 23 1.518412258922364 6.226292654776418 1.6047958673086566 3.464671475193195 0.05767893331746252 0.0 0.0013929374121726124 True 2
8 rf_all 29 14.117015331381916 15.49366794487052 43.53650868837558 81.33508178007524 0.05866772018640992 -0.0009967416035880083 -5.5955198407176e-05 False 4
9 rf_single 23 12.07517666989496 17.672692652721008 34.14562332866605 62.31478818862995 0.058213693721655094 -0.0012244979345549661 -8.259841203689389e-05 False 4
10 structured_single 23 3.088740104976349 5.969109476280061 3.0103379904473164 5.490598706238697 0.05747733327249683 0.0 0.0012048051417407057 True 4

View File

@@ -0,0 +1,149 @@
{
"schema": "frontier-attn-structured-ablation-v1",
"profile": "/home/gahow/phd/aituner/runs/frontier-prefill-kvgrowth-fix-v0/profiles/profile-v5-kvgrowth/attention.csv",
"frontier_checkout": "/tmp/frontier-attn-structured-v0",
"data_audit": {
"tp1": {
"standard_rows": 29,
"single_request_rows": 23,
"target_grid_rows": 10,
"duplicate_feature_groups": 4
},
"tp2": {
"standard_rows": 29,
"single_request_rows": 23,
"target_grid_rows": 10,
"duplicate_feature_groups": 4
},
"tp4": {
"standard_rows": 29,
"single_request_rows": 23,
"target_grid_rows": 10,
"duplicate_feature_groups": 4
}
},
"results": [
{
"candidate": "rf_all",
"training_rows": 29,
"grid_fit_mape_pct": 15.074275515235467,
"single_fit_mape_pct": 19.842994757611535,
"heldout_context_mape_pct": 44.40139318281943,
"heldout_context_max_abs_error_pct": 82.78279487156401,
"prediction_min_ms": 0.06029164119272453,
"q_min_delta_ms": 0.0,
"kv_min_delta_ms": -4.2841601371801374e-05,
"monotone_and_nonnegative": false,
"tp": 1
},
{
"candidate": "rf_single",
"training_rows": 23,
"grid_fit_mape_pct": 11.834992526698676,
"single_fit_mape_pct": 22.893412972496023,
"heldout_context_mape_pct": 34.45356428541224,
"heldout_context_max_abs_error_pct": 62.49334437588834,
"prediction_min_ms": 0.059327708247725125,
"q_min_delta_ms": -0.00022153525203457564,
"kv_min_delta_ms": -1.4336001873005433e-05,
"monotone_and_nonnegative": false,
"tp": 1
},
{
"candidate": "structured_single",
"training_rows": 23,
"grid_fit_mape_pct": 0.8432511364168856,
"single_fit_mape_pct": 2.227110646811972,
"heldout_context_mape_pct": 0.841926169535806,
"heldout_context_max_abs_error_pct": 2.0408978739639134,
"prediction_min_ms": 0.05679146709541477,
"q_min_delta_ms": 0.0,
"kv_min_delta_ms": 0.0016745062683911627,
"monotone_and_nonnegative": true,
"tp": 1
},
{
"candidate": "rf_all",
"training_rows": 29,
"grid_fit_mape_pct": 13.812689689573157,
"single_fit_mape_pct": 17.879870012606048,
"heldout_context_mape_pct": 44.20143320278334,
"heldout_context_max_abs_error_pct": 82.0656368501208,
"prediction_min_ms": 0.06000113548192927,
"q_min_delta_ms": -0.004361070463210395,
"kv_min_delta_ms": -0.009949388915300408,
"monotone_and_nonnegative": false,
"tp": 2
},
{
"candidate": "rf_single",
"training_rows": 23,
"grid_fit_mape_pct": 10.738962684891058,
"single_fit_mape_pct": 20.818820413545826,
"heldout_context_mape_pct": 34.84360402100271,
"heldout_context_max_abs_error_pct": 66.67253880294443,
"prediction_min_ms": 0.059971319361210015,
"q_min_delta_ms": -0.003811210796127021,
"kv_min_delta_ms": -0.009949388915300408,
"monotone_and_nonnegative": false,
"tp": 2
},
{
"candidate": "structured_single",
"training_rows": 23,
"grid_fit_mape_pct": 1.518412258922364,
"single_fit_mape_pct": 6.226292654776418,
"heldout_context_mape_pct": 1.6047958673086566,
"heldout_context_max_abs_error_pct": 3.464671475193195,
"prediction_min_ms": 0.05767893331746252,
"q_min_delta_ms": 0.0,
"kv_min_delta_ms": 0.0013929374121726124,
"monotone_and_nonnegative": true,
"tp": 2
},
{
"candidate": "rf_all",
"training_rows": 29,
"grid_fit_mape_pct": 14.117015331381916,
"single_fit_mape_pct": 15.49366794487052,
"heldout_context_mape_pct": 43.53650868837558,
"heldout_context_max_abs_error_pct": 81.33508178007524,
"prediction_min_ms": 0.05866772018640992,
"q_min_delta_ms": -0.0009967416035880083,
"kv_min_delta_ms": -5.5955198407176e-05,
"monotone_and_nonnegative": false,
"tp": 4
},
{
"candidate": "rf_single",
"training_rows": 23,
"grid_fit_mape_pct": 12.07517666989496,
"single_fit_mape_pct": 17.672692652721008,
"heldout_context_mape_pct": 34.14562332866605,
"heldout_context_max_abs_error_pct": 62.31478818862995,
"prediction_min_ms": 0.058213693721655094,
"q_min_delta_ms": -0.0012244979345549661,
"kv_min_delta_ms": -8.259841203689389e-05,
"monotone_and_nonnegative": false,
"tp": 4
},
{
"candidate": "structured_single",
"training_rows": 23,
"grid_fit_mape_pct": 3.088740104976349,
"single_fit_mape_pct": 5.969109476280061,
"heldout_context_mape_pct": 3.0103379904473164,
"heldout_context_max_abs_error_pct": 5.490598706238697,
"prediction_min_ms": 0.05747733327249683,
"q_min_delta_ms": 0.0,
"kv_min_delta_ms": 0.0012048051417407057,
"monotone_and_nonnegative": true,
"tp": 4
}
],
"checks": {
"heldout_context_mape_le_5pct": true,
"monotone_and_nonnegative": true,
"profile_gate": true
}
}

View File

@@ -0,0 +1,85 @@
cell,metric,quantile,old_bias,new_bias,abs_bias_delta_pp,validity
tp1_rho0p00125,ttft,mean,-0.13461915993830945,-0.06328385143753473,-7.133530850077471,GATE_FAIL_DIAGNOSTIC
tp1_rho0p00125,ttft,p50,-0.1888032883165517,-0.1117698463137391,-7.7033442002812595,GATE_FAIL_DIAGNOSTIC
tp1_rho0p00125,ttft,p90,-0.23223319000679374,-0.0884727580604037,-14.376043194639005,GATE_FAIL_DIAGNOSTIC
tp1_rho0p00125,ttft,p99,-0.16824856840168442,-0.08585570369326061,-8.239286470842382,GATE_FAIL_DIAGNOSTIC
tp1_rho0p00125,tpot,mean,0.1305653876178269,0.14303898259628217,1.2473594978455265,GATE_FAIL_DIAGNOSTIC
tp1_rho0p00125,tpot,p50,0.16837991778063455,0.1688780144991309,0.04980967184963492,GATE_FAIL_DIAGNOSTIC
tp1_rho0p00125,tpot,p90,0.012855564422932954,0.02293841255516459,1.0082848132231637,GATE_FAIL_DIAGNOSTIC
tp1_rho0p00125,tpot,p99,-0.08151080694091946,-0.06480973978913974,-1.6701067151779714,GATE_FAIL_DIAGNOSTIC
tp1_rho0p00125,e2e,mean,0.06539157436639341,0.0838094906426525,1.8417916276259092,GATE_FAIL_DIAGNOSTIC
tp1_rho0p00125,e2e,p50,0.10867934171755954,0.11142188099086506,0.2742539273305519,GATE_FAIL_DIAGNOSTIC
tp1_rho0p00125,e2e,p90,0.10946179160934073,0.11872607178176105,0.9264280172420314,GATE_FAIL_DIAGNOSTIC
tp1_rho0p00125,e2e,p99,-0.0657146472433028,-0.04585643943465822,-1.9858207808644577,GATE_FAIL_DIAGNOSTIC
tp1_rho0p0025,ttft,mean,-0.17677087458166194,-0.09487906467830536,-8.189180990335657,GATE_FAIL_DIAGNOSTIC
tp1_rho0p0025,ttft,p50,0.008311436147272566,0.015508635458377175,0.7197199311104608,GATE_FAIL_DIAGNOSTIC
tp1_rho0p0025,ttft,p90,-0.2608553178054708,-0.129865645852306,-13.098967195316478,GATE_FAIL_DIAGNOSTIC
tp1_rho0p0025,ttft,p99,-0.22116442296103195,-0.14410151018893788,-7.706291277209407,GATE_FAIL_DIAGNOSTIC
tp1_rho0p0025,tpot,mean,0.014380733682430142,0.052302860177306544,3.7922126494876403,GATE_FAIL_DIAGNOSTIC
tp1_rho0p0025,tpot,p50,0.13351145963877706,0.14244154194999165,0.8930082311214588,GATE_FAIL_DIAGNOSTIC
tp1_rho0p0025,tpot,p90,-0.05456950130438105,0.018069551387063856,-3.649994991731719,GATE_FAIL_DIAGNOSTIC
tp1_rho0p0025,tpot,p99,-0.23635406197836167,-0.1790206784227079,-5.733338355565376,GATE_FAIL_DIAGNOSTIC
tp1_rho0p0025,e2e,mean,-0.020164189983441452,0.01563919326555167,-0.4524996717889782,GATE_FAIL_DIAGNOSTIC
tp1_rho0p0025,e2e,p50,0.07632815851795742,0.10505951594320918,2.8731357425251765,GATE_FAIL_DIAGNOSTIC
tp1_rho0p0025,e2e,p90,0.049809009042946335,0.08165607475145953,3.1847065708513194,GATE_FAIL_DIAGNOSTIC
tp1_rho0p0025,e2e,p99,-0.18303183751478602,-0.15333409130015813,-2.96977462146279,GATE_FAIL_DIAGNOSTIC
tp2_rho0p0025,ttft,mean,-0.11161154024124531,-0.045433491316312524,-6.617804892493279,PASS_SUBCRITICAL
tp2_rho0p0025,ttft,p50,-0.18420934047220774,-0.0875990583320057,-9.661028214020204,PASS_SUBCRITICAL
tp2_rho0p0025,ttft,p90,-0.21836264490202395,-0.1142200855839675,-10.414255931805645,PASS_SUBCRITICAL
tp2_rho0p0025,ttft,p99,-0.17283197594971011,-0.07674087497700505,-9.609110097270507,PASS_SUBCRITICAL
tp2_rho0p0025,tpot,mean,0.13711499081487563,0.15073709978689778,1.362210897202215,PASS_SUBCRITICAL
tp2_rho0p0025,tpot,p50,0.17555321305308488,0.18527285925405948,0.9719646200974597,PASS_SUBCRITICAL
tp2_rho0p0025,tpot,p90,0.0591579975137338,0.07863996768588354,1.9481970172149734,PASS_SUBCRITICAL
tp2_rho0p0025,tpot,p99,0.1300483675091633,0.1661995397125918,3.6151172203428503,PASS_SUBCRITICAL
tp2_rho0p0025,e2e,mean,0.10925353865257875,0.12696146154030977,1.7707922887731016,PASS_SUBCRITICAL
tp2_rho0p0025,e2e,p50,0.1339165600755408,0.14796251184179712,1.404595176625631,PASS_SUBCRITICAL
tp2_rho0p0025,e2e,p90,0.10666828724664539,0.12767235986604622,2.1004072619400835,PASS_SUBCRITICAL
tp2_rho0p0025,e2e,p99,-0.07437942975605877,-0.03965040123988098,-3.472902851617779,PASS_SUBCRITICAL
tp2_rho0p005,ttft,mean,-0.14129871969878843,-0.07082500585057615,-7.047371384821228,GATE_FAIL_DIAGNOSTIC
tp2_rho0p005,ttft,p50,-0.2042045530944649,-0.1284088888361358,-7.579566425832909,GATE_FAIL_DIAGNOSTIC
tp2_rho0p005,ttft,p90,-0.19003454588767263,-0.111136573344055,-7.889797254361763,GATE_FAIL_DIAGNOSTIC
tp2_rho0p005,ttft,p99,-0.19373351009539902,-0.09012556161973535,-10.360794847566366,GATE_FAIL_DIAGNOSTIC
tp2_rho0p005,tpot,mean,0.03117337438562004,0.05457336599027876,2.3399991604658723,GATE_FAIL_DIAGNOSTIC
tp2_rho0p005,tpot,p50,0.07668249597302182,0.0872512146277895,1.0568718654767675,GATE_FAIL_DIAGNOSTIC
tp2_rho0p005,tpot,p90,-0.03724827658429621,0.01132352325010614,-2.5924753334190074,GATE_FAIL_DIAGNOSTIC
tp2_rho0p005,tpot,p99,-0.10050436157089844,-0.05948329733667248,-4.102106423422596,GATE_FAIL_DIAGNOSTIC
tp2_rho0p005,e2e,mean,0.023867807624916495,0.050164236759908075,2.629642913499158,GATE_FAIL_DIAGNOSTIC
tp2_rho0p005,e2e,p50,0.07536668131278851,0.0912493994395978,1.588271812680929,GATE_FAIL_DIAGNOSTIC
tp2_rho0p005,e2e,p90,-0.03100470462321266,-0.0004979191794830456,-3.0506785443729614,GATE_FAIL_DIAGNOSTIC
tp2_rho0p005,e2e,p99,-0.04909581632382212,-0.0014369075488634014,-4.765890877495872,GATE_FAIL_DIAGNOSTIC
tp4_rho0p0025,ttft,mean,0.025022574738277282,0.07766646565061346,5.2643890912336175,PASS_SUBCRITICAL
tp4_rho0p0025,ttft,p50,-0.040042171846277425,0.029153379043297147,-1.0888792802980278,PASS_SUBCRITICAL
tp4_rho0p0025,ttft,p90,-0.056230097634382616,0.025222989748299444,-3.100710788608317,PASS_SUBCRITICAL
tp4_rho0p0025,ttft,p99,-0.07535478089127973,0.0205549324689376,-5.479984842234213,PASS_SUBCRITICAL
tp4_rho0p0025,tpot,mean,0.2170232618103144,0.2295170016795841,1.2493739869269715,PASS_SUBCRITICAL
tp4_rho0p0025,tpot,p50,0.22406751004936917,0.22406940610958842,0.00018960602192474862,PASS_SUBCRITICAL
tp4_rho0p0025,tpot,p90,0.1725668492041552,0.1795965483385546,0.7029699134399409,PASS_SUBCRITICAL
tp4_rho0p0025,tpot,p99,0.1603764334794579,0.254199450927048,9.382301744759008,PASS_SUBCRITICAL
tp4_rho0p0025,e2e,mean,0.18328079455378776,0.19190660476104554,0.8625810207257778,PASS_SUBCRITICAL
tp4_rho0p0025,e2e,p50,0.2057109615696404,0.21253242300694286,0.6821461437302473,PASS_SUBCRITICAL
tp4_rho0p0025,e2e,p90,0.1869703879211648,0.19279855916666536,0.5828171245500557,PASS_SUBCRITICAL
tp4_rho0p0025,e2e,p99,0.14838304065885655,0.15160588346499027,0.3222842806133719,PASS_SUBCRITICAL
tp4_rho0p005,ttft,mean,0.028648879997638963,0.08356973519379125,5.492085519615229,PASS_SUBCRITICAL
tp4_rho0p005,ttft,p50,0.028237979190582876,0.09021324737193111,6.197526818134823,PASS_SUBCRITICAL
tp4_rho0p005,ttft,p90,-0.055961238578361966,-0.0006012940489499138,-5.535994452941205,PASS_SUBCRITICAL
tp4_rho0p005,ttft,p99,-0.045491187615110146,0.05253708684194205,0.7045899226831902,PASS_SUBCRITICAL
tp4_rho0p005,tpot,mean,0.1707193936623511,0.1813506819061229,1.0631288243771824,PASS_SUBCRITICAL
tp4_rho0p005,tpot,p50,0.16851374859025317,0.1743324539248605,0.5818705334607321,PASS_SUBCRITICAL
tp4_rho0p005,tpot,p90,0.10492621353626864,0.11816822907155744,1.3242015535288796,PASS_SUBCRITICAL
tp4_rho0p005,tpot,p99,0.31231888769471766,0.3662267201704473,5.390783247572961,PASS_SUBCRITICAL
tp4_rho0p005,e2e,mean,0.151102823467785,0.1630793421767109,1.197651870892591,PASS_SUBCRITICAL
tp4_rho0p005,e2e,p50,0.1594213531394918,0.17360527090575292,1.418391776626113,PASS_SUBCRITICAL
tp4_rho0p005,e2e,p90,0.1266352410718406,0.13652104764058856,0.9885806568747962,PASS_SUBCRITICAL
tp4_rho0p005,e2e,p99,0.15576537699445703,0.17677450343779522,2.100912644333819,PASS_SUBCRITICAL
tp4_rho0p01,ttft,mean,-0.0014652143973501086,0.059650620031540064,5.8185405634189955,PASS_SUBCRITICAL
tp4_rho0p01,ttft,p50,0.22936601881498542,0.24159106387124998,1.2225045056264565,PASS_SUBCRITICAL
tp4_rho0p01,ttft,p90,-0.0646093465409875,-0.011548419893895705,-5.30609266470918,PASS_SUBCRITICAL
tp4_rho0p01,ttft,p99,-0.09621678853313553,-0.015475807392170575,-8.074098114096495,PASS_SUBCRITICAL
tp4_rho0p01,tpot,mean,0.06675609059150077,0.10596101263201793,3.9204922040517163,PASS_SUBCRITICAL
tp4_rho0p01,tpot,p50,0.08477302587551214,0.09703754188844527,1.2264516012933129,PASS_SUBCRITICAL
tp4_rho0p01,tpot,p90,0.0379183273767328,0.08369800201750718,4.577967464077439,PASS_SUBCRITICAL
tp4_rho0p01,tpot,p99,-0.010663954751357074,0.06667813160571406,5.601417685435699,PASS_SUBCRITICAL
tp4_rho0p01,e2e,mean,0.07212904317306096,0.09777288053702092,2.564383736395996,PASS_SUBCRITICAL
tp4_rho0p01,e2e,p50,0.12069215463307655,0.1388377217196832,1.8145567086606653,PASS_SUBCRITICAL
tp4_rho0p01,e2e,p90,0.03317293927357903,0.058060760567474216,2.4887821293895183,PASS_SUBCRITICAL
tp4_rho0p01,e2e,p99,0.0338378271163308,0.06380410696893425,2.996627985260345,PASS_SUBCRITICAL
1 cell metric quantile old_bias new_bias abs_bias_delta_pp validity
2 tp1_rho0p00125 ttft mean -0.13461915993830945 -0.06328385143753473 -7.133530850077471 GATE_FAIL_DIAGNOSTIC
3 tp1_rho0p00125 ttft p50 -0.1888032883165517 -0.1117698463137391 -7.7033442002812595 GATE_FAIL_DIAGNOSTIC
4 tp1_rho0p00125 ttft p90 -0.23223319000679374 -0.0884727580604037 -14.376043194639005 GATE_FAIL_DIAGNOSTIC
5 tp1_rho0p00125 ttft p99 -0.16824856840168442 -0.08585570369326061 -8.239286470842382 GATE_FAIL_DIAGNOSTIC
6 tp1_rho0p00125 tpot mean 0.1305653876178269 0.14303898259628217 1.2473594978455265 GATE_FAIL_DIAGNOSTIC
7 tp1_rho0p00125 tpot p50 0.16837991778063455 0.1688780144991309 0.04980967184963492 GATE_FAIL_DIAGNOSTIC
8 tp1_rho0p00125 tpot p90 0.012855564422932954 0.02293841255516459 1.0082848132231637 GATE_FAIL_DIAGNOSTIC
9 tp1_rho0p00125 tpot p99 -0.08151080694091946 -0.06480973978913974 -1.6701067151779714 GATE_FAIL_DIAGNOSTIC
10 tp1_rho0p00125 e2e mean 0.06539157436639341 0.0838094906426525 1.8417916276259092 GATE_FAIL_DIAGNOSTIC
11 tp1_rho0p00125 e2e p50 0.10867934171755954 0.11142188099086506 0.2742539273305519 GATE_FAIL_DIAGNOSTIC
12 tp1_rho0p00125 e2e p90 0.10946179160934073 0.11872607178176105 0.9264280172420314 GATE_FAIL_DIAGNOSTIC
13 tp1_rho0p00125 e2e p99 -0.0657146472433028 -0.04585643943465822 -1.9858207808644577 GATE_FAIL_DIAGNOSTIC
14 tp1_rho0p0025 ttft mean -0.17677087458166194 -0.09487906467830536 -8.189180990335657 GATE_FAIL_DIAGNOSTIC
15 tp1_rho0p0025 ttft p50 0.008311436147272566 0.015508635458377175 0.7197199311104608 GATE_FAIL_DIAGNOSTIC
16 tp1_rho0p0025 ttft p90 -0.2608553178054708 -0.129865645852306 -13.098967195316478 GATE_FAIL_DIAGNOSTIC
17 tp1_rho0p0025 ttft p99 -0.22116442296103195 -0.14410151018893788 -7.706291277209407 GATE_FAIL_DIAGNOSTIC
18 tp1_rho0p0025 tpot mean 0.014380733682430142 0.052302860177306544 3.7922126494876403 GATE_FAIL_DIAGNOSTIC
19 tp1_rho0p0025 tpot p50 0.13351145963877706 0.14244154194999165 0.8930082311214588 GATE_FAIL_DIAGNOSTIC
20 tp1_rho0p0025 tpot p90 -0.05456950130438105 0.018069551387063856 -3.649994991731719 GATE_FAIL_DIAGNOSTIC
21 tp1_rho0p0025 tpot p99 -0.23635406197836167 -0.1790206784227079 -5.733338355565376 GATE_FAIL_DIAGNOSTIC
22 tp1_rho0p0025 e2e mean -0.020164189983441452 0.01563919326555167 -0.4524996717889782 GATE_FAIL_DIAGNOSTIC
23 tp1_rho0p0025 e2e p50 0.07632815851795742 0.10505951594320918 2.8731357425251765 GATE_FAIL_DIAGNOSTIC
24 tp1_rho0p0025 e2e p90 0.049809009042946335 0.08165607475145953 3.1847065708513194 GATE_FAIL_DIAGNOSTIC
25 tp1_rho0p0025 e2e p99 -0.18303183751478602 -0.15333409130015813 -2.96977462146279 GATE_FAIL_DIAGNOSTIC
26 tp2_rho0p0025 ttft mean -0.11161154024124531 -0.045433491316312524 -6.617804892493279 PASS_SUBCRITICAL
27 tp2_rho0p0025 ttft p50 -0.18420934047220774 -0.0875990583320057 -9.661028214020204 PASS_SUBCRITICAL
28 tp2_rho0p0025 ttft p90 -0.21836264490202395 -0.1142200855839675 -10.414255931805645 PASS_SUBCRITICAL
29 tp2_rho0p0025 ttft p99 -0.17283197594971011 -0.07674087497700505 -9.609110097270507 PASS_SUBCRITICAL
30 tp2_rho0p0025 tpot mean 0.13711499081487563 0.15073709978689778 1.362210897202215 PASS_SUBCRITICAL
31 tp2_rho0p0025 tpot p50 0.17555321305308488 0.18527285925405948 0.9719646200974597 PASS_SUBCRITICAL
32 tp2_rho0p0025 tpot p90 0.0591579975137338 0.07863996768588354 1.9481970172149734 PASS_SUBCRITICAL
33 tp2_rho0p0025 tpot p99 0.1300483675091633 0.1661995397125918 3.6151172203428503 PASS_SUBCRITICAL
34 tp2_rho0p0025 e2e mean 0.10925353865257875 0.12696146154030977 1.7707922887731016 PASS_SUBCRITICAL
35 tp2_rho0p0025 e2e p50 0.1339165600755408 0.14796251184179712 1.404595176625631 PASS_SUBCRITICAL
36 tp2_rho0p0025 e2e p90 0.10666828724664539 0.12767235986604622 2.1004072619400835 PASS_SUBCRITICAL
37 tp2_rho0p0025 e2e p99 -0.07437942975605877 -0.03965040123988098 -3.472902851617779 PASS_SUBCRITICAL
38 tp2_rho0p005 ttft mean -0.14129871969878843 -0.07082500585057615 -7.047371384821228 GATE_FAIL_DIAGNOSTIC
39 tp2_rho0p005 ttft p50 -0.2042045530944649 -0.1284088888361358 -7.579566425832909 GATE_FAIL_DIAGNOSTIC
40 tp2_rho0p005 ttft p90 -0.19003454588767263 -0.111136573344055 -7.889797254361763 GATE_FAIL_DIAGNOSTIC
41 tp2_rho0p005 ttft p99 -0.19373351009539902 -0.09012556161973535 -10.360794847566366 GATE_FAIL_DIAGNOSTIC
42 tp2_rho0p005 tpot mean 0.03117337438562004 0.05457336599027876 2.3399991604658723 GATE_FAIL_DIAGNOSTIC
43 tp2_rho0p005 tpot p50 0.07668249597302182 0.0872512146277895 1.0568718654767675 GATE_FAIL_DIAGNOSTIC
44 tp2_rho0p005 tpot p90 -0.03724827658429621 0.01132352325010614 -2.5924753334190074 GATE_FAIL_DIAGNOSTIC
45 tp2_rho0p005 tpot p99 -0.10050436157089844 -0.05948329733667248 -4.102106423422596 GATE_FAIL_DIAGNOSTIC
46 tp2_rho0p005 e2e mean 0.023867807624916495 0.050164236759908075 2.629642913499158 GATE_FAIL_DIAGNOSTIC
47 tp2_rho0p005 e2e p50 0.07536668131278851 0.0912493994395978 1.588271812680929 GATE_FAIL_DIAGNOSTIC
48 tp2_rho0p005 e2e p90 -0.03100470462321266 -0.0004979191794830456 -3.0506785443729614 GATE_FAIL_DIAGNOSTIC
49 tp2_rho0p005 e2e p99 -0.04909581632382212 -0.0014369075488634014 -4.765890877495872 GATE_FAIL_DIAGNOSTIC
50 tp4_rho0p0025 ttft mean 0.025022574738277282 0.07766646565061346 5.2643890912336175 PASS_SUBCRITICAL
51 tp4_rho0p0025 ttft p50 -0.040042171846277425 0.029153379043297147 -1.0888792802980278 PASS_SUBCRITICAL
52 tp4_rho0p0025 ttft p90 -0.056230097634382616 0.025222989748299444 -3.100710788608317 PASS_SUBCRITICAL
53 tp4_rho0p0025 ttft p99 -0.07535478089127973 0.0205549324689376 -5.479984842234213 PASS_SUBCRITICAL
54 tp4_rho0p0025 tpot mean 0.2170232618103144 0.2295170016795841 1.2493739869269715 PASS_SUBCRITICAL
55 tp4_rho0p0025 tpot p50 0.22406751004936917 0.22406940610958842 0.00018960602192474862 PASS_SUBCRITICAL
56 tp4_rho0p0025 tpot p90 0.1725668492041552 0.1795965483385546 0.7029699134399409 PASS_SUBCRITICAL
57 tp4_rho0p0025 tpot p99 0.1603764334794579 0.254199450927048 9.382301744759008 PASS_SUBCRITICAL
58 tp4_rho0p0025 e2e mean 0.18328079455378776 0.19190660476104554 0.8625810207257778 PASS_SUBCRITICAL
59 tp4_rho0p0025 e2e p50 0.2057109615696404 0.21253242300694286 0.6821461437302473 PASS_SUBCRITICAL
60 tp4_rho0p0025 e2e p90 0.1869703879211648 0.19279855916666536 0.5828171245500557 PASS_SUBCRITICAL
61 tp4_rho0p0025 e2e p99 0.14838304065885655 0.15160588346499027 0.3222842806133719 PASS_SUBCRITICAL
62 tp4_rho0p005 ttft mean 0.028648879997638963 0.08356973519379125 5.492085519615229 PASS_SUBCRITICAL
63 tp4_rho0p005 ttft p50 0.028237979190582876 0.09021324737193111 6.197526818134823 PASS_SUBCRITICAL
64 tp4_rho0p005 ttft p90 -0.055961238578361966 -0.0006012940489499138 -5.535994452941205 PASS_SUBCRITICAL
65 tp4_rho0p005 ttft p99 -0.045491187615110146 0.05253708684194205 0.7045899226831902 PASS_SUBCRITICAL
66 tp4_rho0p005 tpot mean 0.1707193936623511 0.1813506819061229 1.0631288243771824 PASS_SUBCRITICAL
67 tp4_rho0p005 tpot p50 0.16851374859025317 0.1743324539248605 0.5818705334607321 PASS_SUBCRITICAL
68 tp4_rho0p005 tpot p90 0.10492621353626864 0.11816822907155744 1.3242015535288796 PASS_SUBCRITICAL
69 tp4_rho0p005 tpot p99 0.31231888769471766 0.3662267201704473 5.390783247572961 PASS_SUBCRITICAL
70 tp4_rho0p005 e2e mean 0.151102823467785 0.1630793421767109 1.197651870892591 PASS_SUBCRITICAL
71 tp4_rho0p005 e2e p50 0.1594213531394918 0.17360527090575292 1.418391776626113 PASS_SUBCRITICAL
72 tp4_rho0p005 e2e p90 0.1266352410718406 0.13652104764058856 0.9885806568747962 PASS_SUBCRITICAL
73 tp4_rho0p005 e2e p99 0.15576537699445703 0.17677450343779522 2.100912644333819 PASS_SUBCRITICAL
74 tp4_rho0p01 ttft mean -0.0014652143973501086 0.059650620031540064 5.8185405634189955 PASS_SUBCRITICAL
75 tp4_rho0p01 ttft p50 0.22936601881498542 0.24159106387124998 1.2225045056264565 PASS_SUBCRITICAL
76 tp4_rho0p01 ttft p90 -0.0646093465409875 -0.011548419893895705 -5.30609266470918 PASS_SUBCRITICAL
77 tp4_rho0p01 ttft p99 -0.09621678853313553 -0.015475807392170575 -8.074098114096495 PASS_SUBCRITICAL
78 tp4_rho0p01 tpot mean 0.06675609059150077 0.10596101263201793 3.9204922040517163 PASS_SUBCRITICAL
79 tp4_rho0p01 tpot p50 0.08477302587551214 0.09703754188844527 1.2264516012933129 PASS_SUBCRITICAL
80 tp4_rho0p01 tpot p90 0.0379183273767328 0.08369800201750718 4.577967464077439 PASS_SUBCRITICAL
81 tp4_rho0p01 tpot p99 -0.010663954751357074 0.06667813160571406 5.601417685435699 PASS_SUBCRITICAL
82 tp4_rho0p01 e2e mean 0.07212904317306096 0.09777288053702092 2.564383736395996 PASS_SUBCRITICAL
83 tp4_rho0p01 e2e p50 0.12069215463307655 0.1388377217196832 1.8145567086606653 PASS_SUBCRITICAL
84 tp4_rho0p01 e2e p90 0.03317293927357903 0.058060760567474216 2.4887821293895183 PASS_SUBCRITICAL
85 tp4_rho0p01 e2e p99 0.0338378271163308 0.06380410696893425 2.996627985260345 PASS_SUBCRITICAL

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,126 @@
#!/usr/bin/env python3
"""Replay one real-trace cell with the structured-attention experiment commit."""
from __future__ import annotations
import argparse
import importlib.util
import json
import subprocess
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parent
REPO = ROOT.parents[1]
S3_REAL = REPO / "runs/frontier-s3-real-v0"
BASE_REFERENCE = (
REPO
/ "runs/frontier-collective-joint-v0/counterfactual/joint-r2/manifest.json"
)
BASE_COMMIT = "deadc4a321f0baaa534c6ebd17f974123733cdc2"
EXPERIMENT_COMMIT = "1f8900a4ac64e45754b03d0aa7c1dddab65785cf"
PATCH = ROOT / "0001-Experiment-with-structured-attention-prefill-predict.patch"
def load_s3_module():
spec = importlib.util.spec_from_file_location(
"s3_prefix_replay", S3_REAL / "run_frontier_prefix_replay.py"
)
module = importlib.util.module_from_spec(spec)
sys.path.insert(0, str(S3_REAL))
spec.loader.exec_module(module)
return module
def git(checkout: Path, *args: str) -> str:
return subprocess.check_output(
["git", "-C", str(checkout), *args], text=True
).strip()
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--trace", type=Path, required=True)
parser.add_argument("--output-root", type=Path, required=True)
parser.add_argument(
"--config",
choices=("tp4_mns16", "tp2_mns16", "tp1_mns16"),
required=True,
)
parser.add_argument("--label", required=True)
parser.add_argument("--max-tokens", type=int, required=True)
parser.add_argument("--duration-s", type=float)
parser.add_argument("--cache-root", type=Path, required=True)
parser.add_argument(
"--frontier-checkout",
type=Path,
default=Path("/tmp/frontier-attn-structured-v0"),
)
parser.add_argument(
"--attention-profile",
type=Path,
default=REPO
/ "runs/frontier-prefill-kvgrowth-fix-v0/profiles/"
"profile-v5-kvgrowth/attention.csv",
)
args = parser.parse_args()
frontier = args.frontier_checkout.resolve()
profile = args.attention_profile.resolve()
if git(frontier, "rev-parse", "HEAD") != EXPERIMENT_COMMIT:
raise SystemExit(f"unexpected experiment checkout HEAD: {frontier}")
if git(frontier, "rev-parse", "HEAD^") != BASE_COMMIT:
raise SystemExit("experiment commit is not directly based on frozen Frontier")
if git(frontier, "status", "--porcelain"):
raise SystemExit("experiment Frontier checkout must be clean")
if not profile.is_file():
raise SystemExit(f"attention profile missing: {profile}")
reference = json.loads(BASE_REFERENCE.read_text())
reference["frontier_checkout"] = str(frontier)
reference["frontier_commit"] = EXPERIMENT_COMMIT
generated_reference = ROOT / "frontier-reference.json"
generated_reference.write_text(json.dumps(reference, indent=2))
module = load_s3_module()
module.REFERENCE = generated_reference
module.EXPECTED_FRONTIER_COMMIT = EXPERIMENT_COMMIT
original_replace = module.replace_flag
def replace_and_override(argv: list[str], flag: str, value: str) -> None:
original_replace(argv, flag, value)
if flag.endswith("trace_file"):
atten_flag = (
"--random_forrest_execution_time_predictor_config_atten_input_file"
)
original_replace(argv, atten_flag, str(profile))
no_cache = (
"--random_forrest_execution_time_predictor_config_no_cache"
)
if no_cache in argv:
argv.remove(no_cache)
module.replace_flag = replace_and_override
module.parse_args = lambda: args
module.main()
manifest_path = args.output_root / "manifest.json"
manifest = json.loads(manifest_path.read_text())
manifest.update(
{
"schema": "frontier-attn-structured-replay-v1",
"frontier_base_commit": BASE_COMMIT,
"frontier_experiment_commit": EXPERIMENT_COMMIT,
"frontier_patch": str(PATCH.resolve()),
"frontier_patch_sha256": module.sha256(PATCH),
"attention_profile_override": str(profile),
"attention_profile_sha256": module.sha256(profile),
"model_cache_enabled": True,
}
)
manifest_path.write_text(json.dumps(manifest, indent=2))
print(f"structured replay done: {args.output_root}")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,153 @@
# Frontier code-trace campaign handoff
Phase A code prefill+decode 已进入 61min real matrix。data/profile、
max-length、Frontier rho calibration 和 TP2/TP4 paired canary 均已完成;
第一批 TP4 三个 load 与 TP2 low-rho diagnostic 正在 dash1dash4 并行
运行。code prefill-only 的独立 sim calibration 也已完成。
完整设计与 gate 见 [`experiment-card.md`](experiment-card.md)。
## 当前资产与下一步
- development window0513 `[3480,7140)`61min
- held-out window0529 `[2640,6240)`,只在 development 判据冻结后使用;
- profile`profiles/profile-v6-code-longctx/`,覆盖 TP1/2/4 和 131072
KV context
- full paired inputsCPFS
`runs/frontier-code-trace-v0/inputs/full-r0p{0002,0004,0008,0016}-v1/`
- compact provenance`results/calibration-summary.json`
`results/prefill-only-calibration-summary.json`
`results/canary-analysis-tp{2,4}-v*.json`
`results/paired-input-manifests/`
- 当前 A4 wave 1TP4 `rho={0.0002,0.0008,0.0016}` trial 1以及
TP2 `rho=0.0002` trial 1 diagnostic
- TP4 canary 的 TTFT/E2E、prefix hit 与 decode batch 通过TP2
TTFT p90 低估 32.1%,因此 TP2 其余 cell 暂不扩展;
- prefill-onlyTP2 已冻结 `rho={0.0004,0.0008,0.0016}`
TP4 到 `0.0032` 仍亚临界,需追加更高 rho 后冻结 near-knee。
source trace 的远端位置是:
```text
/home/admin/cpfs/wjh/ali-trace/trace-glm5.1-formatted/
```
以下命令保留为从 source 重新构建时的复现入口。
## 1. 审计所有 1h+ code source
在持有 trace 的机器、repo 根目录执行:
```bash
python3 runs/frontier-code-trace-v0/audit_code_trace.py \
--trace-root ~/ali-trace/trace-glm5.1-formatted \
--output runs/frontier-code-trace-v0/inputs/code-audit.json
```
如果目录里混有非 request JSONL先只读列举文件再用多个 `--source` 显式指定。审计输出必须满足:
```text
data_gate = PASS
selected.hash_contract.exact_source_block_size != null
max_model_len_recommendation != null
selected.selected_window_stats.max_model_len_coverage[推荐值].coverage = 1.0
```
全量审计已确认 source block size=512development source window 若 100%
覆盖需要 262144但正式 server cap 以 session-sampled paired cell 的实际
`ISL+OSL max` 向上对齐,不能把 full-window 262144 无条件套到低 rho cell。
审计会单独记录并排除 `input_length<=0``output_length<=0` 的 source
行;这些行只有在 raw trace 同样显示 zero usage/empty response 时才按
“未发生模型执行”处理,不能无记录过滤。
## 2. 物化稳定窗口
```bash
python3 runs/frontier-code-trace-v0/prepare_code_window.py \
--audit runs/frontier-code-trace-v0/inputs/code-audit.json \
--output-root runs/frontier-code-trace-v0/inputs/code-window
```
输出是 6075min `code-raw-window.jsonl` 和 manifest。source 文件不修改。
## 3. 生成 P+D paired trace
若没有 prompt sidecar先生成 shape/prefix-faithful synthetic prompts
```bash
python3 runs/frontier-s3-real-v0/remap_hash_blocks.py \
--input runs/frontier-code-trace-v0/inputs/code-window/code-raw-window.jsonl \
--output-root runs/frontier-code-trace-v0/inputs/code-pd-rho-max \
--source-block-size 512 \
--workload-mode prefill_decode \
--rho 1.0 \
--max-total-tokens 131072 \
--validate-parents
```
命令中的 `512``131072` 必须替换为 audit manifest 值。若存在对齐 prompt sidecar`--prompt``--tokenizer`,并要求 synthetic fallback 为 0。
正式 rho 不能直接用 1.0;先从最大 remap cache 按 session-coherent `sampling_u` 过滤,分别标定 low/mid/near-knee。
## 4. 生成 prefill-only paired trace
对 chat/code 使用同一个转换接口:
```bash
python3 runs/frontier-s3-real-v0/remap_hash_blocks.py \
--input INPUT_WINDOW.jsonl \
--output-root OUTPUT_ROOT \
--source-block-size SOURCE_BLOCK_SIZE \
--workload-mode prefill_only \
--rho RHO \
--max-total-tokens MAX_MODEL_LEN \
--validate-parents
```
该模式会同时把 Frontier `num_decode_tokens`、real request `min/max_tokens` 和 remapped row 的 `output_length` 固定为 1。
## 5. max-model-len 真机 gate
现有 real runner 新增了三个显式环境变量chat 默认行为不变:
```bash
MAX_MODEL_LEN=ACTUAL_CELL_MAX_ROUNDED_UP \
TRACE_INPUT_ROOT=/absolute/path/to/materialized/code-cell \
ALLOW_SYNTHETIC_PROMPTS=true \
OUTPUT_ROOT=/absolute/path/to/new/output \
bash runs/frontier-s3-real-v0/run_full_real.sh RHO_LABEL tp4_mns16 1 PORT
```
- `MAX_MODEL_LEN` 必须覆盖 manifest 中该 paired cell 的实际最大请求;
- `TRACE_INPUT_ROOT` 内必须有 `real_requests.jsonl``manifest.json`
- synthetic prompt 默认拒绝,只有在 experiment card 明确降级 claim 后才设为 `true`
- runner 会在启动前扫描 paired requests若任何 `ISL+OSL` 超 cap 立即失败。
长上下文 server 必须同时设置
`VLLM_ALLOW_LONG_MAX_MODEL_LEN=1`
`--hf-overrides '{"max_position_embeddings":MAX_MODEL_LEN}'`runner 已在
`ALLOW_LONG_CONTEXT_SERVER=true` 时自动处理。长上下文默认使用 host-local
vLLM compile cache并按 topology 复用 FlashInfer workspace启动 compile
不进入 workload latency。
## 6. decode-only
当前 materializer 故意不提供 `decode_only` 选项。已安装 vLLM 0.20.0
包含 `DecodeBenchConnector`,但它在首次 admission 后同步填 dummy KV
fill time 必须与 KV-ready arrival 分离。Frontier `Request` 支持
`num_processed_tokens`,当前 trace generator 尚未从 CSV 注入该值。
只有 real 首步无 prefill、sim ledger 首步为 decode 的 C0 gate 通过后,
才创建 strict decode-only jobs。
## 本地验证
```bash
python3 -m unittest -v \
runs/frontier-code-trace-v0/test_code_trace_preflight.py \
runs/frontier-s3-real-v0/test_remap_hash_blocks.py \
runs/frontier-s3-real-v0/test_select_chat_window.py
python3 -m py_compile \
runs/frontier-code-trace-v0/*.py \
runs/frontier-s3-real-v0/*.py
bash -n runs/frontier-s3-real-v0/run_full_real.sh
```

View File

@@ -0,0 +1,424 @@
#!/usr/bin/env python3
"""Analyze one paired 10-minute code-trace real/sim canary topology."""
from __future__ import annotations
import argparse
import csv
import hashlib
import json
import math
import re
import statistics
from collections import Counter
from pathlib import Path
from typing import Any
METRICS = ("ttft", "tpot", "e2e")
TPOT_MIN_OUTPUT_TOKENS = (2, 8, 32)
SLO_TARGET_PASS_RATE = 0.95
PROM_COUNTERS = ("vllm:prefix_cache_queries_total", "vllm:prefix_cache_hits_total")
csv.field_size_limit(16 * 1024 * 1024)
ITERATION_RE = re.compile(
r"Iteration.*?:\s+"
r"(?P<context_requests>\d+) context requests, "
r"(?P<context_tokens>\d+) context tokens, "
r"(?P<generation_requests>\d+) generation requests, "
r"(?P<generation_tokens>\d+) generation tokens"
)
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser()
parser.add_argument("--input-root", type=Path, required=True)
parser.add_argument("--sim-root", type=Path, required=True)
parser.add_argument("--real-root", type=Path, action="append", required=True)
parser.add_argument("--topology", required=True)
parser.add_argument("--output", type=Path, required=True)
return parser.parse_args()
def sha256(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as stream:
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def percentile(values: list[float], q: float) -> float:
ordered = sorted(values)
position = (len(ordered) - 1) * q
lower = math.floor(position)
upper = math.ceil(position)
if lower == upper:
return ordered[lower]
return ordered[lower] * (upper - position) + ordered[upper] * (position - lower)
def distribution(values: list[float]) -> dict[str, float | int]:
if not values:
raise ValueError("empty distribution")
return {
"count": len(values),
"mean": statistics.fmean(values),
"p50": percentile(values, 0.5),
"p90": percentile(values, 0.9),
"p95": percentile(values, 0.95),
"p99": percentile(values, 0.99),
"max": max(values),
}
def read_csv(path: Path) -> list[dict[str, str]]:
with path.open(newline="") as stream:
return list(csv.DictReader(stream))
def find_one(root: Path, name: str) -> Path:
matches = list(root.glob(f"**/{name}"))
if len(matches) != 1:
raise ValueError(f"expected one {name} below {root}, found {matches}")
return matches[0]
def prom_counter(path: Path, name: str) -> float:
values = []
with path.open() as stream:
for line in stream:
if line.startswith(name + "{") or line.startswith(name + " "):
values.append(float(line.rsplit(maxsplit=1)[1]))
if not values:
raise ValueError(f"{path}: missing Prometheus counter {name}")
return sum(values)
def prefix_cache_delta(root: Path) -> dict[str, float]:
before = root / "metrics/before.prom"
after = root / "metrics/after.prom"
deltas = {
name: prom_counter(after, name) - prom_counter(before, name)
for name in PROM_COUNTERS
}
queries = deltas[PROM_COUNTERS[0]]
hits = deltas[PROM_COUNTERS[1]]
if queries <= 0 or hits < 0 or hits > queries:
raise ValueError(f"{root}: invalid prefix counter deltas {deltas}")
return {
"query_tokens": queries,
"hit_tokens": hits,
"hit_ratio": hits / queries,
}
def real_decode_batch(root: Path) -> dict[str, Any]:
counts: Counter[int] = Counter()
mixed_steps = 0
for path in sorted(root.rglob("server.log")):
with path.open(errors="replace") as stream:
for line in stream:
match = ITERATION_RE.search(line)
if match is None:
continue
context_requests = int(match.group("context_requests"))
generation_requests = int(match.group("generation_requests"))
generation_tokens = int(match.group("generation_tokens"))
if context_requests:
mixed_steps += 1
continue
if generation_requests and generation_tokens == generation_requests:
counts[generation_requests] += 1
if not counts:
return {
"steps": 0,
"mixed_steps_excluded": mixed_steps,
"max": None,
"share_gt_1": None,
"histogram": {},
}
steps = sum(counts.values())
return {
"steps": steps,
"mixed_steps_excluded": mixed_steps,
"max": max(counts),
"share_gt_1": sum(value for key, value in counts.items() if key > 1) / steps,
"histogram": {str(key): value for key, value in sorted(counts.items())},
}
def load_sim(root: Path, trace: list[dict[str, str]], trace_sha: str) -> dict[str, Any]:
manifest = json.loads((root / "manifest.json").read_text())
if manifest["trace_sha256"] != trace_sha:
raise ValueError(f"{root}: sim/input trace SHA mismatch")
rows = read_csv(find_one(root / "metrics", "request_metrics.csv"))
if len(rows) != len(trace):
raise ValueError(f"{root}: sim/input request count mismatch")
values = {
"ttft": [float(row["ttft"]) for row in rows],
"tpot": [float(row["tpot"]) for row in rows if row["tpot"].strip()],
"e2e": [float(row["request_e2e_time"]) for row in rows],
"waiting": [float(row["request_waiting_time_total"]) for row in rows],
}
completions = [
float(trace_row["arrived_at"]) + float(metric_row["request_e2e_time"]) / 1000
for trace_row, metric_row in zip(trace, rows)
]
tail_index = max(range(len(completions)), key=completions.__getitem__)
last_arrival = max(float(row["arrived_at"]) for row in trace)
summary = json.loads((root / "summary.json").read_text())
slo_pass = []
for trace_row, metric_row in zip(trace, rows):
input_tokens = int(trace_row["num_prefill_tokens"])
ttft_threshold_ms = 1000 + 1000 * input_tokens / 8000
tpot = (
float(metric_row["tpot"])
if metric_row["tpot"].strip()
else None
)
slo_pass.append(
float(metric_row["ttft"]) <= ttft_threshold_ms
and (tpot is None or tpot <= 150)
)
return {
"values": values,
"tpot_by_min_output_tokens": {
str(threshold): [
float(metric_row["tpot"])
for trace_row, metric_row in zip(trace, rows)
if int(trace_row["num_decode_tokens"]) >= threshold
and metric_row["tpot"].strip()
]
for threshold in TPOT_MIN_OUTPUT_TOKENS
},
"slo": {
"passed": sum(slo_pass),
"pass_rate": sum(slo_pass) / len(slo_pass),
"feasible": sum(slo_pass) / len(slo_pass) >= SLO_TARGET_PASS_RATE,
},
"summary": summary,
"drain": {
"last_arrival_s": last_arrival,
"last_completion_s": completions[tail_index],
"tail_after_last_arrival_s": completions[tail_index] - last_arrival,
"tail_driver": {
"request_index": tail_index,
"arrival_s": float(trace[tail_index]["arrived_at"]),
"arrival_before_cutoff_s": last_arrival
- float(trace[tail_index]["arrived_at"]),
"input_tokens": int(trace[tail_index]["num_prefill_tokens"]),
"output_tokens": int(trace[tail_index]["num_decode_tokens"]),
"waiting_ms": values["waiting"][tail_index],
"e2e_ms": values["e2e"][tail_index],
},
},
}
def load_real(
root: Path,
input_manifest: dict[str, Any],
trace: list[dict[str, str]],
) -> dict[str, Any]:
result_path = root / "results/result.json"
result = json.loads(result_path.read_text())
if result["contract"]["row_vector_sha256"] != input_manifest["paired_row_vector_sha256"]:
raise ValueError(f"{root}: real/input row digest mismatch")
requests = result["requests"]
if len(requests) != len(trace) or not all(row["success"] for row in requests):
raise ValueError(f"{root}: incomplete or failed real request vector")
for index, (request, trace_row) in enumerate(zip(requests, trace)):
observed = (int(request["input_tokens"]), int(request["requested_output_tokens"]))
expected = (
int(trace_row["num_prefill_tokens"]),
int(trace_row["num_decode_tokens"]),
)
if observed != expected:
raise ValueError(f"{root}: request {index} shape {observed} != {expected}")
values = {
metric: [
float(request[f"{metric}_ms"])
for request in requests
if request.get(f"{metric}_ms") is not None
]
for metric in METRICS
}
completions = [
float(request["admitted_s"]) + float(request["e2e_ms"]) / 1000
for request in requests
]
tail_index = max(range(len(completions)), key=completions.__getitem__)
last_arrival = max(float(request["scheduled_s"]) for request in requests)
return {
"root": str(root),
"result_sha256": sha256(result_path),
"values": values,
"tpot_by_min_output_tokens": {
str(threshold): [
float(request["tpot_ms"])
for request in requests
if int(request["requested_output_tokens"]) >= threshold
and request.get("tpot_ms") is not None
]
for threshold in TPOT_MIN_OUTPUT_TOKENS
},
"slo": {
"passed": sum(bool(request["slo_pass"]) for request in requests),
"pass_rate": sum(bool(request["slo_pass"]) for request in requests)
/ len(requests),
"feasible": sum(bool(request["slo_pass"]) for request in requests)
/ len(requests)
>= SLO_TARGET_PASS_RATE,
},
"summary": result["summary"],
"prefix_cache": prefix_cache_delta(root),
"decode_batch": real_decode_batch(root),
"drain": {
"last_arrival_s": last_arrival,
"last_completion_s": completions[tail_index],
"tail_after_last_arrival_s": completions[tail_index] - last_arrival,
"tail_driver": {
"request_index": tail_index,
"arrival_s": float(requests[tail_index]["scheduled_s"]),
"arrival_before_cutoff_s": last_arrival
- float(requests[tail_index]["scheduled_s"]),
"input_tokens": int(requests[tail_index]["input_tokens"]),
"output_tokens": int(requests[tail_index]["requested_output_tokens"]),
"admission_lag_ms": float(requests[tail_index]["admission_lag_ms"]),
"e2e_ms": float(requests[tail_index]["e2e_ms"]),
},
},
}
def main() -> None:
args = parse_args()
input_manifest = json.loads((args.input_root / "manifest.json").read_text())
trace_path = args.input_root / "frontier.csv"
trace = read_csv(trace_path)
if len(trace) != input_manifest["requests"]:
raise ValueError("input manifest/trace request count mismatch")
trace_sha = sha256(trace_path)
sim = load_sim(args.sim_root, trace, trace_sha)
reals = [load_real(root, input_manifest, trace) for root in args.real_root]
pooled = {
metric: [value for real in reals for value in real["values"][metric]]
for metric in METRICS
}
latency = {}
for metric in METRICS:
real_dist = distribution(pooled[metric])
real_per_trial = [
distribution(real["values"][metric]) for real in reals
]
real_reference = {
statistic: statistics.fmean(
float(trial[statistic]) for trial in real_per_trial
)
for statistic in ("mean", "p50", "p90", "p95", "p99")
}
sim_dist = distribution(sim["values"][metric])
latency[metric] = {
"real": real_dist,
"real_trial_statistic_mean": real_reference,
"sim": sim_dist,
"relative_bias_percent": {
statistic: 100
* (float(sim_dist[statistic]) - real_reference[statistic])
/ real_reference[statistic]
for statistic in ("mean", "p50", "p90", "p95", "p99")
},
"real_per_trial": real_per_trial,
}
tpot_sensitivity = {}
for threshold in TPOT_MIN_OUTPUT_TOKENS:
key = str(threshold)
real_per_trial = [
distribution(real["tpot_by_min_output_tokens"][key])
for real in reals
]
real_values = [
value
for real in reals
for value in real["tpot_by_min_output_tokens"][key]
]
sim_values = sim["tpot_by_min_output_tokens"][key]
real_dist = distribution(real_values)
real_reference = {
statistic: statistics.fmean(
float(trial[statistic]) for trial in real_per_trial
)
for statistic in ("mean", "p50", "p90", "p95", "p99")
}
sim_dist = distribution(sim_values)
tpot_sensitivity[key] = {
"real": real_dist,
"real_trial_statistic_mean": real_reference,
"real_per_trial": real_per_trial,
"sim": sim_dist,
"relative_bias_percent": {
statistic: 100
* (float(sim_dist[statistic]) - real_reference[statistic])
/ real_reference[statistic]
for statistic in ("mean", "p50", "p90", "p95", "p99")
},
}
payload = {
"schema": "frontier-code-trace-canary-analysis-v1",
"topology": args.topology,
"requests_per_trial": len(trace),
"trials": len(reals),
"input": {
"manifest": str(args.input_root / "manifest.json"),
"paired_row_vector_sha256": input_manifest["paired_row_vector_sha256"],
"frontier_csv_sha256": trace_sha,
},
"latency_ms": latency,
"tpot_by_min_output_tokens": tpot_sensitivity,
"slo": {
"definition": {
"ttft_ms": "1000 + 1000 * input_tokens / 8000",
"tpot_ms": 150,
"target_pass_rate": SLO_TARGET_PASS_RATE,
},
"real_per_trial": [real["slo"] for real in reals],
"sim": sim["slo"],
"feasibility_flip": any(
real["slo"]["feasible"] != sim["slo"]["feasible"]
for real in reals
),
},
"prefix_cache": {
"real_per_trial": [real["prefix_cache"] for real in reals],
"real_hit_ratio_mean": statistics.fmean(
real["prefix_cache"]["hit_ratio"] for real in reals
),
"sim": sim["summary"]["prefix_cache"],
},
"drain": {
"interpretation": (
"Report the max-completion request explicitly; a response that "
"arrived well before the cutoff can create a long drain tail "
"without implying queue accumulation."
),
"real_per_trial": [real["drain"] for real in reals],
"sim": sim["drain"],
},
"sim_decode_batch": sim["summary"]["decode_batch"],
"real_decode_batch_per_trial": [real["decode_batch"] for real in reals],
"real_artifacts": [
{
"root": real["root"],
"result_sha256": real["result_sha256"],
}
for real in reals
],
}
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n")
print(json.dumps({"output": str(args.output), "topology": args.topology}))
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,382 @@
#!/usr/bin/env python3
"""Audit long code traces before choosing a replay window and max model length."""
from __future__ import annotations
import argparse
import json
import math
import statistics
from collections import Counter
from pathlib import Path
from typing import Any, Iterable, Sequence
BLOCK_SIZE_CANDIDATES = (16, 32, 64, 128, 256, 512, 1024)
MAX_MODEL_LEN_CANDIDATES = (40960, 65536, 98304, 131072, 262144)
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser()
parser.add_argument("--trace-root", type=Path)
parser.add_argument("--source", type=Path, action="append")
parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--min-minutes", type=int, default=60)
parser.add_argument("--max-minutes", type=int, default=75)
parser.add_argument("--bin-seconds", type=int, default=60)
parser.add_argument("--max-acceptable-gap-s", type=float, default=5.0)
parser.add_argument("--model-position-limit", type=int, default=262144)
return parser.parse_args()
def percentile(values: Sequence[int | float], fraction: float) -> float | None:
if not values:
return None
ordered = sorted(float(value) for value in values)
position = (len(ordered) - 1) * fraction
lower = math.floor(position)
upper = math.ceil(position)
if lower == upper:
return ordered[lower]
return ordered[lower] * (upper - position) + ordered[upper] * (position - lower)
def distribution(values: Sequence[int | float]) -> dict[str, int | float | None]:
return {
"count": len(values),
"min": min(values) if values else None,
"p50": percentile(values, 0.50),
"p90": percentile(values, 0.90),
"p95": percentile(values, 0.95),
"p99": percentile(values, 0.99),
"max": max(values) if values else None,
"mean": statistics.fmean(values) if values else None,
}
def parse_hash_ids(value: Any) -> list[Any]:
if isinstance(value, list):
return value
if isinstance(value, str):
stripped = value.strip()
if not stripped:
return []
if stripped.startswith("["):
decoded = json.loads(stripped)
if not isinstance(decoded, list):
raise ValueError("hash_ids JSON must decode to a list")
return decoded
delimiter = "|" if "|" in stripped else ","
return [part for part in stripped.split(delimiter) if part.strip()]
if value is None:
return []
return [value]
def iter_jsonl(path: Path) -> Iterable[tuple[int, dict[str, Any]]]:
with path.open() as stream:
for line_number, line in enumerate(stream, 1):
if not line.strip():
continue
row = json.loads(line)
if not isinstance(row, dict):
raise ValueError(f"{path}:{line_number}: row must be an object")
yield line_number, row
def choose_window(
*,
counts: Sequence[int],
max_gaps: Sequence[float],
first_timestamp: float,
min_minutes: int,
max_minutes: int,
bin_seconds: int,
max_acceptable_gap_s: float,
) -> dict[str, Any] | None:
candidates = []
for minutes in range(max_minutes, min_minutes - 1, -1):
bins = math.ceil(minutes * 60 / bin_seconds)
for start_bin in range(0, len(counts) - bins + 1):
selected = counts[start_bin : start_bin + bins]
mean = statistics.fmean(selected)
cv = statistics.pstdev(selected) / mean if mean else math.inf
max_gap = max(max_gaps[start_bin : start_bin + bins], default=0.0)
candidates.append(
{
"_score": (
max_gap > max_acceptable_gap_s,
cv,
max_gap,
-minutes,
start_bin,
),
"start_bin": start_bin,
"minutes": minutes,
"count_mean_per_bin": mean,
"count_cv": cv,
"count_min_per_bin": min(selected),
"count_max_per_bin": max(selected),
"max_gap_s": max_gap,
}
)
if not candidates:
return None
chosen = min(candidates, key=lambda item: item["_score"])
chosen.pop("_score")
chosen["start_timestamp"] = first_timestamp + chosen["start_bin"] * bin_seconds
chosen["end_timestamp"] = chosen["start_timestamp"] + chosen["minutes"] * 60
return chosen
def scan_source(path: Path, args: argparse.Namespace) -> dict[str, Any]:
rows = 0
source_rows = 0
invalid_zero_token_rows = 0
invalid_zero_token_examples: list[dict[str, Any]] = []
first_timestamp = None
last_timestamp = None
previous_timestamp = None
counts: Counter[int] = Counter()
max_gaps: dict[int, float] = {}
input_lengths: list[int] = []
output_lengths: list[int] = []
total_lengths: list[int] = []
hash_rows = 0
hash_matches = Counter()
prompt_rows = 0
sampling_rows = 0
schema_keys: Counter[str] = Counter()
for line_number, row in iter_jsonl(path):
source_rows += 1
missing = [
key
for key in ("timestamp", "input_length", "output_length")
if key not in row
]
if missing:
raise ValueError(f"{path}:{line_number}: missing required fields {missing}")
timestamp = float(row["timestamp"])
input_tokens = int(row["input_length"])
output_tokens = int(row["output_length"])
schema_keys.update(row.keys())
if input_tokens <= 0 or output_tokens <= 0:
invalid_zero_token_rows += 1
if len(invalid_zero_token_examples) < 20:
invalid_zero_token_examples.append(
{
"line_number": line_number,
"chat_id": row.get("chat_id"),
"timestamp": timestamp,
"input_length": input_tokens,
"output_length": output_tokens,
}
)
continue
if first_timestamp is None:
first_timestamp = timestamp
if previous_timestamp is not None and timestamp < previous_timestamp:
raise ValueError(
f"{path}:{line_number}: timestamp {timestamp} < {previous_timestamp}"
)
bin_index = math.floor((timestamp - first_timestamp) / args.bin_seconds)
counts[bin_index] += 1
if previous_timestamp is not None:
previous_bin = math.floor(
(previous_timestamp - first_timestamp) / args.bin_seconds
)
max_gaps[previous_bin] = max(
max_gaps.get(previous_bin, 0.0),
timestamp - previous_timestamp,
)
input_lengths.append(input_tokens)
output_lengths.append(output_tokens)
total_lengths.append(input_tokens + output_tokens)
hashes = parse_hash_ids(row.get("hash_ids"))
if hashes:
hash_rows += 1
for block_size in BLOCK_SIZE_CANDIDATES:
if len(hashes) == math.ceil(input_tokens / block_size):
hash_matches[block_size] += 1
prompt_rows += int(
isinstance(row.get("prompt"), (str, list)) and bool(row.get("prompt"))
)
sampling_rows += int("sampling_u" in row)
rows += 1
previous_timestamp = timestamp
last_timestamp = timestamp
if not rows or first_timestamp is None or last_timestamp is None:
raise ValueError(f"{path}: empty trace")
total_bins = math.floor((last_timestamp - first_timestamp) / args.bin_seconds) + 1
chosen = choose_window(
counts=[counts[index] for index in range(total_bins)],
max_gaps=[max_gaps.get(index, 0.0) for index in range(total_bins)],
first_timestamp=first_timestamp,
min_minutes=args.min_minutes,
max_minutes=args.max_minutes,
bin_seconds=args.bin_seconds,
max_acceptable_gap_s=args.max_acceptable_gap_s,
)
return {
"source": str(path.resolve()),
"rows": rows,
"source_rows": source_rows,
"invalid_zero_token_rows": invalid_zero_token_rows,
"invalid_zero_token_fraction": invalid_zero_token_rows / source_rows,
"invalid_zero_token_examples": invalid_zero_token_examples,
"first_timestamp": first_timestamp,
"last_timestamp": last_timestamp,
"span_s": last_timestamp - first_timestamp,
"request_rate_per_s": rows / max(last_timestamp - first_timestamp, 1.0),
"input_length": distribution(input_lengths),
"output_length": distribution(output_lengths),
"total_length": distribution(total_lengths),
"over_max_model_len": {
str(limit): {
"requests": sum(value > limit for value in total_lengths),
"fraction": sum(value > limit for value in total_lengths) / rows,
}
for limit in MAX_MODEL_LEN_CANDIDATES
},
"hash_contract": {
"rows_with_hash_ids": hash_rows,
"candidate_exact_match_rows": {
str(size): hash_matches[size] for size in BLOCK_SIZE_CANDIDATES
},
"exact_source_block_size": next(
(
size
for size in BLOCK_SIZE_CANDIDATES
if hash_rows and hash_matches[size] == hash_rows
),
None,
),
},
"prompt_rows": prompt_rows,
"sampling_u_rows": sampling_rows,
"schema_field_counts": dict(sorted(schema_keys.items())),
"stable_window": chosen,
}
def scan_window(source: Path, window: dict[str, Any]) -> dict[str, Any]:
start = float(window["start_timestamp"])
end = float(window["end_timestamp"])
inputs: list[int] = []
outputs: list[int] = []
totals: list[int] = []
for _, row in iter_jsonl(source):
timestamp = float(row["timestamp"])
if timestamp < start:
continue
if timestamp >= end:
break
input_tokens = int(row["input_length"])
output_tokens = int(row["output_length"])
if input_tokens <= 0 or output_tokens <= 0:
continue
inputs.append(input_tokens)
outputs.append(output_tokens)
totals.append(input_tokens + output_tokens)
return {
"requests": len(totals),
"input_length": distribution(inputs),
"output_length": distribution(outputs),
"total_length": distribution(totals),
"max_model_len_coverage": {
str(limit): {
"covered_requests": sum(value <= limit for value in totals),
"excluded_requests": sum(value > limit for value in totals),
"coverage": sum(value <= limit for value in totals) / len(totals),
}
for limit in MAX_MODEL_LEN_CANDIDATES
},
}
def resolve_sources(args: argparse.Namespace) -> list[Path]:
if args.source:
return [path.resolve() for path in args.source]
if args.trace_root is None:
raise ValueError("provide --trace-root or one or more --source")
sources = sorted(
path.resolve()
for path in args.trace_root.glob("*.jsonl")
if "prompt" not in path.stem.lower()
)
if not sources:
raise FileNotFoundError(f"no non-prompt JSONL files under {args.trace_root}")
return sources
def main() -> None:
args = parse_args()
if not 0 < args.min_minutes <= args.max_minutes:
raise ValueError("require 0 < min_minutes <= max_minutes")
sources = resolve_sources(args)
files = [scan_source(path, args) for path in sources]
eligible = [item for item in files if item["stable_window"] is not None]
if not eligible:
chosen = None
data_gate = "BLOCKED_NO_1H_WINDOW"
else:
chosen = min(
eligible,
key=lambda item: (
item["stable_window"]["max_gap_s"] > args.max_acceptable_gap_s,
item["stable_window"]["count_cv"],
-item["stable_window"]["minutes"],
item["source"],
),
)
chosen["selected_window_stats"] = scan_window(
Path(chosen["source"]), chosen["stable_window"]
)
exact_block_size = chosen["hash_contract"]["exact_source_block_size"]
max_total = chosen["selected_window_stats"]["total_length"]["max"]
data_gate = (
"PASS"
if exact_block_size is not None
and max_total is not None
and max_total <= args.model_position_limit
else "BLOCKED_HASH_OR_POSITION_CONTRACT"
)
recommendation = None
if chosen is not None:
maximum = chosen["selected_window_stats"]["total_length"]["max"]
recommendation = next(
(
limit
for limit in MAX_MODEL_LEN_CANDIDATES
if maximum <= limit <= args.model_position_limit
),
None,
)
payload = {
"schema": "frontier-code-trace-audit-v1",
"trace_root": str(args.trace_root.resolve()) if args.trace_root else None,
"sources": [str(path) for path in sources],
"window_policy": {
"min_minutes": args.min_minutes,
"max_minutes": args.max_minutes,
"bin_seconds": args.bin_seconds,
"max_acceptable_gap_s": args.max_acceptable_gap_s,
"selection": "lowest density CV after rejecting anomalous-gap windows",
},
"model_position_limit": args.model_position_limit,
"files": files,
"selected": chosen,
"max_model_len_recommendation": recommendation,
"data_gate": data_gate,
"runtime_gate": (
"PENDING: vLLM startup must prove enough KV blocks and nonzero "
"max concurrency at the recommended max_model_len for each TP"
),
}
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n")
print(json.dumps({"data_gate": data_gate, "output": str(args.output)}))
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,79 @@
#!/usr/bin/env python3
"""Check fresh-process repeat stability for the code long-context grid."""
from __future__ import annotations
import argparse
import json
from pathlib import Path
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser()
parser.add_argument("--first", type=Path, nargs="+", required=True)
parser.add_argument("--second", type=Path, nargs="+", required=True)
parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--max-relative-difference", type=float, default=0.05)
return parser.parse_args()
def load(paths: list[Path]) -> dict[tuple[int, str], dict]:
rows: dict[tuple[int, str], dict] = {}
for path in paths:
payload = json.loads(path.read_text())
for row in payload["rows"]:
if row.get("error"):
raise ValueError(
f"{path}: failed profile row {row['config']['batch_spec']}"
)
key = (
int(row["tensor_parallel_size"]),
str(row["config"]["batch_spec"]),
)
if key in rows:
raise ValueError(f"duplicate row {key}")
rows[key] = row
return rows
def main() -> None:
args = parse_args()
first = load(args.first)
second = load(args.second)
if first.keys() != second.keys():
raise ValueError(
f"repeat key mismatch: first_only={sorted(first.keys()-second.keys())}, "
f"second_only={sorted(second.keys()-first.keys())}"
)
comparisons = []
for key in sorted(first):
left = float(first[key]["mean_time"])
right = float(second[key]["mean_time"])
relative = abs(left - right) / ((left + right) / 2)
comparisons.append(
{
"tp": key[0],
"batch_spec": key[1],
"first_mean_s": left,
"second_mean_s": right,
"relative_difference": relative,
"pass": relative <= args.max_relative_difference,
}
)
maximum = max(item["relative_difference"] for item in comparisons)
payload = {
"schema": "frontier-code-longctx-repeat-check-v1",
"threshold": args.max_relative_difference,
"maximum_relative_difference": maximum,
"status": "PASS" if maximum <= args.max_relative_difference else "FAIL",
"comparisons": comparisons,
}
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n")
print(json.dumps({"status": payload["status"], "max": maximum}))
if payload["status"] != "PASS":
raise SystemExit(1)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,346 @@
# EXP-CODE-TRACE从 chat 1h trace 扩展到 code 与 phase-separated replay
> **状态RUNNINGPhase A code P+D。** A0 数据/profile、A1
> max-length smoke、A2 paired canary 与 A3 sim calibration 已完成;
> A4 第一批 61min real jobs 正在 `dash1`--`dash4` 运行。禁止使用 `dash0`。
## 目标与成功定义
当前 1h+ 证据只覆盖 Qwen3-30B-A3B 的生产 chat trace、prefill+decodeP+D和亚临界负载。本 campaign 分两步扩展:
1. **主任务:** 使用 `~/ali-trace/trace-glm5.1-formatted/` 中的 1h+ code trace先完成 P+D real-vs-Frontier 回放;
2. **后续 phase matrix** 对 chat/code 都补 prefill-only 和严格 decode-only。
本轮不是只看“能否跑完”。每个正式 cell 必须满足:同一 request vector、同一 arrival、同一 token shape、同一 prefix/initial-KV 合约、real 零失败、无持续 backlog并同时报告 TTFT/TPOT/E2E、queue/batch、KV/prefix state 与 5min 分窗漂移。
## 当前决策快照
| 项目 | 当前结论 | 下一 gate |
|---|---|---|
| code P+D 数据 | 61min development window、long-context profile-v6、四个 paired full 输入已冻结 | 第一批 1h real 运行中 |
| TP4 负载 | low/mid/near-knee=`rho 0.0002/0.0008/0.0016`paired canary 的 TTFT/E2E、KV、batch 通过 | 三个 load 的 trial 1 运行中 |
| TP2 负载 | `rho 0.0004` canary 暴露 TTFT p90 `-32.1%` bad case | 只跑 `rho 0.0002` 1h diagnostic暂不铺满 |
| code prefill-only | TP2 已冻结 `rho 0.0004/0.0008/0.0016``0.0032` 过载TP4 到 `0.0032` 仍亚临界 | TP4 追加更高 rho 边界 |
| strict decode-only | vLLM 0.20.0 有 `DecodeBenchConnector`Frontier trace generator 尚不能注入 initial computed tokens | C0 contract canary未进入正式结果 |
## 三种 workload mode 的冻结定义
| Mode | 保留 | 改写 | 主指标 | 明确不声称 |
|---|---|---|---|---|
| P+D | 原 ISL/OSL、arrival、session/prefix | 仅做 source block→16-token runtime block 映射 | TTFT、TPOT、E2E、hit ratio、batch/queue | 不代表 PD 分离 |
| prefill-only | 原 ISL、arrival、session/prefix | OSL 固定为 1real `min_tokens=max_tokens=1`sim decode tokens=1 | TTFT、prefill service/tokens/s、prefix hit、queue | TPOT 不定义1-token decode 只用于完成请求 |
| strict decode-only | 原 OSL、context length、arrival burst | arrival 定义为 **KV-ready time**;请求进入 decode 时已有 ISL 长度的 initial KV | TPOT、decode tokens/s、batch/queue、preemption | 不包含 prefill 与 KV transfer latency不把短 prompt proxy 称为 decode-only |
strict decode-only 必须同时具备:
- realvLLM `DecodeBenchConnector`(或等价、经验证的 initial-KV 注入);
- simFrontier request 在 admission 时已拥有相同长度/块布局的 computed KV
- 两侧都不在 decode critical path 重做 prefill
- arrival 以 KV-ready time 对齐。若只保留原 trace 的相对到达形状,结论限定为 decode engine compute/scheduling fidelity。
在该合约完成前,只允许跑并标注为 **decode-dominant proxy**,不能进入 strict decode-only 结果表。
## 为什么 code P+D 不能直接复用 chat 配置
已知历史探查显示 code trace ISL p90 约 81.9k,约 32.6% 请求超过旧 `40960` 上限;真实数值必须由本 campaign 重新审计。至少有四个独立适配面:
1. **Serving cap** `max_model_len` 必须覆盖 `ISL+OSL`,不能只看 ISL也不能静默丢掉超长请求
2. **KV capacity** Qwen3-30B 模型 position limit 为 262144但 TP1/2/4 在 H20 上是否有足够 KV blocks 是 runtime gate不由 config.json 自动保证;
3. **Prefix block** code source hash 预计为 512-token blockchat harness 原先固定 64→16
4. **Profile support** 当前修复后的 attention profile 只覆盖到约 32k KV context。即使 vLLM 能跑 128kFrontier 对 32k128k 仍会出 profile 支撑域;在补 long-context 网格前只能做诊断 replay不能做 fidelity claim。
## Hypotheses
- **H-code-generalizes** 在补齐 long-context profile 支撑域后code P+D 的 TTFT/TPOT/E2E 分布统计偏差仍处于当前 chat 量级,且 1h 残差不发散。
- **H-longctx-gap** code 的主要新增 gap 来自 32k 以上 KV-context 外推;补到 trace p99/max 对应的网格后TTFT bias 随 ISL 的二次项显著收敛。
- **H-phase-specific** prefill-only 主要暴露 long-context/profile gapstrict decode-only 主要暴露 batch-conditioned whole-layer service 与 scheduler fixed-point gap。二者不能用 P+D 的误差抵消来互相证明准确。
## Preflight gates按顺序任一失败即停止后续真机矩阵
### G0数据位置与 provenance
- 只读列举 `trace-glm5.1-formatted/*.jsonl`,记录文件大小与 SHA256
- 确认至少两个独立日期段:一个作为 development一个 held-out
- 本机当前没有该目录;仓库历史记录的远端位置为
`/home/admin/cpfs/wjh/ali-trace/trace-glm5.1-formatted/`。恢复机器后先确认 `~/ali-trace/...` 是否为同一路径/软链,不能假设。
### G11h window、schema 与 block contract
运行 `audit_code_trace.py`,要求:
- timestamp 单调,存在 6075min 连续稳定窗口;
- `timestamp/input_length/output_length` 全行存在;
- `input_length<=0``output_length<=0` 的行不进入 replay但必须计数并保留样例。已抽查的两条 0→0 行在 raw trace 中同时满足 `usage.total_tokens=0``response_message={}`,属于未发生模型执行的 source request不是 full-cache decode
- `hash_ids` 数量与某个 source block size 在全行严格满足
`ceil(ISL/source_block_size)`;预计值 512但以审计结果为准
- 记录 ISL/OSL/ISL+OSL 的 p50/p90/p95/p99/max、gap、request rate、prompt/sampling 字段覆盖。
选择窗口后用 `prepare_code_window.py` 物化只读派生文件,并按 session root 生成确定性的 `sampling_u`。另一日期段不参与 rho 与 profile 选择。
### G2`max_model_len` data gate
source-window audit 先用 `40960/65536/98304/131072/262144` 给出完整
窗口上界;真实 server 则使用能 **100% 覆盖该 rho 实际 paired requests
`ISL+OSL`** 的最小 16-token 对齐值。规则:
- sampling 只按 session-coherent `sampling_u`,不得按 token length 过滤;
- full source window 的 cap 用于记录 workload envelope不强迫低 rho cell
为未被抽中的 outlier 预留 KV capacity
- 若某 paired cell max≤131072使用 131072 或更小的对齐值;超过
131072 时按该 cell 实际 max 向上对齐,而不是直接跳到 262144
- Frontier 的 trace max tokens、predictor max tokens/request、vLLM `--max-model-len` 三处使用同一个 manifest 值。
### G3prompt 与 prefix fidelity
优先级:
1. 有对齐 prompt text sidecar用 Qwen tokenizer 重分词,要求 token length 与 trace ISL 全行一致;
2. trace 内已有 prompt text/token IDs同样做长度与 hash relation 检查;
3. 两者都没有:允许用 source hash 确定性展开为 synthetic Qwen token IDs但结果降级为 **length/arrival/prefix-shape faithful**,不声称 prompt-content 或 MoE routing faithful。
不论走哪条路径source→16 映射冲突、runtime identity collision、parent prefix violation 都必须为 0。P+D/prefill-only 两侧 prefix caching 同开;先用 510min TP4/MNS16 做 hit-ratio audit。
### G4long-context profile support
现有 profile-v5 的 KV context 上界约 32k对 code 不足。根据 development window 的 uncached-ISL 分布生成 profile-v6-code-longctx
- full chunk`q8k`context 至少覆盖 40k/56k/72k/88k/104k/120k/128k
- tail chunk从真实 `ISL mod 8192` 的 p50/p90 选择 24k/46k 代表点;
- TP1/2/4 分开采集,复测 `q1ks8k/q8ks32k` anchor
- 每点至少两次 fresh-process repeatCV≤5%anchor drift≤10%
- profile max context 必须 ≥ development window p99正式 max claim 要求 ≥ max。若只覆盖 p99max 以上请求单独列为 out-of-support不进入总体准确度数字。
这是 code P+D 正式 fidelity 的硬 gate。可以先用旧 profile 跑 diagnostic sim 来估 load但不得与真机组成最终 gap。
### G5vLLM max-length/KV runtime gate
对每个候选 topology先 TP4再 TP2TP1 后置):
1. fresh server以 manifest cap 启动;
2. 记录 vLLM 版本、model config、GPU KV blocks、maximum concurrency、启动日志
3. 发 3 个单请求ISL p50、p99、maxOSL=1usage 必须逐 token 对齐;
4. 发 5min sampled P+D canary零 OOM/timeout/preemption storm
5. 只有 maximum concurrency>1 且 canary drain tail≤窗口时长 10% 才进入 rho calibration。
`max_model_len` 变大不等于每个请求都预占最大 KV但会改变启动合法性与可表达的单请求上界实际 KV 压力仍由并发 token state 决定。
对本模型,`VLLM_ALLOW_LONG_MAX_MODEL_LEN=1` 只放宽 scheduler/config
校验,不会扩展模型内部 RoPE cacheserver 还必须显式传
`--hf-overrides '{"max_position_embeddings":147456}'`。runner 对
`MAX_MODEL_LEN>40960` 自动同时设置这两层。长上下文 job 默认使用
host-local vLLM compile cacheFlashInfer workspace 按 topology 复用,
避免每个 rho/trial 重编译同一组 fused-MoE kernels。两者只影响启动
不进入 replay latency。
### G6每种 mode 独立标定 rho
不能复用 P+D rho
- P+D 同时按 raw/prefix-adjusted prefill tokens/s 与 decode tokens/s 看 knee
- prefill-only 因 OSL=1重新按 prefill work 标定;
- strict decode-only 因无 prefill按 decode tokens/s 和 batch fixed point 标定。
每种 workload×mode 选择 `low/mid/near-knee` 三点;正式点必须亚临界:全请求完成、无持续 backlog、drain tail≤10%、waiting p99 不单调随时间增长。跨 knee 点若运行,只作为 overload boundary不支持“不发散”结论。
`drain tail` 必须同时列出最后完成请求的 arrival、ISL、OSL、waiting
和 E2E。早于 cutoff 到达但 OSL 很长的请求可以在最后 arrival 后继续
decode这属于 intrinsic response tail不等价于 arrival cutoff 时仍有
持续增长的 queue backlog。亚临界判断以 queue/waiting trajectory 和
tail driver 分解共同决定,不能只用一个 drain 秒数。
### G7strict decode-only capability gate
先在 10min synthetic trace 上验证:
- real connector 确认没有执行 prefill kernel
- Frontier ledger 第一个阶段就是 decodecomputed tokens=ISL
- 相同 context length 下两侧 KV block count 一致;
- connector preload/transfer 时间独立记账,不混入 TPOT
- decode batch telemetry 能覆盖 b1 到目标 batch。
若 vLLM 0.20 community stack 没有等价 connector严格 case 保持 BLOCKED可另跑 decode-dominant proxy但单独命名和汇报。
## 正式实验矩阵与推进顺序
### Phase Acode P+D第一优先级
1. **A0 CPU/data** G0G4
2. **A1 max-len smoke** TP4→TP2TP1 只在 KV gate 通过后加入;
3. **A2 paired 10min canary** TP4/MNS16low rhoreal+sim
4. **A3 calibration** 各 rho 只先跑 sim冻结 low/mid/near-knee
5. **A4 full** TP4/MNS16、TP2/MNS16 × 3 rho × 2 trial × 6075min
6. **A5 held-out** 只在 development window 判据冻结后,对第二日期段跑 TP4 的 mid/near-knee。
若某 topology 的 near-knee 过载,像现有 chat TP2/ρ0.01 一样排除,不为凑齐矩阵强跑。
当前状态A0A3 完成。A4 第一批为 TP4
`rho={0.0002,0.0008,0.0016}` trial 1以及 TP2 `rho=0.0002`
trial 1 diagnostic其余 TP2 cell 等该 diagnostic 验证 canary bad case
后再决定是否扩展。
### Phase Bchat/code prefill-only
- 复用各自已物化 window只把 OSL 改为 1
- primaryTP4/MNS16、TP2/MNS16 × 3 独立 rho × 2 trial
- 报 TTFT/CDF/quantiles、prefill tokens/s、prefix hit、waiting、chunk/context 分带 residual
- TPOT 记为 N/AE2E 仅作为“一 token completion”辅助值
- code 必须继续使用 profile-v6 long-contextchat 使用已验证 profile-v5。
### Phase Cchat/code strict decode-only
先做 batch-sensitive screening再决定是否铺满
- **C0 capability canary** 两 workload × TP4 × MNS{16,128}10min
- **C1 core full** TP{2,4} × MNS{16,128} × rho{low,near-knee} × 2 trial
- **C2 conditional expansion** 只有当 C1 的 batch 分布从 b≤8 跨到 b>8或 accuracy gap 随 MNS 改变>5pp才补 MNS{32,64} 与 mid rho。
decode profile/serving anchors 至少覆盖实际 batch p99。当前 whole-layer grid 只对少数 b≤8 有证据,且 b6 有长尾;在 MNS128 case 前必须补 b{1,2,4,8,16,32,64,128} 或实际访问 bucket不能把 b8 常数外推到 b128。
已安装 vLLM 0.20.0 的 `DecodeBenchConnector` 会在首次 schedule 时把
`request.num_tokens-num_computed_tokens-1` 个 token 标为 external
同步向已分配的每层 KV blocks 写 dummy non-zero values再从最后一个
prompt token 开始 forward。因此它适合测大 context 下的 decode
compute/scheduling但 connector fill 发生在 client admission 之后:
fill time 必须单独记账并从 KV-ready arrival/TPOT 口径中排除。
dummy KV 也不提供真实 prompt-content 或 MoE-routing fidelity。
Frontier commit `deadc4a3``Request` 已支持构造
`num_processed_tokens`,但 `TraceReplayRequestGenerator` 不读取该列;
因此 sim 侧仍需一个显式、可测试的 `initial_computed_tokens` trace
contract。C0 必须同时证明 real 首个 model step 是 decode、Frontier
首个 ledger stage 是 decode之后才能解除 strict decode-only 的 BLOCKED。
## 指标与判据
共同口径:
- 分布统计偏差:`(sim statistic-real statistic)/real statistic`,不是 per-request MAPE
- mean/p50/p90/p99 与 empirical CDF
- 5min 分窗,前 15min warmup 不进漂移 slope
- batch histogram、time-weighted running/waiting、drain tail、preemption
- 两 trial pooled 结果和 trial-to-trial noise floor 分开报告。
判据分两层:
1. **准确度:** primary latency mean/p90/p99 的 |bias|≤15% 为强通过1530% 为有界但需标注 correction>30% 立 bad case任何 topology 排序或 SLO feasibility 翻转都单独判 failure不能被平均值掩盖。
2. **长时稳定:** `|residual TheilSen slope|×12 / real noise floor < 1` 为 H-BOUNDED只适用于亚临界 cell。
mode-specific
- P+DTTFT/TPOT/E2E 全部 primary
- prefill-onlyTTFT primaryTPOT N/A
- strict decode-onlyTPOT primaryTTFT 仅表示 admission/connector overhead不进入 compute-fidelity gate。
## 成本与调度
- Phase A core12 个 6075min jobs2 topology×3 load×2 trial约 15 host-hours按 TP 加权约 45 H20-GPU-hours加 24 个 smoke/canary
- Phase B 两 workload24 个 full jobs按相同 75min 上界约 90 H20-GPU-hours
- Phase C 不一次铺满。C0 4 个 10min canaryC1 32 个 full jobsC2 按触发条件追加。
每个 job fresh server只在 `dash1``dash4` 全 8 卡 idle/healthy 时启动。即使 TP2/TP4 job 只用部分 GPU也不在同一 host 并跑,避免 fresh-server 空窗竞态。每一批使用新的 jobs TOML现有 dispatcher 非幂等。
## 预期产物
- `inputs/code-audit.json``inputs/code-window/window-manifest.json`
- P+D/prefill-only 的 paired `frontier.csv``real_requests.jsonl` 与 manifest
- profile-v6-code-longctx raw/merged profile 与 variance report
- 每 cell real/sim request metrics、server telemetry、stage ledger
- `results/code-pd-fidelity.md`
- 最终 `chat/code × P+D/prefill-only/decode-only` compatibility table。
## 已知边界
- code trace 来自 GLM5.1 业务serving model 是 Qwen3-30B若无原 prompt text测试只能保持 shape/prefix 结构,不能证明内容相关 routing fidelity
- `max_model_len=128k/256k` 解决的是接入上界,不自动解决 32k 以上 profile 外推;
- strict decode-only 只测 decode engine完整 PD 分离还需要单独建模 prefill、KV transfer、backpressure 与 KV-ready arrival。
## 执行记录2026-07-23
- fleet probedash1dash4 均为 8×H2032 张卡 memory.used=0、
utilization=0、无 compute process、uncorrected ECC=0
- 两个 formatted trace 都严格满足 512-token source hash contract
- 05132,108,130 个有效请求、6090 个 zero-usage source 行;稳定
development window=`[3480,7140)`61min、1,078,928 请求;
- 05291,977,423 个有效请求、6031 个 zero-usage source 行;冻结为
held-out稳定候选 window=`[2640,6240)`
- development windowISL p50/p90/p99/max =
20,051/88,224/125,803/202,371OSL p50/p90/p99/max =
78/758/6449/131,072`ISL+OSL max=202,745`
- full-window 131072 coverage=99.399%262144 coverage=100%。但
session sampling 的候选 `rho<=0.0032` 实际 max total=137,016因此
primary server cap 将按最终 cell max 对齐,不为未抽中的 202k outlier
直接预留 262k
- source 无 Qwen-aligned prompt/token IDs。raw canonical prompt 使用 GLM
token contract不能同时保持 Qwen token content 与 trace ISL本 campaign
采用 synthetic Qwen tokens 保持 length/hash/prefix shape并降级内容 claim。
- selected rho=0.0032 中有 1355 个可检查 parent linkstail rewrite
p50/p90/p95/p99/max=1/1/4/57/169 个 source blocks说明 coder
`parent_chat_id` 不等价于 append-only prompt。source hash 序列作为 prefix
truthsynthetic content block 生成后再计算 parent-sensitive runtime
identities避免“相同内容块出现在不同前缀后”造成 Frontier false hit。
- 远端 Qwen3-30B `config.json` 的原生 position limit 是 40960
(`rope_theta=1e6`,无 rope_scaling)。147456 profile smoke 在显式
`VLLM_ALLOW_LONG_MAX_MODEL_LEN=1` 下成功;该 override 只支持
performance/shape fidelity不形成生成质量或模型长上下文正确性 claim
并作为 provenance 中的显式实验变量。
- profile-v6-code-longctx 覆盖 TP1/2/4、KV context 到 131072
33 个 long-context rows两次 fresh-process repeat 的最大相对差
4.648%,旧 anchor drift 最大 1.7%。attention profile SHA256 =
`fbcf7e1f95789a6f6d771e24d1fc60958b7daf04eb0db260d27869a19d71d550`
- TP4 `max_model_len=147456` smoke 已在 dash4 通过。server 日志同时确认
`max_model_len=147456``hf_overrides.max_position_embeddings=147456`
20,051+78、119,702+68、136,774+242 三个 shape 均成功。对应
TTFT=853.29/12,699.04/3,820.55msTPOT=16.65/7.21/8.13ms。
最长请求的非单调 TTFT 来自 cold compile/cache state因此这里只作为
runtime support gate不作为 profile accuracy 数据。
- calibration 全部使用同一 profile-v6 SHA。TP4 的 `rho=0.0016`
decode batch max=16、drain=21.07s,仍通过 10% 亚临界 gateTP2 的
`rho=0.0016` waiting p50=332.97s、drain=976.86s,明确过载并排除。
完整 compact table 在 `results/calibration-summary.json`
- TP2/TP4 的 `rho={0.0002,0.0004,0.0008,0.0016}` 61min full paired
inputs 已在 CPFS 物化;每个 paired row digest 和 Frontier CSV SHA
均与 calibration input 逐项一致。manifest 副本在
`results/paired-input-manifests/`。最大一个目录约 901MiB不把大型
token arrays 提交进 Git。
- code prefill-only 的最大 calibration cache 已物化:
3477 requests、总 prefill 115,828,371 tokens、OSL 全为 1
paired digest=`40865068e02414612ba1cd4595894e20e85e01fd34d73f8660185552d531ecea`
- 多 host 并发 server startup 暴露出 shared CPFS AOT cache 和每-job
FlashInfer JIT 的 apparatus cost。它发生在 readiness 前,不进入 TTFT
runner commit `d5bb974` 改为长上下文默认使用 host-local vLLM cache
并按 topology 复用 FlashInfer workspace。
- 第一轮 paired real canary 的 3 次旧 client 运行都只在同一个
`106709+197` 请求失败,根因是 `return_token_ids` 把 100k+ prompt
vector 放进单条 SSE event超过 aiohttp 默认 512KiB line limit。
commit `e1f2557` 把 exact client read buffer 提到 8MiB700KiB
单-event runtime 对照和随后 TP4×2、TP2×1 的 53/53 replay 均通过。
- TP4 canary 的 real-vs-sim prefix hit ratio =
`0.239908/0.239973`pure-decode batch max 都为 4
`share(b>1)=15.87%/15.69%`real 两 trialvs `16.35%`sim
TTFT mean/p50/p90/p95/p99 bias =
`-8.3/-4.1/-8.8/-13.1/-6.3%`E2E =
`+2.2/+6.1/+11.0/+1.4/-1.3%`。长 drain 的同一
`61976+21361` 请求 real=92.61/92.34s、sim=91.37s,不是 backlog。
- TP4 若把 OSL=4 请求纳入 TPOTmean/p99 bias 会被单个
`~213ms/token` 样本放大到 `-29.9%/-71.5%`OSL≥8 后
mean/p50/p90/p95/p99 bias =
`+8.7/+8.2/+0.6/+13.4/+3.7%`。因此 raw TPOT 仍保留,但正式报告必须
同时给 OSL threshold sensitivity不能把短输出的三段 inter-token
interval 当作稳定 decode service。
- TP4 canary 两 trial 的 real SLO pass rate 都是 `50/53=94.34%`
sim 为 `52/53=98.11%`,在 95% feasibility threshold 上发生翻转;
这由两个临界 TTFT 请求和上述 OSL=4 请求共同造成,作为明确 bad case
进入 1h 检验,不能被总体 latency gap 掩盖。
- TP2 canary 的 cache/batch/drain 仍对齐,但 TTFT p90 bias=`-32.1%`
E2E p90/p95=`-18.4%/-23.8%`。因此先只启动 low-rho 1h diagnostic
不直接铺满 TP2 六个正式 jobs。
- code prefill-only 已完成 10-cell Frontier calibration。TP2
`rho=0.0032` drain=1384.59s、waiting p50=751.01s,明确过载;
`0.0004/0.0008/0.0016` 冻结为 low/mid/near-knee。TP4 到
`rho=0.0032` 仍只有 9.09s drain暂称 highest-tested追加更高 rho
后才冻结 near-knee。compact table 在
`results/prefill-only-calibration-summary.json`
- 2026-07-23 18:37 UTC 启动 A4 wave 1dash1=`TP4/rho0.0002/t1`
dash3=`TP4/rho0.0008/t1`、dash4=`TP4/rho0.0016/t1`
dash2=`TP2/rho0.0002/t1 diagnostic`;四台启动前再次确认 8×H20
memory/utilization=0、无 compute process、uncorrected ECC=0。

View File

@@ -0,0 +1,116 @@
#!/usr/bin/env python3
"""Materialize the stable code window selected by audit_code_trace.py."""
from __future__ import annotations
import argparse
import hashlib
import json
from pathlib import Path
from typing import Any
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser()
parser.add_argument("--audit", type=Path, required=True)
parser.add_argument("--output-root", type=Path, required=True)
parser.add_argument("--sample-seed", type=int, default=20260723)
return parser.parse_args()
def session_uniform(seed: int, window_id: str, session_root: Any) -> float:
payload = json.dumps(
{"seed": seed, "window_id": window_id, "session_root": session_root},
sort_keys=True,
separators=(",", ":"),
).encode()
return int.from_bytes(hashlib.blake2b(payload, digest_size=8).digest(), "big") / (
1 << 64
)
def sha256(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as stream:
for chunk in iter(lambda: stream.read(1 << 20), b""):
digest.update(chunk)
return digest.hexdigest()
def main() -> None:
args = parse_args()
audit = json.loads(args.audit.read_text())
if audit["data_gate"] != "PASS":
raise ValueError(f"trace data gate is not PASS: {audit['data_gate']}")
selected = audit["selected"]
source = Path(selected["source"])
window = selected["stable_window"]
start = float(window["start_timestamp"])
end = float(window["end_timestamp"])
if args.output_root.exists():
raise ValueError(f"refusing to overwrite {args.output_root}")
args.output_root.mkdir(parents=True)
destination = args.output_root / "code-raw-window.jsonl"
root_of: dict[Any, Any] = {}
request_count = 0
with source.open() as input_stream, destination.open("w") as output_stream:
for source_index, line in enumerate(input_stream):
if not line.strip():
continue
row = json.loads(line)
timestamp = float(row["timestamp"])
if timestamp < start:
continue
if timestamp >= end:
break
if int(row["input_length"]) <= 0 or int(row["output_length"]) <= 0:
continue
chat = row.get("chat_id", source_index)
parent = row.get("parent_chat_id")
has_parent = parent not in (None, "", -1, "-1")
session_root = root_of.get(parent, parent) if has_parent else chat
root_of[chat] = session_root
materialized = {
**row,
"source_index": source_index,
"session_root": session_root,
"sampling_u": session_uniform(
args.sample_seed,
f"code-{start:.6f}-{end:.6f}",
session_root,
),
}
output_stream.write(
json.dumps(materialized, ensure_ascii=False, separators=(",", ":"))
+ "\n"
)
request_count += 1
expected = int(selected["selected_window_stats"]["requests"])
if request_count != expected:
raise ValueError(f"window request mismatch: materialized={request_count}, audit={expected}")
manifest = {
"schema": "frontier-code-window-v1",
"audit": str(args.audit.resolve()),
"audit_sha256": sha256(args.audit),
"source": str(source.resolve()),
"source_block_size": selected["hash_contract"]["exact_source_block_size"],
"target_block_size": 16,
"start_timestamp": start,
"end_timestamp": end,
"duration_s": end - start,
"requests": request_count,
"sample_seed": args.sample_seed,
"sampling_rule": "session-coherent deterministic sampling_u",
"max_model_len": audit["max_model_len_recommendation"],
"window_stats": selected["selected_window_stats"],
"raw_window": str(destination.resolve()),
"raw_window_sha256": sha256(destination),
}
(args.output_root / "window-manifest.json").write_text(
json.dumps(manifest, indent=2, sort_keys=True) + "\n"
)
print(json.dumps({"requests": request_count, "output_root": str(args.output_root)}))
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,277 @@
time_stats.attn_input_reshape.min,time_stats.attn_input_reshape.max,time_stats.attn_input_reshape.mean,time_stats.attn_input_reshape.median,time_stats.attn_input_reshape.std,time_stats.attn_kv_cache_save.min,time_stats.attn_kv_cache_save.max,time_stats.attn_kv_cache_save.mean,time_stats.attn_kv_cache_save.median,time_stats.attn_kv_cache_save.std,time_stats.attn_prefill.min,time_stats.attn_prefill.max,time_stats.attn_prefill.mean,time_stats.attn_prefill.median,time_stats.attn_prefill.std,time_stats.attn_decode.min,time_stats.attn_decode.max,time_stats.attn_decode.mean,time_stats.attn_decode.median,time_stats.attn_decode.std,time_stats.attn_output_reshape.min,time_stats.attn_output_reshape.max,time_stats.attn_output_reshape.mean,time_stats.attn_output_reshape.median,time_stats.attn_output_reshape.std,n_embd,n_q_head,n_kv_head,block_size,num_tensor_parallel_workers,max_model_len,batch_size,prefill_chunk_size,kv_cache_size,is_prefill,attention_backend,is_mixed_batch,mode,seq_lens,total_tokens,max_seq_len,min_seq_len,avg_seq_len,equal_seq_len,seq_len_variance,seq_len_std,seq_len_cv,is_chunked_prefill_sample,chunk_start_token,chunk_end_token,total_prefill_tokens,profiling_precision,model_arch,quant_signature,measurement_type,is_true_mixed_batch,prefill_seq_lens,prefill_kv_cache_sizes,decode_kv_cache_sizes,num_prefill_seqs,num_decode_seqs,decode_batch_size,total_batch_size,total_decode_tokens,decode_avg_kv_cache_size,batch_composition_ratio,batch_spec,projection_policy
0.0,0.0,0.0,0.0,0.0,0.01414399966597557,0.028863999992609024,0.019705599918961526,0.01771199982613325,0.005157200849836681,0.047968000173568726,0.07046400010585785,0.05810240097343922,0.05810240097343922,0.007477463486041561,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,64,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[64],64,64,64,64.0,True,0.0,0.0,0.0,False,0.0,64.0,64,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q64,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.04947200044989586,0.020412799902260303,0.01635199971497059,0.010107497379722417,0.046560000628232956,0.08323200047016144,0.05587520003318787,0.05587520003318787,0.011126758739503428,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,128,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[128],128,128,128,128.0,True,0.0,0.0,0.0,False,0.0,128.0,128,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01484800036996603,0.022207999601960182,0.017033600155264138,0.015312000177800655,0.002819235991970241,0.05104000121355057,0.07692799717187881,0.056396800279617305,0.056396800279617305,0.007481982178637539,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,256,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[256],256,256,256,256.0,True,0.0,0.0,0.0,False,0.0,256.0,256,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q256,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015072000212967396,0.022272000089287758,0.01706880023702979,0.01616000011563301,0.002460889579319197,0.06931199878454208,0.0838719978928566,0.07432000041007995,0.07432000041007995,0.004777766433175866,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,512,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,False,0.0,512.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.018592000007629395,0.028543999418616295,0.02095999978482723,0.019183999858796597,0.003198175496053494,0.12179200351238251,0.15408000349998474,0.1307712011039257,0.1307712011039257,0.00858807797538298,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,1024,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[1024],1024,1024,1024,1024.0,True,0.0,0.0,0.0,False,0.0,1024.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.027775999158620834,0.03385600075125694,0.030131200328469276,0.029680000618100166,0.0021152558103575215,0.32678401470184326,0.3450239896774292,0.33396480381488797,0.33396480381488797,0.0045872424917606375,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,2048,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,2048.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04438399896025658,0.05084799975156784,0.046540799736976626,0.04531199857592583,0.002277905811237223,1.0959680080413818,1.1151360273361206,1.0999775886535645,1.0999775886535645,0.005694403246120485,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.078015998005867,0.08691199868917465,0.08114239946007729,0.08019199967384338,0.00292795706334475,4.070400238037109,4.113152027130127,4.087088012695312,4.087088012695312,0.013660567012509554,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.016063999384641647,0.05167999863624573,0.022115200012922286,0.017583999782800674,0.010340094822340818,0.05196800082921982,0.09011200070381165,0.06328320093452933,0.06328320093452933,0.012557341255467452,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,64,448.0,True,FLASH_ATTN,False,vllm020_batch_spec,[64],64,64,64,64.0,True,0.0,0.0,0.0,True,448.0,512.0,64,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q64s512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01583999954164028,0.026623999699950218,0.018927999772131443,0.017376000061631203,0.003514650316260619,0.06681600213050842,0.07993599772453308,0.0725280001759529,0.0725280001759529,0.004343502558613716,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,128,896.0,True,FLASH_ATTN,False,vllm020_batch_spec,[128],128,128,128,128.0,True,0.0,0.0,0.0,True,896.0,1024.0,128,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q128s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01648000068962574,0.030880000442266464,0.01945280022919178,0.017967999912798405,0.004096211183007485,0.1311360001564026,0.1546880006790161,0.13908160030841826,0.13908160030841826,0.007511906874366178,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,256,1792.0,True,FLASH_ATTN,False,vllm020_batch_spec,[256],256,256,256,256.0,True,0.0,0.0,0.0,True,1792.0,2048.0,256,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q256s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.017855999991297722,0.03558399900794029,0.020851199887692927,0.018559999763965607,0.005235911594130716,0.32950401306152344,0.350271999835968,0.33912960588932034,0.33912960588932034,0.006027400986663648,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,512,3584.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,True,3584.0,4096.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.019328000023961067,0.040608000010252,0.022790400311350822,0.020704000256955624,0.006113051965778337,1.1415679454803467,1.1518720388412476,1.144483208656311,1.144483208656311,0.0032332311374389127,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,1024,7168.0,True,FLASH_ATTN,False,vllm020_batch_spec,[1024],1024,1024,1024,1024.0,True,0.0,0.0,0.0,True,7168.0,8192.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1ks8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015807999297976494,0.030688000842928886,0.019596799835562707,0.01774400006979704,0.004343384771033462,0.0,0.0,0.0,0.0,0.0,0.049056001007556915,0.07580800354480743,0.05948160067200661,0.05948160067200661,0.009031541471446955,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01603199914097786,0.02486399933695793,0.01923839971423149,0.018079999834299088,0.0032282528537266424,0.0,0.0,0.0,0.0,0.0,0.05142400041222572,0.07353600114583969,0.059328000620007516,0.059328000620007516,0.0073307807735143084,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.016543999314308167,0.03977600112557411,0.021379199624061585,0.018511999398469925,0.006593576176246171,0.0,0.0,0.0,0.0,0.0,0.0488319993019104,0.06435199826955795,0.05479039996862411,0.05479039996862411,0.005672522998491864,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01635199971497059,0.02844800055027008,0.019267200119793416,0.017952000722289085,0.0035068687666949577,0.0,0.0,0.0,0.0,0.0,0.049855999648571014,0.07798399776220322,0.05986879989504815,0.05986879989504815,0.01043914754878828,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.016383999958634377,0.026079999282956123,0.01923519968986511,0.017791999503970146,0.0032161974331284568,0.0,0.0,0.0,0.0,0.0,0.058111999183893204,0.1045759990811348,0.06708480007946492,0.06708480007946492,0.013479022462646494,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014720000326633453,0.04057599976658821,0.019100800156593323,0.015455999877303839,0.007512281243011577,0.0,0.0,0.0,0.0,0.0,0.05363199859857559,0.07782399654388428,0.06090559959411622,0.06090559959411622,0.007544620176348091,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014431999996304512,0.02191999927163124,0.016684799920767546,0.01563199982047081,0.0024293118621811216,0.0,0.0,0.0,0.0,0.0,0.0629120022058487,0.07891199737787247,0.06891520097851753,0.06891520097851753,0.005472695665695425,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014560000039637089,0.038943998515605927,0.018313600029796363,0.01561600062996149,0.007127270260115769,0.0,0.0,0.0,0.0,0.0,0.08675199747085571,0.10662399977445602,0.09391999915242194,0.09391999915242194,0.006988099589086635,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014527999795973301,0.054687999188899994,0.021439999900758268,0.01539199985563755,0.012052764849597775,0.0,0.0,0.0,0.0,0.0,0.13488000631332397,0.1528639942407608,0.1431359991431236,0.1431359991431236,0.005436271464033599,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.041919998824596405,0.01899839974939823,0.015343999955803156,0.007989843526623287,0.0,0.0,0.0,0.0,0.0,0.06176000088453293,0.08374399691820145,0.06747519969940186,0.06747519969940186,0.0066067747128778,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.016256000846624374,0.11353600025177002,0.042761600017547606,0.028960000723600388,0.029104301538020762,0.0,0.0,0.0,0.0,0.0,0.09734400361776352,0.14422400295734406,0.11392960175871848,0.11392960175871848,0.013198594600417867,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014688000082969666,0.034143999218940735,0.018918400071561335,0.01643200032413006,0.005500943993080684,0.0,0.0,0.0,0.0,0.0,0.12918399274349213,0.15087999403476715,0.13807999789714814,0.13807999789714814,0.007658330538677587,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.016063999384641647,0.03641600161790848,0.0198208000510931,0.01780799962580204,0.0057264128169845765,0.0,0.0,0.0,0.0,0.0,0.22099199891090393,0.23904000222682953,0.2293503984808922,0.2293503984808922,0.004861342907006028,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014751999638974667,0.035840000957250595,0.018908800091594458,0.015792000107467175,0.0064374817924757475,0.0,0.0,0.0,0.0,0.0,0.10134399682283401,0.12201599776744843,0.10896319895982742,0.10896319895982742,0.006336330809165179,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.013856000266969204,0.03417599946260452,0.017846399918198586,0.014800000004470348,0.006495007539635255,0.0,0.0,0.0,0.0,0.0,0.13126400113105774,0.15561600029468536,0.1389280006289482,0.1389280006289482,0.008381472811075022,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014431999996304512,0.03519999980926514,0.019168000388890504,0.01600000075995922,0.005995522477654695,0.0,0.0,0.0,0.0,0.0,0.21728000044822693,0.2343679964542389,0.2231455981731415,0.2231455981731415,0.004720730646739123,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014047999866306782,0.03670400008559227,0.018441599886864425,0.015023999847471714,0.006596535162793127,0.0,0.0,0.0,0.0,0.0,0.39321601390838623,0.4524799883365631,0.4058080047369003,0.4058080047369003,0.01578349755088941,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014336000196635723,0.022143999114632607,0.016912000067532063,0.015168000012636185,0.0028156089295136347,0.0,0.0,0.0,0.0,0.0,0.15587200224399567,0.3079040050506592,0.17838079929351805,0.17838079929351805,0.04355865575265927,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014271999709308147,0.02112000063061714,0.015516800060868263,0.01488000014796853,0.001940870731593904,0.0,0.0,0.0,0.0,0.0,0.21587200462818146,0.23561599850654602,0.22250880002975468,0.22250880002975468,0.006181951170646666,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015552000142633915,0.039264000952243805,0.023609600123018028,0.02131200022995472,0.007236979625711548,0.0,0.0,0.0,0.0,0.0,0.408735990524292,0.470335990190506,0.4336863994598388,0.4336863994598388,0.01844662383160074,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.025407999753952026,0.016227199975401164,0.014960000291466713,0.0031617375441736185,0.0,0.0,0.0,0.0,0.0,0.7412800192832947,0.7627840042114258,0.7464000046253203,0.7464000046253203,0.006112167448837547,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01462399959564209,0.02812799997627735,0.020652799773961304,0.021359999664127827,0.004308706957613102,0.028383498565450627,0.039859687970646644,0.032492258074592426,0.032492258074592426,0.00453597266208597,0.029312501176103633,0.04116431058787463,0.03355574193327539,0.03355574193327539,0.004684436757701648,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,72,,,,,,,,False,,,64,BF16,generic,none,CUDA_EVENT,True,[64],[0],"[512, 512, 512, 512, 512, 512, 512, 512]",1,8,8,9,8,512.0,0.1111111111111111,q64_8q1s512,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.015552000142633915,0.034143999218940735,0.023171199765056372,0.024255999363958836,0.005908614918096971,0.03333159243114438,0.038935341782478095,0.03580428402241854,0.03580428402241854,0.002082270297044095,0.03633240903369937,0.04244065945634484,0.03902771507087562,0.03902771507087562,0.0022697354261490147,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,136,,,,,,,,False,,,128,BF16,generic,none,CUDA_EVENT,True,[128],[0],"[1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024]",1,8,8,9,8,1024.0,0.1111111111111111,q128_8q1s1k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.02611199952661991,0.019635199941694735,0.018112000077962875,0.0038298050749257795,0.04189529417991216,0.057484239920526384,0.04744885718421094,0.04744885718421094,0.004779830748455743,0.051672703037266184,0.07089975418195164,0.05852234134479412,0.05852234134479412,0.005895334539786378,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,144,,,,,,,,False,,,128,BF16,generic,none,CUDA_EVENT,True,[128],[0],"[1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024]",1,16,16,17,16,1024.0,0.058823529411764705,q128_16q1s1k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014720000326633453,0.029311999678611755,0.018662399891763926,0.01673599984496832,0.0044017112162725355,0.04322973959325901,0.05049827064705393,0.04535414343408448,0.04535414343408448,0.0022944156000240697,0.08733025617719538,0.10201372836398578,0.09162185574241775,0.09162185574241775,0.004635047631846023,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,272,,,,,,,,False,,,256,BF16,generic,none,CUDA_EVENT,True,[256],[0],"[2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048]",1,16,16,17,16,2048.0,0.058823529411764705,q256_16q1s2k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014592000283300877,0.026367999613285065,0.016336000058799982,0.01515199989080429,0.003415353455946402,0.06031842775160765,0.06618323188375198,0.06274415549817247,0.06274415549817247,0.001985647289644255,0.14768157653992678,0.16204076249052324,0.15362064543185072,0.15362064543185072,0.004861590945216247,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,288,,,,,,,,False,,,256,BF16,generic,none,CUDA_EVENT,True,[256],[0],"[2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048]",1,32,32,33,32,2048.0,0.030303030303030304,q256_32q1s2k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014751999638974667,0.02454400062561035,0.017167999967932702,0.016191999427974224,0.0028685016454498436,0.09128700688359712,0.09689150775996329,0.09360396051475776,0.09360396051475776,0.0013477542744916764,0.2740889887523892,0.2909164873209846,0.2810456356995366,0.2810456356995366,0.004046628526808561,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,544,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,32,32,33,32,4096.0,0.030303030303030304,q512_32q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.01881599985063076,0.03097599931061268,0.024598400108516216,0.024848000146448612,0.0037539437391565975,0.1035249255866932,0.10603132147437412,0.10399797220840973,0.10399797220840973,0.0007025109429056814,0.5652750707893444,0.5789606938874117,0.5678580377517648,0.5678580377517648,0.003835906384194897,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,576,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,64,64,65,64,4096.0,0.015384615384615385,q512_64q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.018624000251293182,0.043487999588251114,0.024460799992084503,0.019600000232458115,0.008922081629781394,0.196169204945307,0.2076140047945799,0.2008444429250931,0.2008444429250931,0.00394350953292805,1.1196708009293268,1.1849940417370974,1.1463555573610091,1.1463555573610091,0.022508285530529783,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1088,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192]",1,64,64,65,64,8192.0,0.015384615384615385,q1k_64q1s8k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.027135999873280525,0.03667199984192848,0.02945920005440712,0.028256000019609928,0.0029446678408169553,0.37757279619664613,0.3898113624476372,0.38075520430942184,0.38075520430942184,0.0036600203831042254,0.2522831942132049,0.2604606493092597,0.25440958703617433,0.25440958703617433,0.0024455194930253126,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,32,32,33,32,4096.0,0.030303030303030304,q2k_32q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.0435199998319149,0.049695998430252075,0.04502719938755036,0.04395199939608574,0.002035785660169308,1.1048984388245497,1.1189053886476108,1.1112371236754128,1.1112371236754128,0.0050220696724624985,0.1395495077239122,0.14131859607071193,0.14035008841031085,0.14035008841031085,0.0006342911944856293,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,4112,,,,,,,,False,,,4096,BF16,generic,none,CUDA_EVENT,True,[4096],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,16,16,17,16,4096.0,0.058823529411764705,q4k_16q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.018783999606966972,0.024288000538945198,0.020627199858427047,0.019567999988794327,0.0019457052717059358,0.09455999732017517,0.12185599654912949,0.10207359939813614,0.10207359939813614,0.00753346544014261,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,2,1024,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512]",1024,512,512,512.0,True,0.0,0.0,0.0,False,0.0,1024.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,2q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.027295999228954315,0.034752000123262405,0.02943360023200512,0.028528000228106976,0.0023159160681123767,0.14416000247001648,0.15904000401496887,0.14979200065135959,0.14979200065135959,0.00480624675866005,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,4,2048,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512]",2048,512,512,512.0,True,0.0,0.0,0.0,False,0.0,2048.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,4q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.043455999344587326,0.05084799975156784,0.045500800386071204,0.04411200061440468,0.002490556062831049,0.24454399943351746,0.25865599513053894,0.2504959970712662,0.2504959970712662,0.003778494140301353,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512, 512, 512, 512, 512]",4096,512,512,512.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.07673600316047668,0.08374399691820145,0.07970559895038605,0.07873599976301193,0.0022907976135004057,0.445248007774353,0.4758400022983551,0.45409599840641024,0.45409599840641024,0.009133230340383306,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512]",8192,512,512,512.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.043296001851558685,0.051072001457214355,0.044972800090909,0.04399999976158142,0.002451216824441962,0.6147199869155884,0.6290879845619202,0.6204927921295166,0.6204927921295166,0.004344846739788608,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,2,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[2048, 2048]",4096,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,2q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.07737600058317184,0.09123200178146362,0.08094720020890236,0.07980800047516823,0.004065264798368611,1.1744320392608643,1.1887680292129517,1.178323209285736,1.178323209285736,0.004395876469319665,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,4,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[2048, 2048, 2048, 2048]",8192,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,4q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014688000082969666,0.03254399821162224,0.01810879958793521,0.01611199975013733,0.005058478889747962,0.0,0.0,0.0,0.0,0.0,0.06339199841022491,0.0841279998421669,0.07019519805908202,0.07019519805908202,0.006070126182471763,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01484800036996603,0.030271999537944794,0.018124799989163876,0.015711999498307705,0.004696400814632773,0.0,0.0,0.0,0.0,0.0,0.2739199995994568,0.2922239899635315,0.28281279802322384,0.28281279802322384,0.00633037978599564,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014495999552309513,0.022431999444961548,0.016844799742102623,0.015471999999135733,0.0027624803153841917,0.0,0.0,0.0,0.0,0.0,0.3909119963645935,0.4160960018634796,0.39996159672737125,0.39996159672737125,0.007037174636804898,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014240000396966934,0.035071998834609985,0.018764800019562246,0.015583999920636415,0.006305594700523817,0.0,0.0,0.0,0.0,0.0,0.7383679747581482,0.7597119808197021,0.7430047929286957,0.7430047929286957,0.006192645960911466,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.028991999104619026,0.019401599932461978,0.01665600063279271,0.005434953793780546,0.0,0.0,0.0,0.0,0.0,1.427008032798767,1.4517120122909546,1.4361984014511109,1.4361984014511109,0.008802755821028187,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014271999709308147,0.02844800055027008,0.01820160010829568,0.014944000169634819,0.004942002036909087,0.0,0.0,0.0,0.0,0.0,0.08361600339412689,0.11027199774980545,0.09058240056037903,0.09058240056037903,0.00872938604423622,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014112000353634357,0.023104000836610794,0.01648960020393133,0.015632000286132097,0.0027628828340349578,0.0,0.0,0.0,0.0,0.0,0.5063040256500244,0.5497599840164185,0.5168287932872773,0.5168287932872773,0.013557229606313293,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01616000011563301,0.029055999591946602,0.018780799955129622,0.017215999774634838,0.003794434947431941,0.0,0.0,0.0,0.0,0.0,0.7439360022544861,0.7719680070877075,0.7502080142498017,0.7502080142498017,0.00839205598262567,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015968000516295433,0.03046399913728237,0.018822400271892546,0.01726400014013052,0.004212536831927695,0.0,0.0,0.0,0.0,0.0,1.4256000518798828,1.449504017829895,1.4325888037681578,1.4325888037681578,0.0065448028108711165,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014271999709308147,0.027712000533938408,0.01633920017629862,0.014944000169634819,0.0038488220071983326,0.0,0.0,0.0,0.0,0.0,2.798719882965088,3.0184640884399414,2.8293471813201903,2.8293471813201903,0.06375662767704417,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014720000326633453,0.021407999098300934,0.01637439979240298,0.014992000069469213,0.0024463594774459265,0.0,0.0,0.0,0.0,0.0,0.09347199648618698,0.10473600029945374,0.09820479974150656,0.09820479974150656,0.004061668684999473,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014303999952971935,0.04790399968624115,0.020179199893027543,0.016447999514639378,0.009635010283506967,0.0,0.0,0.0,0.0,0.0,0.6221439838409424,0.6444799900054932,0.6277984082698822,0.6277984082698822,0.007234140179397069,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014271999709308147,0.03276799991726875,0.01921279989182949,0.015232000034302473,0.0065506148511265076,0.0,0.0,0.0,0.0,0.0,0.9111359715461731,0.9307839870452881,0.9151648044586183,0.9151648044586183,0.005528712062927265,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014175999909639359,0.02707199938595295,0.016512000095099212,0.014688000082969666,0.0038989080209321414,0.0,0.0,0.0,0.0,0.0,1.7645119428634644,1.7849279642105103,1.7684095859527589,1.7684095859527589,0.005876963340184193,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014240000396966934,0.029184000566601753,0.016825600154697896,0.014752000104635954,0.00457910514148248,0.0,0.0,0.0,0.0,0.0,3.4781761169433594,3.5388801097869873,3.490892815589905,3.490892815589905,0.01818910600692522,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014976000413298607,0.03139200061559677,0.01898880014196038,0.016848000697791576,0.005083715524393418,0.13752702814163098,0.14117629917511842,0.13848196486571557,0.13848196486571557,0.0009880564964713527,0.523336967410432,0.5372236808930884,0.5269708253945994,0.5269708253945994,0.00375988994658546,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,520,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,8,8,9,8,16384.0,0.1111111111111111,q512_8q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.01833599992096424,0.02707199938595295,0.020995199866592883,0.01976000051945448,0.0028392656352050185,0.17817885890237703,0.18622915300900206,0.18109068484526028,0.18109068484526028,0.0027040006080457624,0.5449571488834201,0.5695788427872015,0.5538629212834322,0.5538629212834322,0.008270141985514729,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1040,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,16,16,17,16,16384.0,0.058823529411764705,q1k_16q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.02703999914228916,0.03308799862861633,0.028883199393749236,0.02759999968111515,0.002285758578932753,0.4688266550410815,0.48301665772804186,0.4727750380198454,0.4727750380198454,0.004396396389258256,1.0430453981052807,1.0746153117238624,1.051829759343436,1.051829759343436,0.009781101336187197,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,32,32,33,32,16384.0,0.030303030303030304,q2k_32q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.04368000105023384,0.059328000992536545,0.04809600040316582,0.04617599956691265,0.005158792915002645,1.3509461459747514,1.3759066409666496,1.3589365122072,1.3589365122072,0.008008877716183384,0.9213738861449998,0.9383974724214122,0.9268234851606594,0.9268234851606594,0.005462224239661061,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,4112,,,,,,,,False,,,4096,BF16,generic,none,CUDA_EVENT,True,[4096],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,16,16,17,16,32768.0,0.058823529411764705,q4k_16q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.026944000273942947,0.040511999279260635,0.030291200056672095,0.02817599941045046,0.004386386489305873,0.5071830964059985,0.5142792136001758,0.5096392092471257,0.5096392092471257,0.0022059272685869546,2.175632932188972,2.2060727208328075,2.186168772243963,2.186168772243963,0.009462633998570079,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,32,32,33,32,32768.0,0.030303030303030304,q2k_32q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.018303999677300453,0.03017600066959858,0.02039040010422468,0.018943999893963337,0.0034612051983613614,0.21341429693945616,0.21552776300295579,0.21393496609766816,0.21393496609766816,0.0005736773262009222,4.617401495144284,4.663128147488988,4.628666619290515,4.628666619290515,0.012412001359412127,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1088,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,64,64,65,64,32768.0,0.015384615384615385,q1k_64q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014720000326633453,0.02364799939095974,0.01809599995613098,0.016352000646293163,0.0035481127058959038,0.04822399839758873,0.08566399663686752,0.05961279980838299,0.05961279980838299,0.011445665413968877,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,64,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[64],64,64,64,64.0,True,0.0,0.0,0.0,False,0.0,64.0,64,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q64,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014175999909639359,0.020864000543951988,0.01668160008266568,0.015664000064134598,0.0025769063833097584,0.049695998430252075,0.08057600259780884,0.05882879942655563,0.05882879942655563,0.009515126108519331,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,128,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[128],128,128,128,128.0,True,0.0,0.0,0.0,False,0.0,128.0,128,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.028704000636935234,0.017430400010198355,0.015056000091135502,0.004349294613335555,0.049855999648571014,0.07366400212049484,0.05459520071744919,0.05459520071744919,0.0069610920757925574,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,256,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[256],256,256,256,256.0,True,0.0,0.0,0.0,False,0.0,256.0,256,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q256,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014336000196635723,0.03161599859595299,0.017788799852132796,0.015711999963968992,0.005005385723318659,0.06102399900555611,0.08179199695587158,0.06650560013949873,0.06650560013949873,0.006947995595801105,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,512,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,False,0.0,512.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.04416000097990036,0.019200000166893005,0.015488000120967627,0.008561241323364038,0.08054400235414505,0.09196799993515015,0.08607039973139763,0.08607039973139763,0.004145329035483754,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,1024,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[1024],1024,1024,1024,1024.0,True,0.0,0.0,0.0,False,0.0,1024.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.017855999991297722,0.023391999304294586,0.019667199812829494,0.018463999964296818,0.0022577669687832585,0.18729600310325623,0.20585599541664124,0.19546559900045393,0.19546559900045393,0.006339068824303663,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,2048,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,2048.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.026240000501275063,0.03446400165557861,0.028297600522637367,0.027088000439107418,0.0025148441522922374,0.5754240155220032,0.5889919996261597,0.5800191938877105,0.5800191938877105,0.003858869273829596,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04182400181889534,0.047200001776218414,0.043036799877882004,0.0423360001295805,0.0017073405772076728,2.063199996948242,2.0787200927734375,2.067151999473572,2.067151999473572,0.004959651271127835,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.05648000165820122,0.02270399993285537,0.017935999669134617,0.012377915150727689,0.049536000937223434,0.07196799665689468,0.056015999615192415,0.056015999615192415,0.0070476637552742884,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,64,448.0,True,FLASH_ATTN,False,vllm020_batch_spec,[64],64,64,64,64.0,True,0.0,0.0,0.0,True,448.0,512.0,64,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q64s512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01408000010997057,0.021344000473618507,0.01611520005390048,0.014928000047802925,0.0025499884993961702,0.07072000205516815,0.2642880082130432,0.1588256008923054,0.1588256008923054,0.053220347086102376,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,128,896.0,True,FLASH_ATTN,False,vllm020_batch_spec,[128],128,128,128,128.0,True,0.0,0.0,0.0,True,896.0,1024.0,128,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q128s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01360000018030405,0.03014400042593479,0.016975999902933837,0.015056000091135502,0.004715159697364111,0.08393599838018417,0.11036799848079681,0.09160000011324881,0.09160000011324881,0.007911669434472792,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,256,1792.0,True,FLASH_ATTN,False,vllm020_batch_spec,[256],256,256,256,256.0,True,0.0,0.0,0.0,True,1792.0,2048.0,256,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q256s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014495999552309513,0.020767999812960625,0.016233599931001663,0.015392000321298838,0.002202615907995427,0.1998399943113327,0.22070400416851044,0.20855360180139543,0.20855360180139543,0.00689307200230667,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,512,3584.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,True,3584.0,4096.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01484800036996603,0.033440001308918,0.018684800155460833,0.016048000194132328,0.00553991005639174,0.6043199896812439,0.635807991027832,0.6126143991947173,0.6126143991947173,0.008745933953408096,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,1024,7168.0,True,FLASH_ATTN,False,vllm020_batch_spec,[1024],1024,1024,1024,1024.0,True,0.0,0.0,0.0,True,7168.0,8192.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1ks8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.013824000023305416,0.03359999880194664,0.017811199743300678,0.014944000169634819,0.005926140483565472,0.0,0.0,0.0,0.0,0.0,0.045471999794244766,0.07539200037717819,0.054758400097489356,0.054758400097489356,0.010253548506101549,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014047999866306782,0.02409599907696247,0.01641279999166727,0.01473599998280406,0.003347158638713987,0.0,0.0,0.0,0.0,0.0,0.0461760014295578,0.07529599964618683,0.05460800044238568,0.05460800044238568,0.009748937798340135,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014271999709308147,0.028416000306606293,0.018611199874430894,0.01598400017246604,0.005209875093863647,0.0,0.0,0.0,0.0,0.0,0.048128001391887665,0.07897599786520004,0.061353600397706036,0.061353600397706036,0.010488153655157845,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014112000353634357,0.025728000327944756,0.016092800162732603,0.01512000011280179,0.003300888123052605,0.0,0.0,0.0,0.0,0.0,0.04864000156521797,0.07648000121116638,0.05810560062527656,0.05810560062527656,0.009473544218404408,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.03551999852061272,0.018588799890130757,0.016064000315964222,0.006071057437992447,0.0,0.0,0.0,0.0,0.0,0.04822399839758873,0.07862400263547897,0.05660480037331582,0.05660480037331582,0.009565394213730401,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014303999952971935,0.028831999748945236,0.01699519995599985,0.015024000313133001,0.004285000744868188,0.0,0.0,0.0,0.0,0.0,0.04854400083422661,0.06719999760389328,0.05778240002691746,0.05778240002691746,0.006852125554679805,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.0144640002399683,0.024927999824285507,0.0161183999851346,0.01521599991247058,0.0029774808524673907,0.0,0.0,0.0,0.0,0.0,0.05379199981689453,0.08966399729251862,0.06270079985260964,0.06270079985260964,0.009983591277092696,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014431999996304512,0.03001599945127964,0.017648000083863736,0.01547200046479702,0.004585126309484848,0.0,0.0,0.0,0.0,0.0,0.061664000153541565,0.07692799717187881,0.06704320013523103,0.06704320013523103,0.005500191798490629,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014175999909639359,0.026367999613285065,0.016947199776768684,0.015039999969303608,0.0037951566103550205,0.0,0.0,0.0,0.0,0.0,0.08799999952316284,0.111455999314785,0.0964031994342804,0.0964031994342804,0.007397541615558088,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01369599997997284,0.021247999742627144,0.015359999984502793,0.014640000183135271,0.0020942770950814317,0.0,0.0,0.0,0.0,0.0,0.051711998879909515,0.07065600156784058,0.058054400235414506,0.058054400235414506,0.006633034815910025,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014208000153303146,0.0307839997112751,0.017148799914866685,0.015039999969303608,0.004882374686312284,0.0,0.0,0.0,0.0,0.0,0.061919998377561576,0.07843200117349625,0.06628479920327664,0.06628479920327664,0.004962852689801192,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.013887999579310417,0.02969600073993206,0.017500799987465142,0.015232000034302473,0.00467273319705314,0.0,0.0,0.0,0.0,0.0,0.08819200098514557,0.11097600311040878,0.09493440166115762,0.09493440166115762,0.007509235042985577,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.022112000733613968,0.01751680001616478,0.016047999262809753,0.0030306621792915785,0.0,0.0,0.0,0.0,0.0,0.13065600395202637,0.15110400319099426,0.13857279866933822,0.13857279866933822,0.00750137841249771,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.013919999822974205,0.04012800008058548,0.01892479993402958,0.01550400024279952,0.007459545267816961,0.0,0.0,0.0,0.0,0.0,0.06278400123119354,0.08259200304746628,0.0704512007534504,0.0704512007534504,0.005979055984382744,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014336000196635723,0.02236800082027912,0.016332800220698118,0.014928000047802925,0.002784044363186731,0.0,0.0,0.0,0.0,0.0,0.1003199964761734,0.1279360055923462,0.10921279862523078,0.10921279862523078,0.00862769617046716,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.013919999822974205,0.0225600004196167,0.016128000058233737,0.014800000004470348,0.002958953332547708,0.0,0.0,0.0,0.0,0.0,0.13116799294948578,0.14812800288200378,0.13783999979496003,0.13783999979496003,0.005696384361148053,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014208000153303146,0.029983999207615852,0.017468800116330386,0.01508800033479929,0.0047379550962483065,0.0,0.0,0.0,0.0,0.0,0.217631995677948,0.2447360008955002,0.22715839892625808,0.22715839892625808,0.008831828221847138,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014047999866306782,0.02304000034928322,0.016512000095099212,0.014512000139802694,0.0035026774828624254,0.0,0.0,0.0,0.0,0.0,0.11020799726247787,0.12307199835777283,0.11600959971547126,0.11600959971547126,0.004667637266950902,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014112000353634357,0.042080000042915344,0.017510399967432023,0.014607999939471483,0.00823117883530167,0.0,0.0,0.0,0.0,0.0,0.15702399611473083,0.17587199807167053,0.16399359852075576,0.16399359852075576,0.006588299074676393,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.013919999822974205,0.0208320003002882,0.015299199987202883,0.01462399959564209,0.001916768086505386,0.0,0.0,0.0,0.0,0.0,0.21779200434684753,0.2415360063314438,0.2260768011212349,0.2260768011212349,0.007251352080685236,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014015999622642994,0.021088000386953354,0.015619200188666582,0.01473599998280406,0.002013227452302141,0.0,0.0,0.0,0.0,0.0,0.3959999978542328,0.4152640104293823,0.40332479774951924,0.40332479774951924,0.006942401431914052,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014175999909639359,0.03379200026392937,0.020595200080424547,0.019600000232458115,0.006548754248881344,0.026192623739694512,0.040006882507168426,0.029155180178492036,0.029155180178492036,0.00408828748438241,0.024591375524545756,0.037561119537986146,0.027372820355088746,0.027372820355088746,0.0038383559348575944,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,72,,,,,,,,False,,,64,BF16,generic,none,CUDA_EVENT,True,[64],[0],"[512, 512, 512, 512, 512, 512, 512, 512]",1,8,8,9,8,512.0,0.1111111111111111,q64_8q1s512,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.013856000266969204,0.020896000787615776,0.015318400040268899,0.014431999996304512,0.0019640312960926966,0.029349018208693862,0.06236262941356679,0.03714313592014963,0.03714313592014963,0.009618313791872569,0.028826981462526914,0.06125337308649041,0.036482463672774496,0.036482463672774496,0.009447230957011863,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,136,,,,,,,,False,,,128,BF16,generic,none,CUDA_EVENT,True,[128],[0],"[1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024]",1,8,8,9,8,1024.0,0.1111111111111111,q128_8q1s1k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.013856000266969204,0.032127998769283295,0.017286399938166143,0.014479999896138906,0.005536495446636193,0.03273085874558354,0.04394578491937082,0.03563837490653034,0.03563837490653034,0.003429601730256567,0.03488514202593899,0.04683821345079978,0.037984025407085474,0.037984025407085474,0.0036553316361902684,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,144,,,,,,,,False,,,128,BF16,generic,none,CUDA_EVENT,True,[128],[0],"[1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024]",1,16,16,17,16,1024.0,0.058823529411764705,q128_16q1s1k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.013856000266969204,0.021503999829292297,0.015078400075435639,0.01425600005313754,0.002238435975478849,0.042100813549974185,0.051784144690147756,0.045378692890289625,0.045378692890289625,0.003184109940650555,0.05111518843748309,0.06287185663927425,0.05509490773570219,0.05509490773570219,0.0038658725544299132,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,272,,,,,,,,False,,,256,BF16,generic,none,CUDA_EVENT,True,[256],[0],"[2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048]",1,16,16,17,16,2048.0,0.058823529411764705,q256_16q1s2k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.02687999978661537,0.017667199857532977,0.01566399959847331,0.003864811846387173,0.048475323773821306,0.05599956978723733,0.05123499252968345,0.05123499252968345,0.002427379820423941,0.08429268135885387,0.09737642835214408,0.0890914090616175,0.0890914090616175,0.0042209177332077005,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,288,,,,,,,,False,,,256,BF16,generic,none,CUDA_EVENT,True,[256],[0],"[2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048]",1,32,32,33,32,2048.0,0.030303030303030304,q256_32q1s2k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.01462399959564209,0.03359999880194664,0.019804799742996693,0.016032000072300434,0.006750519908145474,0.07121508474579985,0.07292308367136396,0.07194410845270183,0.07194410845270183,0.0005500065915943844,0.14760091249712767,0.15114092373010238,0.14911189243564577,0.14911189243564577,0.0011399477384397005,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,544,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,32,32,33,32,4096.0,0.030303030303030304,q512_32q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.021856000646948814,0.016835200227797033,0.015263999812304974,0.00283854448975065,0.08146338272142935,0.08560865714384096,0.0834932625520734,0.0834932625520734,0.0012259635905347874,0.2782486219401307,0.29240733788179385,0.2851819365989658,0.2851819365989658,0.004187435731481665,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,576,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,64,64,65,64,4096.0,0.015384615384615385,q512_64q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.04495999962091446,0.02143679987639189,0.015856000129133463,0.009709987872683342,0.1194459208702178,0.12302524755181463,0.12053885718696096,0.12053885718696096,0.001005118119341339,0.5597220650459199,0.5764947443228802,0.5648435507167102,0.5648435507167102,0.004709970715400788,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1088,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192]",1,64,64,65,64,8192.0,0.015384615384615385,q1k_64q1s8k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.018112000077962875,0.026496000587940216,0.019318400137126445,0.018432000651955605,0.0024249605112359905,0.19770253574610402,0.222811797868348,0.2054811520619412,0.2054811520619412,0.008000944254868043,0.13941746080159492,0.157124211777114,0.14490284788179197,0.14490284788179197,0.005642170080515992,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,32,32,33,32,4096.0,0.030303030303030304,q2k_32q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.02595200017094612,0.03155200183391571,0.02727359998971224,0.026575999334454536,0.0016260948764843166,0.6014684881116659,0.6095472988212,0.6046757700946376,0.6046757700946376,0.0023537435563177303,0.11325152264578045,0.11477269562841545,0.11385542721485654,0.11385542721485654,0.0004431903698023628,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,4112,,,,,,,,False,,,4096,BF16,generic,none,CUDA_EVENT,True,[4096],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,16,16,17,16,4096.0,0.058823529411764705,q4k_16q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014303999952971935,0.03359999880194664,0.01814719969406724,0.015375999733805656,0.005678506740662529,0.06774400174617767,0.07891199737787247,0.07312640026211739,0.07312640026211739,0.003826583051422731,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,2,1024,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512]",1024,512,512,512.0,True,0.0,0.0,0.0,False,0.0,1024.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,2q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01788800023496151,0.028543999418616295,0.021340799890458582,0.019504000432789326,0.003718627037219618,0.09548799693584442,0.1327359974384308,0.10823359936475753,0.10823359936475753,0.013230635292043864,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,4,2048,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512]",2048,512,512,512.0,True,0.0,0.0,0.0,False,0.0,2048.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,4q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.02627200074493885,0.030368000268936157,0.027184000052511693,0.02643200010061264,0.0013545679205210022,0.14364799857139587,0.1597760021686554,0.15008639842271806,0.15008639842271806,0.005367723415171222,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512, 512, 512, 512, 512]",4096,512,512,512.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04150399938225746,0.04726399853825569,0.042950399965047834,0.04224000126123428,0.001720147501765378,0.2433920055627823,0.28963199257850647,0.2549152016639709,0.2549152016639709,0.013450991019687407,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512]",8192,512,512,512.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.025887999683618546,0.03407999873161316,0.028303999826312064,0.026367999613285065,0.0028877495368841042,0.325439989566803,0.3441599905490875,0.33442879617214205,0.33442879617214205,0.005693597811441922,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,2,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[2048, 2048]",4096,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,2q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04137599840760231,0.05084799975156784,0.044828799366950986,0.043087998405098915,0.003560363865797613,0.6110399961471558,0.6421759724617004,0.6204223990440368,0.6204223990440368,0.011214490370794758,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,4,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[2048, 2048, 2048, 2048]",8192,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,4q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014816000126302242,0.02364799939095974,0.016668799985200166,0.01592000015079975,0.0025515238179941247,0.0,0.0,0.0,0.0,0.0,0.0544000007212162,0.09014400094747543,0.06511679962277411,0.06511679962277411,0.013005726711295521,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014527999795973301,0.0208320003002882,0.016604799963533878,0.01566399959847331,0.0021353594266203244,0.0,0.0,0.0,0.0,0.0,0.17871999740600586,0.1961279958486557,0.18568639904260634,0.18568639904260634,0.004566142885274747,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014431999996304512,0.02006400004029274,0.015619200002402068,0.01508800033479929,0.001572984783715635,0.0,0.0,0.0,0.0,0.0,0.2730880081653595,0.29721599817276,0.2815328001976013,0.2815328001976013,0.007016534727616923,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.02755199931561947,0.018678399827331306,0.015919999685138464,0.004323555372382858,0.0,0.0,0.0,0.0,0.0,0.3893119990825653,0.44041600823402405,0.40225600004196166,0.40225600004196166,0.013442998165884852,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014303999952971935,0.02659199945628643,0.016748800035566093,0.015343999955803156,0.0036412341868394082,0.0,0.0,0.0,0.0,0.0,0.7404800057411194,0.9689919948577881,0.8451807916164398,0.8451807916164398,0.08637455920036795,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014751999638974667,0.02160000056028366,0.016752000153064727,0.01521599991247058,0.0025167650350367246,0.0,0.0,0.0,0.0,0.0,0.06412799656391144,0.08956799656152725,0.07242240011692047,0.07242240011692047,0.009797120588901621,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014495999552309513,0.022911999374628067,0.016137600038200618,0.015343999955803156,0.002330911555470513,0.0,0.0,0.0,0.0,0.0,0.3128319978713989,0.3282879889011383,0.3203647971153259,0.3203647971153259,0.004598568238907996,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.03574400022625923,0.01919359974563122,0.01583999954164028,0.006642521913489215,0.0,0.0,0.0,0.0,0.0,0.5050879716873169,0.5311999917030334,0.511932796239853,0.511932796239853,0.007150929647317677,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.02595200017094612,0.016992000117897987,0.015456000342965126,0.003390731937723684,0.0,0.0,0.0,0.0,0.0,0.7385600209236145,0.7681919932365417,0.7483008027076722,0.7483008027076722,0.010214373356208012,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014527999795973301,0.038975998759269714,0.01952639976516366,0.015327999833971262,0.007695927285621061,0.0,0.0,0.0,0.0,0.0,1.4228800535202026,1.5237760543823242,1.4435008168220522,1.4435008168220522,0.0342955291725742,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014495999552309513,0.03907199949026108,0.01793599994853139,0.015248000156134367,0.007167103363234667,0.0,0.0,0.0,0.0,0.0,0.06947200000286102,0.08857599645853043,0.07469440028071403,0.07469440028071403,0.00583249952929747,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014495999552309513,0.02175999991595745,0.016710399929434062,0.015327999833971262,0.002705383977079183,0.0,0.0,0.0,0.0,0.0,0.388480007648468,0.4073280096054077,0.3966591984033584,0.3966591984033584,0.004487315516095964,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.0144640002399683,0.022048000246286392,0.017011200170964004,0.015392000321298838,0.002776899649845722,0.0,0.0,0.0,0.0,0.0,0.622048020362854,0.6347839832305908,0.6262047946453094,0.6262047946453094,0.004593360122958751,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014592000283300877,0.02937600016593933,0.01847040019929409,0.015344000421464443,0.005043809142122341,0.0,0.0,0.0,0.0,0.0,0.9105280041694641,0.9304640293121338,0.914108806848526,0.914108806848526,0.00563919268816644,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01462399959564209,0.039903998374938965,0.01791359977796674,0.015440000221133232,0.007367547649297217,0.0,0.0,0.0,0.0,0.0,1.764799952507019,1.7965760231018066,1.770739197731018,1.770739197731018,0.009082122770320404,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015168000012636185,0.03187200054526329,0.018553599901497363,0.016127999871969223,0.0050628774268006264,0.16963527081512533,0.1737871320906266,0.17088926838108623,0.17088926838108623,0.0011703114258121128,0.47362872483230506,0.48522089403242025,0.47712993814280913,0.47712993814280913,0.0032675581298664976,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,520,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,8,8,9,8,16384.0,0.1111111111111111,q512_8q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.03868800029158592,0.01921919994056225,0.016959999687969685,0.006691309533030663,0.1589600576212269,0.16117782913137854,0.15969057520605917,0.15969057520605917,0.0006212984448748182,0.5199519263456005,0.5272061673552852,0.5223414198520004,0.5223414198520004,0.002032242112153396,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1040,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,16,16,17,16,16384.0,0.058823529411764705,q1k_16q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.018400000408291817,0.024320000782608986,0.019910399988293647,0.0191040001809597,0.001841532763984352,0.25650752966102763,0.27066608538463055,0.259736894547936,0.259736894547936,0.00462927315947332,0.5278764825612624,0.5570139063374764,0.5345223138928448,0.5345223138928448,0.009526755161797178,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,32,32,33,32,16384.0,0.030303030303030304,q2k_32q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.026048000901937485,0.03728000074625015,0.028636799938976765,0.02705600019544363,0.003313953649011259,0.9270006318443208,0.9712965120641314,0.93480767601568,0.93480767601568,0.012612551450284608,0.8181833128578277,0.8572794566782395,0.8250739157811953,0.8250739157811953,0.01113200873299586,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,4112,,,,,,,,False,,,4096,BF16,generic,none,CUDA_EVENT,True,[4096],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,16,16,17,16,32768.0,0.058823529411764705,q4k_16q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.018079999834299088,0.037696000188589096,0.022092800214886667,0.019024000503122807,0.005986278786633071,0.2839857165542317,0.2876110048757805,0.2849506939696605,0.2849506939696605,0.0009995178425916847,1.0871823008331585,1.101060989333509,1.0908765231323903,1.0908765231323903,0.0038264533900426207,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,32,32,33,32,32768.0,0.030303030303030304,q2k_32q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.015072000212967396,0.025567999109625816,0.01726400014013052,0.015728000551462173,0.0031324465940990903,0.137270464802061,0.13799613818579368,0.13751499486424038,0.13751499486424038,0.00025221761530472904,2.3021855212208884,2.314355908780759,2.3062865750744197,2.3062865750744197,0.004229983070201486,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1088,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,64,64,65,64,32768.0,0.015384615384615385,q1k_64q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014560000039637089,0.022784000262618065,0.016435200069099664,0.015519999898970127,0.0023688301421469523,0.04879999905824661,0.09139200299978256,0.06054079942405224,0.06054079942405224,0.012152608702448775,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,64,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[64],64,64,64,64.0,True,0.0,0.0,0.0,False,0.0,64.0,64,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q64,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014816000126302242,0.02844800055027008,0.017008000146597625,0.015696000307798386,0.003941466294662679,0.047807998955249786,0.07356800138950348,0.05626560002565384,0.05626560002565384,0.00842179125412236,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,128,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[128],128,128,128,128.0,True,0.0,0.0,0.0,False,0.0,128.0,128,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014688000082969666,0.053408000618219376,0.019353600218892097,0.01532800029963255,0.01138302400841626,0.048448000103235245,0.0785600021481514,0.0556256003677845,0.0556256003677845,0.009348274502616908,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,256,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[256],256,256,256,256.0,True,0.0,0.0,0.0,False,0.0,256.0,256,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q256,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015039999969303608,0.03139200061559677,0.01823679991066456,0.016159999649971724,0.004860071238302512,0.055424001067876816,0.08505599945783615,0.0640383992344141,0.0640383992344141,0.00921448636178623,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,512,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,False,0.0,512.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014751999638974667,0.026528000831604004,0.017324799951165915,0.01536000007763505,0.0036004822686428305,0.07660800218582153,0.0942080020904541,0.08209280073642732,0.08209280073642732,0.00515895587669752,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,1024,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[1024],1024,1024,1024,1024.0,True,0.0,0.0,0.0,False,0.0,1024.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.021247999742627144,0.016403199825435876,0.01532800029963255,0.001995897022647934,0.11395200341939926,0.15113599598407745,0.12431039959192276,0.12431039959192276,0.011164431123683732,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,2048,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,2048.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.017184000462293625,0.028672000393271446,0.019865600019693376,0.017823999747633934,0.0035798491315929977,0.3171840012073517,0.3341119885444641,0.3261695951223373,0.3261695951223373,0.005111046452878918,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.024639999493956566,0.030400000512599945,0.02656640000641346,0.02556800004094839,0.0020189846603237303,1.0648640394210815,1.0828479528427124,1.0715327858924866,1.0715327858924866,0.005558639263049851,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014976000413298607,0.021695999428629875,0.017305599898099898,0.01593599934130907,0.0025718046517268054,0.04956800118088722,0.07932800054550171,0.06228480041027069,0.06228480041027069,0.01027160349757827,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,64,448.0,True,FLASH_ATTN,False,vllm020_batch_spec,[64],64,64,64,64.0,True,0.0,0.0,0.0,True,448.0,512.0,64,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q64s512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01539199985563755,0.03580800071358681,0.019686400331556796,0.017136000096797943,0.005918387811304003,0.05158400163054466,0.10713600367307663,0.06364160068333148,0.06364160068333148,0.015832957809696766,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,128,896.0,True,FLASH_ATTN,False,vllm020_batch_spec,[128],128,128,128,128.0,True,0.0,0.0,0.0,True,896.0,1024.0,128,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q128s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015168000012636185,0.02223999984562397,0.016672000009566545,0.015887999907135963,0.0020934945946034563,0.06521599739789963,0.08902399986982346,0.07520959973335266,0.07520959973335266,0.007913840684102929,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,256,1792.0,True,FLASH_ATTN,False,vllm020_batch_spec,[256],256,256,256,256.0,True,0.0,0.0,0.0,True,1792.0,2048.0,256,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q256s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015231999568641186,0.04028800129890442,0.019971200078725816,0.01646399963647127,0.007351305886577286,0.14467200636863708,0.16844800114631653,0.15363519936800005,0.15363519936800005,0.008174435899956223,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,512,3584.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,True,3584.0,4096.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015519999898970127,0.02304000034928322,0.01775679988786578,0.016784000210464,0.0025077243712082584,0.33740800619125366,0.35343998670578003,0.3445120006799698,0.3445120006799698,0.00445648463590358,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,1024,7168.0,True,FLASH_ATTN,False,vllm020_batch_spec,[1024],1024,1024,1024,1024.0,True,0.0,0.0,0.0,True,7168.0,8192.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1ks8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015647999942302704,0.02393599972128868,0.01809599995613098,0.01654400024563074,0.002998393102466254,0.0,0.0,0.0,0.0,0.0,0.0504320003092289,0.0843840017914772,0.060083200410008426,0.060083200410008426,0.00986959318572296,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01603199914097786,0.030047999694943428,0.019510399922728537,0.017680000513792038,0.004065185265963332,0.0,0.0,0.0,0.0,0.0,0.05004800111055374,0.07036799937486649,0.059315200522542,0.059315200522542,0.006537768821329647,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01484800036996603,0.02393599972128868,0.018035199772566558,0.016512000001966953,0.0030667689116777724,0.0,0.0,0.0,0.0,0.0,0.05100800096988678,0.06735999882221222,0.058387200161814694,0.058387200161814694,0.005832787739241381,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01500799972563982,0.026208000257611275,0.017500799987465142,0.01646399963647127,0.0030666103306165714,0.0,0.0,0.0,0.0,0.0,0.05023999884724617,0.07254400104284286,0.057254400476813315,0.057254400476813315,0.006890853653068151,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014688000082969666,0.027327999472618103,0.0179776000790298,0.01648000068962574,0.003783417447531392,0.0,0.0,0.0,0.0,0.0,0.04819199815392494,0.07100799679756165,0.05621119923889638,0.05621119923889638,0.007824391484124725,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015647999942302704,0.036448001861572266,0.019136000238358975,0.016704000532627106,0.006076530095624803,0.0,0.0,0.0,0.0,0.0,0.047488000243902206,0.08441600203514099,0.0588383998721838,0.0588383998721838,0.011275940726332522,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015135999768972397,0.033952001482248306,0.02119360016658902,0.018000000156462193,0.007136646614305304,0.0,0.0,0.0,0.0,0.0,0.04931199923157692,0.06992000341415405,0.05755840018391609,0.05755840018391609,0.007297587694315226,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.034272000193595886,0.0204415999352932,0.017311999574303627,0.00695404701803013,0.0,0.0,0.0,0.0,0.0,0.05215999856591225,0.0735040009021759,0.06117440015077591,0.06117440015077591,0.007381118436894922,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015359999611973763,0.03743999823927879,0.02012479966506362,0.01775999926030636,0.006181045509079057,0.0,0.0,0.0,0.0,0.0,0.06355199962854385,0.08508799970149994,0.07019200026988984,0.07019200026988984,0.0062246580442117845,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015552000142633915,0.032607998698949814,0.01959999995306134,0.017487999983131886,0.004882475068460749,0.0,0.0,0.0,0.0,0.0,0.04918399825692177,0.0865280032157898,0.05973760038614274,0.05973760038614274,0.011546847942113974,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015168000012636185,0.02675200067460537,0.017430400010198355,0.016560000367462635,0.003243135094239819,0.0,0.0,0.0,0.0,0.0,0.052671998739242554,0.08367999643087387,0.06238719932734965,0.06238719932734965,0.011207824780630104,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01500799972563982,0.03612799942493439,0.019327999837696553,0.01688000001013279,0.006058251425153085,0.0,0.0,0.0,0.0,0.0,0.06364800035953522,0.07878399640321732,0.06935679838061332,0.06935679838061332,0.004515588328677643,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015104000456631184,0.03711999952793121,0.019670399930328132,0.017152000218629837,0.006165994646522264,0.0,0.0,0.0,0.0,0.0,0.09071999788284302,0.11507199704647064,0.10252480059862136,0.10252480059862136,0.008782544051535657,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015039999969303608,0.026528000831604004,0.018396800104528665,0.01601599995046854,0.004279788048986283,0.0,0.0,0.0,0.0,0.0,0.05331199988722801,0.07977599650621414,0.06076480001211167,0.06076480001211167,0.008761579122069606,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015359999611973763,0.02643200010061264,0.01726400004699826,0.015855999663472176,0.003210008417242029,0.0,0.0,0.0,0.0,0.0,0.062431998550891876,0.08246400207281113,0.06970879957079888,0.06970879957079888,0.007016341676068233,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014911999925971031,0.02223999984562397,0.0166015999391675,0.015887999907135963,0.00204913969129354,0.0,0.0,0.0,0.0,0.0,0.10127999633550644,0.1141119971871376,0.10621120035648347,0.10621120035648347,0.004029318311754605,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015359999611973763,0.03363199904561043,0.019305599946528675,0.016671999357640743,0.0055445582307981234,0.0,0.0,0.0,0.0,0.0,0.1319040060043335,0.15839999914169312,0.14040640145540234,0.14040640145540234,0.007998041073596942,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014368000440299511,0.04447999969124794,0.018908800091594458,0.0157279996201396,0.0086814937461721,0.0,0.0,0.0,0.0,0.0,0.06428799778223038,0.08982399851083755,0.07349760085344315,0.07349760085344315,0.008605284512197258,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015104000456631184,0.028672000393271446,0.01890560006722808,0.016080000437796116,0.004797584627815021,0.0,0.0,0.0,0.0,0.0,0.11123199760913849,0.14115199446678162,0.11942399889230729,0.11942399889230729,0.008513394706791027,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014271999709308147,0.024927999824285507,0.016915200091898442,0.015216000378131866,0.003374147499442635,0.0,0.0,0.0,0.0,0.0,0.15887999534606934,0.18111999332904816,0.16934399753808976,0.16934399753808976,0.007415181260761123,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014336000196635723,0.026944000273942947,0.017737600207328796,0.015199999790638685,0.00415598714375255,0.0,0.0,0.0,0.0,0.0,0.2192319929599762,0.23472000658512115,0.22809920012950893,0.22809920012950893,0.004730841376327335,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014688000082969666,0.05052800104022026,0.020320000313222408,0.015520000364631414,0.010604522177924678,0.024270629882498958,0.03340247625954076,0.028446344104128624,0.028446344104128624,0.0032330369099793834,0.023697370291069768,0.03261352722995356,0.027774456325453972,0.027774456325453972,0.0031566742680923386,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,72,,,,,,,,False,,,64,BF16,generic,none,CUDA_EVENT,True,[64],[0],"[512, 512, 512, 512, 512, 512, 512, 512]",1,8,8,9,8,512.0,0.1111111111111111,q64_8q1s512,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.015359999611973763,0.033663999289274216,0.019987199828028678,0.018240000121295452,0.005522929139253732,0.0259194055660947,0.03719755183990719,0.029916029687899703,0.029916029687899703,0.003966666590554318,0.027104595853449518,0.03889844644729374,0.03128396954022015,0.03128396954022015,0.004148046317967881,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,136,,,,,,,,False,,,128,BF16,generic,none,CUDA_EVENT,True,[128],[0],"[1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024]",1,8,8,9,8,1024.0,0.1111111111111111,q128_8q1s1k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014592000283300877,0.03299200162291527,0.019840000104159115,0.016207999549806118,0.006546343008726814,0.02587869595769926,0.03763167265431482,0.030174939058162802,0.030174939058162802,0.0037303581782277364,0.026473304070195713,0.03849632587654989,0.03086826083864964,0.03086826083864964,0.003816069654529222,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,144,,,,,,,,False,,,128,BF16,generic,none,CUDA_EVENT,True,[128],[0],"[1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024]",1,16,16,17,16,1024.0,0.058823529411764705,q128_16q1s1k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.013824000023305416,0.021023999899625778,0.016304000187665223,0.015584000386297703,0.0023236165806545476,0.031810621525966996,0.04156949936878106,0.0346749350032807,0.0346749350032807,0.003130209181143985,0.035677378270900374,0.04662250161636451,0.038889864871740294,0.038889864871740294,0.0035107033960828727,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,272,,,,,,,,False,,,256,BF16,generic,none,CUDA_EVENT,True,[256],[0],"[2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048]",1,16,16,17,16,2048.0,0.058823529411764705,q256_16q1s2k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014527999795973301,0.02175999991595745,0.01569600012153387,0.015008000191301107,0.0020882666168036863,0.038211712107062853,0.07212229256520057,0.04364509673334097,0.04364509673334097,0.009671953074025085,0.0476442859917874,0.08992570455184198,0.05441890342616107,0.05441890342616107,0.01205947791784038,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,288,,,,,,,,False,,,256,BF16,generic,none,CUDA_EVENT,True,[256],[0],"[2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048]",1,32,32,33,32,2048.0,0.030303030303030304,q256_32q1s2k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014368000440299511,0.033824000507593155,0.017628800217062236,0.014864000026136637,0.005791495403857738,0.05214261250030033,0.06266261508706669,0.05677791295527661,0.05677791295527661,0.0030298223079651514,0.08648138506878382,0.10392938682791132,0.09416928531647478,0.09416928531647478,0.005025126612204489,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,544,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,32,32,33,32,4096.0,0.030303030303030304,q512_32q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014527999795973301,0.031199999153614044,0.017900799959897996,0.015343999955803156,0.004974475661316249,0.06411958891421111,0.07405276123263956,0.06793047918211377,0.06793047918211377,0.0033918658556700006,0.14058441263169497,0.16236323591492058,0.1489399211274489,0.1489399211274489,0.00743678300375353,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,576,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,64,64,65,64,4096.0,0.015384615384615385,q512_64q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014720000326633453,0.03855999931693077,0.018495999928563833,0.015647999942302704,0.006918530056761627,0.09872985549401277,0.10683454583043753,0.10185316839884552,0.10185316839884552,0.0020802254037042074,0.2743261390166379,0.2968454508984596,0.2830044295482752,0.2830044295482752,0.005780016596065092,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1088,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192]",1,64,64,65,64,8192.0,0.015384615384615385,q1k_64q1s8k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014911999925971031,0.030400000512599945,0.01977920001372695,0.01726400014013052,0.005350883464206169,0.13918872472233365,0.16279522855335207,0.1501912864839173,0.1501912864839173,0.008992185529011513,0.11892328861766266,0.13909276049083735,0.1283239123428725,0.1283239123428725,0.007682951884956939,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,32,32,33,32,4096.0,0.030303030303030304,q2k_32q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.01727999933063984,0.02223999984562397,0.01819519978016615,0.017680000513792038,0.0014397298270620873,0.3728044181625443,0.38295504353701676,0.3761264483787333,0.3761264483787333,0.0033660272332490977,0.07967557017401883,0.08184495665371805,0.08038555277807365,0.08038555277807365,0.0007193856241088455,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,4112,,,,,,,,False,,,4096,BF16,generic,none,CUDA_EVENT,True,[4096],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,16,16,17,16,4096.0,0.058823529411764705,q4k_16q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.01500799972563982,0.034591998904943466,0.01941439984366298,0.01756799966096878,0.005541380584373771,0.05951999872922897,0.08268799632787704,0.06715519949793816,0.06715519949793816,0.006657802116288364,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,2,1024,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512]",1024,512,512,512.0,True,0.0,0.0,0.0,False,0.0,1024.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,2q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015104000456631184,0.03283200040459633,0.019180799927562477,0.016543999314308167,0.0052187911233635975,0.07199999690055847,0.08675199747085571,0.07749439924955369,0.07749439924955369,0.004849874741156772,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,4,2048,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512]",2048,512,512,512.0,True,0.0,0.0,0.0,False,0.0,2048.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,4q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01724799908697605,0.02284800074994564,0.01928640007972717,0.018400000408291817,0.0020641390157565697,0.09676799923181534,0.12310399860143663,0.10618879944086074,0.10618879944086074,0.008982738207839057,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512, 512, 512, 512, 512]",4096,512,512,512.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.024224000051617622,0.03315199911594391,0.027436799928545953,0.027328000403940678,0.002679154874546328,0.14716799557209015,0.16412800550460815,0.15470399856567385,0.15470399856567385,0.005423454433987419,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512]",8192,512,512,512.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01833599992096424,0.04806400090456009,0.026252799853682517,0.022672000341117382,0.00878438440429033,0.1844799965620041,0.2072959989309311,0.19359359890222552,0.19359359890222552,0.007663539820967659,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,2,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[2048, 2048]",4096,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,2q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.024480000138282776,0.04979199916124344,0.029081599973142146,0.025679999962449074,0.007227422044689197,0.34147199988365173,0.37968000769615173,0.3583200007677078,0.3583200007677078,0.011804422100726231,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,4,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[2048, 2048, 2048, 2048]",8192,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,4q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014720000326633453,0.023744000121951103,0.017753600236028434,0.016368000768125057,0.003061040281616705,0.0,0.0,0.0,0.0,0.0,0.05686400085687637,0.1090880036354065,0.06629760004580021,0.06629760004580021,0.014803727026812366,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.021088000386953354,0.01647359998896718,0.015840000472962856,0.0017964402433206,0.0,0.0,0.0,0.0,0.0,0.08540800213813782,0.09961599856615067,0.09146559983491898,0.09146559983491898,0.00447423706371901,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014560000039637089,0.04156799986958504,0.018512000143527985,0.015584000386297703,0.007849701681877904,0.0,0.0,0.0,0.0,0.0,0.17948800325393677,0.194815993309021,0.1866239994764328,0.1866239994764328,0.004156328241374654,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.029632000252604485,0.019449600111693145,0.016736000776290894,0.005126811425136928,0.0,0.0,0.0,0.0,0.0,0.27529600262641907,0.2898879945278168,0.2825664013624191,0.2825664013624191,0.004542801609674093,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.031328000128269196,0.019759999960660933,0.016367999836802483,0.006266303740589241,0.0,0.0,0.0,0.0,0.0,0.39190399646759033,0.4079039990901947,0.3987520009279252,0.3987520009279252,0.0037367727509362725,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01539199985563755,0.040031999349594116,0.0212032001465559,0.01649599988013506,0.008068547381446682,0.0,0.0,0.0,0.0,0.0,0.06588800251483917,0.07897599786520004,0.0703904002904892,0.0703904002904892,0.004214826869769361,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014495999552309513,0.022495999932289124,0.016006399970501663,0.01515199989080429,0.002265581884342885,0.0,0.0,0.0,0.0,0.0,0.1231679990887642,0.13526399433612823,0.1288223996758461,0.1288223996758461,0.004328976434897094,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014592000283300877,0.023231999948620796,0.01705600004643202,0.016048000194132328,0.002810904312746391,0.0,0.0,0.0,0.0,0.0,0.3158079981803894,0.33129599690437317,0.32348800003528594,0.32348800003528594,0.00417599076136664,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014527999795973301,0.021888000890612602,0.01708160014823079,0.01609600055962801,0.0025289234621475062,0.0,0.0,0.0,0.0,0.0,0.5103679895401001,0.5200319886207581,0.5132320046424866,0.5132320046424866,0.0035759433157749533,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014688000082969666,0.032416000962257385,0.019353600032627583,0.016176000237464905,0.0061458798960684425,0.0,0.0,0.0,0.0,0.0,0.7400320172309875,0.7516480088233948,0.7449311971664427,0.7449311971664427,0.004573461776168607,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.0144640002399683,0.02006400004029274,0.01573119992390275,0.014992000069469213,0.001654446341262991,0.0,0.0,0.0,0.0,0.0,0.07097599655389786,0.09932799637317657,0.07828159928321837,0.07828159928321837,0.009464602895232642,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01484800036996603,0.02319999970495701,0.016883199848234654,0.015775999519973993,0.0026693906156048403,0.0,0.0,0.0,0.0,0.0,0.14176000654697418,0.1598079949617386,0.15063679963350293,0.15063679963350293,0.005741662969679433,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01484800036996603,0.03203200176358223,0.022224000189453363,0.02247999981045723,0.005996176183124981,0.0,0.0,0.0,0.0,0.0,0.39180800318717957,0.41046398878097534,0.4003200054168701,0.4003200054168701,0.005552433764307601,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015072000212967396,0.028192000463604927,0.017152000125497578,0.015536000020802021,0.003847486121602065,0.0,0.0,0.0,0.0,0.0,0.6239359974861145,0.6367359757423401,0.6285343945026398,0.6285343945026398,0.004006084286606002,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.026688000187277794,0.01718079997226596,0.015488000120967627,0.0037132536813398722,0.0,0.0,0.0,0.0,0.0,0.9141119718551636,0.9721279740333557,0.927455997467041,0.927455997467041,0.02148767779232235,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014751999638974667,0.047520000487565994,0.028710400220006704,0.0266720000654459,0.010774688401598903,0.06094816381288764,0.06816969726785131,0.0644060660218149,0.0644060660218149,0.002203106589073183,0.087051838094461,0.09736630405680231,0.09199073574791852,0.09199073574791852,0.0031466818046499644,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,520,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,8,8,9,8,16384.0,0.1111111111111111,q512_8q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.014911999925971031,0.02489599958062172,0.017427200078964235,0.01595200039446354,0.003071906489592143,0.19733789497223914,0.2016295616652644,0.19860388994013833,0.19860388994013833,0.001512389485439305,0.44861409134062713,0.45837046456077934,0.45149211526120153,0.45149211526120153,0.0034381598874302305,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1040,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,16,16,17,16,16384.0,0.058823529411764705,q1k_16q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.015039999969303608,0.03046399913728237,0.020643199887126686,0.016944000497460365,0.006320148675595729,0.2157924314537054,0.22215709640166995,0.21769400765743155,0.21769400765743155,0.0019679921844645526,0.4905115822753901,0.5049789194903589,0.4948340005651485,0.4948340005651485,0.0044733865493072344,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,32,32,33,32,16384.0,0.030303030303030304,q2k_32q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.017152000218629837,0.02425600029528141,0.01865920014679432,0.018112000077962875,0.0019454553198986453,0.7430866512973927,0.7578834738720781,0.7455351005236347,0.7455351005236347,0.00416692487556653,0.7369773831645824,0.7516525540362471,0.7394057025273603,0.7394057025273603,0.004132666607961175,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,4112,,,,,,,,False,,,4096,BF16,generic,none,CUDA_EVENT,True,[4096],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,16,16,17,16,32768.0,0.058823529411764705,q4k_16q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.015263999812304974,0.02470399998128414,0.018009600043296815,0.01657600048929453,0.0031544532774534346,0.25251173919752334,0.25566890792461106,0.2536729054481981,0.2536729054481981,0.0010279562245342983,1.0425282722084215,1.0555630628624373,1.0473223013846877,1.0473223013846877,0.004244053880724073,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,32,32,33,32,32768.0,0.030303030303030304,q2k_32q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.01500799972563982,0.027807999402284622,0.01912960009649396,0.01643200032413006,0.004688164695934997,0.1251379565220268,0.14341821167748953,0.12742497577885914,0.12742497577885914,0.005353812167343766,1.1355340167064276,1.3014137556763312,1.1562870179154463,1.1562870179154463,0.048581869194943325,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1088,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,64,64,65,64,32768.0,0.015384615384615385,q1k_64q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
0.0,0.0,0.0,0.0,0.0,0.07446400076150894,0.08246400207281113,0.07736000046133995,0.07713599875569344,0.002301031358835964,11.935359954833984,12.102368354797363,11.96896333694458,11.96896333694458,0.05162549713370252,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,8192,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,8192.0,16384.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.07664000242948532,0.08118399977684021,0.07796800062060356,0.077504001557827,0.0013681081693640953,19.84774398803711,20.15795135498047,19.915702438354494,19.915702438354494,0.0912299324030902,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,8192,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,16384.0,24576.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks24k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.0753600001335144,0.08207999914884567,0.07715519964694977,0.07595199719071388,0.0024238358447475007,27.781503677368164,27.9836483001709,27.844886589050287,27.844886589050287,0.061307010154819624,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,8192,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,24576.0,32768.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04339199885725975,0.050016000866889954,0.04504639990627766,0.04391999915242195,0.002339637813151782,5.1544318199157715,5.269279956817627,5.16938238143921,5.16938238143921,0.03363105774712194,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,4096,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,8192.0,12288.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks12k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04291199892759323,0.058848001062870026,0.045657599717378615,0.04383999854326248,0.004522763233004815,9.256383895874023,9.27734375,9.261776161193849,9.261776161193849,0.0062232312957398745,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,4096,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,16384.0,20480.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks20k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.043296001851558685,0.04982399940490723,0.045123199746012685,0.04387199878692627,0.0023369398894319345,13.365216255187988,13.634464263916016,13.398719978332519,13.398719978332519,0.0787651922310212,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,4096,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,24576.0,28672.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks28k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.02703999914228916,0.0297279991209507,0.027692800015211107,0.02723200060427189,0.000851127720159845,2.387968063354492,2.4014720916748047,2.3920736074447637,2.3920736074447637,0.003552645719470337,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,2048,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,8192.0,10240.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks10k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.02691200003027916,0.029823999851942062,0.027689599990844728,0.027583999559283257,0.0007783369959809023,4.440767765045166,4.4521918296813965,4.444268751144409,4.444268751144409,0.004022325406840951,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,2048,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,16384.0,18432.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks18k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.02703999914228916,0.03888000175356865,0.030262400023639204,0.02801600005477667,0.0037877256209144718,6.500351905822754,6.600607872009277,6.524902391433716,6.524902391433716,0.03321667087212851,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,2048,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,24576.0,26624.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks26k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.07865600287914276,0.08668799698352814,0.08130879923701287,0.07993599772453308,0.0027016201268628523,35.717376708984375,35.843265533447266,35.771837615966795,35.771837615966795,0.04024815379149301,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,8192,32768.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,32768.0,40960.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04179200157523155,0.04864000156521797,0.0436256006360054,0.04267200082540512,0.002191476789190783,6.103871822357178,6.18287992477417,6.118390369415283,6.118390369415283,0.022924175047267112,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,8192,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,8192.0,16384.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04163200035691261,0.05987200140953064,0.04625920057296753,0.04403200000524521,0.005544136325859629,10.207136154174805,11.073247909545898,10.326777648925782,10.326777648925782,0.25139335734700863,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,8192,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,16384.0,24576.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks24k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04137599840760231,0.05027199909090996,0.04355199970304966,0.042399998754262924,0.0026572716942034787,14.303423881530762,14.343520164489746,14.311049747467042,14.311049747467042,0.01137511287661936,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,8192,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,24576.0,32768.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.026016000658273697,0.03718400001525879,0.03212799951434135,0.03270399942994118,0.003953184738275612,2.6534719467163086,2.6875839233398438,2.661257576942444,2.661257576942444,0.009372641904054613,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,4096,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,8192.0,12288.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks12k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.025887999683618546,0.03347200155258179,0.027168000116944313,0.02649599965661764,0.0021492000999850562,4.70630407333374,4.7400641441345215,4.714492845535279,4.714492845535279,0.009349196197962609,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,4096,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,16384.0,20480.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks20k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.025631999596953392,0.030719999223947525,0.02736639976501465,0.02711999975144863,0.0013921874495504173,6.759712219238281,6.831999778747559,6.774675178527832,6.774675178527832,0.021507023842020512,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,4096,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,24576.0,28672.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks28k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.017855999991297722,0.024831999093294144,0.01991359982639551,0.018655999563634396,0.0023971416189370203,1.3446400165557861,1.3609600067138672,1.3508928060531615,1.3508928060531615,0.006186954712541735,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,2048,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,8192.0,10240.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks10k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.0180479995906353,0.022752000018954277,0.019507200084626676,0.01896000001579523,0.0015079573553847933,2.5208001136779785,2.5887041091918945,2.534086418151855,2.534086418151855,0.01891953046799999,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,2048,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,16384.0,18432.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks18k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.018144000321626663,0.03222399950027466,0.02052800003439188,0.018864000216126442,0.004114364828004189,3.6922879219055176,3.760576009750366,3.706630396842957,3.706630396842957,0.019888657921309623,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,2048,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,24576.0,26624.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks26k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04169600084424019,0.04822399839758873,0.0439775999635458,0.04334400035440922,0.002101844179466219,18.41494369506836,18.502975463867188,18.445004844665526,18.445004844665526,0.026939913540867878,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,8192,32768.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,32768.0,40960.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.023615999147295952,0.03267199918627739,0.026281599886715412,0.0248800003901124,0.003167969318232417,3.1188158988952637,3.1837120056152344,3.133536005020142,3.133536005020142,0.018365009771617643,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,8192,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,8192.0,16384.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.02380800060927868,0.03283200040459633,0.025670399703085423,0.024639999493956566,0.0026095917641781453,5.1729278564453125,5.243135929107666,5.192828798294067,5.192828798294067,0.01984737475315108,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,8192,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,16384.0,24576.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks24k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.024032000452280045,0.03167999908328056,0.02665280010551214,0.02550400048494339,0.002510131946267113,7.223167896270752,7.257152080535889,7.231532812118529,7.231532812118529,0.009418898846297825,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,8192,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,24576.0,32768.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01696000061929226,0.023679999634623528,0.01866880003362894,0.017455999739468098,0.002277860345236006,1.478559970855713,1.496127963066101,1.48257919549942,1.48257919549942,0.005278358037488471,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,4096,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,8192.0,12288.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks12k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.017023999243974686,0.030688000842928886,0.02055360022932291,0.018240000121295452,0.004065815389665311,2.651711940765381,2.6691839694976807,2.657548785209656,2.657548785209656,0.005897264516465086,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,4096,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,16384.0,20480.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks20k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.016767999157309532,0.024351999163627625,0.01840319987386465,0.017167999409139156,0.0025234367788448957,3.823551893234253,3.8631999492645264,3.8335999727249135,3.8335999727249135,0.011858923917429712,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,4096,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,24576.0,28672.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks28k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.013952000066637993,0.03622400015592575,0.018214400112628936,0.014719999860972166,0.007277897466521675,0.7039039731025696,0.7163199782371521,0.7077983975410462,0.7077983975410462,0.004048424224033295,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,2048,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,8192.0,10240.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks10k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.015104000456631184,0.03097599931061268,0.020870400313287973,0.02054399996995926,0.005422197456220526,1.2929600477218628,1.313088059425354,1.3006752014160157,1.3006752014160157,0.00790622446009414,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,2048,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,16384.0,18432.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks18k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.03232000023126602,0.019132800027728082,0.01673599984496832,0.0050857949387988245,1.8775999546051025,1.9125759601593018,1.8867840051651,1.8867840051651,0.01196043526228758,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,2048,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,24576.0,26624.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks26k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.023584000766277313,0.03126399964094162,0.02633600030094385,0.025200000032782555,0.0025710957466677864,9.283391952514648,9.48249626159668,9.331705665588379,9.331705665588379,0.06366970422148335,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,8192,32768.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,32768.0,40960.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.07552000135183334,0.0851840004324913,0.07839040011167527,0.07703999802470207,0.0030805904904383677,43.62025451660156,43.89299011230469,43.68390693664551,43.68390693664551,0.08458163192147439,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,40960.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,40960.0,49152.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks48k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.07718399912118912,0.0854400023818016,0.07895359992980958,0.07791999727487564,0.0024425064759109735,59.421791076660156,59.73017501831055,59.5021312713623,59.5021312713623,0.10367381308510447,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,57344.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,57344.0,65536.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks64k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.07875200361013412,0.09055999666452408,0.08208959847688675,0.08087999746203423,0.0033614630055883482,75.2852783203125,75.55481719970703,75.38758392333985,75.38758392333985,0.08870109119325344,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,73728.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,73728.0,81920.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks80k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.07788799703121185,0.09888000041246414,0.0819871999323368,0.08008000254631042,0.005835618489111142,91.13442993164062,91.45164489746094,91.24225158691405,91.24225158691405,0.1009018902177725,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,90112.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,90112.0,98304.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks96k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.07596799731254578,0.08790399879217148,0.07967040091753005,0.07808000221848488,0.0036209252740976605,106.89055633544922,107.34063720703125,106.9560287475586,106.9560287475586,0.13032874360676439,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,106496.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,106496.0,114688.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks112k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.07648000121116638,0.08534400165081024,0.07928640022873878,0.07787200063467026,0.0028305104107121735,122.71724700927734,122.9840316772461,122.8334243774414,122.8334243774414,0.09625274868603513,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,122880.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,122880.0,131072.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks128k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.07664000242948532,0.08774399757385254,0.08081279993057251,0.07972799986600876,0.003487270970977775,130.6565399169922,131.09359741210938,130.79229431152345,130.79229431152345,0.13138911961272914,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,131072.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,131072.0,139264.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks136k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.01532800029963255,0.040063999593257904,0.019459199998527764,0.01601599995046854,0.007496165774486326,9.443936347961426,9.649087905883789,9.489968109130858,9.489968109130858,0.05961646816296781,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,512,130560.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,True,130560.0,131072.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512s128k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.02691200003027916,0.040832001715898514,0.02942080032080412,0.027888000011444092,0.004040048279091587,16.753759384155273,16.90662384033203,16.78766403198242,16.78766403198242,0.04236470233451937,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,2048,65536.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,65536.0,67584.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks66k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04416000097990036,0.04867200180888176,0.04565120078623295,0.04468800127506256,0.0018079971851529544,50.29715347290039,50.50300979614258,50.35054740905761,50.35054740905761,0.0724480602714254,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,4096,98304.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,98304.0,102400.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks100k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.060447998344898224,0.06684800237417221,0.06276160031557083,0.061824001371860504,0.0024045636532566625,96.07142639160156,96.53209686279297,96.1898666381836,96.1898666381836,0.1365794879996899,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,6144,131072.0,True,FLASH_ATTN,False,vllm020_batch_spec,[6144],6144,6144,6144,6144.0,True,0.0,0.0,0.0,True,131072.0,137216.0,6144,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q6ks134k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04217600077390671,0.05100800096988678,0.04406719990074635,0.042847998440265656,0.002656672983576728,22.5166072845459,22.986656188964844,22.63068161010742,22.63068161010742,0.13840494695459715,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,40960.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,40960.0,49152.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks48k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04182400181889534,0.047200001776218414,0.043644800409674646,0.04262400045990944,0.001927741871473188,30.725727081298828,30.787456512451172,30.74285774230957,30.74285774230957,0.019110324860515015,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,57344.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,57344.0,65536.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks64k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04185599833726883,0.05049600079655647,0.043852799385786054,0.04289599880576134,0.0024573088797598033,38.930206298828125,38.98448181152344,38.94538269042969,38.94538269042969,0.017734743323340796,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,73728.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,73728.0,81920.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks80k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04227200150489807,0.046911999583244324,0.04412479996681214,0.04327999986708164,0.001811858548765738,47.13151931762695,47.245887756347656,47.148198699951166,47.148198699951166,0.033297799765613076,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,90112.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,90112.0,98304.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks96k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04182400181889534,0.06195199862122536,0.04529919996857643,0.043136000633239746,0.005751677082163394,55.338497161865234,55.64672088623047,55.42207336425782,55.42207336425782,0.11010194068839292,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,106496.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,106496.0,114688.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks112k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.04182400181889534,0.054687999188899994,0.04418559968471527,0.042767999693751335,0.0036820249064621804,63.549087524414055,63.80697631835938,63.62084197998047,63.62084197998047,0.08872184973119365,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,122880.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,122880.0,131072.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks128k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.0461760014295578,0.06566400080919266,0.05008000023663044,0.04843199998140335,0.0055350567765421,67.64147186279297,67.91651153564453,67.70619888305666,67.70619888305666,0.0993520482760623,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,131072.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,131072.0,139264.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks136k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.023711999878287315,0.03859199956059456,0.02689919974654913,0.0244159996509552,0.004626699769228834,4.775519847869873,4.829855918884277,4.790303993225097,4.790303993225097,0.01597276061319242,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,512,130560.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,True,130560.0,131072.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512s128k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.024224000051617622,0.0453759990632534,0.027676799893379213,0.025200000032782555,0.006154880946840619,9.579392433166504,9.63871955871582,9.59802885055542,9.59802885055542,0.016811667352984866,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,2048,65536.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,65536.0,67584.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks66k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.02630399912595749,0.03593600168824196,0.02780479993671179,0.027040000073611736,0.0027569987978884785,25.21686363220215,25.383039474487305,25.250255966186526,25.250255966186526,0.04761963064809279,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,4096,98304.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,98304.0,102400.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks100k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.034015998244285583,0.041728001087903976,0.03665280006825924,0.0352960005402565,0.002710617869728372,48.101280212402344,48.49884796142578,48.17219200134278,48.17219200134278,0.11195495368044632,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,6144,131072.0,True,FLASH_ATTN,False,vllm020_batch_spec,[6144],6144,6144,6144,6144.0,True,0.0,0.0,0.0,True,131072.0,137216.0,6144,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q6ks134k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.024191999807953835,0.039583999663591385,0.027286400087177753,0.025200000032782555,0.004620118437533604,11.32140827178955,11.496224403381348,11.354345512390138,11.354345512390138,0.050117766215439966,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,40960.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,40960.0,49152.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks48k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.024032000452280045,0.03232000023126602,0.026406400091946124,0.02518399991095066,0.002792696117638337,15.42249584197998,15.639776229858397,15.451993656158448,15.451993656158448,0.06304729308178388,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,57344.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,57344.0,65536.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks64k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.024224000051617622,0.03488000109791756,0.026147199980914592,0.02459200005978346,0.003303059079404486,19.5251522064209,19.70569610595703,19.54985942840576,19.54985942840576,0.052248914965338324,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,73728.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,73728.0,81920.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks80k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.024159999564290047,0.03046399913728237,0.0255103999748826,0.024656000547111034,0.00195717586616911,23.630016326904297,23.927040100097656,23.672822570800783,23.672822570800783,0.08715594728873713,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,90112.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,90112.0,98304.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks96k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.02380800060927868,0.03574400022625923,0.02678720001131296,0.024720000103116035,0.003678231234048846,27.736671447753906,27.79840087890625,27.747158622741697,27.747158622741697,0.018952179343788657,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,106496.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,106496.0,114688.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks112k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.024224000051617622,0.042080000042915344,0.02844479978084564,0.02527999971061945,0.006562862750709003,31.83427238464356,32.08835220336914,31.86994876861572,31.86994876861572,0.07444245241186317,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,122880.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,122880.0,131072.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks128k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.02377600036561489,0.03139200061559677,0.026396799832582474,0.025071999989449978,0.0025962810796740263,33.88313674926758,33.96985626220703,33.89708137512208,33.89708137512208,0.02518493598134421,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,131072.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,131072.0,139264.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks136k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014944000169634819,0.022624000906944275,0.017174400109797715,0.015536000020802021,0.0028331860349340154,3.179744005203247,3.192960023880005,3.182867193222046,3.182867193222046,0.0035239767128429386,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,512,130560.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,True,130560.0,131072.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512s128k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.02985600009560585,0.017318400088697672,0.015568000264465809,0.004479309643844885,4.811935901641846,4.826848030090332,4.8173023700714115,4.8173023700714115,0.00594674080057053,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,2048,65536.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,65536.0,67584.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks66k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.017184000462293625,0.03916800022125244,0.021356799826025962,0.01935999933630228,0.0062801287963799276,14.374591827392578,14.635007858276367,14.41227512359619,14.41227512359619,0.07569270440104603,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,4096,98304.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,98304.0,102400.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks100k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
0.0,0.0,0.0,0.0,0.0,0.020479999482631683,0.033215999603271484,0.023340800032019614,0.02092800009995699,0.004304974035052475,24.088096618652344,24.18492889404297,24.11243553161621,24.11243553161621,0.02797845886790172,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,6144,131072.0,True,FLASH_ATTN,False,vllm020_batch_spec,[6144],6144,6144,6144,6144.0,True,0.0,0.0,0.0,True,131072.0,137216.0,6144,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q6ks134k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
1 time_stats.attn_input_reshape.min time_stats.attn_input_reshape.max time_stats.attn_input_reshape.mean time_stats.attn_input_reshape.median time_stats.attn_input_reshape.std time_stats.attn_kv_cache_save.min time_stats.attn_kv_cache_save.max time_stats.attn_kv_cache_save.mean time_stats.attn_kv_cache_save.median time_stats.attn_kv_cache_save.std time_stats.attn_prefill.min time_stats.attn_prefill.max time_stats.attn_prefill.mean time_stats.attn_prefill.median time_stats.attn_prefill.std time_stats.attn_decode.min time_stats.attn_decode.max time_stats.attn_decode.mean time_stats.attn_decode.median time_stats.attn_decode.std time_stats.attn_output_reshape.min time_stats.attn_output_reshape.max time_stats.attn_output_reshape.mean time_stats.attn_output_reshape.median time_stats.attn_output_reshape.std n_embd n_q_head n_kv_head block_size num_tensor_parallel_workers max_model_len batch_size prefill_chunk_size kv_cache_size is_prefill attention_backend is_mixed_batch mode seq_lens total_tokens max_seq_len min_seq_len avg_seq_len equal_seq_len seq_len_variance seq_len_std seq_len_cv is_chunked_prefill_sample chunk_start_token chunk_end_token total_prefill_tokens profiling_precision model_arch quant_signature measurement_type is_true_mixed_batch prefill_seq_lens prefill_kv_cache_sizes decode_kv_cache_sizes num_prefill_seqs num_decode_seqs decode_batch_size total_batch_size total_decode_tokens decode_avg_kv_cache_size batch_composition_ratio batch_spec projection_policy
2 0.0 0.0 0.0 0.0 0.0 0.01414399966597557 0.028863999992609024 0.019705599918961526 0.01771199982613325 0.005157200849836681 0.047968000173568726 0.07046400010585785 0.05810240097343922 0.05810240097343922 0.007477463486041561 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 64 0.0 True FLASH_ATTN False vllm020_batch_spec [64] 64 64 64 64.0 True 0.0 0.0 0.0 False 0.0 64.0 64 BF16 generic none CUDA_EVENT False q64 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
3 0.0 0.0 0.0 0.0 0.0 0.014399999752640724 0.04947200044989586 0.020412799902260303 0.01635199971497059 0.010107497379722417 0.046560000628232956 0.08323200047016144 0.05587520003318787 0.05587520003318787 0.011126758739503428 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 128 0.0 True FLASH_ATTN False vllm020_batch_spec [128] 128 128 128 128.0 True 0.0 0.0 0.0 False 0.0 128.0 128 BF16 generic none CUDA_EVENT False q128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
4 0.0 0.0 0.0 0.0 0.0 0.01484800036996603 0.022207999601960182 0.017033600155264138 0.015312000177800655 0.002819235991970241 0.05104000121355057 0.07692799717187881 0.056396800279617305 0.056396800279617305 0.007481982178637539 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 256 0.0 True FLASH_ATTN False vllm020_batch_spec [256] 256 256 256 256.0 True 0.0 0.0 0.0 False 0.0 256.0 256 BF16 generic none CUDA_EVENT False q256 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
5 0.0 0.0 0.0 0.0 0.0 0.015072000212967396 0.022272000089287758 0.01706880023702979 0.01616000011563301 0.002460889579319197 0.06931199878454208 0.0838719978928566 0.07432000041007995 0.07432000041007995 0.004777766433175866 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 512 0.0 True FLASH_ATTN False vllm020_batch_spec [512] 512 512 512 512.0 True 0.0 0.0 0.0 False 0.0 512.0 512 BF16 generic none CUDA_EVENT False q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
6 0.0 0.0 0.0 0.0 0.0 0.018592000007629395 0.028543999418616295 0.02095999978482723 0.019183999858796597 0.003198175496053494 0.12179200351238251 0.15408000349998474 0.1307712011039257 0.1307712011039257 0.00858807797538298 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 1024 0.0 True FLASH_ATTN False vllm020_batch_spec [1024] 1024 1024 1024 1024.0 True 0.0 0.0 0.0 False 0.0 1024.0 1024 BF16 generic none CUDA_EVENT False q1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
7 0.0 0.0 0.0 0.0 0.0 0.027775999158620834 0.03385600075125694 0.030131200328469276 0.029680000618100166 0.0021152558103575215 0.32678401470184326 0.3450239896774292 0.33396480381488797 0.33396480381488797 0.0045872424917606375 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 2048 0.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 False 0.0 2048.0 2048 BF16 generic none CUDA_EVENT False q2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
8 0.0 0.0 0.0 0.0 0.0 0.04438399896025658 0.05084799975156784 0.046540799736976626 0.04531199857592583 0.002277905811237223 1.0959680080413818 1.1151360273361206 1.0999775886535645 1.0999775886535645 0.005694403246120485 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 4096 0.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 False 0.0 4096.0 4096 BF16 generic none CUDA_EVENT False q4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
9 0.0 0.0 0.0 0.0 0.0 0.078015998005867 0.08691199868917465 0.08114239946007729 0.08019199967384338 0.00292795706334475 4.070400238037109 4.113152027130127 4.087088012695312 4.087088012695312 0.013660567012509554 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 8192 0.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 False 0.0 8192.0 8192 BF16 generic none CUDA_EVENT False q8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
10 0.0 0.0 0.0 0.0 0.0 0.016063999384641647 0.05167999863624573 0.022115200012922286 0.017583999782800674 0.010340094822340818 0.05196800082921982 0.09011200070381165 0.06328320093452933 0.06328320093452933 0.012557341255467452 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 64 448.0 True FLASH_ATTN False vllm020_batch_spec [64] 64 64 64 64.0 True 0.0 0.0 0.0 True 448.0 512.0 64 BF16 generic none CUDA_EVENT False q64s512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
11 0.0 0.0 0.0 0.0 0.0 0.01583999954164028 0.026623999699950218 0.018927999772131443 0.017376000061631203 0.003514650316260619 0.06681600213050842 0.07993599772453308 0.0725280001759529 0.0725280001759529 0.004343502558613716 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 128 896.0 True FLASH_ATTN False vllm020_batch_spec [128] 128 128 128 128.0 True 0.0 0.0 0.0 True 896.0 1024.0 128 BF16 generic none CUDA_EVENT False q128s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
12 0.0 0.0 0.0 0.0 0.0 0.01648000068962574 0.030880000442266464 0.01945280022919178 0.017967999912798405 0.004096211183007485 0.1311360001564026 0.1546880006790161 0.13908160030841826 0.13908160030841826 0.007511906874366178 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 256 1792.0 True FLASH_ATTN False vllm020_batch_spec [256] 256 256 256 256.0 True 0.0 0.0 0.0 True 1792.0 2048.0 256 BF16 generic none CUDA_EVENT False q256s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
13 0.0 0.0 0.0 0.0 0.0 0.017855999991297722 0.03558399900794029 0.020851199887692927 0.018559999763965607 0.005235911594130716 0.32950401306152344 0.350271999835968 0.33912960588932034 0.33912960588932034 0.006027400986663648 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 512 3584.0 True FLASH_ATTN False vllm020_batch_spec [512] 512 512 512 512.0 True 0.0 0.0 0.0 True 3584.0 4096.0 512 BF16 generic none CUDA_EVENT False q512s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
14 0.0 0.0 0.0 0.0 0.0 0.019328000023961067 0.040608000010252 0.022790400311350822 0.020704000256955624 0.006113051965778337 1.1415679454803467 1.1518720388412476 1.144483208656311 1.144483208656311 0.0032332311374389127 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 1024 7168.0 True FLASH_ATTN False vllm020_batch_spec [1024] 1024 1024 1024 1024.0 True 0.0 0.0 0.0 True 7168.0 8192.0 1024 BF16 generic none CUDA_EVENT False q1ks8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
15 0.0 0.0 0.0 0.0 0.0 0.015807999297976494 0.030688000842928886 0.019596799835562707 0.01774400006979704 0.004343384771033462 0.0 0.0 0.0 0.0 0.0 0.049056001007556915 0.07580800354480743 0.05948160067200661 0.05948160067200661 0.009031541471446955 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1] 1 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
16 0.0 0.0 0.0 0.0 0.0 0.01603199914097786 0.02486399933695793 0.01923839971423149 0.018079999834299088 0.0032282528537266424 0.0 0.0 0.0 0.0 0.0 0.05142400041222572 0.07353600114583969 0.059328000620007516 0.059328000620007516 0.0073307807735143084 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 8 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
17 0.0 0.0 0.0 0.0 0.0 0.016543999314308167 0.03977600112557411 0.021379199624061585 0.018511999398469925 0.006593576176246171 0.0 0.0 0.0 0.0 0.0 0.0488319993019104 0.06435199826955795 0.05479039996862411 0.05479039996862411 0.005672522998491864 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 16 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
18 0.0 0.0 0.0 0.0 0.0 0.01635199971497059 0.02844800055027008 0.019267200119793416 0.017952000722289085 0.0035068687666949577 0.0 0.0 0.0 0.0 0.0 0.049855999648571014 0.07798399776220322 0.05986879989504815 0.05986879989504815 0.01043914754878828 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 32 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
19 0.0 0.0 0.0 0.0 0.0 0.016383999958634377 0.026079999282956123 0.01923519968986511 0.017791999503970146 0.0032161974331284568 0.0 0.0 0.0 0.0 0.0 0.058111999183893204 0.1045759990811348 0.06708480007946492 0.06708480007946492 0.013479022462646494 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 64 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
20 0.0 0.0 0.0 0.0 0.0 0.014720000326633453 0.04057599976658821 0.019100800156593323 0.015455999877303839 0.007512281243011577 0.0 0.0 0.0 0.0 0.0 0.05363199859857559 0.07782399654388428 0.06090559959411622 0.06090559959411622 0.007544620176348091 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 8 0 1024.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
21 0.0 0.0 0.0 0.0 0.0 0.014431999996304512 0.02191999927163124 0.016684799920767546 0.01563199982047081 0.0024293118621811216 0.0 0.0 0.0 0.0 0.0 0.0629120022058487 0.07891199737787247 0.06891520097851753 0.06891520097851753 0.005472695665695425 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 16 0 1024.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
22 0.0 0.0 0.0 0.0 0.0 0.014560000039637089 0.038943998515605927 0.018313600029796363 0.01561600062996149 0.007127270260115769 0.0 0.0 0.0 0.0 0.0 0.08675199747085571 0.10662399977445602 0.09391999915242194 0.09391999915242194 0.006988099589086635 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 32 0 1024.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
23 0.0 0.0 0.0 0.0 0.0 0.014527999795973301 0.054687999188899994 0.021439999900758268 0.01539199985563755 0.012052764849597775 0.0 0.0 0.0 0.0 0.0 0.13488000631332397 0.1528639942407608 0.1431359991431236 0.1431359991431236 0.005436271464033599 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 64 0 1024.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
24 0.0 0.0 0.0 0.0 0.0 0.014399999752640724 0.041919998824596405 0.01899839974939823 0.015343999955803156 0.007989843526623287 0.0 0.0 0.0 0.0 0.0 0.06176000088453293 0.08374399691820145 0.06747519969940186 0.06747519969940186 0.0066067747128778 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 8 0 2048.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
25 0.0 0.0 0.0 0.0 0.0 0.016256000846624374 0.11353600025177002 0.042761600017547606 0.028960000723600388 0.029104301538020762 0.0 0.0 0.0 0.0 0.0 0.09734400361776352 0.14422400295734406 0.11392960175871848 0.11392960175871848 0.013198594600417867 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 16 0 2048.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
26 0.0 0.0 0.0 0.0 0.0 0.014688000082969666 0.034143999218940735 0.018918400071561335 0.01643200032413006 0.005500943993080684 0.0 0.0 0.0 0.0 0.0 0.12918399274349213 0.15087999403476715 0.13807999789714814 0.13807999789714814 0.007658330538677587 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 32 0 2048.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
27 0.0 0.0 0.0 0.0 0.0 0.016063999384641647 0.03641600161790848 0.0198208000510931 0.01780799962580204 0.0057264128169845765 0.0 0.0 0.0 0.0 0.0 0.22099199891090393 0.23904000222682953 0.2293503984808922 0.2293503984808922 0.004861342907006028 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 64 0 2048.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
28 0.0 0.0 0.0 0.0 0.0 0.014751999638974667 0.035840000957250595 0.018908800091594458 0.015792000107467175 0.0064374817924757475 0.0 0.0 0.0 0.0 0.0 0.10134399682283401 0.12201599776744843 0.10896319895982742 0.10896319895982742 0.006336330809165179 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 8 0 4096.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
29 0.0 0.0 0.0 0.0 0.0 0.013856000266969204 0.03417599946260452 0.017846399918198586 0.014800000004470348 0.006495007539635255 0.0 0.0 0.0 0.0 0.0 0.13126400113105774 0.15561600029468536 0.1389280006289482 0.1389280006289482 0.008381472811075022 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 16 0 4096.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
30 0.0 0.0 0.0 0.0 0.0 0.014431999996304512 0.03519999980926514 0.019168000388890504 0.01600000075995922 0.005995522477654695 0.0 0.0 0.0 0.0 0.0 0.21728000044822693 0.2343679964542389 0.2231455981731415 0.2231455981731415 0.004720730646739123 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 32 0 4096.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
31 0.0 0.0 0.0 0.0 0.0 0.014047999866306782 0.03670400008559227 0.018441599886864425 0.015023999847471714 0.006596535162793127 0.0 0.0 0.0 0.0 0.0 0.39321601390838623 0.4524799883365631 0.4058080047369003 0.4058080047369003 0.01578349755088941 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 64 0 4096.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
32 0.0 0.0 0.0 0.0 0.0 0.014336000196635723 0.022143999114632607 0.016912000067532063 0.015168000012636185 0.0028156089295136347 0.0 0.0 0.0 0.0 0.0 0.15587200224399567 0.3079040050506592 0.17838079929351805 0.17838079929351805 0.04355865575265927 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 8 0 8192.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
33 0.0 0.0 0.0 0.0 0.0 0.014271999709308147 0.02112000063061714 0.015516800060868263 0.01488000014796853 0.001940870731593904 0.0 0.0 0.0 0.0 0.0 0.21587200462818146 0.23561599850654602 0.22250880002975468 0.22250880002975468 0.006181951170646666 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 16 0 8192.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
34 0.0 0.0 0.0 0.0 0.0 0.015552000142633915 0.039264000952243805 0.023609600123018028 0.02131200022995472 0.007236979625711548 0.0 0.0 0.0 0.0 0.0 0.408735990524292 0.470335990190506 0.4336863994598388 0.4336863994598388 0.01844662383160074 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 32 0 8192.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
35 0.0 0.0 0.0 0.0 0.0 0.014399999752640724 0.025407999753952026 0.016227199975401164 0.014960000291466713 0.0031617375441736185 0.0 0.0 0.0 0.0 0.0 0.7412800192832947 0.7627840042114258 0.7464000046253203 0.7464000046253203 0.006112167448837547 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 64 0 8192.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
36 0.0 0.0 0.0 0.0 0.0 0.01462399959564209 0.02812799997627735 0.020652799773961304 0.021359999664127827 0.004308706957613102 0.028383498565450627 0.039859687970646644 0.032492258074592426 0.032492258074592426 0.00453597266208597 0.029312501176103633 0.04116431058787463 0.03355574193327539 0.03355574193327539 0.004684436757701648 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 9 0 0 True FLASH_ATTN False true_mixed_fused_projected 72 False 64 BF16 generic none CUDA_EVENT True [64] [0] [512, 512, 512, 512, 512, 512, 512, 512] 1 8 8 9 8 512.0 0.1111111111111111 q64_8q1s512 fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
37 0.0 0.0 0.0 0.0 0.0 0.015552000142633915 0.034143999218940735 0.023171199765056372 0.024255999363958836 0.005908614918096971 0.03333159243114438 0.038935341782478095 0.03580428402241854 0.03580428402241854 0.002082270297044095 0.03633240903369937 0.04244065945634484 0.03902771507087562 0.03902771507087562 0.0022697354261490147 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 9 0 0 True FLASH_ATTN False true_mixed_fused_projected 136 False 128 BF16 generic none CUDA_EVENT True [128] [0] [1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024] 1 8 8 9 8 1024.0 0.1111111111111111 q128_8q1s1k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
38 0.0 0.0 0.0 0.0 0.0 0.014655999839305878 0.02611199952661991 0.019635199941694735 0.018112000077962875 0.0038298050749257795 0.04189529417991216 0.057484239920526384 0.04744885718421094 0.04744885718421094 0.004779830748455743 0.051672703037266184 0.07089975418195164 0.05852234134479412 0.05852234134479412 0.005895334539786378 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 144 False 128 BF16 generic none CUDA_EVENT True [128] [0] [1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024] 1 16 16 17 16 1024.0 0.058823529411764705 q128_16q1s1k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
39 0.0 0.0 0.0 0.0 0.0 0.014720000326633453 0.029311999678611755 0.018662399891763926 0.01673599984496832 0.0044017112162725355 0.04322973959325901 0.05049827064705393 0.04535414343408448 0.04535414343408448 0.0022944156000240697 0.08733025617719538 0.10201372836398578 0.09162185574241775 0.09162185574241775 0.004635047631846023 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 272 False 256 BF16 generic none CUDA_EVENT True [256] [0] [2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048] 1 16 16 17 16 2048.0 0.058823529411764705 q256_16q1s2k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
40 0.0 0.0 0.0 0.0 0.0 0.014592000283300877 0.026367999613285065 0.016336000058799982 0.01515199989080429 0.003415353455946402 0.06031842775160765 0.06618323188375198 0.06274415549817247 0.06274415549817247 0.001985647289644255 0.14768157653992678 0.16204076249052324 0.15362064543185072 0.15362064543185072 0.004861590945216247 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 288 False 256 BF16 generic none CUDA_EVENT True [256] [0] [2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048] 1 32 32 33 32 2048.0 0.030303030303030304 q256_32q1s2k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
41 0.0 0.0 0.0 0.0 0.0 0.014751999638974667 0.02454400062561035 0.017167999967932702 0.016191999427974224 0.0028685016454498436 0.09128700688359712 0.09689150775996329 0.09360396051475776 0.09360396051475776 0.0013477542744916764 0.2740889887523892 0.2909164873209846 0.2810456356995366 0.2810456356995366 0.004046628526808561 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 544 False 512 BF16 generic none CUDA_EVENT True [512] [0] [4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096] 1 32 32 33 32 4096.0 0.030303030303030304 q512_32q1s4k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
42 0.0 0.0 0.0 0.0 0.0 0.01881599985063076 0.03097599931061268 0.024598400108516216 0.024848000146448612 0.0037539437391565975 0.1035249255866932 0.10603132147437412 0.10399797220840973 0.10399797220840973 0.0007025109429056814 0.5652750707893444 0.5789606938874117 0.5678580377517648 0.5678580377517648 0.003835906384194897 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 65 0 0 True FLASH_ATTN False true_mixed_fused_projected 576 False 512 BF16 generic none CUDA_EVENT True [512] [0] [4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096] 1 64 64 65 64 4096.0 0.015384615384615385 q512_64q1s4k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
43 0.0 0.0 0.0 0.0 0.0 0.018624000251293182 0.043487999588251114 0.024460799992084503 0.019600000232458115 0.008922081629781394 0.196169204945307 0.2076140047945799 0.2008444429250931 0.2008444429250931 0.00394350953292805 1.1196708009293268 1.1849940417370974 1.1463555573610091 1.1463555573610091 0.022508285530529783 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 65 0 0 True FLASH_ATTN False true_mixed_fused_projected 1088 False 1024 BF16 generic none CUDA_EVENT True [1024] [0] [8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192] 1 64 64 65 64 8192.0 0.015384615384615385 q1k_64q1s8k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
44 0.0 0.0 0.0 0.0 0.0 0.027135999873280525 0.03667199984192848 0.02945920005440712 0.028256000019609928 0.0029446678408169553 0.37757279619664613 0.3898113624476372 0.38075520430942184 0.38075520430942184 0.0036600203831042254 0.2522831942132049 0.2604606493092597 0.25440958703617433 0.25440958703617433 0.0024455194930253126 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 2080 False 2048 BF16 generic none CUDA_EVENT True [2048] [0] [4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096] 1 32 32 33 32 4096.0 0.030303030303030304 q2k_32q1s4k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
45 0.0 0.0 0.0 0.0 0.0 0.0435199998319149 0.049695998430252075 0.04502719938755036 0.04395199939608574 0.002035785660169308 1.1048984388245497 1.1189053886476108 1.1112371236754128 1.1112371236754128 0.0050220696724624985 0.1395495077239122 0.14131859607071193 0.14035008841031085 0.14035008841031085 0.0006342911944856293 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 4112 False 4096 BF16 generic none CUDA_EVENT True [4096] [0] [4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096] 1 16 16 17 16 4096.0 0.058823529411764705 q4k_16q1s4k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
46 0.0 0.0 0.0 0.0 0.0 0.018783999606966972 0.024288000538945198 0.020627199858427047 0.019567999988794327 0.0019457052717059358 0.09455999732017517 0.12185599654912949 0.10207359939813614 0.10207359939813614 0.00753346544014261 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 2 1024 0.0 True FLASH_ATTN False vllm020_batch_spec [512, 512] 1024 512 512 512.0 True 0.0 0.0 0.0 False 0.0 1024.0 1024 BF16 generic none CUDA_EVENT False 2q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
47 0.0 0.0 0.0 0.0 0.0 0.027295999228954315 0.034752000123262405 0.02943360023200512 0.028528000228106976 0.0023159160681123767 0.14416000247001648 0.15904000401496887 0.14979200065135959 0.14979200065135959 0.00480624675866005 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 4 2048 0.0 True FLASH_ATTN False vllm020_batch_spec [512, 512, 512, 512] 2048 512 512 512.0 True 0.0 0.0 0.0 False 0.0 2048.0 2048 BF16 generic none CUDA_EVENT False 4q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
48 0.0 0.0 0.0 0.0 0.0 0.043455999344587326 0.05084799975156784 0.045500800386071204 0.04411200061440468 0.002490556062831049 0.24454399943351746 0.25865599513053894 0.2504959970712662 0.2504959970712662 0.003778494140301353 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 8 4096 0.0 True FLASH_ATTN False vllm020_batch_spec [512, 512, 512, 512, 512, 512, 512, 512] 4096 512 512 512.0 True 0.0 0.0 0.0 False 0.0 4096.0 4096 BF16 generic none CUDA_EVENT False 8q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
49 0.0 0.0 0.0 0.0 0.0 0.07673600316047668 0.08374399691820145 0.07970559895038605 0.07873599976301193 0.0022907976135004057 0.445248007774353 0.4758400022983551 0.45409599840641024 0.45409599840641024 0.009133230340383306 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 16 8192 0.0 True FLASH_ATTN False vllm020_batch_spec [512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512] 8192 512 512 512.0 True 0.0 0.0 0.0 False 0.0 8192.0 8192 BF16 generic none CUDA_EVENT False 16q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
50 0.0 0.0 0.0 0.0 0.0 0.043296001851558685 0.051072001457214355 0.044972800090909 0.04399999976158142 0.002451216824441962 0.6147199869155884 0.6290879845619202 0.6204927921295166 0.6204927921295166 0.004344846739788608 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 2 4096 0.0 True FLASH_ATTN False vllm020_batch_spec [2048, 2048] 4096 2048 2048 2048.0 True 0.0 0.0 0.0 False 0.0 4096.0 4096 BF16 generic none CUDA_EVENT False 2q2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
51 0.0 0.0 0.0 0.0 0.0 0.07737600058317184 0.09123200178146362 0.08094720020890236 0.07980800047516823 0.004065264798368611 1.1744320392608643 1.1887680292129517 1.178323209285736 1.178323209285736 0.004395876469319665 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 4 8192 0.0 True FLASH_ATTN False vllm020_batch_spec [2048, 2048, 2048, 2048] 8192 2048 2048 2048.0 True 0.0 0.0 0.0 False 0.0 8192.0 8192 BF16 generic none CUDA_EVENT False 4q2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
52 0.0 0.0 0.0 0.0 0.0 0.014688000082969666 0.03254399821162224 0.01810879958793521 0.01611199975013733 0.005058478889747962 0.0 0.0 0.0 0.0 0.0 0.06339199841022491 0.0841279998421669 0.07019519805908202 0.07019519805908202 0.006070126182471763 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1] 1 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
53 0.0 0.0 0.0 0.0 0.0 0.01484800036996603 0.030271999537944794 0.018124799989163876 0.015711999498307705 0.004696400814632773 0.0 0.0 0.0 0.0 0.0 0.2739199995994568 0.2922239899635315 0.28281279802322384 0.28281279802322384 0.00633037978599564 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 8 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
54 0.0 0.0 0.0 0.0 0.0 0.014495999552309513 0.022431999444961548 0.016844799742102623 0.015471999999135733 0.0027624803153841917 0.0 0.0 0.0 0.0 0.0 0.3909119963645935 0.4160960018634796 0.39996159672737125 0.39996159672737125 0.007037174636804898 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 16 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
55 0.0 0.0 0.0 0.0 0.0 0.014240000396966934 0.035071998834609985 0.018764800019562246 0.015583999920636415 0.006305594700523817 0.0 0.0 0.0 0.0 0.0 0.7383679747581482 0.7597119808197021 0.7430047929286957 0.7430047929286957 0.006192645960911466 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 32 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
56 0.0 0.0 0.0 0.0 0.0 0.014399999752640724 0.028991999104619026 0.019401599932461978 0.01665600063279271 0.005434953793780546 0.0 0.0 0.0 0.0 0.0 1.427008032798767 1.4517120122909546 1.4361984014511109 1.4361984014511109 0.008802755821028187 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 64 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
57 0.0 0.0 0.0 0.0 0.0 0.014271999709308147 0.02844800055027008 0.01820160010829568 0.014944000169634819 0.004942002036909087 0.0 0.0 0.0 0.0 0.0 0.08361600339412689 0.11027199774980545 0.09058240056037903 0.09058240056037903 0.00872938604423622 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1] 1 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
58 0.0 0.0 0.0 0.0 0.0 0.014112000353634357 0.023104000836610794 0.01648960020393133 0.015632000286132097 0.0027628828340349578 0.0 0.0 0.0 0.0 0.0 0.5063040256500244 0.5497599840164185 0.5168287932872773 0.5168287932872773 0.013557229606313293 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 8 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
59 0.0 0.0 0.0 0.0 0.0 0.01616000011563301 0.029055999591946602 0.018780799955129622 0.017215999774634838 0.003794434947431941 0.0 0.0 0.0 0.0 0.0 0.7439360022544861 0.7719680070877075 0.7502080142498017 0.7502080142498017 0.00839205598262567 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 16 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
60 0.0 0.0 0.0 0.0 0.0 0.015968000516295433 0.03046399913728237 0.018822400271892546 0.01726400014013052 0.004212536831927695 0.0 0.0 0.0 0.0 0.0 1.4256000518798828 1.449504017829895 1.4325888037681578 1.4325888037681578 0.0065448028108711165 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 32 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
61 0.0 0.0 0.0 0.0 0.0 0.014271999709308147 0.027712000533938408 0.01633920017629862 0.014944000169634819 0.0038488220071983326 0.0 0.0 0.0 0.0 0.0 2.798719882965088 3.0184640884399414 2.8293471813201903 2.8293471813201903 0.06375662767704417 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 64 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
62 0.0 0.0 0.0 0.0 0.0 0.014720000326633453 0.021407999098300934 0.01637439979240298 0.014992000069469213 0.0024463594774459265 0.0 0.0 0.0 0.0 0.0 0.09347199648618698 0.10473600029945374 0.09820479974150656 0.09820479974150656 0.004061668684999473 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1] 1 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
63 0.0 0.0 0.0 0.0 0.0 0.014303999952971935 0.04790399968624115 0.020179199893027543 0.016447999514639378 0.009635010283506967 0.0 0.0 0.0 0.0 0.0 0.6221439838409424 0.6444799900054932 0.6277984082698822 0.6277984082698822 0.007234140179397069 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 8 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
64 0.0 0.0 0.0 0.0 0.0 0.014271999709308147 0.03276799991726875 0.01921279989182949 0.015232000034302473 0.0065506148511265076 0.0 0.0 0.0 0.0 0.0 0.9111359715461731 0.9307839870452881 0.9151648044586183 0.9151648044586183 0.005528712062927265 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 16 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
65 0.0 0.0 0.0 0.0 0.0 0.014175999909639359 0.02707199938595295 0.016512000095099212 0.014688000082969666 0.0038989080209321414 0.0 0.0 0.0 0.0 0.0 1.7645119428634644 1.7849279642105103 1.7684095859527589 1.7684095859527589 0.005876963340184193 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 32 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
66 0.0 0.0 0.0 0.0 0.0 0.014240000396966934 0.029184000566601753 0.016825600154697896 0.014752000104635954 0.00457910514148248 0.0 0.0 0.0 0.0 0.0 3.4781761169433594 3.5388801097869873 3.490892815589905 3.490892815589905 0.01818910600692522 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 64 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
67 0.0 0.0 0.0 0.0 0.0 0.014976000413298607 0.03139200061559677 0.01898880014196038 0.016848000697791576 0.005083715524393418 0.13752702814163098 0.14117629917511842 0.13848196486571557 0.13848196486571557 0.0009880564964713527 0.523336967410432 0.5372236808930884 0.5269708253945994 0.5269708253945994 0.00375988994658546 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 9 0 0 True FLASH_ATTN False true_mixed_fused_projected 520 False 512 BF16 generic none CUDA_EVENT True [512] [0] [16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384] 1 8 8 9 8 16384.0 0.1111111111111111 q512_8q1s16k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
68 0.0 0.0 0.0 0.0 0.0 0.01833599992096424 0.02707199938595295 0.020995199866592883 0.01976000051945448 0.0028392656352050185 0.17817885890237703 0.18622915300900206 0.18109068484526028 0.18109068484526028 0.0027040006080457624 0.5449571488834201 0.5695788427872015 0.5538629212834322 0.5538629212834322 0.008270141985514729 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 1040 False 1024 BF16 generic none CUDA_EVENT True [1024] [0] [16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384] 1 16 16 17 16 16384.0 0.058823529411764705 q1k_16q1s16k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
69 0.0 0.0 0.0 0.0 0.0 0.02703999914228916 0.03308799862861633 0.028883199393749236 0.02759999968111515 0.002285758578932753 0.4688266550410815 0.48301665772804186 0.4727750380198454 0.4727750380198454 0.004396396389258256 1.0430453981052807 1.0746153117238624 1.051829759343436 1.051829759343436 0.009781101336187197 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 2080 False 2048 BF16 generic none CUDA_EVENT True [2048] [0] [16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384] 1 32 32 33 32 16384.0 0.030303030303030304 q2k_32q1s16k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
70 0.0 0.0 0.0 0.0 0.0 0.04368000105023384 0.059328000992536545 0.04809600040316582 0.04617599956691265 0.005158792915002645 1.3509461459747514 1.3759066409666496 1.3589365122072 1.3589365122072 0.008008877716183384 0.9213738861449998 0.9383974724214122 0.9268234851606594 0.9268234851606594 0.005462224239661061 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 4112 False 4096 BF16 generic none CUDA_EVENT True [4096] [0] [32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768] 1 16 16 17 16 32768.0 0.058823529411764705 q4k_16q1s32k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
71 0.0 0.0 0.0 0.0 0.0 0.026944000273942947 0.040511999279260635 0.030291200056672095 0.02817599941045046 0.004386386489305873 0.5071830964059985 0.5142792136001758 0.5096392092471257 0.5096392092471257 0.0022059272685869546 2.175632932188972 2.2060727208328075 2.186168772243963 2.186168772243963 0.009462633998570079 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 2080 False 2048 BF16 generic none CUDA_EVENT True [2048] [0] [32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768] 1 32 32 33 32 32768.0 0.030303030303030304 q2k_32q1s32k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
72 0.0 0.0 0.0 0.0 0.0 0.018303999677300453 0.03017600066959858 0.02039040010422468 0.018943999893963337 0.0034612051983613614 0.21341429693945616 0.21552776300295579 0.21393496609766816 0.21393496609766816 0.0005736773262009222 4.617401495144284 4.663128147488988 4.628666619290515 4.628666619290515 0.012412001359412127 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 65 0 0 True FLASH_ATTN False true_mixed_fused_projected 1088 False 1024 BF16 generic none CUDA_EVENT True [1024] [0] [32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768] 1 64 64 65 64 32768.0 0.015384615384615385 q1k_64q1s32k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
73 0.0 0.0 0.0 0.0 0.0 0.014720000326633453 0.02364799939095974 0.01809599995613098 0.016352000646293163 0.0035481127058959038 0.04822399839758873 0.08566399663686752 0.05961279980838299 0.05961279980838299 0.011445665413968877 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 64 0.0 True FLASH_ATTN False vllm020_batch_spec [64] 64 64 64 64.0 True 0.0 0.0 0.0 False 0.0 64.0 64 BF16 generic none CUDA_EVENT False q64 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
74 0.0 0.0 0.0 0.0 0.0 0.014175999909639359 0.020864000543951988 0.01668160008266568 0.015664000064134598 0.0025769063833097584 0.049695998430252075 0.08057600259780884 0.05882879942655563 0.05882879942655563 0.009515126108519331 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 128 0.0 True FLASH_ATTN False vllm020_batch_spec [128] 128 128 128 128.0 True 0.0 0.0 0.0 False 0.0 128.0 128 BF16 generic none CUDA_EVENT False q128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
75 0.0 0.0 0.0 0.0 0.0 0.014399999752640724 0.028704000636935234 0.017430400010198355 0.015056000091135502 0.004349294613335555 0.049855999648571014 0.07366400212049484 0.05459520071744919 0.05459520071744919 0.0069610920757925574 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 256 0.0 True FLASH_ATTN False vllm020_batch_spec [256] 256 256 256 256.0 True 0.0 0.0 0.0 False 0.0 256.0 256 BF16 generic none CUDA_EVENT False q256 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
76 0.0 0.0 0.0 0.0 0.0 0.014336000196635723 0.03161599859595299 0.017788799852132796 0.015711999963968992 0.005005385723318659 0.06102399900555611 0.08179199695587158 0.06650560013949873 0.06650560013949873 0.006947995595801105 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 512 0.0 True FLASH_ATTN False vllm020_batch_spec [512] 512 512 512 512.0 True 0.0 0.0 0.0 False 0.0 512.0 512 BF16 generic none CUDA_EVENT False q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
77 0.0 0.0 0.0 0.0 0.0 0.014655999839305878 0.04416000097990036 0.019200000166893005 0.015488000120967627 0.008561241323364038 0.08054400235414505 0.09196799993515015 0.08607039973139763 0.08607039973139763 0.004145329035483754 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 1024 0.0 True FLASH_ATTN False vllm020_batch_spec [1024] 1024 1024 1024 1024.0 True 0.0 0.0 0.0 False 0.0 1024.0 1024 BF16 generic none CUDA_EVENT False q1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
78 0.0 0.0 0.0 0.0 0.0 0.017855999991297722 0.023391999304294586 0.019667199812829494 0.018463999964296818 0.0022577669687832585 0.18729600310325623 0.20585599541664124 0.19546559900045393 0.19546559900045393 0.006339068824303663 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 2048 0.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 False 0.0 2048.0 2048 BF16 generic none CUDA_EVENT False q2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
79 0.0 0.0 0.0 0.0 0.0 0.026240000501275063 0.03446400165557861 0.028297600522637367 0.027088000439107418 0.0025148441522922374 0.5754240155220032 0.5889919996261597 0.5800191938877105 0.5800191938877105 0.003858869273829596 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 4096 0.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 False 0.0 4096.0 4096 BF16 generic none CUDA_EVENT False q4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
80 0.0 0.0 0.0 0.0 0.0 0.04182400181889534 0.047200001776218414 0.043036799877882004 0.0423360001295805 0.0017073405772076728 2.063199996948242 2.0787200927734375 2.067151999473572 2.067151999473572 0.004959651271127835 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 8192 0.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 False 0.0 8192.0 8192 BF16 generic none CUDA_EVENT False q8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
81 0.0 0.0 0.0 0.0 0.0 0.014399999752640724 0.05648000165820122 0.02270399993285537 0.017935999669134617 0.012377915150727689 0.049536000937223434 0.07196799665689468 0.056015999615192415 0.056015999615192415 0.0070476637552742884 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 64 448.0 True FLASH_ATTN False vllm020_batch_spec [64] 64 64 64 64.0 True 0.0 0.0 0.0 True 448.0 512.0 64 BF16 generic none CUDA_EVENT False q64s512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
82 0.0 0.0 0.0 0.0 0.0 0.01408000010997057 0.021344000473618507 0.01611520005390048 0.014928000047802925 0.0025499884993961702 0.07072000205516815 0.2642880082130432 0.1588256008923054 0.1588256008923054 0.053220347086102376 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 128 896.0 True FLASH_ATTN False vllm020_batch_spec [128] 128 128 128 128.0 True 0.0 0.0 0.0 True 896.0 1024.0 128 BF16 generic none CUDA_EVENT False q128s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
83 0.0 0.0 0.0 0.0 0.0 0.01360000018030405 0.03014400042593479 0.016975999902933837 0.015056000091135502 0.004715159697364111 0.08393599838018417 0.11036799848079681 0.09160000011324881 0.09160000011324881 0.007911669434472792 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 256 1792.0 True FLASH_ATTN False vllm020_batch_spec [256] 256 256 256 256.0 True 0.0 0.0 0.0 True 1792.0 2048.0 256 BF16 generic none CUDA_EVENT False q256s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
84 0.0 0.0 0.0 0.0 0.0 0.014495999552309513 0.020767999812960625 0.016233599931001663 0.015392000321298838 0.002202615907995427 0.1998399943113327 0.22070400416851044 0.20855360180139543 0.20855360180139543 0.00689307200230667 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 512 3584.0 True FLASH_ATTN False vllm020_batch_spec [512] 512 512 512 512.0 True 0.0 0.0 0.0 True 3584.0 4096.0 512 BF16 generic none CUDA_EVENT False q512s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
85 0.0 0.0 0.0 0.0 0.0 0.01484800036996603 0.033440001308918 0.018684800155460833 0.016048000194132328 0.00553991005639174 0.6043199896812439 0.635807991027832 0.6126143991947173 0.6126143991947173 0.008745933953408096 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 1024 7168.0 True FLASH_ATTN False vllm020_batch_spec [1024] 1024 1024 1024 1024.0 True 0.0 0.0 0.0 True 7168.0 8192.0 1024 BF16 generic none CUDA_EVENT False q1ks8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
86 0.0 0.0 0.0 0.0 0.0 0.013824000023305416 0.03359999880194664 0.017811199743300678 0.014944000169634819 0.005926140483565472 0.0 0.0 0.0 0.0 0.0 0.045471999794244766 0.07539200037717819 0.054758400097489356 0.054758400097489356 0.010253548506101549 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1] 1 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
87 0.0 0.0 0.0 0.0 0.0 0.014047999866306782 0.02409599907696247 0.01641279999166727 0.01473599998280406 0.003347158638713987 0.0 0.0 0.0 0.0 0.0 0.0461760014295578 0.07529599964618683 0.05460800044238568 0.05460800044238568 0.009748937798340135 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 8 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
88 0.0 0.0 0.0 0.0 0.0 0.014271999709308147 0.028416000306606293 0.018611199874430894 0.01598400017246604 0.005209875093863647 0.0 0.0 0.0 0.0 0.0 0.048128001391887665 0.07897599786520004 0.061353600397706036 0.061353600397706036 0.010488153655157845 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 16 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
89 0.0 0.0 0.0 0.0 0.0 0.014112000353634357 0.025728000327944756 0.016092800162732603 0.01512000011280179 0.003300888123052605 0.0 0.0 0.0 0.0 0.0 0.04864000156521797 0.07648000121116638 0.05810560062527656 0.05810560062527656 0.009473544218404408 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 32 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
90 0.0 0.0 0.0 0.0 0.0 0.014655999839305878 0.03551999852061272 0.018588799890130757 0.016064000315964222 0.006071057437992447 0.0 0.0 0.0 0.0 0.0 0.04822399839758873 0.07862400263547897 0.05660480037331582 0.05660480037331582 0.009565394213730401 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 64 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
91 0.0 0.0 0.0 0.0 0.0 0.014303999952971935 0.028831999748945236 0.01699519995599985 0.015024000313133001 0.004285000744868188 0.0 0.0 0.0 0.0 0.0 0.04854400083422661 0.06719999760389328 0.05778240002691746 0.05778240002691746 0.006852125554679805 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 8 0 1024.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
92 0.0 0.0 0.0 0.0 0.0 0.0144640002399683 0.024927999824285507 0.0161183999851346 0.01521599991247058 0.0029774808524673907 0.0 0.0 0.0 0.0 0.0 0.05379199981689453 0.08966399729251862 0.06270079985260964 0.06270079985260964 0.009983591277092696 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 16 0 1024.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
93 0.0 0.0 0.0 0.0 0.0 0.014431999996304512 0.03001599945127964 0.017648000083863736 0.01547200046479702 0.004585126309484848 0.0 0.0 0.0 0.0 0.0 0.061664000153541565 0.07692799717187881 0.06704320013523103 0.06704320013523103 0.005500191798490629 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 32 0 1024.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
94 0.0 0.0 0.0 0.0 0.0 0.014175999909639359 0.026367999613285065 0.016947199776768684 0.015039999969303608 0.0037951566103550205 0.0 0.0 0.0 0.0 0.0 0.08799999952316284 0.111455999314785 0.0964031994342804 0.0964031994342804 0.007397541615558088 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 64 0 1024.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
95 0.0 0.0 0.0 0.0 0.0 0.01369599997997284 0.021247999742627144 0.015359999984502793 0.014640000183135271 0.0020942770950814317 0.0 0.0 0.0 0.0 0.0 0.051711998879909515 0.07065600156784058 0.058054400235414506 0.058054400235414506 0.006633034815910025 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 8 0 2048.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
96 0.0 0.0 0.0 0.0 0.0 0.014208000153303146 0.0307839997112751 0.017148799914866685 0.015039999969303608 0.004882374686312284 0.0 0.0 0.0 0.0 0.0 0.061919998377561576 0.07843200117349625 0.06628479920327664 0.06628479920327664 0.004962852689801192 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 16 0 2048.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
97 0.0 0.0 0.0 0.0 0.0 0.013887999579310417 0.02969600073993206 0.017500799987465142 0.015232000034302473 0.00467273319705314 0.0 0.0 0.0 0.0 0.0 0.08819200098514557 0.11097600311040878 0.09493440166115762 0.09493440166115762 0.007509235042985577 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 32 0 2048.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
98 0.0 0.0 0.0 0.0 0.0 0.014399999752640724 0.022112000733613968 0.01751680001616478 0.016047999262809753 0.0030306621792915785 0.0 0.0 0.0 0.0 0.0 0.13065600395202637 0.15110400319099426 0.13857279866933822 0.13857279866933822 0.00750137841249771 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 64 0 2048.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
99 0.0 0.0 0.0 0.0 0.0 0.013919999822974205 0.04012800008058548 0.01892479993402958 0.01550400024279952 0.007459545267816961 0.0 0.0 0.0 0.0 0.0 0.06278400123119354 0.08259200304746628 0.0704512007534504 0.0704512007534504 0.005979055984382744 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 8 0 4096.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
100 0.0 0.0 0.0 0.0 0.0 0.014336000196635723 0.02236800082027912 0.016332800220698118 0.014928000047802925 0.002784044363186731 0.0 0.0 0.0 0.0 0.0 0.1003199964761734 0.1279360055923462 0.10921279862523078 0.10921279862523078 0.00862769617046716 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 16 0 4096.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
101 0.0 0.0 0.0 0.0 0.0 0.013919999822974205 0.0225600004196167 0.016128000058233737 0.014800000004470348 0.002958953332547708 0.0 0.0 0.0 0.0 0.0 0.13116799294948578 0.14812800288200378 0.13783999979496003 0.13783999979496003 0.005696384361148053 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 32 0 4096.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
102 0.0 0.0 0.0 0.0 0.0 0.014208000153303146 0.029983999207615852 0.017468800116330386 0.01508800033479929 0.0047379550962483065 0.0 0.0 0.0 0.0 0.0 0.217631995677948 0.2447360008955002 0.22715839892625808 0.22715839892625808 0.008831828221847138 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 64 0 4096.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
103 0.0 0.0 0.0 0.0 0.0 0.014047999866306782 0.02304000034928322 0.016512000095099212 0.014512000139802694 0.0035026774828624254 0.0 0.0 0.0 0.0 0.0 0.11020799726247787 0.12307199835777283 0.11600959971547126 0.11600959971547126 0.004667637266950902 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 8 0 8192.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
104 0.0 0.0 0.0 0.0 0.0 0.014112000353634357 0.042080000042915344 0.017510399967432023 0.014607999939471483 0.00823117883530167 0.0 0.0 0.0 0.0 0.0 0.15702399611473083 0.17587199807167053 0.16399359852075576 0.16399359852075576 0.006588299074676393 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 16 0 8192.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
105 0.0 0.0 0.0 0.0 0.0 0.013919999822974205 0.0208320003002882 0.015299199987202883 0.01462399959564209 0.001916768086505386 0.0 0.0 0.0 0.0 0.0 0.21779200434684753 0.2415360063314438 0.2260768011212349 0.2260768011212349 0.007251352080685236 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 32 0 8192.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
106 0.0 0.0 0.0 0.0 0.0 0.014015999622642994 0.021088000386953354 0.015619200188666582 0.01473599998280406 0.002013227452302141 0.0 0.0 0.0 0.0 0.0 0.3959999978542328 0.4152640104293823 0.40332479774951924 0.40332479774951924 0.006942401431914052 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 64 0 8192.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
107 0.0 0.0 0.0 0.0 0.0 0.014175999909639359 0.03379200026392937 0.020595200080424547 0.019600000232458115 0.006548754248881344 0.026192623739694512 0.040006882507168426 0.029155180178492036 0.029155180178492036 0.00408828748438241 0.024591375524545756 0.037561119537986146 0.027372820355088746 0.027372820355088746 0.0038383559348575944 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 9 0 0 True FLASH_ATTN False true_mixed_fused_projected 72 False 64 BF16 generic none CUDA_EVENT True [64] [0] [512, 512, 512, 512, 512, 512, 512, 512] 1 8 8 9 8 512.0 0.1111111111111111 q64_8q1s512 fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
108 0.0 0.0 0.0 0.0 0.0 0.013856000266969204 0.020896000787615776 0.015318400040268899 0.014431999996304512 0.0019640312960926966 0.029349018208693862 0.06236262941356679 0.03714313592014963 0.03714313592014963 0.009618313791872569 0.028826981462526914 0.06125337308649041 0.036482463672774496 0.036482463672774496 0.009447230957011863 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 9 0 0 True FLASH_ATTN False true_mixed_fused_projected 136 False 128 BF16 generic none CUDA_EVENT True [128] [0] [1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024] 1 8 8 9 8 1024.0 0.1111111111111111 q128_8q1s1k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
109 0.0 0.0 0.0 0.0 0.0 0.013856000266969204 0.032127998769283295 0.017286399938166143 0.014479999896138906 0.005536495446636193 0.03273085874558354 0.04394578491937082 0.03563837490653034 0.03563837490653034 0.003429601730256567 0.03488514202593899 0.04683821345079978 0.037984025407085474 0.037984025407085474 0.0036553316361902684 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 144 False 128 BF16 generic none CUDA_EVENT True [128] [0] [1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024] 1 16 16 17 16 1024.0 0.058823529411764705 q128_16q1s1k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
110 0.0 0.0 0.0 0.0 0.0 0.013856000266969204 0.021503999829292297 0.015078400075435639 0.01425600005313754 0.002238435975478849 0.042100813549974185 0.051784144690147756 0.045378692890289625 0.045378692890289625 0.003184109940650555 0.05111518843748309 0.06287185663927425 0.05509490773570219 0.05509490773570219 0.0038658725544299132 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 272 False 256 BF16 generic none CUDA_EVENT True [256] [0] [2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048] 1 16 16 17 16 2048.0 0.058823529411764705 q256_16q1s2k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
111 0.0 0.0 0.0 0.0 0.0 0.014399999752640724 0.02687999978661537 0.017667199857532977 0.01566399959847331 0.003864811846387173 0.048475323773821306 0.05599956978723733 0.05123499252968345 0.05123499252968345 0.002427379820423941 0.08429268135885387 0.09737642835214408 0.0890914090616175 0.0890914090616175 0.0042209177332077005 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 288 False 256 BF16 generic none CUDA_EVENT True [256] [0] [2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048] 1 32 32 33 32 2048.0 0.030303030303030304 q256_32q1s2k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
112 0.0 0.0 0.0 0.0 0.0 0.01462399959564209 0.03359999880194664 0.019804799742996693 0.016032000072300434 0.006750519908145474 0.07121508474579985 0.07292308367136396 0.07194410845270183 0.07194410845270183 0.0005500065915943844 0.14760091249712767 0.15114092373010238 0.14911189243564577 0.14911189243564577 0.0011399477384397005 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 544 False 512 BF16 generic none CUDA_EVENT True [512] [0] [4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096] 1 32 32 33 32 4096.0 0.030303030303030304 q512_32q1s4k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
113 0.0 0.0 0.0 0.0 0.0 0.014399999752640724 0.021856000646948814 0.016835200227797033 0.015263999812304974 0.00283854448975065 0.08146338272142935 0.08560865714384096 0.0834932625520734 0.0834932625520734 0.0012259635905347874 0.2782486219401307 0.29240733788179385 0.2851819365989658 0.2851819365989658 0.004187435731481665 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 65 0 0 True FLASH_ATTN False true_mixed_fused_projected 576 False 512 BF16 generic none CUDA_EVENT True [512] [0] [4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096] 1 64 64 65 64 4096.0 0.015384615384615385 q512_64q1s4k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
114 0.0 0.0 0.0 0.0 0.0 0.014399999752640724 0.04495999962091446 0.02143679987639189 0.015856000129133463 0.009709987872683342 0.1194459208702178 0.12302524755181463 0.12053885718696096 0.12053885718696096 0.001005118119341339 0.5597220650459199 0.5764947443228802 0.5648435507167102 0.5648435507167102 0.004709970715400788 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 65 0 0 True FLASH_ATTN False true_mixed_fused_projected 1088 False 1024 BF16 generic none CUDA_EVENT True [1024] [0] [8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192] 1 64 64 65 64 8192.0 0.015384615384615385 q1k_64q1s8k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
115 0.0 0.0 0.0 0.0 0.0 0.018112000077962875 0.026496000587940216 0.019318400137126445 0.018432000651955605 0.0024249605112359905 0.19770253574610402 0.222811797868348 0.2054811520619412 0.2054811520619412 0.008000944254868043 0.13941746080159492 0.157124211777114 0.14490284788179197 0.14490284788179197 0.005642170080515992 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 2080 False 2048 BF16 generic none CUDA_EVENT True [2048] [0] [4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096] 1 32 32 33 32 4096.0 0.030303030303030304 q2k_32q1s4k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
116 0.0 0.0 0.0 0.0 0.0 0.02595200017094612 0.03155200183391571 0.02727359998971224 0.026575999334454536 0.0016260948764843166 0.6014684881116659 0.6095472988212 0.6046757700946376 0.6046757700946376 0.0023537435563177303 0.11325152264578045 0.11477269562841545 0.11385542721485654 0.11385542721485654 0.0004431903698023628 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 4112 False 4096 BF16 generic none CUDA_EVENT True [4096] [0] [4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096] 1 16 16 17 16 4096.0 0.058823529411764705 q4k_16q1s4k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
117 0.0 0.0 0.0 0.0 0.0 0.014303999952971935 0.03359999880194664 0.01814719969406724 0.015375999733805656 0.005678506740662529 0.06774400174617767 0.07891199737787247 0.07312640026211739 0.07312640026211739 0.003826583051422731 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 2 1024 0.0 True FLASH_ATTN False vllm020_batch_spec [512, 512] 1024 512 512 512.0 True 0.0 0.0 0.0 False 0.0 1024.0 1024 BF16 generic none CUDA_EVENT False 2q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
118 0.0 0.0 0.0 0.0 0.0 0.01788800023496151 0.028543999418616295 0.021340799890458582 0.019504000432789326 0.003718627037219618 0.09548799693584442 0.1327359974384308 0.10823359936475753 0.10823359936475753 0.013230635292043864 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 4 2048 0.0 True FLASH_ATTN False vllm020_batch_spec [512, 512, 512, 512] 2048 512 512 512.0 True 0.0 0.0 0.0 False 0.0 2048.0 2048 BF16 generic none CUDA_EVENT False 4q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
119 0.0 0.0 0.0 0.0 0.0 0.02627200074493885 0.030368000268936157 0.027184000052511693 0.02643200010061264 0.0013545679205210022 0.14364799857139587 0.1597760021686554 0.15008639842271806 0.15008639842271806 0.005367723415171222 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 8 4096 0.0 True FLASH_ATTN False vllm020_batch_spec [512, 512, 512, 512, 512, 512, 512, 512] 4096 512 512 512.0 True 0.0 0.0 0.0 False 0.0 4096.0 4096 BF16 generic none CUDA_EVENT False 8q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
120 0.0 0.0 0.0 0.0 0.0 0.04150399938225746 0.04726399853825569 0.042950399965047834 0.04224000126123428 0.001720147501765378 0.2433920055627823 0.28963199257850647 0.2549152016639709 0.2549152016639709 0.013450991019687407 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 16 8192 0.0 True FLASH_ATTN False vllm020_batch_spec [512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512] 8192 512 512 512.0 True 0.0 0.0 0.0 False 0.0 8192.0 8192 BF16 generic none CUDA_EVENT False 16q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
121 0.0 0.0 0.0 0.0 0.0 0.025887999683618546 0.03407999873161316 0.028303999826312064 0.026367999613285065 0.0028877495368841042 0.325439989566803 0.3441599905490875 0.33442879617214205 0.33442879617214205 0.005693597811441922 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 2 4096 0.0 True FLASH_ATTN False vllm020_batch_spec [2048, 2048] 4096 2048 2048 2048.0 True 0.0 0.0 0.0 False 0.0 4096.0 4096 BF16 generic none CUDA_EVENT False 2q2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
122 0.0 0.0 0.0 0.0 0.0 0.04137599840760231 0.05084799975156784 0.044828799366950986 0.043087998405098915 0.003560363865797613 0.6110399961471558 0.6421759724617004 0.6204223990440368 0.6204223990440368 0.011214490370794758 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 4 8192 0.0 True FLASH_ATTN False vllm020_batch_spec [2048, 2048, 2048, 2048] 8192 2048 2048 2048.0 True 0.0 0.0 0.0 False 0.0 8192.0 8192 BF16 generic none CUDA_EVENT False 4q2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
123 0.0 0.0 0.0 0.0 0.0 0.014816000126302242 0.02364799939095974 0.016668799985200166 0.01592000015079975 0.0025515238179941247 0.0 0.0 0.0 0.0 0.0 0.0544000007212162 0.09014400094747543 0.06511679962277411 0.06511679962277411 0.013005726711295521 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1] 1 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
124 0.0 0.0 0.0 0.0 0.0 0.014527999795973301 0.0208320003002882 0.016604799963533878 0.01566399959847331 0.0021353594266203244 0.0 0.0 0.0 0.0 0.0 0.17871999740600586 0.1961279958486557 0.18568639904260634 0.18568639904260634 0.004566142885274747 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 8 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
125 0.0 0.0 0.0 0.0 0.0 0.014431999996304512 0.02006400004029274 0.015619200002402068 0.01508800033479929 0.001572984783715635 0.0 0.0 0.0 0.0 0.0 0.2730880081653595 0.29721599817276 0.2815328001976013 0.2815328001976013 0.007016534727616923 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 16 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
126 0.0 0.0 0.0 0.0 0.0 0.014879999682307243 0.02755199931561947 0.018678399827331306 0.015919999685138464 0.004323555372382858 0.0 0.0 0.0 0.0 0.0 0.3893119990825653 0.44041600823402405 0.40225600004196166 0.40225600004196166 0.013442998165884852 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 32 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
127 0.0 0.0 0.0 0.0 0.0 0.014303999952971935 0.02659199945628643 0.016748800035566093 0.015343999955803156 0.0036412341868394082 0.0 0.0 0.0 0.0 0.0 0.7404800057411194 0.9689919948577881 0.8451807916164398 0.8451807916164398 0.08637455920036795 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 64 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
128 0.0 0.0 0.0 0.0 0.0 0.014751999638974667 0.02160000056028366 0.016752000153064727 0.01521599991247058 0.0025167650350367246 0.0 0.0 0.0 0.0 0.0 0.06412799656391144 0.08956799656152725 0.07242240011692047 0.07242240011692047 0.009797120588901621 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1] 1 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
129 0.0 0.0 0.0 0.0 0.0 0.014495999552309513 0.022911999374628067 0.016137600038200618 0.015343999955803156 0.002330911555470513 0.0 0.0 0.0 0.0 0.0 0.3128319978713989 0.3282879889011383 0.3203647971153259 0.3203647971153259 0.004598568238907996 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 8 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
130 0.0 0.0 0.0 0.0 0.0 0.014655999839305878 0.03574400022625923 0.01919359974563122 0.01583999954164028 0.006642521913489215 0.0 0.0 0.0 0.0 0.0 0.5050879716873169 0.5311999917030334 0.511932796239853 0.511932796239853 0.007150929647317677 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 16 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
131 0.0 0.0 0.0 0.0 0.0 0.014655999839305878 0.02595200017094612 0.016992000117897987 0.015456000342965126 0.003390731937723684 0.0 0.0 0.0 0.0 0.0 0.7385600209236145 0.7681919932365417 0.7483008027076722 0.7483008027076722 0.010214373356208012 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 32 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
132 0.0 0.0 0.0 0.0 0.0 0.014527999795973301 0.038975998759269714 0.01952639976516366 0.015327999833971262 0.007695927285621061 0.0 0.0 0.0 0.0 0.0 1.4228800535202026 1.5237760543823242 1.4435008168220522 1.4435008168220522 0.0342955291725742 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 64 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
133 0.0 0.0 0.0 0.0 0.0 0.014495999552309513 0.03907199949026108 0.01793599994853139 0.015248000156134367 0.007167103363234667 0.0 0.0 0.0 0.0 0.0 0.06947200000286102 0.08857599645853043 0.07469440028071403 0.07469440028071403 0.00583249952929747 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1] 1 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
134 0.0 0.0 0.0 0.0 0.0 0.014495999552309513 0.02175999991595745 0.016710399929434062 0.015327999833971262 0.002705383977079183 0.0 0.0 0.0 0.0 0.0 0.388480007648468 0.4073280096054077 0.3966591984033584 0.3966591984033584 0.004487315516095964 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 8 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
135 0.0 0.0 0.0 0.0 0.0 0.0144640002399683 0.022048000246286392 0.017011200170964004 0.015392000321298838 0.002776899649845722 0.0 0.0 0.0 0.0 0.0 0.622048020362854 0.6347839832305908 0.6262047946453094 0.6262047946453094 0.004593360122958751 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 16 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
136 0.0 0.0 0.0 0.0 0.0 0.014592000283300877 0.02937600016593933 0.01847040019929409 0.015344000421464443 0.005043809142122341 0.0 0.0 0.0 0.0 0.0 0.9105280041694641 0.9304640293121338 0.914108806848526 0.914108806848526 0.00563919268816644 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 32 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
137 0.0 0.0 0.0 0.0 0.0 0.01462399959564209 0.039903998374938965 0.01791359977796674 0.015440000221133232 0.007367547649297217 0.0 0.0 0.0 0.0 0.0 1.764799952507019 1.7965760231018066 1.770739197731018 1.770739197731018 0.009082122770320404 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 64 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
138 0.0 0.0 0.0 0.0 0.0 0.015168000012636185 0.03187200054526329 0.018553599901497363 0.016127999871969223 0.0050628774268006264 0.16963527081512533 0.1737871320906266 0.17088926838108623 0.17088926838108623 0.0011703114258121128 0.47362872483230506 0.48522089403242025 0.47712993814280913 0.47712993814280913 0.0032675581298664976 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 9 0 0 True FLASH_ATTN False true_mixed_fused_projected 520 False 512 BF16 generic none CUDA_EVENT True [512] [0] [16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384] 1 8 8 9 8 16384.0 0.1111111111111111 q512_8q1s16k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
139 0.0 0.0 0.0 0.0 0.0 0.014879999682307243 0.03868800029158592 0.01921919994056225 0.016959999687969685 0.006691309533030663 0.1589600576212269 0.16117782913137854 0.15969057520605917 0.15969057520605917 0.0006212984448748182 0.5199519263456005 0.5272061673552852 0.5223414198520004 0.5223414198520004 0.002032242112153396 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 1040 False 1024 BF16 generic none CUDA_EVENT True [1024] [0] [16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384] 1 16 16 17 16 16384.0 0.058823529411764705 q1k_16q1s16k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
140 0.0 0.0 0.0 0.0 0.0 0.018400000408291817 0.024320000782608986 0.019910399988293647 0.0191040001809597 0.001841532763984352 0.25650752966102763 0.27066608538463055 0.259736894547936 0.259736894547936 0.00462927315947332 0.5278764825612624 0.5570139063374764 0.5345223138928448 0.5345223138928448 0.009526755161797178 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 2080 False 2048 BF16 generic none CUDA_EVENT True [2048] [0] [16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384] 1 32 32 33 32 16384.0 0.030303030303030304 q2k_32q1s16k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
141 0.0 0.0 0.0 0.0 0.0 0.026048000901937485 0.03728000074625015 0.028636799938976765 0.02705600019544363 0.003313953649011259 0.9270006318443208 0.9712965120641314 0.93480767601568 0.93480767601568 0.012612551450284608 0.8181833128578277 0.8572794566782395 0.8250739157811953 0.8250739157811953 0.01113200873299586 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 4112 False 4096 BF16 generic none CUDA_EVENT True [4096] [0] [32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768] 1 16 16 17 16 32768.0 0.058823529411764705 q4k_16q1s32k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
142 0.0 0.0 0.0 0.0 0.0 0.018079999834299088 0.037696000188589096 0.022092800214886667 0.019024000503122807 0.005986278786633071 0.2839857165542317 0.2876110048757805 0.2849506939696605 0.2849506939696605 0.0009995178425916847 1.0871823008331585 1.101060989333509 1.0908765231323903 1.0908765231323903 0.0038264533900426207 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 2080 False 2048 BF16 generic none CUDA_EVENT True [2048] [0] [32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768] 1 32 32 33 32 32768.0 0.030303030303030304 q2k_32q1s32k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
143 0.0 0.0 0.0 0.0 0.0 0.015072000212967396 0.025567999109625816 0.01726400014013052 0.015728000551462173 0.0031324465940990903 0.137270464802061 0.13799613818579368 0.13751499486424038 0.13751499486424038 0.00025221761530472904 2.3021855212208884 2.314355908780759 2.3062865750744197 2.3062865750744197 0.004229983070201486 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 65 0 0 True FLASH_ATTN False true_mixed_fused_projected 1088 False 1024 BF16 generic none CUDA_EVENT True [1024] [0] [32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768] 1 64 64 65 64 32768.0 0.015384615384615385 q1k_64q1s32k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
144 0.0 0.0 0.0 0.0 0.0 0.014560000039637089 0.022784000262618065 0.016435200069099664 0.015519999898970127 0.0023688301421469523 0.04879999905824661 0.09139200299978256 0.06054079942405224 0.06054079942405224 0.012152608702448775 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 64 0.0 True FLASH_ATTN False vllm020_batch_spec [64] 64 64 64 64.0 True 0.0 0.0 0.0 False 0.0 64.0 64 BF16 generic none CUDA_EVENT False q64 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
145 0.0 0.0 0.0 0.0 0.0 0.014816000126302242 0.02844800055027008 0.017008000146597625 0.015696000307798386 0.003941466294662679 0.047807998955249786 0.07356800138950348 0.05626560002565384 0.05626560002565384 0.00842179125412236 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 128 0.0 True FLASH_ATTN False vllm020_batch_spec [128] 128 128 128 128.0 True 0.0 0.0 0.0 False 0.0 128.0 128 BF16 generic none CUDA_EVENT False q128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
146 0.0 0.0 0.0 0.0 0.0 0.014688000082969666 0.053408000618219376 0.019353600218892097 0.01532800029963255 0.01138302400841626 0.048448000103235245 0.0785600021481514 0.0556256003677845 0.0556256003677845 0.009348274502616908 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 256 0.0 True FLASH_ATTN False vllm020_batch_spec [256] 256 256 256 256.0 True 0.0 0.0 0.0 False 0.0 256.0 256 BF16 generic none CUDA_EVENT False q256 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
147 0.0 0.0 0.0 0.0 0.0 0.015039999969303608 0.03139200061559677 0.01823679991066456 0.016159999649971724 0.004860071238302512 0.055424001067876816 0.08505599945783615 0.0640383992344141 0.0640383992344141 0.00921448636178623 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 512 0.0 True FLASH_ATTN False vllm020_batch_spec [512] 512 512 512 512.0 True 0.0 0.0 0.0 False 0.0 512.0 512 BF16 generic none CUDA_EVENT False q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
148 0.0 0.0 0.0 0.0 0.0 0.014751999638974667 0.026528000831604004 0.017324799951165915 0.01536000007763505 0.0036004822686428305 0.07660800218582153 0.0942080020904541 0.08209280073642732 0.08209280073642732 0.00515895587669752 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 1024 0.0 True FLASH_ATTN False vllm020_batch_spec [1024] 1024 1024 1024 1024.0 True 0.0 0.0 0.0 False 0.0 1024.0 1024 BF16 generic none CUDA_EVENT False q1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
149 0.0 0.0 0.0 0.0 0.0 0.014879999682307243 0.021247999742627144 0.016403199825435876 0.01532800029963255 0.001995897022647934 0.11395200341939926 0.15113599598407745 0.12431039959192276 0.12431039959192276 0.011164431123683732 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 2048 0.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 False 0.0 2048.0 2048 BF16 generic none CUDA_EVENT False q2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
150 0.0 0.0 0.0 0.0 0.0 0.017184000462293625 0.028672000393271446 0.019865600019693376 0.017823999747633934 0.0035798491315929977 0.3171840012073517 0.3341119885444641 0.3261695951223373 0.3261695951223373 0.005111046452878918 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 4096 0.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 False 0.0 4096.0 4096 BF16 generic none CUDA_EVENT False q4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
151 0.0 0.0 0.0 0.0 0.0 0.024639999493956566 0.030400000512599945 0.02656640000641346 0.02556800004094839 0.0020189846603237303 1.0648640394210815 1.0828479528427124 1.0715327858924866 1.0715327858924866 0.005558639263049851 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 8192 0.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 False 0.0 8192.0 8192 BF16 generic none CUDA_EVENT False q8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
152 0.0 0.0 0.0 0.0 0.0 0.014976000413298607 0.021695999428629875 0.017305599898099898 0.01593599934130907 0.0025718046517268054 0.04956800118088722 0.07932800054550171 0.06228480041027069 0.06228480041027069 0.01027160349757827 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 64 448.0 True FLASH_ATTN False vllm020_batch_spec [64] 64 64 64 64.0 True 0.0 0.0 0.0 True 448.0 512.0 64 BF16 generic none CUDA_EVENT False q64s512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
153 0.0 0.0 0.0 0.0 0.0 0.01539199985563755 0.03580800071358681 0.019686400331556796 0.017136000096797943 0.005918387811304003 0.05158400163054466 0.10713600367307663 0.06364160068333148 0.06364160068333148 0.015832957809696766 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 128 896.0 True FLASH_ATTN False vllm020_batch_spec [128] 128 128 128 128.0 True 0.0 0.0 0.0 True 896.0 1024.0 128 BF16 generic none CUDA_EVENT False q128s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
154 0.0 0.0 0.0 0.0 0.0 0.015168000012636185 0.02223999984562397 0.016672000009566545 0.015887999907135963 0.0020934945946034563 0.06521599739789963 0.08902399986982346 0.07520959973335266 0.07520959973335266 0.007913840684102929 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 256 1792.0 True FLASH_ATTN False vllm020_batch_spec [256] 256 256 256 256.0 True 0.0 0.0 0.0 True 1792.0 2048.0 256 BF16 generic none CUDA_EVENT False q256s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
155 0.0 0.0 0.0 0.0 0.0 0.015231999568641186 0.04028800129890442 0.019971200078725816 0.01646399963647127 0.007351305886577286 0.14467200636863708 0.16844800114631653 0.15363519936800005 0.15363519936800005 0.008174435899956223 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 512 3584.0 True FLASH_ATTN False vllm020_batch_spec [512] 512 512 512 512.0 True 0.0 0.0 0.0 True 3584.0 4096.0 512 BF16 generic none CUDA_EVENT False q512s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
156 0.0 0.0 0.0 0.0 0.0 0.015519999898970127 0.02304000034928322 0.01775679988786578 0.016784000210464 0.0025077243712082584 0.33740800619125366 0.35343998670578003 0.3445120006799698 0.3445120006799698 0.00445648463590358 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 1024 7168.0 True FLASH_ATTN False vllm020_batch_spec [1024] 1024 1024 1024 1024.0 True 0.0 0.0 0.0 True 7168.0 8192.0 1024 BF16 generic none CUDA_EVENT False q1ks8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
157 0.0 0.0 0.0 0.0 0.0 0.015647999942302704 0.02393599972128868 0.01809599995613098 0.01654400024563074 0.002998393102466254 0.0 0.0 0.0 0.0 0.0 0.0504320003092289 0.0843840017914772 0.060083200410008426 0.060083200410008426 0.00986959318572296 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1] 1 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
158 0.0 0.0 0.0 0.0 0.0 0.01603199914097786 0.030047999694943428 0.019510399922728537 0.017680000513792038 0.004065185265963332 0.0 0.0 0.0 0.0 0.0 0.05004800111055374 0.07036799937486649 0.059315200522542 0.059315200522542 0.006537768821329647 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 8 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
159 0.0 0.0 0.0 0.0 0.0 0.01484800036996603 0.02393599972128868 0.018035199772566558 0.016512000001966953 0.0030667689116777724 0.0 0.0 0.0 0.0 0.0 0.05100800096988678 0.06735999882221222 0.058387200161814694 0.058387200161814694 0.005832787739241381 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 16 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
160 0.0 0.0 0.0 0.0 0.0 0.01500799972563982 0.026208000257611275 0.017500799987465142 0.01646399963647127 0.0030666103306165714 0.0 0.0 0.0 0.0 0.0 0.05023999884724617 0.07254400104284286 0.057254400476813315 0.057254400476813315 0.006890853653068151 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 32 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
161 0.0 0.0 0.0 0.0 0.0 0.014688000082969666 0.027327999472618103 0.0179776000790298 0.01648000068962574 0.003783417447531392 0.0 0.0 0.0 0.0 0.0 0.04819199815392494 0.07100799679756165 0.05621119923889638 0.05621119923889638 0.007824391484124725 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 64 0 128.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s128 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
162 0.0 0.0 0.0 0.0 0.0 0.015647999942302704 0.036448001861572266 0.019136000238358975 0.016704000532627106 0.006076530095624803 0.0 0.0 0.0 0.0 0.0 0.047488000243902206 0.08441600203514099 0.0588383998721838 0.0588383998721838 0.011275940726332522 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 8 0 1024.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
163 0.0 0.0 0.0 0.0 0.0 0.015135999768972397 0.033952001482248306 0.02119360016658902 0.018000000156462193 0.007136646614305304 0.0 0.0 0.0 0.0 0.0 0.04931199923157692 0.06992000341415405 0.05755840018391609 0.05755840018391609 0.007297587694315226 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 16 0 1024.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
164 0.0 0.0 0.0 0.0 0.0 0.014879999682307243 0.034272000193595886 0.0204415999352932 0.017311999574303627 0.00695404701803013 0.0 0.0 0.0 0.0 0.0 0.05215999856591225 0.0735040009021759 0.06117440015077591 0.06117440015077591 0.007381118436894922 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 32 0 1024.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
165 0.0 0.0 0.0 0.0 0.0 0.015359999611973763 0.03743999823927879 0.02012479966506362 0.01775999926030636 0.006181045509079057 0.0 0.0 0.0 0.0 0.0 0.06355199962854385 0.08508799970149994 0.07019200026988984 0.07019200026988984 0.0062246580442117845 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 64 0 1024.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s1k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
166 0.0 0.0 0.0 0.0 0.0 0.015552000142633915 0.032607998698949814 0.01959999995306134 0.017487999983131886 0.004882475068460749 0.0 0.0 0.0 0.0 0.0 0.04918399825692177 0.0865280032157898 0.05973760038614274 0.05973760038614274 0.011546847942113974 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 8 0 2048.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
167 0.0 0.0 0.0 0.0 0.0 0.015168000012636185 0.02675200067460537 0.017430400010198355 0.016560000367462635 0.003243135094239819 0.0 0.0 0.0 0.0 0.0 0.052671998739242554 0.08367999643087387 0.06238719932734965 0.06238719932734965 0.011207824780630104 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 16 0 2048.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
168 0.0 0.0 0.0 0.0 0.0 0.01500799972563982 0.03612799942493439 0.019327999837696553 0.01688000001013279 0.006058251425153085 0.0 0.0 0.0 0.0 0.0 0.06364800035953522 0.07878399640321732 0.06935679838061332 0.06935679838061332 0.004515588328677643 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 32 0 2048.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
169 0.0 0.0 0.0 0.0 0.0 0.015104000456631184 0.03711999952793121 0.019670399930328132 0.017152000218629837 0.006165994646522264 0.0 0.0 0.0 0.0 0.0 0.09071999788284302 0.11507199704647064 0.10252480059862136 0.10252480059862136 0.008782544051535657 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 64 0 2048.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
170 0.0 0.0 0.0 0.0 0.0 0.015039999969303608 0.026528000831604004 0.018396800104528665 0.01601599995046854 0.004279788048986283 0.0 0.0 0.0 0.0 0.0 0.05331199988722801 0.07977599650621414 0.06076480001211167 0.06076480001211167 0.008761579122069606 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 8 0 4096.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
171 0.0 0.0 0.0 0.0 0.0 0.015359999611973763 0.02643200010061264 0.01726400004699826 0.015855999663472176 0.003210008417242029 0.0 0.0 0.0 0.0 0.0 0.062431998550891876 0.08246400207281113 0.06970879957079888 0.06970879957079888 0.007016341676068233 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 16 0 4096.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
172 0.0 0.0 0.0 0.0 0.0 0.014911999925971031 0.02223999984562397 0.0166015999391675 0.015887999907135963 0.00204913969129354 0.0 0.0 0.0 0.0 0.0 0.10127999633550644 0.1141119971871376 0.10621120035648347 0.10621120035648347 0.004029318311754605 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 32 0 4096.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
173 0.0 0.0 0.0 0.0 0.0 0.015359999611973763 0.03363199904561043 0.019305599946528675 0.016671999357640743 0.0055445582307981234 0.0 0.0 0.0 0.0 0.0 0.1319040060043335 0.15839999914169312 0.14040640145540234 0.14040640145540234 0.007998041073596942 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 64 0 4096.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s4k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
174 0.0 0.0 0.0 0.0 0.0 0.014368000440299511 0.04447999969124794 0.018908800091594458 0.0157279996201396 0.0086814937461721 0.0 0.0 0.0 0.0 0.0 0.06428799778223038 0.08982399851083755 0.07349760085344315 0.07349760085344315 0.008605284512197258 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 8 0 8192.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
175 0.0 0.0 0.0 0.0 0.0 0.015104000456631184 0.028672000393271446 0.01890560006722808 0.016080000437796116 0.004797584627815021 0.0 0.0 0.0 0.0 0.0 0.11123199760913849 0.14115199446678162 0.11942399889230729 0.11942399889230729 0.008513394706791027 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 16 0 8192.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
176 0.0 0.0 0.0 0.0 0.0 0.014271999709308147 0.024927999824285507 0.016915200091898442 0.015216000378131866 0.003374147499442635 0.0 0.0 0.0 0.0 0.0 0.15887999534606934 0.18111999332904816 0.16934399753808976 0.16934399753808976 0.007415181260761123 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 32 0 8192.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
177 0.0 0.0 0.0 0.0 0.0 0.014336000196635723 0.026944000273942947 0.017737600207328796 0.015199999790638685 0.00415598714375255 0.0 0.0 0.0 0.0 0.0 0.2192319929599762 0.23472000658512115 0.22809920012950893 0.22809920012950893 0.004730841376327335 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 64 0 8192.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s8k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
178 0.0 0.0 0.0 0.0 0.0 0.014688000082969666 0.05052800104022026 0.020320000313222408 0.015520000364631414 0.010604522177924678 0.024270629882498958 0.03340247625954076 0.028446344104128624 0.028446344104128624 0.0032330369099793834 0.023697370291069768 0.03261352722995356 0.027774456325453972 0.027774456325453972 0.0031566742680923386 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 9 0 0 True FLASH_ATTN False true_mixed_fused_projected 72 False 64 BF16 generic none CUDA_EVENT True [64] [0] [512, 512, 512, 512, 512, 512, 512, 512] 1 8 8 9 8 512.0 0.1111111111111111 q64_8q1s512 fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
179 0.0 0.0 0.0 0.0 0.0 0.015359999611973763 0.033663999289274216 0.019987199828028678 0.018240000121295452 0.005522929139253732 0.0259194055660947 0.03719755183990719 0.029916029687899703 0.029916029687899703 0.003966666590554318 0.027104595853449518 0.03889844644729374 0.03128396954022015 0.03128396954022015 0.004148046317967881 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 9 0 0 True FLASH_ATTN False true_mixed_fused_projected 136 False 128 BF16 generic none CUDA_EVENT True [128] [0] [1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024] 1 8 8 9 8 1024.0 0.1111111111111111 q128_8q1s1k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
180 0.0 0.0 0.0 0.0 0.0 0.014592000283300877 0.03299200162291527 0.019840000104159115 0.016207999549806118 0.006546343008726814 0.02587869595769926 0.03763167265431482 0.030174939058162802 0.030174939058162802 0.0037303581782277364 0.026473304070195713 0.03849632587654989 0.03086826083864964 0.03086826083864964 0.003816069654529222 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 144 False 128 BF16 generic none CUDA_EVENT True [128] [0] [1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024] 1 16 16 17 16 1024.0 0.058823529411764705 q128_16q1s1k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
181 0.0 0.0 0.0 0.0 0.0 0.013824000023305416 0.021023999899625778 0.016304000187665223 0.015584000386297703 0.0023236165806545476 0.031810621525966996 0.04156949936878106 0.0346749350032807 0.0346749350032807 0.003130209181143985 0.035677378270900374 0.04662250161636451 0.038889864871740294 0.038889864871740294 0.0035107033960828727 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 272 False 256 BF16 generic none CUDA_EVENT True [256] [0] [2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048] 1 16 16 17 16 2048.0 0.058823529411764705 q256_16q1s2k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
182 0.0 0.0 0.0 0.0 0.0 0.014527999795973301 0.02175999991595745 0.01569600012153387 0.015008000191301107 0.0020882666168036863 0.038211712107062853 0.07212229256520057 0.04364509673334097 0.04364509673334097 0.009671953074025085 0.0476442859917874 0.08992570455184198 0.05441890342616107 0.05441890342616107 0.01205947791784038 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 288 False 256 BF16 generic none CUDA_EVENT True [256] [0] [2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048] 1 32 32 33 32 2048.0 0.030303030303030304 q256_32q1s2k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
183 0.0 0.0 0.0 0.0 0.0 0.014368000440299511 0.033824000507593155 0.017628800217062236 0.014864000026136637 0.005791495403857738 0.05214261250030033 0.06266261508706669 0.05677791295527661 0.05677791295527661 0.0030298223079651514 0.08648138506878382 0.10392938682791132 0.09416928531647478 0.09416928531647478 0.005025126612204489 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 544 False 512 BF16 generic none CUDA_EVENT True [512] [0] [4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096] 1 32 32 33 32 4096.0 0.030303030303030304 q512_32q1s4k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
184 0.0 0.0 0.0 0.0 0.0 0.014527999795973301 0.031199999153614044 0.017900799959897996 0.015343999955803156 0.004974475661316249 0.06411958891421111 0.07405276123263956 0.06793047918211377 0.06793047918211377 0.0033918658556700006 0.14058441263169497 0.16236323591492058 0.1489399211274489 0.1489399211274489 0.00743678300375353 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 65 0 0 True FLASH_ATTN False true_mixed_fused_projected 576 False 512 BF16 generic none CUDA_EVENT True [512] [0] [4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096] 1 64 64 65 64 4096.0 0.015384615384615385 q512_64q1s4k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
185 0.0 0.0 0.0 0.0 0.0 0.014720000326633453 0.03855999931693077 0.018495999928563833 0.015647999942302704 0.006918530056761627 0.09872985549401277 0.10683454583043753 0.10185316839884552 0.10185316839884552 0.0020802254037042074 0.2743261390166379 0.2968454508984596 0.2830044295482752 0.2830044295482752 0.005780016596065092 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 65 0 0 True FLASH_ATTN False true_mixed_fused_projected 1088 False 1024 BF16 generic none CUDA_EVENT True [1024] [0] [8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192] 1 64 64 65 64 8192.0 0.015384615384615385 q1k_64q1s8k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
186 0.0 0.0 0.0 0.0 0.0 0.014911999925971031 0.030400000512599945 0.01977920001372695 0.01726400014013052 0.005350883464206169 0.13918872472233365 0.16279522855335207 0.1501912864839173 0.1501912864839173 0.008992185529011513 0.11892328861766266 0.13909276049083735 0.1283239123428725 0.1283239123428725 0.007682951884956939 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 2080 False 2048 BF16 generic none CUDA_EVENT True [2048] [0] [4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096] 1 32 32 33 32 4096.0 0.030303030303030304 q2k_32q1s4k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
187 0.0 0.0 0.0 0.0 0.0 0.01727999933063984 0.02223999984562397 0.01819519978016615 0.017680000513792038 0.0014397298270620873 0.3728044181625443 0.38295504353701676 0.3761264483787333 0.3761264483787333 0.0033660272332490977 0.07967557017401883 0.08184495665371805 0.08038555277807365 0.08038555277807365 0.0007193856241088455 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 4112 False 4096 BF16 generic none CUDA_EVENT True [4096] [0] [4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096] 1 16 16 17 16 4096.0 0.058823529411764705 q4k_16q1s4k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
188 0.0 0.0 0.0 0.0 0.0 0.01500799972563982 0.034591998904943466 0.01941439984366298 0.01756799966096878 0.005541380584373771 0.05951999872922897 0.08268799632787704 0.06715519949793816 0.06715519949793816 0.006657802116288364 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 2 1024 0.0 True FLASH_ATTN False vllm020_batch_spec [512, 512] 1024 512 512 512.0 True 0.0 0.0 0.0 False 0.0 1024.0 1024 BF16 generic none CUDA_EVENT False 2q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
189 0.0 0.0 0.0 0.0 0.0 0.015104000456631184 0.03283200040459633 0.019180799927562477 0.016543999314308167 0.0052187911233635975 0.07199999690055847 0.08675199747085571 0.07749439924955369 0.07749439924955369 0.004849874741156772 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 4 2048 0.0 True FLASH_ATTN False vllm020_batch_spec [512, 512, 512, 512] 2048 512 512 512.0 True 0.0 0.0 0.0 False 0.0 2048.0 2048 BF16 generic none CUDA_EVENT False 4q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
190 0.0 0.0 0.0 0.0 0.0 0.01724799908697605 0.02284800074994564 0.01928640007972717 0.018400000408291817 0.0020641390157565697 0.09676799923181534 0.12310399860143663 0.10618879944086074 0.10618879944086074 0.008982738207839057 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 8 4096 0.0 True FLASH_ATTN False vllm020_batch_spec [512, 512, 512, 512, 512, 512, 512, 512] 4096 512 512 512.0 True 0.0 0.0 0.0 False 0.0 4096.0 4096 BF16 generic none CUDA_EVENT False 8q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
191 0.0 0.0 0.0 0.0 0.0 0.024224000051617622 0.03315199911594391 0.027436799928545953 0.027328000403940678 0.002679154874546328 0.14716799557209015 0.16412800550460815 0.15470399856567385 0.15470399856567385 0.005423454433987419 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 16 8192 0.0 True FLASH_ATTN False vllm020_batch_spec [512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512] 8192 512 512 512.0 True 0.0 0.0 0.0 False 0.0 8192.0 8192 BF16 generic none CUDA_EVENT False 16q512 measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
192 0.0 0.0 0.0 0.0 0.0 0.01833599992096424 0.04806400090456009 0.026252799853682517 0.022672000341117382 0.00878438440429033 0.1844799965620041 0.2072959989309311 0.19359359890222552 0.19359359890222552 0.007663539820967659 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 2 4096 0.0 True FLASH_ATTN False vllm020_batch_spec [2048, 2048] 4096 2048 2048 2048.0 True 0.0 0.0 0.0 False 0.0 4096.0 4096 BF16 generic none CUDA_EVENT False 2q2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
193 0.0 0.0 0.0 0.0 0.0 0.024480000138282776 0.04979199916124344 0.029081599973142146 0.025679999962449074 0.007227422044689197 0.34147199988365173 0.37968000769615173 0.3583200007677078 0.3583200007677078 0.011804422100726231 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 4 8192 0.0 True FLASH_ATTN False vllm020_batch_spec [2048, 2048, 2048, 2048] 8192 2048 2048 2048.0 True 0.0 0.0 0.0 False 0.0 8192.0 8192 BF16 generic none CUDA_EVENT False 4q2k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
194 0.0 0.0 0.0 0.0 0.0 0.014720000326633453 0.023744000121951103 0.017753600236028434 0.016368000768125057 0.003061040281616705 0.0 0.0 0.0 0.0 0.0 0.05686400085687637 0.1090880036354065 0.06629760004580021 0.06629760004580021 0.014803727026812366 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1] 1 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
195 0.0 0.0 0.0 0.0 0.0 0.014879999682307243 0.021088000386953354 0.01647359998896718 0.015840000472962856 0.0017964402433206 0.0 0.0 0.0 0.0 0.0 0.08540800213813782 0.09961599856615067 0.09146559983491898 0.09146559983491898 0.00447423706371901 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 8 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
196 0.0 0.0 0.0 0.0 0.0 0.014560000039637089 0.04156799986958504 0.018512000143527985 0.015584000386297703 0.007849701681877904 0.0 0.0 0.0 0.0 0.0 0.17948800325393677 0.194815993309021 0.1866239994764328 0.1866239994764328 0.004156328241374654 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 16 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
197 0.0 0.0 0.0 0.0 0.0 0.014879999682307243 0.029632000252604485 0.019449600111693145 0.016736000776290894 0.005126811425136928 0.0 0.0 0.0 0.0 0.0 0.27529600262641907 0.2898879945278168 0.2825664013624191 0.2825664013624191 0.004542801609674093 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 32 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
198 0.0 0.0 0.0 0.0 0.0 0.014399999752640724 0.031328000128269196 0.019759999960660933 0.016367999836802483 0.006266303740589241 0.0 0.0 0.0 0.0 0.0 0.39190399646759033 0.4079039990901947 0.3987520009279252 0.3987520009279252 0.0037367727509362725 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 64 0 16384.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
199 0.0 0.0 0.0 0.0 0.0 0.01539199985563755 0.040031999349594116 0.0212032001465559 0.01649599988013506 0.008068547381446682 0.0 0.0 0.0 0.0 0.0 0.06588800251483917 0.07897599786520004 0.0703904002904892 0.0703904002904892 0.004214826869769361 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1] 1 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
200 0.0 0.0 0.0 0.0 0.0 0.014495999552309513 0.022495999932289124 0.016006399970501663 0.01515199989080429 0.002265581884342885 0.0 0.0 0.0 0.0 0.0 0.1231679990887642 0.13526399433612823 0.1288223996758461 0.1288223996758461 0.004328976434897094 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 8 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
201 0.0 0.0 0.0 0.0 0.0 0.014592000283300877 0.023231999948620796 0.01705600004643202 0.016048000194132328 0.002810904312746391 0.0 0.0 0.0 0.0 0.0 0.3158079981803894 0.33129599690437317 0.32348800003528594 0.32348800003528594 0.00417599076136664 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 16 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
202 0.0 0.0 0.0 0.0 0.0 0.014527999795973301 0.021888000890612602 0.01708160014823079 0.01609600055962801 0.0025289234621475062 0.0 0.0 0.0 0.0 0.0 0.5103679895401001 0.5200319886207581 0.5132320046424866 0.5132320046424866 0.0035759433157749533 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 32 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
203 0.0 0.0 0.0 0.0 0.0 0.014688000082969666 0.032416000962257385 0.019353600032627583 0.016176000237464905 0.0061458798960684425 0.0 0.0 0.0 0.0 0.0 0.7400320172309875 0.7516480088233948 0.7449311971664427 0.7449311971664427 0.004573461776168607 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 64 0 32768.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
204 0.0 0.0 0.0 0.0 0.0 0.0144640002399683 0.02006400004029274 0.01573119992390275 0.014992000069469213 0.001654446341262991 0.0 0.0 0.0 0.0 0.0 0.07097599655389786 0.09932799637317657 0.07828159928321837 0.07828159928321837 0.009464602895232642 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1] 1 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
205 0.0 0.0 0.0 0.0 0.0 0.01484800036996603 0.02319999970495701 0.016883199848234654 0.015775999519973993 0.0026693906156048403 0.0 0.0 0.0 0.0 0.0 0.14176000654697418 0.1598079949617386 0.15063679963350293 0.15063679963350293 0.005741662969679433 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 8 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1] 8 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 8q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
206 0.0 0.0 0.0 0.0 0.0 0.01484800036996603 0.03203200176358223 0.022224000189453363 0.02247999981045723 0.005996176183124981 0.0 0.0 0.0 0.0 0.0 0.39180800318717957 0.41046398878097534 0.4003200054168701 0.4003200054168701 0.005552433764307601 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 16 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 16 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 16q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
207 0.0 0.0 0.0 0.0 0.0 0.015072000212967396 0.028192000463604927 0.017152000125497578 0.015536000020802021 0.003847486121602065 0.0 0.0 0.0 0.0 0.0 0.6239359974861145 0.6367359757423401 0.6285343945026398 0.6285343945026398 0.004006084286606002 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 32 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 32 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 32q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
208 0.0 0.0 0.0 0.0 0.0 0.014879999682307243 0.026688000187277794 0.01718079997226596 0.015488000120967627 0.0037132536813398722 0.0 0.0 0.0 0.0 0.0 0.9141119718551636 0.9721279740333557 0.927455997467041 0.927455997467041 0.02148767779232235 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 64 0 40960.0 False FLASH_ATTN False vllm020_batch_spec [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1] 64 1 1 1.0 True 0.0 0.0 0.0 False 0 0 0 BF16 generic none CUDA_EVENT False 64q1s40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
209 0.0 0.0 0.0 0.0 0.0 0.014751999638974667 0.047520000487565994 0.028710400220006704 0.0266720000654459 0.010774688401598903 0.06094816381288764 0.06816969726785131 0.0644060660218149 0.0644060660218149 0.002203106589073183 0.087051838094461 0.09736630405680231 0.09199073574791852 0.09199073574791852 0.0031466818046499644 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 9 0 0 True FLASH_ATTN False true_mixed_fused_projected 520 False 512 BF16 generic none CUDA_EVENT True [512] [0] [16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384] 1 8 8 9 8 16384.0 0.1111111111111111 q512_8q1s16k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
210 0.0 0.0 0.0 0.0 0.0 0.014911999925971031 0.02489599958062172 0.017427200078964235 0.01595200039446354 0.003071906489592143 0.19733789497223914 0.2016295616652644 0.19860388994013833 0.19860388994013833 0.001512389485439305 0.44861409134062713 0.45837046456077934 0.45149211526120153 0.45149211526120153 0.0034381598874302305 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 1040 False 1024 BF16 generic none CUDA_EVENT True [1024] [0] [16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384] 1 16 16 17 16 16384.0 0.058823529411764705 q1k_16q1s16k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
211 0.0 0.0 0.0 0.0 0.0 0.015039999969303608 0.03046399913728237 0.020643199887126686 0.016944000497460365 0.006320148675595729 0.2157924314537054 0.22215709640166995 0.21769400765743155 0.21769400765743155 0.0019679921844645526 0.4905115822753901 0.5049789194903589 0.4948340005651485 0.4948340005651485 0.0044733865493072344 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 2080 False 2048 BF16 generic none CUDA_EVENT True [2048] [0] [16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384] 1 32 32 33 32 16384.0 0.030303030303030304 q2k_32q1s16k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
212 0.0 0.0 0.0 0.0 0.0 0.017152000218629837 0.02425600029528141 0.01865920014679432 0.018112000077962875 0.0019454553198986453 0.7430866512973927 0.7578834738720781 0.7455351005236347 0.7455351005236347 0.00416692487556653 0.7369773831645824 0.7516525540362471 0.7394057025273603 0.7394057025273603 0.004132666607961175 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 17 0 0 True FLASH_ATTN False true_mixed_fused_projected 4112 False 4096 BF16 generic none CUDA_EVENT True [4096] [0] [32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768] 1 16 16 17 16 32768.0 0.058823529411764705 q4k_16q1s32k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
213 0.0 0.0 0.0 0.0 0.0 0.015263999812304974 0.02470399998128414 0.018009600043296815 0.01657600048929453 0.0031544532774534346 0.25251173919752334 0.25566890792461106 0.2536729054481981 0.2536729054481981 0.0010279562245342983 1.0425282722084215 1.0555630628624373 1.0473223013846877 1.0473223013846877 0.004244053880724073 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 33 0 0 True FLASH_ATTN False true_mixed_fused_projected 2080 False 2048 BF16 generic none CUDA_EVENT True [2048] [0] [32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768] 1 32 32 33 32 32768.0 0.030303030303030304 q2k_32q1s32k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
214 0.0 0.0 0.0 0.0 0.0 0.01500799972563982 0.027807999402284622 0.01912960009649396 0.01643200032413006 0.004688164695934997 0.1251379565220268 0.14341821167748953 0.12742497577885914 0.12742497577885914 0.005353812167343766 1.1355340167064276 1.3014137556763312 1.1562870179154463 1.1562870179154463 0.048581869194943325 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 65 0 0 True FLASH_ATTN False true_mixed_fused_projected 1088 False 1024 BF16 generic none CUDA_EVENT True [1024] [0] [32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768] 1 64 64 65 64 32768.0 0.015384615384615385 q1k_64q1s32k fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
215 0.0 0.0 0.0 0.0 0.0 0.07446400076150894 0.08246400207281113 0.07736000046133995 0.07713599875569344 0.002301031358835964 11.935359954833984 12.102368354797363 11.96896333694458 11.96896333694458 0.05162549713370252 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 8192 8192.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 8192.0 16384.0 8192 BF16 generic none CUDA_EVENT False q8ks16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
216 0.0 0.0 0.0 0.0 0.0 0.07664000242948532 0.08118399977684021 0.07796800062060356 0.077504001557827 0.0013681081693640953 19.84774398803711 20.15795135498047 19.915702438354494 19.915702438354494 0.0912299324030902 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 8192 16384.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 16384.0 24576.0 8192 BF16 generic none CUDA_EVENT False q8ks24k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
217 0.0 0.0 0.0 0.0 0.0 0.0753600001335144 0.08207999914884567 0.07715519964694977 0.07595199719071388 0.0024238358447475007 27.781503677368164 27.9836483001709 27.844886589050287 27.844886589050287 0.061307010154819624 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 8192 24576.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 24576.0 32768.0 8192 BF16 generic none CUDA_EVENT False q8ks32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
218 0.0 0.0 0.0 0.0 0.0 0.04339199885725975 0.050016000866889954 0.04504639990627766 0.04391999915242195 0.002339637813151782 5.1544318199157715 5.269279956817627 5.16938238143921 5.16938238143921 0.03363105774712194 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 4096 8192.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 True 8192.0 12288.0 4096 BF16 generic none CUDA_EVENT False q4ks12k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
219 0.0 0.0 0.0 0.0 0.0 0.04291199892759323 0.058848001062870026 0.045657599717378615 0.04383999854326248 0.004522763233004815 9.256383895874023 9.27734375 9.261776161193849 9.261776161193849 0.0062232312957398745 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 4096 16384.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 True 16384.0 20480.0 4096 BF16 generic none CUDA_EVENT False q4ks20k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
220 0.0 0.0 0.0 0.0 0.0 0.043296001851558685 0.04982399940490723 0.045123199746012685 0.04387199878692627 0.0023369398894319345 13.365216255187988 13.634464263916016 13.398719978332519 13.398719978332519 0.0787651922310212 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 4096 24576.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 True 24576.0 28672.0 4096 BF16 generic none CUDA_EVENT False q4ks28k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
221 0.0 0.0 0.0 0.0 0.0 0.02703999914228916 0.0297279991209507 0.027692800015211107 0.02723200060427189 0.000851127720159845 2.387968063354492 2.4014720916748047 2.3920736074447637 2.3920736074447637 0.003552645719470337 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 2048 8192.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 True 8192.0 10240.0 2048 BF16 generic none CUDA_EVENT False q2ks10k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
222 0.0 0.0 0.0 0.0 0.0 0.02691200003027916 0.029823999851942062 0.027689599990844728 0.027583999559283257 0.0007783369959809023 4.440767765045166 4.4521918296813965 4.444268751144409 4.444268751144409 0.004022325406840951 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 2048 16384.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 True 16384.0 18432.0 2048 BF16 generic none CUDA_EVENT False q2ks18k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
223 0.0 0.0 0.0 0.0 0.0 0.02703999914228916 0.03888000175356865 0.030262400023639204 0.02801600005477667 0.0037877256209144718 6.500351905822754 6.600607872009277 6.524902391433716 6.524902391433716 0.03321667087212851 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 2048 24576.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 True 24576.0 26624.0 2048 BF16 generic none CUDA_EVENT False q2ks26k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
224 0.0 0.0 0.0 0.0 0.0 0.07865600287914276 0.08668799698352814 0.08130879923701287 0.07993599772453308 0.0027016201268628523 35.717376708984375 35.843265533447266 35.771837615966795 35.771837615966795 0.04024815379149301 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 40960 1 8192 32768.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 32768.0 40960.0 8192 BF16 generic none CUDA_EVENT False q8ks40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
225 0.0 0.0 0.0 0.0 0.0 0.04179200157523155 0.04864000156521797 0.0436256006360054 0.04267200082540512 0.002191476789190783 6.103871822357178 6.18287992477417 6.118390369415283 6.118390369415283 0.022924175047267112 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 8192 8192.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 8192.0 16384.0 8192 BF16 generic none CUDA_EVENT False q8ks16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
226 0.0 0.0 0.0 0.0 0.0 0.04163200035691261 0.05987200140953064 0.04625920057296753 0.04403200000524521 0.005544136325859629 10.207136154174805 11.073247909545898 10.326777648925782 10.326777648925782 0.25139335734700863 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 8192 16384.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 16384.0 24576.0 8192 BF16 generic none CUDA_EVENT False q8ks24k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
227 0.0 0.0 0.0 0.0 0.0 0.04137599840760231 0.05027199909090996 0.04355199970304966 0.042399998754262924 0.0026572716942034787 14.303423881530762 14.343520164489746 14.311049747467042 14.311049747467042 0.01137511287661936 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 8192 24576.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 24576.0 32768.0 8192 BF16 generic none CUDA_EVENT False q8ks32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
228 0.0 0.0 0.0 0.0 0.0 0.026016000658273697 0.03718400001525879 0.03212799951434135 0.03270399942994118 0.003953184738275612 2.6534719467163086 2.6875839233398438 2.661257576942444 2.661257576942444 0.009372641904054613 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 4096 8192.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 True 8192.0 12288.0 4096 BF16 generic none CUDA_EVENT False q4ks12k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
229 0.0 0.0 0.0 0.0 0.0 0.025887999683618546 0.03347200155258179 0.027168000116944313 0.02649599965661764 0.0021492000999850562 4.70630407333374 4.7400641441345215 4.714492845535279 4.714492845535279 0.009349196197962609 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 4096 16384.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 True 16384.0 20480.0 4096 BF16 generic none CUDA_EVENT False q4ks20k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
230 0.0 0.0 0.0 0.0 0.0 0.025631999596953392 0.030719999223947525 0.02736639976501465 0.02711999975144863 0.0013921874495504173 6.759712219238281 6.831999778747559 6.774675178527832 6.774675178527832 0.021507023842020512 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 4096 24576.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 True 24576.0 28672.0 4096 BF16 generic none CUDA_EVENT False q4ks28k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
231 0.0 0.0 0.0 0.0 0.0 0.017855999991297722 0.024831999093294144 0.01991359982639551 0.018655999563634396 0.0023971416189370203 1.3446400165557861 1.3609600067138672 1.3508928060531615 1.3508928060531615 0.006186954712541735 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 2048 8192.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 True 8192.0 10240.0 2048 BF16 generic none CUDA_EVENT False q2ks10k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
232 0.0 0.0 0.0 0.0 0.0 0.0180479995906353 0.022752000018954277 0.019507200084626676 0.01896000001579523 0.0015079573553847933 2.5208001136779785 2.5887041091918945 2.534086418151855 2.534086418151855 0.01891953046799999 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 2048 16384.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 True 16384.0 18432.0 2048 BF16 generic none CUDA_EVENT False q2ks18k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
233 0.0 0.0 0.0 0.0 0.0 0.018144000321626663 0.03222399950027466 0.02052800003439188 0.018864000216126442 0.004114364828004189 3.6922879219055176 3.760576009750366 3.706630396842957 3.706630396842957 0.019888657921309623 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 2048 24576.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 True 24576.0 26624.0 2048 BF16 generic none CUDA_EVENT False q2ks26k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
234 0.0 0.0 0.0 0.0 0.0 0.04169600084424019 0.04822399839758873 0.0439775999635458 0.04334400035440922 0.002101844179466219 18.41494369506836 18.502975463867188 18.445004844665526 18.445004844665526 0.026939913540867878 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 40960 1 8192 32768.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 32768.0 40960.0 8192 BF16 generic none CUDA_EVENT False q8ks40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
235 0.0 0.0 0.0 0.0 0.0 0.023615999147295952 0.03267199918627739 0.026281599886715412 0.0248800003901124 0.003167969318232417 3.1188158988952637 3.1837120056152344 3.133536005020142 3.133536005020142 0.018365009771617643 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 8192 8192.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 8192.0 16384.0 8192 BF16 generic none CUDA_EVENT False q8ks16k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
236 0.0 0.0 0.0 0.0 0.0 0.02380800060927868 0.03283200040459633 0.025670399703085423 0.024639999493956566 0.0026095917641781453 5.1729278564453125 5.243135929107666 5.192828798294067 5.192828798294067 0.01984737475315108 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 8192 16384.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 16384.0 24576.0 8192 BF16 generic none CUDA_EVENT False q8ks24k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
237 0.0 0.0 0.0 0.0 0.0 0.024032000452280045 0.03167999908328056 0.02665280010551214 0.02550400048494339 0.002510131946267113 7.223167896270752 7.257152080535889 7.231532812118529 7.231532812118529 0.009418898846297825 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 8192 24576.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 24576.0 32768.0 8192 BF16 generic none CUDA_EVENT False q8ks32k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
238 0.0 0.0 0.0 0.0 0.0 0.01696000061929226 0.023679999634623528 0.01866880003362894 0.017455999739468098 0.002277860345236006 1.478559970855713 1.496127963066101 1.48257919549942 1.48257919549942 0.005278358037488471 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 4096 8192.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 True 8192.0 12288.0 4096 BF16 generic none CUDA_EVENT False q4ks12k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
239 0.0 0.0 0.0 0.0 0.0 0.017023999243974686 0.030688000842928886 0.02055360022932291 0.018240000121295452 0.004065815389665311 2.651711940765381 2.6691839694976807 2.657548785209656 2.657548785209656 0.005897264516465086 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 4096 16384.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 True 16384.0 20480.0 4096 BF16 generic none CUDA_EVENT False q4ks20k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
240 0.0 0.0 0.0 0.0 0.0 0.016767999157309532 0.024351999163627625 0.01840319987386465 0.017167999409139156 0.0025234367788448957 3.823551893234253 3.8631999492645264 3.8335999727249135 3.8335999727249135 0.011858923917429712 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 4096 24576.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 True 24576.0 28672.0 4096 BF16 generic none CUDA_EVENT False q4ks28k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
241 0.0 0.0 0.0 0.0 0.0 0.013952000066637993 0.03622400015592575 0.018214400112628936 0.014719999860972166 0.007277897466521675 0.7039039731025696 0.7163199782371521 0.7077983975410462 0.7077983975410462 0.004048424224033295 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 2048 8192.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 True 8192.0 10240.0 2048 BF16 generic none CUDA_EVENT False q2ks10k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
242 0.0 0.0 0.0 0.0 0.0 0.015104000456631184 0.03097599931061268 0.020870400313287973 0.02054399996995926 0.005422197456220526 1.2929600477218628 1.313088059425354 1.3006752014160157 1.3006752014160157 0.00790622446009414 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 2048 16384.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 True 16384.0 18432.0 2048 BF16 generic none CUDA_EVENT False q2ks18k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
243 0.0 0.0 0.0 0.0 0.0 0.014655999839305878 0.03232000023126602 0.019132800027728082 0.01673599984496832 0.0050857949387988245 1.8775999546051025 1.9125759601593018 1.8867840051651 1.8867840051651 0.01196043526228758 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 2048 24576.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 True 24576.0 26624.0 2048 BF16 generic none CUDA_EVENT False q2ks26k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
244 0.0 0.0 0.0 0.0 0.0 0.023584000766277313 0.03126399964094162 0.02633600030094385 0.025200000032782555 0.0025710957466677864 9.283391952514648 9.48249626159668 9.331705665588379 9.331705665588379 0.06366970422148335 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 40960 1 8192 32768.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 32768.0 40960.0 8192 BF16 generic none CUDA_EVENT False q8ks40k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
245 0.0 0.0 0.0 0.0 0.0 0.07552000135183334 0.0851840004324913 0.07839040011167527 0.07703999802470207 0.0030805904904383677 43.62025451660156 43.89299011230469 43.68390693664551 43.68390693664551 0.08458163192147439 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 147456 1 8192 40960.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 40960.0 49152.0 8192 BF16 generic none CUDA_EVENT False q8ks48k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
246 0.0 0.0 0.0 0.0 0.0 0.07718399912118912 0.0854400023818016 0.07895359992980958 0.07791999727487564 0.0024425064759109735 59.421791076660156 59.73017501831055 59.5021312713623 59.5021312713623 0.10367381308510447 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 147456 1 8192 57344.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 57344.0 65536.0 8192 BF16 generic none CUDA_EVENT False q8ks64k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
247 0.0 0.0 0.0 0.0 0.0 0.07875200361013412 0.09055999666452408 0.08208959847688675 0.08087999746203423 0.0033614630055883482 75.2852783203125 75.55481719970703 75.38758392333985 75.38758392333985 0.08870109119325344 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 147456 1 8192 73728.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 73728.0 81920.0 8192 BF16 generic none CUDA_EVENT False q8ks80k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
248 0.0 0.0 0.0 0.0 0.0 0.07788799703121185 0.09888000041246414 0.0819871999323368 0.08008000254631042 0.005835618489111142 91.13442993164062 91.45164489746094 91.24225158691405 91.24225158691405 0.1009018902177725 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 147456 1 8192 90112.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 90112.0 98304.0 8192 BF16 generic none CUDA_EVENT False q8ks96k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
249 0.0 0.0 0.0 0.0 0.0 0.07596799731254578 0.08790399879217148 0.07967040091753005 0.07808000221848488 0.0036209252740976605 106.89055633544922 107.34063720703125 106.9560287475586 106.9560287475586 0.13032874360676439 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 147456 1 8192 106496.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 106496.0 114688.0 8192 BF16 generic none CUDA_EVENT False q8ks112k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
250 0.0 0.0 0.0 0.0 0.0 0.07648000121116638 0.08534400165081024 0.07928640022873878 0.07787200063467026 0.0028305104107121735 122.71724700927734 122.9840316772461 122.8334243774414 122.8334243774414 0.09625274868603513 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 147456 1 8192 122880.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 122880.0 131072.0 8192 BF16 generic none CUDA_EVENT False q8ks128k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
251 0.0 0.0 0.0 0.0 0.0 0.07664000242948532 0.08774399757385254 0.08081279993057251 0.07972799986600876 0.003487270970977775 130.6565399169922 131.09359741210938 130.79229431152345 130.79229431152345 0.13138911961272914 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 147456 1 8192 131072.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 131072.0 139264.0 8192 BF16 generic none CUDA_EVENT False q8ks136k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
252 0.0 0.0 0.0 0.0 0.0 0.01532800029963255 0.040063999593257904 0.019459199998527764 0.01601599995046854 0.007496165774486326 9.443936347961426 9.649087905883789 9.489968109130858 9.489968109130858 0.05961646816296781 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 147456 1 512 130560.0 True FLASH_ATTN False vllm020_batch_spec [512] 512 512 512 512.0 True 0.0 0.0 0.0 True 130560.0 131072.0 512 BF16 generic none CUDA_EVENT False q512s128k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
253 0.0 0.0 0.0 0.0 0.0 0.02691200003027916 0.040832001715898514 0.02942080032080412 0.027888000011444092 0.004040048279091587 16.753759384155273 16.90662384033203 16.78766403198242 16.78766403198242 0.04236470233451937 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 147456 1 2048 65536.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 True 65536.0 67584.0 2048 BF16 generic none CUDA_EVENT False q2ks66k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
254 0.0 0.0 0.0 0.0 0.0 0.04416000097990036 0.04867200180888176 0.04565120078623295 0.04468800127506256 0.0018079971851529544 50.29715347290039 50.50300979614258 50.35054740905761 50.35054740905761 0.0724480602714254 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 147456 1 4096 98304.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 True 98304.0 102400.0 4096 BF16 generic none CUDA_EVENT False q4ks100k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
255 0.0 0.0 0.0 0.0 0.0 0.060447998344898224 0.06684800237417221 0.06276160031557083 0.061824001371860504 0.0024045636532566625 96.07142639160156 96.53209686279297 96.1898666381836 96.1898666381836 0.1365794879996899 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 1 147456 1 6144 131072.0 True FLASH_ATTN False vllm020_batch_spec [6144] 6144 6144 6144 6144.0 True 0.0 0.0 0.0 True 131072.0 137216.0 6144 BF16 generic none CUDA_EVENT False q6ks134k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
256 0.0 0.0 0.0 0.0 0.0 0.04217600077390671 0.05100800096988678 0.04406719990074635 0.042847998440265656 0.002656672983576728 22.5166072845459 22.986656188964844 22.63068161010742 22.63068161010742 0.13840494695459715 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 147456 1 8192 40960.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 40960.0 49152.0 8192 BF16 generic none CUDA_EVENT False q8ks48k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
257 0.0 0.0 0.0 0.0 0.0 0.04182400181889534 0.047200001776218414 0.043644800409674646 0.04262400045990944 0.001927741871473188 30.725727081298828 30.787456512451172 30.74285774230957 30.74285774230957 0.019110324860515015 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 147456 1 8192 57344.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 57344.0 65536.0 8192 BF16 generic none CUDA_EVENT False q8ks64k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
258 0.0 0.0 0.0 0.0 0.0 0.04185599833726883 0.05049600079655647 0.043852799385786054 0.04289599880576134 0.0024573088797598033 38.930206298828125 38.98448181152344 38.94538269042969 38.94538269042969 0.017734743323340796 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 147456 1 8192 73728.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 73728.0 81920.0 8192 BF16 generic none CUDA_EVENT False q8ks80k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
259 0.0 0.0 0.0 0.0 0.0 0.04227200150489807 0.046911999583244324 0.04412479996681214 0.04327999986708164 0.001811858548765738 47.13151931762695 47.245887756347656 47.148198699951166 47.148198699951166 0.033297799765613076 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 147456 1 8192 90112.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 90112.0 98304.0 8192 BF16 generic none CUDA_EVENT False q8ks96k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
260 0.0 0.0 0.0 0.0 0.0 0.04182400181889534 0.06195199862122536 0.04529919996857643 0.043136000633239746 0.005751677082163394 55.338497161865234 55.64672088623047 55.42207336425782 55.42207336425782 0.11010194068839292 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 147456 1 8192 106496.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 106496.0 114688.0 8192 BF16 generic none CUDA_EVENT False q8ks112k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
261 0.0 0.0 0.0 0.0 0.0 0.04182400181889534 0.054687999188899994 0.04418559968471527 0.042767999693751335 0.0036820249064621804 63.549087524414055 63.80697631835938 63.62084197998047 63.62084197998047 0.08872184973119365 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 147456 1 8192 122880.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 122880.0 131072.0 8192 BF16 generic none CUDA_EVENT False q8ks128k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
262 0.0 0.0 0.0 0.0 0.0 0.0461760014295578 0.06566400080919266 0.05008000023663044 0.04843199998140335 0.0055350567765421 67.64147186279297 67.91651153564453 67.70619888305666 67.70619888305666 0.0993520482760623 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 147456 1 8192 131072.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 131072.0 139264.0 8192 BF16 generic none CUDA_EVENT False q8ks136k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
263 0.0 0.0 0.0 0.0 0.0 0.023711999878287315 0.03859199956059456 0.02689919974654913 0.0244159996509552 0.004626699769228834 4.775519847869873 4.829855918884277 4.790303993225097 4.790303993225097 0.01597276061319242 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 147456 1 512 130560.0 True FLASH_ATTN False vllm020_batch_spec [512] 512 512 512 512.0 True 0.0 0.0 0.0 True 130560.0 131072.0 512 BF16 generic none CUDA_EVENT False q512s128k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
264 0.0 0.0 0.0 0.0 0.0 0.024224000051617622 0.0453759990632534 0.027676799893379213 0.025200000032782555 0.006154880946840619 9.579392433166504 9.63871955871582 9.59802885055542 9.59802885055542 0.016811667352984866 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 147456 1 2048 65536.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 True 65536.0 67584.0 2048 BF16 generic none CUDA_EVENT False q2ks66k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
265 0.0 0.0 0.0 0.0 0.0 0.02630399912595749 0.03593600168824196 0.02780479993671179 0.027040000073611736 0.0027569987978884785 25.21686363220215 25.383039474487305 25.250255966186526 25.250255966186526 0.04761963064809279 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 147456 1 4096 98304.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 True 98304.0 102400.0 4096 BF16 generic none CUDA_EVENT False q4ks100k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
266 0.0 0.0 0.0 0.0 0.0 0.034015998244285583 0.041728001087903976 0.03665280006825924 0.0352960005402565 0.002710617869728372 48.101280212402344 48.49884796142578 48.17219200134278 48.17219200134278 0.11195495368044632 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 2 147456 1 6144 131072.0 True FLASH_ATTN False vllm020_batch_spec [6144] 6144 6144 6144 6144.0 True 0.0 0.0 0.0 True 131072.0 137216.0 6144 BF16 generic none CUDA_EVENT False q6ks134k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
267 0.0 0.0 0.0 0.0 0.0 0.024191999807953835 0.039583999663591385 0.027286400087177753 0.025200000032782555 0.004620118437533604 11.32140827178955 11.496224403381348 11.354345512390138 11.354345512390138 0.050117766215439966 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 147456 1 8192 40960.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 40960.0 49152.0 8192 BF16 generic none CUDA_EVENT False q8ks48k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
268 0.0 0.0 0.0 0.0 0.0 0.024032000452280045 0.03232000023126602 0.026406400091946124 0.02518399991095066 0.002792696117638337 15.42249584197998 15.639776229858397 15.451993656158448 15.451993656158448 0.06304729308178388 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 147456 1 8192 57344.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 57344.0 65536.0 8192 BF16 generic none CUDA_EVENT False q8ks64k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
269 0.0 0.0 0.0 0.0 0.0 0.024224000051617622 0.03488000109791756 0.026147199980914592 0.02459200005978346 0.003303059079404486 19.5251522064209 19.70569610595703 19.54985942840576 19.54985942840576 0.052248914965338324 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 147456 1 8192 73728.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 73728.0 81920.0 8192 BF16 generic none CUDA_EVENT False q8ks80k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
270 0.0 0.0 0.0 0.0 0.0 0.024159999564290047 0.03046399913728237 0.0255103999748826 0.024656000547111034 0.00195717586616911 23.630016326904297 23.927040100097656 23.672822570800783 23.672822570800783 0.08715594728873713 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 147456 1 8192 90112.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 90112.0 98304.0 8192 BF16 generic none CUDA_EVENT False q8ks96k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
271 0.0 0.0 0.0 0.0 0.0 0.02380800060927868 0.03574400022625923 0.02678720001131296 0.024720000103116035 0.003678231234048846 27.736671447753906 27.79840087890625 27.747158622741697 27.747158622741697 0.018952179343788657 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 147456 1 8192 106496.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 106496.0 114688.0 8192 BF16 generic none CUDA_EVENT False q8ks112k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
272 0.0 0.0 0.0 0.0 0.0 0.024224000051617622 0.042080000042915344 0.02844479978084564 0.02527999971061945 0.006562862750709003 31.83427238464356 32.08835220336914 31.86994876861572 31.86994876861572 0.07444245241186317 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 147456 1 8192 122880.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 122880.0 131072.0 8192 BF16 generic none CUDA_EVENT False q8ks128k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
273 0.0 0.0 0.0 0.0 0.0 0.02377600036561489 0.03139200061559677 0.026396799832582474 0.025071999989449978 0.0025962810796740263 33.88313674926758 33.96985626220703 33.89708137512208 33.89708137512208 0.02518493598134421 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 147456 1 8192 131072.0 True FLASH_ATTN False vllm020_batch_spec [8192] 8192 8192 8192 8192.0 True 0.0 0.0 0.0 True 131072.0 139264.0 8192 BF16 generic none CUDA_EVENT False q8ks136k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
274 0.0 0.0 0.0 0.0 0.0 0.014944000169634819 0.022624000906944275 0.017174400109797715 0.015536000020802021 0.0028331860349340154 3.179744005203247 3.192960023880005 3.182867193222046 3.182867193222046 0.0035239767128429386 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 147456 1 512 130560.0 True FLASH_ATTN False vllm020_batch_spec [512] 512 512 512 512.0 True 0.0 0.0 0.0 True 130560.0 131072.0 512 BF16 generic none CUDA_EVENT False q512s128k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
275 0.0 0.0 0.0 0.0 0.0 0.014655999839305878 0.02985600009560585 0.017318400088697672 0.015568000264465809 0.004479309643844885 4.811935901641846 4.826848030090332 4.8173023700714115 4.8173023700714115 0.00594674080057053 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 147456 1 2048 65536.0 True FLASH_ATTN False vllm020_batch_spec [2048] 2048 2048 2048 2048.0 True 0.0 0.0 0.0 True 65536.0 67584.0 2048 BF16 generic none CUDA_EVENT False q2ks66k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
276 0.0 0.0 0.0 0.0 0.0 0.017184000462293625 0.03916800022125244 0.021356799826025962 0.01935999933630228 0.0062801287963799276 14.374591827392578 14.635007858276367 14.41227512359619 14.41227512359619 0.07569270440104603 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 147456 1 4096 98304.0 True FLASH_ATTN False vllm020_batch_spec [4096] 4096 4096 4096 4096.0 True 0.0 0.0 0.0 True 98304.0 102400.0 4096 BF16 generic none CUDA_EVENT False q4ks100k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
277 0.0 0.0 0.0 0.0 0.0 0.020479999482631683 0.033215999603271484 0.023340800032019614 0.02092800009995699 0.004304974035052475 24.088096618652344 24.18492889404297 24.11243553161621 24.11243553161621 0.02797845886790172 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 0.0 2048 32 4 16 4 147456 1 6144 131072.0 True FLASH_ATTN False vllm020_batch_spec [6144] 6144 6144 6144 6144.0 True 0.0 0.0 0.0 True 131072.0 137216.0 6144 BF16 generic none CUDA_EVENT False q6ks134k measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median

View File

@@ -0,0 +1,37 @@
time_stats.emb.min,time_stats.emb.max,time_stats.emb.mean,time_stats.emb.median,time_stats.emb.std,time_stats.input_layernorm.min,time_stats.input_layernorm.max,time_stats.input_layernorm.mean,time_stats.input_layernorm.median,time_stats.input_layernorm.std,time_stats.attn_pre_proj.min,time_stats.attn_pre_proj.max,time_stats.attn_pre_proj.mean,time_stats.attn_pre_proj.median,time_stats.attn_pre_proj.std,time_stats.attn_rope.min,time_stats.attn_rope.max,time_stats.attn_rope.mean,time_stats.attn_rope.median,time_stats.attn_rope.std,time_stats.attn_post_proj.min,time_stats.attn_post_proj.max,time_stats.attn_post_proj.mean,time_stats.attn_post_proj.median,time_stats.attn_post_proj.std,time_stats.post_attention_layernorm.min,time_stats.post_attention_layernorm.max,time_stats.post_attention_layernorm.mean,time_stats.post_attention_layernorm.median,time_stats.post_attention_layernorm.std,n_head,n_kv_head,n_embd,n_expanded_embd,vocab_size,use_gated_mlp,use_qk_norm,attn_output_gate,num_tokens,num_tensor_parallel_workers,padded_n_embd,padded_n_expanded_embd,model_arch,is_step2_mini,share_expert_dim,share_q_dim,measurement_type,profiling_precision,quant_signature
0.029184000566601753,0.06780800223350525,0.03157280012965202,0.030736000277101994,0.005874173435341216,0.033215999603271484,0.04825599864125252,0.03443359974771738,0.0337119996547699,0.0031825678429048183,1.438431978225708,1.505568027496338,1.446228802204132,1.4429279565811157,0.014128607642643025,0.538752019405365,0.5440319776535034,0.5416463971138,0.5420799851417542,0.0016099306164432847,1.0073280334472656,1.0163840055465698,1.0098415970802308,1.0081279873847961,0.0032980457197329728,0.04022400081157684,0.041728001087903976,0.04091359991580248,0.04081599973142147,0.0004728505416767937,32,4,2048,768,151936,True,True,False,8192,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
0.01360000018030405,0.06784000247716904,0.017755200061947106,0.0179840000346303,0.00837681421401443,0.01836800016462803,0.03651199862360954,0.019606399815529585,0.018719999119639397,0.0038821958848767424,0.7512000203132629,0.8208960294723511,0.7566704005002975,0.75382399559021,0.014793092586577971,0.28995200991630554,0.29337599873542786,0.29135999977588656,0.29150401055812836,0.000998381071258815,0.5149760246276855,0.5169600248336792,0.5160208016633987,0.5158880054950714,0.0005038652783291977,0.02051199972629547,0.021503999829292297,0.021067200042307378,0.021104000508785248,0.00023542855306902367,32,4,2048,768,151936,True,True,False,4096,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
0.0080960001796484,0.04281599819660187,0.011092799971811474,0.011039999779313803,0.005489135752949302,0.012223999947309494,0.026335999369621277,0.013187199970707298,0.01247999956831336,0.0030236510562153375,0.3928639888763428,0.45372799038887024,0.3976895987987518,0.39528000354766846,0.012905176716136006,0.1547199934720993,0.15884800255298615,0.15712319910526276,0.1573439985513687,0.001165121945135037,0.26633599400520325,0.268095999956131,0.26719200164079665,0.2671840041875839,0.0004242740199415328,0.013024000450968742,0.013887999579310417,0.013489600038155913,0.013520000036805868,0.0002382817554057738,32,4,2048,768,151936,True,True,False,2048,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
0.00825599953532219,0.04755200073122978,0.02398160002194345,0.01961600035429001,0.00874683840888523,0.018880000337958336,0.03868800029158592,0.021590400114655496,0.0208320003002882,0.004057015751746236,0.24316799640655518,0.2710399925708771,0.2521967992186546,0.25065599381923676,0.007998418528894075,0.09644799679517746,0.19120000302791595,0.10407840013504029,0.09963199868798256,0.020043722414992166,0.13600000739097595,0.18892799317836761,0.15760480016469955,0.15760000050067902,0.0112223677907762,0.009664000011980534,0.010015999898314476,0.009836799977347255,0.009824000298976898,9.016971952948177e-05,32,4,2048,768,151936,True,True,False,1024,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
0.017376000061631203,0.044704001396894455,0.026318399980664254,0.027312000282108784,0.007565344034680866,0.01836800016462803,0.03014400042593479,0.021276800055056812,0.020655999891459942,0.002632977855097951,0.1438719928264618,0.17132799327373505,0.152497598528862,0.15012799948453903,0.007947150319625347,0.1430719941854477,0.19305600225925446,0.16630879789590836,0.1685439944267273,0.014628024163894684,0.08899199962615967,0.10467199981212616,0.0943599995225668,0.09374399855732918,0.003472669494074113,0.007615999784320593,0.007935999892652035,0.007769599952735007,0.0077760000713169575,7.680004540222077e-05,32,4,2048,768,151936,True,True,False,512,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
0.016992000862956047,0.0544000007212162,0.02573199989274144,0.026016000658273697,0.00803323401720434,0.018432000651955605,0.02348800003528595,0.020648000109940768,0.02062400057911873,0.0013939985047930988,0.10220800340175629,0.1361600011587143,0.11850560046732425,0.11684799939393997,0.010637754898360304,0.16710400581359863,0.21110400557518005,0.19078560024499894,0.19409599900245667,0.013714352646558832,0.06265600025653839,0.0740479975938797,0.06842879951000214,0.0690080001950264,0.0031292240446560557,0.00979200005531311,0.033440001308918,0.01864320016466081,0.017280000261962414,0.006957998188876383,32,4,2048,768,151936,True,True,False,256,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
0.017152000218629837,0.04396799951791763,0.025907999789342284,0.02598400041460991,0.007714554737915267,0.018400000408291817,0.03667199984192848,0.024132800102233887,0.02112000063061714,0.006120348733354326,0.10678400099277496,0.1363839954137802,0.1193264003843069,0.11583999916911125,0.009838647443214228,0.17017599940299988,0.22748799622058868,0.18853759989142418,0.18433599919080734,0.015728078443174653,0.04569600149989128,0.06652799993753433,0.05192639995366335,0.05151999928057194,0.004516605694976449,0.021856000646948814,0.026335999369621277,0.02384479995816946,0.023599999956786633,0.0011301153013314744,32,4,2048,768,151936,True,True,False,128,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
0.017343999817967415,1.0683200359344482,0.05748240072280168,0.029504000209271908,0.16366824814254827,0.018688000738620758,0.3317759931087494,0.03685439983382821,0.021151999942958355,0.06767422466731164,0.10255999863147736,0.9434880018234253,0.16412640027701855,0.11956800147891045,0.17964606281728834,0.1714559942483902,2.1306240558624268,0.3011296011507511,0.1926399990916252,0.42310127734378766,0.03574400022625923,0.6859520077705383,0.08389280084520578,0.04279999993741512,0.14321647071615612,0.020479999482631683,0.1831360012292862,0.033024000097066165,0.023856000043451786,0.03470987082429836,32,4,2048,768,151936,True,True,False,64,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
0.016095999628305435,0.05142400041222572,0.026363200135529043,0.02676799986511469,0.008637725852473854,0.01849599927663803,0.03577600046992302,0.022193600237369538,0.020848000422120094,0.004615266181181815,0.10540799796581268,0.15014399588108063,0.12211520001292228,0.11896000057458878,0.012772013396624768,0.17315199971199036,0.21478399634361267,0.1881632000207901,0.1873439997434616,0.011761164657572015,0.03481600061058998,0.058079998940229416,0.042200000025331974,0.03969600051641464,0.006461281521769989,0.02143999934196472,0.038816001266241074,0.02466559996828437,0.023856000043451786,0.0035418033748569927,32,4,2048,768,151936,True,True,False,32,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
0.015904000028967857,0.03753599897027016,0.023127999808639287,0.02527999971061945,0.0054230027890305385,0.018400000408291817,0.03001599945127964,0.020488000102341176,0.020096000283956528,0.002585688394137697,0.10355199873447418,0.1438400000333786,0.11536479964852334,0.11124800145626068,0.011136028294919255,0.16502399742603302,0.2072959989309311,0.18646399974822997,0.19075199961662292,0.013296437761247597,0.03142400085926056,0.05215999856591225,0.03888959977775812,0.03750399872660637,0.0057399302067536314,0.020128000527620316,0.04064000025391579,0.023937600292265417,0.023648000322282314,0.004144197400898248,32,4,2048,768,151936,True,True,False,16,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
0.015200000256299973,0.037151999771595,0.02144480012357235,0.02195199951529503,0.005576364091933697,0.017952000722289085,0.0226879995316267,0.019934400077909233,0.019952000118792057,0.0011812499470458758,0.10063999891281128,0.13468800485134125,0.11595199964940547,0.1207519993185997,0.011974301621092394,0.16332800686359406,0.20748800039291382,0.18162400051951408,0.1767839938402176,0.01474120743720334,0.03222399950027466,0.043455999344587326,0.03829439990222454,0.03859200142323971,0.0031378131240041122,0.020959999412298203,0.03587200120091438,0.023795200139284135,0.023423999547958374,0.0031339489695198443,32,4,2048,768,151936,True,True,False,8,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
0.014944000169634819,0.04022400081157684,0.02162720002233982,0.022463999688625336,0.006017219317026815,0.018464000895619392,0.026528000831604004,0.021264000236988066,0.020896000787615776,0.001934095017586082,0.10156799852848053,0.1703999936580658,0.12565439902245998,0.1244799979031086,0.01641776722785409,0.1653759926557541,0.23865599930286407,0.1969360001385212,0.19223999977111816,0.02058903050274278,0.03254399821162224,0.06752000004053116,0.0443536002188921,0.041519999504089355,0.00984829071098989,0.020096000283956528,0.040031999349594116,0.026934400014579297,0.02478400059044361,0.00562071702648593,32,4,2048,768,151936,True,True,False,1,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.7531200051307678,0.8461440205574036,0.7618160009384155,0.7576479911804199,0.019464233617982506,0.2922559976577759,0.2985599935054779,0.2953856036067009,0.29576000571250916,0.001575901077689139,0.510047972202301,0.5140479803085327,0.5121696025133133,0.5123839974403381,0.0012263137313476844,,,,,,32,4,2048,768,151936,True,True,False,8192,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.39529600739479065,0.47279998660087585,0.40216960161924364,0.39825600385665894,0.0162749796240819,0.15625600516796112,0.16211199760437012,0.15959519892930984,0.15988799929618835,0.0011493418942396922,0.2635200023651123,0.2642880082130432,0.2639120012521744,0.26392000913619995,0.0001903593401384135,,,,,,32,4,2048,768,151936,True,True,False,4096,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.2531839907169342,0.32419198751449585,0.26924319565296173,0.26049599051475525,0.01657194262368116,0.10051199793815613,0.2375359982252121,0.12411200068891048,0.10311999917030334,0.039466165428540506,0.15881599485874176,0.19289599359035492,0.16896959990262986,0.1640480011701584,0.009162416086418127,,,,,,32,4,2048,768,151936,True,True,False,2048,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.15302400290966034,0.1780800074338913,0.16284480094909667,0.16113600134849548,0.00740355661356144,0.14521600306034088,0.20233599841594696,0.17807039842009545,0.1796799972653389,0.013908448560094403,0.0907519981265068,0.09750399738550186,0.09460479989647866,0.09478399902582169,0.0017965487667361475,,,,,,32,4,2048,768,151936,True,True,False,1024,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.11740799993276596,0.17958399653434753,0.13262080028653145,0.12878400087356567,0.014146060532423056,0.17468799650669098,0.21161599457263947,0.1910431995987892,0.18966399878263474,0.01270903647700971,0.05926400050520897,0.07577600330114365,0.06530559975653887,0.06404799968004227,0.00479458462514625,,,,,,32,4,2048,768,151936,True,True,False,512,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.11260800063610077,0.17209599912166595,0.13610880002379416,0.13809599727392197,0.017675922458993677,0.18729600310325623,0.26822400093078613,0.20815680101513861,0.2078079953789711,0.01700718593658771,0.04499199986457825,0.06537599861621857,0.0507551996037364,0.04787199944257736,0.005727123232692158,,,,,,32,4,2048,768,151936,True,True,False,256,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.11257600039243698,0.23452800512313843,0.13866880126297473,0.13964799791574478,0.026590047697062115,0.1828799992799759,0.24751999974250793,0.20735519900918006,0.20670399814844131,0.01500438131586336,0.03654399886727333,0.05593600124120712,0.0408239996060729,0.03892800025641918,0.004864409450710354,,,,,,32,4,2048,768,151936,True,True,False,128,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.11734399944543839,0.18726399540901184,0.13890240006148816,0.13308800011873245,0.019086167340006257,0.16991999745368958,0.2443840056657791,0.19185120090842248,0.19257599860429764,0.018017536159219673,0.03033600002527237,0.04956800118088722,0.038387199863791466,0.03750400058925152,0.00500897018702887,,,,,,32,4,2048,768,151936,True,True,False,64,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.11420799791812897,0.18115200102329254,0.13938880078494548,0.1393439993262291,0.018141313962922536,0.1693439930677414,0.221343994140625,0.19187839925289155,0.19438399374485016,0.01430752121272052,0.03017600066959858,0.06019200012087822,0.03866560012102127,0.0364960003644228,0.0074199790031205266,,,,,,32,4,2048,768,151936,True,True,False,32,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.11276800185441971,0.16223999857902527,0.1331360016018152,0.13305599987506866,0.01432493740801146,0.17187200486660004,0.23625600337982178,0.19287680014967917,0.1913280040025711,0.01754628831589704,0.029823999851942062,0.07574400305747986,0.03900959976017475,0.036927999928593636,0.009277465083962879,,,,,,32,4,2048,768,151936,True,True,False,16,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.1130559965968132,0.19814400374889374,0.13379519879817964,0.12531199678778648,0.020074427302248836,0.16841599345207214,0.21139200031757355,0.1893615983426571,0.19092799723148346,0.012418720676762711,0.029503999277949333,0.07558400183916092,0.040144000016152856,0.03728000074625015,0.010223239196498205,,,,,,32,4,2048,768,151936,True,True,False,8,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.115167997777462,0.2640640139579773,0.13805920109152794,0.13118399679660797,0.03132849683515479,0.17103999853134155,0.2977280020713806,0.19899839907884598,0.1976960003376007,0.028928046763067948,0.0297279991209507,0.05990400165319443,0.03834720011800528,0.0363520011305809,0.0070885330713495905,,,,,,32,4,2048,768,151936,True,True,False,1,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.3940800130367279,0.46483200788497925,0.39923040121793746,0.3957759886980057,0.015108903906233569,0.16022400557994843,0.1634880006313324,0.16203359961509706,0.16228799521923065,0.0010220720912414007,0.26073598861694336,0.2627840042114258,0.26138080209493636,0.2613760083913803,0.00041640363494172775,,,,,,32,4,2048,768,151936,True,True,False,8192,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.2536959946155548,0.27161601185798645,0.25960480123758317,0.2577280104160309,0.005142017404245015,0.09849599748849869,0.10281600058078766,0.10057279989123344,0.10063999891281128,0.0008836232237268523,0.13913600146770477,0.17606399953365326,0.15875840038061143,0.15988799929618835,0.010058952112389172,,,,,,32,4,2048,768,151936,True,True,False,4096,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.15142400562763214,0.1780479997396469,0.16139679849147798,0.1602879986166954,0.0070656055731385115,0.14601600170135498,0.23343999683856964,0.17913119941949845,0.179967999458313,0.023065274309176566,0.09388799965381622,0.12108799815177917,0.10317599996924401,0.10073599964380264,0.007364345618470178,,,,,,32,4,2048,768,151936,True,True,False,2048,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.12198399752378464,0.18406400084495544,0.13770400024950505,0.13232000172138214,0.01623164885687898,0.1773120015859604,0.20688000321388245,0.1870912007987499,0.18535999953746796,0.00825628058448234,0.05766399949789047,0.06739199906587601,0.06187200043350458,0.061216000467538834,0.003004312467037863,,,,,,32,4,2048,768,151936,True,True,False,1024,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.11737599968910217,0.18614399433135986,0.13109439946711063,0.12771200388669968,0.01372168725934724,0.17587199807167053,0.2699519991874695,0.19926720038056372,0.19075199961662292,0.02327478982490538,0.041728001087903976,0.06224000081419945,0.05000480003654957,0.049375999718904495,0.006185850944273961,,,,,,32,4,2048,768,151936,True,True,False,512,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.1159679964184761,0.1610880047082901,0.12988320142030715,0.12494400516152382,0.011865375783184065,0.1767680048942566,0.27379199862480164,0.21035519987344742,0.21275199949741364,0.023206964017550125,0.037087999284267426,0.062144000083208084,0.04439679980278015,0.04283200018107891,0.006155226392511492,,,,,,32,4,2048,768,151936,True,True,False,256,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.12015999853610992,0.158720001578331,0.13573280088603495,0.1343199983239174,0.010666830836014414,0.1737920045852661,0.22127999365329742,0.20261440128087999,0.20321600139141083,0.011153965849067301,0.03587200120091438,0.04944000020623207,0.038265600241720675,0.037328001111745834,0.0030431580113575636,,,,,,32,4,2048,768,151936,True,True,False,128,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.1212799996137619,0.15139199793338776,0.13183839991688728,0.13014400005340576,0.00851207547022104,0.17257599532604218,0.2250880002975464,0.1872655987739563,0.18193599581718445,0.013638284975248818,0.02969600073993206,0.0525440014898777,0.03840640028938651,0.03444799967110157,0.007598511754254174,,,,,,32,4,2048,768,151936,True,True,False,64,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.11270400136709213,0.1624639928340912,0.13508000001311302,0.13180799782276154,0.014243564451606775,0.17052799463272095,0.23715199530124664,0.1953311987221241,0.1966560035943985,0.01500670718978398,0.030239999294281006,0.05270399898290634,0.03807039987295866,0.03742399998009205,0.004906165602021684,,,,,,32,4,2048,768,151936,True,True,False,32,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.12319999933242798,0.17919999361038208,0.14059039913117885,0.14156799763441086,0.016217662982107223,0.1701119989156723,0.21987199783325195,0.18903039917349815,0.1870879977941513,0.014117854507841239,0.030368000268936157,0.06406400352716446,0.04045119984075427,0.03710399940609932,0.00878942205983628,,,,,,32,4,2048,768,151936,True,True,False,16,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.11296000331640244,0.16761599481105804,0.13843199908733367,0.13814399391412735,0.014925286935582404,0.1711679995059967,0.21561600267887115,0.1906527981162071,0.1876479983329773,0.011803992622137974,0.0306560005992651,0.09216000139713287,0.041129599791020155,0.036847999319434166,0.013739721468154687,,,,,,32,4,2048,768,151936,True,True,False,8,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
,,,,,,,,,,0.11929599940776825,0.2008959949016571,0.14535359852015972,0.13971199840307236,0.024769473061576282,0.16803200542926788,0.25123199820518494,0.19573760256171227,0.19366399943828583,0.023196011081705884,0.03049599938094616,0.08089599758386612,0.04739360017701984,0.0453919991850853,0.01286389846889463,,,,,,32,4,2048,768,151936,True,True,False,1,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
1 time_stats.emb.min time_stats.emb.max time_stats.emb.mean time_stats.emb.median time_stats.emb.std time_stats.input_layernorm.min time_stats.input_layernorm.max time_stats.input_layernorm.mean time_stats.input_layernorm.median time_stats.input_layernorm.std time_stats.attn_pre_proj.min time_stats.attn_pre_proj.max time_stats.attn_pre_proj.mean time_stats.attn_pre_proj.median time_stats.attn_pre_proj.std time_stats.attn_rope.min time_stats.attn_rope.max time_stats.attn_rope.mean time_stats.attn_rope.median time_stats.attn_rope.std time_stats.attn_post_proj.min time_stats.attn_post_proj.max time_stats.attn_post_proj.mean time_stats.attn_post_proj.median time_stats.attn_post_proj.std time_stats.post_attention_layernorm.min time_stats.post_attention_layernorm.max time_stats.post_attention_layernorm.mean time_stats.post_attention_layernorm.median time_stats.post_attention_layernorm.std n_head n_kv_head n_embd n_expanded_embd vocab_size use_gated_mlp use_qk_norm attn_output_gate num_tokens num_tensor_parallel_workers padded_n_embd padded_n_expanded_embd model_arch is_step2_mini share_expert_dim share_q_dim measurement_type profiling_precision quant_signature
2 0.029184000566601753 0.06780800223350525 0.03157280012965202 0.030736000277101994 0.005874173435341216 0.033215999603271484 0.04825599864125252 0.03443359974771738 0.0337119996547699 0.0031825678429048183 1.438431978225708 1.505568027496338 1.446228802204132 1.4429279565811157 0.014128607642643025 0.538752019405365 0.5440319776535034 0.5416463971138 0.5420799851417542 0.0016099306164432847 1.0073280334472656 1.0163840055465698 1.0098415970802308 1.0081279873847961 0.0032980457197329728 0.04022400081157684 0.041728001087903976 0.04091359991580248 0.04081599973142147 0.0004728505416767937 32 4 2048 768 151936 True True False 8192 1 2048 768 generic False CUDA_EVENT BF16 none
3 0.01360000018030405 0.06784000247716904 0.017755200061947106 0.0179840000346303 0.00837681421401443 0.01836800016462803 0.03651199862360954 0.019606399815529585 0.018719999119639397 0.0038821958848767424 0.7512000203132629 0.8208960294723511 0.7566704005002975 0.75382399559021 0.014793092586577971 0.28995200991630554 0.29337599873542786 0.29135999977588656 0.29150401055812836 0.000998381071258815 0.5149760246276855 0.5169600248336792 0.5160208016633987 0.5158880054950714 0.0005038652783291977 0.02051199972629547 0.021503999829292297 0.021067200042307378 0.021104000508785248 0.00023542855306902367 32 4 2048 768 151936 True True False 4096 1 2048 768 generic False CUDA_EVENT BF16 none
4 0.0080960001796484 0.04281599819660187 0.011092799971811474 0.011039999779313803 0.005489135752949302 0.012223999947309494 0.026335999369621277 0.013187199970707298 0.01247999956831336 0.0030236510562153375 0.3928639888763428 0.45372799038887024 0.3976895987987518 0.39528000354766846 0.012905176716136006 0.1547199934720993 0.15884800255298615 0.15712319910526276 0.1573439985513687 0.001165121945135037 0.26633599400520325 0.268095999956131 0.26719200164079665 0.2671840041875839 0.0004242740199415328 0.013024000450968742 0.013887999579310417 0.013489600038155913 0.013520000036805868 0.0002382817554057738 32 4 2048 768 151936 True True False 2048 1 2048 768 generic False CUDA_EVENT BF16 none
5 0.00825599953532219 0.04755200073122978 0.02398160002194345 0.01961600035429001 0.00874683840888523 0.018880000337958336 0.03868800029158592 0.021590400114655496 0.0208320003002882 0.004057015751746236 0.24316799640655518 0.2710399925708771 0.2521967992186546 0.25065599381923676 0.007998418528894075 0.09644799679517746 0.19120000302791595 0.10407840013504029 0.09963199868798256 0.020043722414992166 0.13600000739097595 0.18892799317836761 0.15760480016469955 0.15760000050067902 0.0112223677907762 0.009664000011980534 0.010015999898314476 0.009836799977347255 0.009824000298976898 9.016971952948177e-05 32 4 2048 768 151936 True True False 1024 1 2048 768 generic False CUDA_EVENT BF16 none
6 0.017376000061631203 0.044704001396894455 0.026318399980664254 0.027312000282108784 0.007565344034680866 0.01836800016462803 0.03014400042593479 0.021276800055056812 0.020655999891459942 0.002632977855097951 0.1438719928264618 0.17132799327373505 0.152497598528862 0.15012799948453903 0.007947150319625347 0.1430719941854477 0.19305600225925446 0.16630879789590836 0.1685439944267273 0.014628024163894684 0.08899199962615967 0.10467199981212616 0.0943599995225668 0.09374399855732918 0.003472669494074113 0.007615999784320593 0.007935999892652035 0.007769599952735007 0.0077760000713169575 7.680004540222077e-05 32 4 2048 768 151936 True True False 512 1 2048 768 generic False CUDA_EVENT BF16 none
7 0.016992000862956047 0.0544000007212162 0.02573199989274144 0.026016000658273697 0.00803323401720434 0.018432000651955605 0.02348800003528595 0.020648000109940768 0.02062400057911873 0.0013939985047930988 0.10220800340175629 0.1361600011587143 0.11850560046732425 0.11684799939393997 0.010637754898360304 0.16710400581359863 0.21110400557518005 0.19078560024499894 0.19409599900245667 0.013714352646558832 0.06265600025653839 0.0740479975938797 0.06842879951000214 0.0690080001950264 0.0031292240446560557 0.00979200005531311 0.033440001308918 0.01864320016466081 0.017280000261962414 0.006957998188876383 32 4 2048 768 151936 True True False 256 1 2048 768 generic False CUDA_EVENT BF16 none
8 0.017152000218629837 0.04396799951791763 0.025907999789342284 0.02598400041460991 0.007714554737915267 0.018400000408291817 0.03667199984192848 0.024132800102233887 0.02112000063061714 0.006120348733354326 0.10678400099277496 0.1363839954137802 0.1193264003843069 0.11583999916911125 0.009838647443214228 0.17017599940299988 0.22748799622058868 0.18853759989142418 0.18433599919080734 0.015728078443174653 0.04569600149989128 0.06652799993753433 0.05192639995366335 0.05151999928057194 0.004516605694976449 0.021856000646948814 0.026335999369621277 0.02384479995816946 0.023599999956786633 0.0011301153013314744 32 4 2048 768 151936 True True False 128 1 2048 768 generic False CUDA_EVENT BF16 none
9 0.017343999817967415 1.0683200359344482 0.05748240072280168 0.029504000209271908 0.16366824814254827 0.018688000738620758 0.3317759931087494 0.03685439983382821 0.021151999942958355 0.06767422466731164 0.10255999863147736 0.9434880018234253 0.16412640027701855 0.11956800147891045 0.17964606281728834 0.1714559942483902 2.1306240558624268 0.3011296011507511 0.1926399990916252 0.42310127734378766 0.03574400022625923 0.6859520077705383 0.08389280084520578 0.04279999993741512 0.14321647071615612 0.020479999482631683 0.1831360012292862 0.033024000097066165 0.023856000043451786 0.03470987082429836 32 4 2048 768 151936 True True False 64 1 2048 768 generic False CUDA_EVENT BF16 none
10 0.016095999628305435 0.05142400041222572 0.026363200135529043 0.02676799986511469 0.008637725852473854 0.01849599927663803 0.03577600046992302 0.022193600237369538 0.020848000422120094 0.004615266181181815 0.10540799796581268 0.15014399588108063 0.12211520001292228 0.11896000057458878 0.012772013396624768 0.17315199971199036 0.21478399634361267 0.1881632000207901 0.1873439997434616 0.011761164657572015 0.03481600061058998 0.058079998940229416 0.042200000025331974 0.03969600051641464 0.006461281521769989 0.02143999934196472 0.038816001266241074 0.02466559996828437 0.023856000043451786 0.0035418033748569927 32 4 2048 768 151936 True True False 32 1 2048 768 generic False CUDA_EVENT BF16 none
11 0.015904000028967857 0.03753599897027016 0.023127999808639287 0.02527999971061945 0.0054230027890305385 0.018400000408291817 0.03001599945127964 0.020488000102341176 0.020096000283956528 0.002585688394137697 0.10355199873447418 0.1438400000333786 0.11536479964852334 0.11124800145626068 0.011136028294919255 0.16502399742603302 0.2072959989309311 0.18646399974822997 0.19075199961662292 0.013296437761247597 0.03142400085926056 0.05215999856591225 0.03888959977775812 0.03750399872660637 0.0057399302067536314 0.020128000527620316 0.04064000025391579 0.023937600292265417 0.023648000322282314 0.004144197400898248 32 4 2048 768 151936 True True False 16 1 2048 768 generic False CUDA_EVENT BF16 none
12 0.015200000256299973 0.037151999771595 0.02144480012357235 0.02195199951529503 0.005576364091933697 0.017952000722289085 0.0226879995316267 0.019934400077909233 0.019952000118792057 0.0011812499470458758 0.10063999891281128 0.13468800485134125 0.11595199964940547 0.1207519993185997 0.011974301621092394 0.16332800686359406 0.20748800039291382 0.18162400051951408 0.1767839938402176 0.01474120743720334 0.03222399950027466 0.043455999344587326 0.03829439990222454 0.03859200142323971 0.0031378131240041122 0.020959999412298203 0.03587200120091438 0.023795200139284135 0.023423999547958374 0.0031339489695198443 32 4 2048 768 151936 True True False 8 1 2048 768 generic False CUDA_EVENT BF16 none
13 0.014944000169634819 0.04022400081157684 0.02162720002233982 0.022463999688625336 0.006017219317026815 0.018464000895619392 0.026528000831604004 0.021264000236988066 0.020896000787615776 0.001934095017586082 0.10156799852848053 0.1703999936580658 0.12565439902245998 0.1244799979031086 0.01641776722785409 0.1653759926557541 0.23865599930286407 0.1969360001385212 0.19223999977111816 0.02058903050274278 0.03254399821162224 0.06752000004053116 0.0443536002188921 0.041519999504089355 0.00984829071098989 0.020096000283956528 0.040031999349594116 0.026934400014579297 0.02478400059044361 0.00562071702648593 32 4 2048 768 151936 True True False 1 1 2048 768 generic False CUDA_EVENT BF16 none
14 0.7531200051307678 0.8461440205574036 0.7618160009384155 0.7576479911804199 0.019464233617982506 0.2922559976577759 0.2985599935054779 0.2953856036067009 0.29576000571250916 0.001575901077689139 0.510047972202301 0.5140479803085327 0.5121696025133133 0.5123839974403381 0.0012263137313476844 32 4 2048 768 151936 True True False 8192 2 2048 768 generic False CUDA_EVENT BF16 none
15 0.39529600739479065 0.47279998660087585 0.40216960161924364 0.39825600385665894 0.0162749796240819 0.15625600516796112 0.16211199760437012 0.15959519892930984 0.15988799929618835 0.0011493418942396922 0.2635200023651123 0.2642880082130432 0.2639120012521744 0.26392000913619995 0.0001903593401384135 32 4 2048 768 151936 True True False 4096 2 2048 768 generic False CUDA_EVENT BF16 none
16 0.2531839907169342 0.32419198751449585 0.26924319565296173 0.26049599051475525 0.01657194262368116 0.10051199793815613 0.2375359982252121 0.12411200068891048 0.10311999917030334 0.039466165428540506 0.15881599485874176 0.19289599359035492 0.16896959990262986 0.1640480011701584 0.009162416086418127 32 4 2048 768 151936 True True False 2048 2 2048 768 generic False CUDA_EVENT BF16 none
17 0.15302400290966034 0.1780800074338913 0.16284480094909667 0.16113600134849548 0.00740355661356144 0.14521600306034088 0.20233599841594696 0.17807039842009545 0.1796799972653389 0.013908448560094403 0.0907519981265068 0.09750399738550186 0.09460479989647866 0.09478399902582169 0.0017965487667361475 32 4 2048 768 151936 True True False 1024 2 2048 768 generic False CUDA_EVENT BF16 none
18 0.11740799993276596 0.17958399653434753 0.13262080028653145 0.12878400087356567 0.014146060532423056 0.17468799650669098 0.21161599457263947 0.1910431995987892 0.18966399878263474 0.01270903647700971 0.05926400050520897 0.07577600330114365 0.06530559975653887 0.06404799968004227 0.00479458462514625 32 4 2048 768 151936 True True False 512 2 2048 768 generic False CUDA_EVENT BF16 none
19 0.11260800063610077 0.17209599912166595 0.13610880002379416 0.13809599727392197 0.017675922458993677 0.18729600310325623 0.26822400093078613 0.20815680101513861 0.2078079953789711 0.01700718593658771 0.04499199986457825 0.06537599861621857 0.0507551996037364 0.04787199944257736 0.005727123232692158 32 4 2048 768 151936 True True False 256 2 2048 768 generic False CUDA_EVENT BF16 none
20 0.11257600039243698 0.23452800512313843 0.13866880126297473 0.13964799791574478 0.026590047697062115 0.1828799992799759 0.24751999974250793 0.20735519900918006 0.20670399814844131 0.01500438131586336 0.03654399886727333 0.05593600124120712 0.0408239996060729 0.03892800025641918 0.004864409450710354 32 4 2048 768 151936 True True False 128 2 2048 768 generic False CUDA_EVENT BF16 none
21 0.11734399944543839 0.18726399540901184 0.13890240006148816 0.13308800011873245 0.019086167340006257 0.16991999745368958 0.2443840056657791 0.19185120090842248 0.19257599860429764 0.018017536159219673 0.03033600002527237 0.04956800118088722 0.038387199863791466 0.03750400058925152 0.00500897018702887 32 4 2048 768 151936 True True False 64 2 2048 768 generic False CUDA_EVENT BF16 none
22 0.11420799791812897 0.18115200102329254 0.13938880078494548 0.1393439993262291 0.018141313962922536 0.1693439930677414 0.221343994140625 0.19187839925289155 0.19438399374485016 0.01430752121272052 0.03017600066959858 0.06019200012087822 0.03866560012102127 0.0364960003644228 0.0074199790031205266 32 4 2048 768 151936 True True False 32 2 2048 768 generic False CUDA_EVENT BF16 none
23 0.11276800185441971 0.16223999857902527 0.1331360016018152 0.13305599987506866 0.01432493740801146 0.17187200486660004 0.23625600337982178 0.19287680014967917 0.1913280040025711 0.01754628831589704 0.029823999851942062 0.07574400305747986 0.03900959976017475 0.036927999928593636 0.009277465083962879 32 4 2048 768 151936 True True False 16 2 2048 768 generic False CUDA_EVENT BF16 none
24 0.1130559965968132 0.19814400374889374 0.13379519879817964 0.12531199678778648 0.020074427302248836 0.16841599345207214 0.21139200031757355 0.1893615983426571 0.19092799723148346 0.012418720676762711 0.029503999277949333 0.07558400183916092 0.040144000016152856 0.03728000074625015 0.010223239196498205 32 4 2048 768 151936 True True False 8 2 2048 768 generic False CUDA_EVENT BF16 none
25 0.115167997777462 0.2640640139579773 0.13805920109152794 0.13118399679660797 0.03132849683515479 0.17103999853134155 0.2977280020713806 0.19899839907884598 0.1976960003376007 0.028928046763067948 0.0297279991209507 0.05990400165319443 0.03834720011800528 0.0363520011305809 0.0070885330713495905 32 4 2048 768 151936 True True False 1 2 2048 768 generic False CUDA_EVENT BF16 none
26 0.3940800130367279 0.46483200788497925 0.39923040121793746 0.3957759886980057 0.015108903906233569 0.16022400557994843 0.1634880006313324 0.16203359961509706 0.16228799521923065 0.0010220720912414007 0.26073598861694336 0.2627840042114258 0.26138080209493636 0.2613760083913803 0.00041640363494172775 32 4 2048 768 151936 True True False 8192 4 2048 768 generic False CUDA_EVENT BF16 none
27 0.2536959946155548 0.27161601185798645 0.25960480123758317 0.2577280104160309 0.005142017404245015 0.09849599748849869 0.10281600058078766 0.10057279989123344 0.10063999891281128 0.0008836232237268523 0.13913600146770477 0.17606399953365326 0.15875840038061143 0.15988799929618835 0.010058952112389172 32 4 2048 768 151936 True True False 4096 4 2048 768 generic False CUDA_EVENT BF16 none
28 0.15142400562763214 0.1780479997396469 0.16139679849147798 0.1602879986166954 0.0070656055731385115 0.14601600170135498 0.23343999683856964 0.17913119941949845 0.179967999458313 0.023065274309176566 0.09388799965381622 0.12108799815177917 0.10317599996924401 0.10073599964380264 0.007364345618470178 32 4 2048 768 151936 True True False 2048 4 2048 768 generic False CUDA_EVENT BF16 none
29 0.12198399752378464 0.18406400084495544 0.13770400024950505 0.13232000172138214 0.01623164885687898 0.1773120015859604 0.20688000321388245 0.1870912007987499 0.18535999953746796 0.00825628058448234 0.05766399949789047 0.06739199906587601 0.06187200043350458 0.061216000467538834 0.003004312467037863 32 4 2048 768 151936 True True False 1024 4 2048 768 generic False CUDA_EVENT BF16 none
30 0.11737599968910217 0.18614399433135986 0.13109439946711063 0.12771200388669968 0.01372168725934724 0.17587199807167053 0.2699519991874695 0.19926720038056372 0.19075199961662292 0.02327478982490538 0.041728001087903976 0.06224000081419945 0.05000480003654957 0.049375999718904495 0.006185850944273961 32 4 2048 768 151936 True True False 512 4 2048 768 generic False CUDA_EVENT BF16 none
31 0.1159679964184761 0.1610880047082901 0.12988320142030715 0.12494400516152382 0.011865375783184065 0.1767680048942566 0.27379199862480164 0.21035519987344742 0.21275199949741364 0.023206964017550125 0.037087999284267426 0.062144000083208084 0.04439679980278015 0.04283200018107891 0.006155226392511492 32 4 2048 768 151936 True True False 256 4 2048 768 generic False CUDA_EVENT BF16 none
32 0.12015999853610992 0.158720001578331 0.13573280088603495 0.1343199983239174 0.010666830836014414 0.1737920045852661 0.22127999365329742 0.20261440128087999 0.20321600139141083 0.011153965849067301 0.03587200120091438 0.04944000020623207 0.038265600241720675 0.037328001111745834 0.0030431580113575636 32 4 2048 768 151936 True True False 128 4 2048 768 generic False CUDA_EVENT BF16 none
33 0.1212799996137619 0.15139199793338776 0.13183839991688728 0.13014400005340576 0.00851207547022104 0.17257599532604218 0.2250880002975464 0.1872655987739563 0.18193599581718445 0.013638284975248818 0.02969600073993206 0.0525440014898777 0.03840640028938651 0.03444799967110157 0.007598511754254174 32 4 2048 768 151936 True True False 64 4 2048 768 generic False CUDA_EVENT BF16 none
34 0.11270400136709213 0.1624639928340912 0.13508000001311302 0.13180799782276154 0.014243564451606775 0.17052799463272095 0.23715199530124664 0.1953311987221241 0.1966560035943985 0.01500670718978398 0.030239999294281006 0.05270399898290634 0.03807039987295866 0.03742399998009205 0.004906165602021684 32 4 2048 768 151936 True True False 32 4 2048 768 generic False CUDA_EVENT BF16 none
35 0.12319999933242798 0.17919999361038208 0.14059039913117885 0.14156799763441086 0.016217662982107223 0.1701119989156723 0.21987199783325195 0.18903039917349815 0.1870879977941513 0.014117854507841239 0.030368000268936157 0.06406400352716446 0.04045119984075427 0.03710399940609932 0.00878942205983628 32 4 2048 768 151936 True True False 16 4 2048 768 generic False CUDA_EVENT BF16 none
36 0.11296000331640244 0.16761599481105804 0.13843199908733367 0.13814399391412735 0.014925286935582404 0.1711679995059967 0.21561600267887115 0.1906527981162071 0.1876479983329773 0.011803992622137974 0.0306560005992651 0.09216000139713287 0.041129599791020155 0.036847999319434166 0.013739721468154687 32 4 2048 768 151936 True True False 8 4 2048 768 generic False CUDA_EVENT BF16 none
37 0.11929599940776825 0.2008959949016571 0.14535359852015972 0.13971199840307236 0.024769473061576282 0.16803200542926788 0.25123199820518494 0.19573760256171227 0.19366399943828583 0.023196011081705884 0.03049599938094616 0.08089599758386612 0.04739360017701984 0.0453919991850853 0.01286389846889463 32 4 2048 768 151936 True True False 1 4 2048 768 generic False CUDA_EVENT BF16 none

View File

@@ -0,0 +1,21 @@
{
"schema": "frontier-profile-v6-code-longctx-v1",
"base": "runs/frontier-prefill-kvgrowth-fix-v0/profiles/profile-v5-kvgrowth/attention.csv",
"base_sha256": "ff32c38975e68565c85770d85b92bc310ab1d8e34fd02327bef7fb305a9ffae3",
"raw_inputs": {
"runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/raw/flashattn-code-longctx-tp1.json": "3937c623adcea3d379b82bb7ab63d3b293f917fa59fa6fd6a147581d6a252cb6",
"runs/frontier-code-trace-v0/profiles/raw/tp2-r1-v2/raw/flashattn-code-longctx-tp2.json": "e48aa6e6d039bdb68759701c9976f78ad818d6778b6c40b14823a0a4e852c52d",
"runs/frontier-code-trace-v0/profiles/raw/tp4-r1-v2/raw/flashattn-code-longctx-tp4.json": "759d33decc09d229d07818df8161ed41180cab575304f335ad4d5e7a917cbc93"
},
"appended_rows": 33,
"max_model_len": 147456,
"anchor_checks": [
"anchor q1ks8k/TP1: v4=1.1445ms new=1.1426ms rel_diff=0.2%",
"anchor q512s4k/TP1: v4=0.3391ms new=0.3367ms rel_diff=0.7%",
"anchor q1ks8k/TP2: v4=0.6126ms new=0.6065ms rel_diff=1.0%",
"anchor q512s4k/TP2: v4=0.2086ms new=0.2079ms rel_diff=0.3%",
"anchor q1ks8k/TP4: v4=0.3445ms new=0.3470ms rel_diff=0.7%",
"anchor q512s4k/TP4: v4=0.1536ms new=0.1510ms rel_diff=1.7%"
],
"output_sha256": "fbcf7e1f95789a6f6d771e24d1fc60958b7daf04eb0db260d27869a19d71d550"
}

View File

@@ -0,0 +1,73 @@
time_stats.moe_gating_linear.min,time_stats.moe_gating_linear.max,time_stats.moe_gating_linear.mean,time_stats.moe_gating_linear.median,time_stats.moe_gating_linear.std,time_stats.moe_gating_routing_topk.min,time_stats.moe_gating_routing_topk.max,time_stats.moe_gating_routing_topk.mean,time_stats.moe_gating_routing_topk.median,time_stats.moe_gating_routing_topk.std,time_stats.moe_shuffling.min,time_stats.moe_shuffling.max,time_stats.moe_shuffling.mean,time_stats.moe_shuffling.median,time_stats.moe_shuffling.std,time_stats.moe_grouped_gemm.min,time_stats.moe_grouped_gemm.max,time_stats.moe_grouped_gemm.mean,time_stats.moe_grouped_gemm.median,time_stats.moe_grouped_gemm.std,num_tokens,num_experts,num_experts_per_device,expert_parallel_size,routing_runtime_path,routing_assignment_policy,routing_weight_policy,routing_uses_router_logits,gating_runtime_context,gating_runtime_context_impl,router_topk,hidden_dim,expert_hidden_dim,use_gated,num_tensor_parallel_workers,total_routed_tokens,model_expansion_ratio,tokens_per_expert_avg,tokens_to_experts_ratio,expert_utilization,min_load_ratio,load_imbalance_cv,max_load_ratio,load_entropy,load_gini_coefficient,load_distribution,seed,moe_grouped_gemm_backend,measurement_type,profiling_precision,model_arch,quant_signature,router_median_nonadditivity_ratio,projection_policy
0.02502400055527687,0.06092799827456474,0.033839999698102474,0.028672000393271446,0.010315255343709818,0.019360000267624855,0.0352960005402565,0.023424000293016434,0.022064000368118286,0.0038898102289194572,0.0,0.0,0.0,0.0,0.0,0.33926400542259216,0.405023992061615,0.36780479848384856,0.36507199704647064,0.01690507644474779,1,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,8,0.375,0.0625,0.0625,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0233364439829928,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.025280000641942024,0.05132799968123436,0.030641599837690593,0.027343999594449997,0.007562409232763304,0.020160000771284103,0.052960000932216644,0.024145600199699403,0.021424000151455402,0.007450198645599985,0.0,0.0,0.0,0.0,0.0,1.1943039894104004,1.286784052848816,1.228384006023407,1.2273280024528503,0.02832547242381263,8,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,64,0.375,0.5,0.5,0.3984375,0.0,1.346291201783626,4.0,5.59375,0.661865234375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9806430689981738,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024831999093294144,0.046560000628232956,0.030371200107038022,0.02763199992477894,0.005878487205028602,0.020320000126957893,0.04560000076889992,0.02601920012384653,0.02270400058478117,0.00751129965906799,0.0,0.0,0.0,0.0,0.0,1.679744005203247,1.766144037246704,1.7095808148384095,1.7015680074691772,0.02438921262998535,16,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,128,0.375,1.0,1.0,0.625,0.0,1.015504800579495,5.0,6.15516433212955,0.529052734375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9103623678483975,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024288000538945198,0.049375999718904495,0.03086080001667142,0.0267359996214509,0.0070864531384997225,0.020479999482631683,0.030912000685930252,0.02274719988927245,0.021359999664127827,0.0029966813650672505,0.0,0.0,0.0,0.0,0.0,2.1576640605926514,2.2921600341796875,2.2097824096679686,2.188944101333618,0.045572321842012986,32,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,256,0.375,2.0,2.0,0.875,0.0,0.6343057228182637,2.5,6.64370748444639,0.35369873046875,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9610778571819444,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02393599972128868,0.04342399910092354,0.029380799923092126,0.026688000187277794,0.0057374782000526574,0.020031999796628952,0.036607999354600906,0.022487999964505435,0.020911999978125095,0.0038135203062103235,0.0,0.0,0.0,0.0,0.0,2.422368049621582,2.5130879878997803,2.4516672134399413,2.434159994125366,0.03278900287381846,64,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,512,0.375,4.0,4.0,0.984375,0.0,0.4921254921257382,2.25,6.817190042344769,0.272369384765625,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9952941013961014,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.025407999753952026,0.05510399863123894,0.031430399790406224,0.02798399981111288,0.007335050273982725,0.020479999482631683,0.03561599925160408,0.02275839988142252,0.021551999263465405,0.003545718365165811,0.0,0.0,0.0,0.0,0.0,2.2217600345611572,2.289599895477295,2.2571327924728393,2.263375997543335,0.021660416089449488,128,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,1024,0.375,8.0,8.0,1.0,0.125,0.3486861500690843,1.875,6.908192310183997,0.197662353515625,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9273256282883522,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.0244159996509552,0.05395200103521347,0.03188959984108806,0.026031999848783016,0.00943995927387483,0.02051199972629547,0.036607999354600906,0.023247999791055917,0.02147199958562851,0.003910623837069648,0.0,0.0,0.0,0.0,0.0,2.18668794631958,2.318079948425293,2.2218016147613526,2.211087942123413,0.035380897213135316,256,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,2048,0.375,16.0,16.0,1.0,0.4375,0.2525504668006971,1.875,6.953347743053017,0.1410369873046875,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0380599882396606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024512000381946564,0.04620800167322159,0.02884640023112297,0.026320000179111958,0.005516557583355925,0.02054399996995926,0.03846399858593941,0.023401600029319524,0.021263999864459038,0.0044742944198265374,0.0,0.0,0.0,0.0,0.0,2.2291839122772217,2.3929600715637207,2.2908096313476562,2.2804640531539917,0.04479348924786221,512,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,4096,0.375,32.0,32.0,1.0,0.65625,0.15765965680164504,1.5625,6.98229848728205,0.08779525756835938,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9569603278386984,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02412799932062626,0.06940799951553345,0.030817600432783365,0.026800000108778477,0.010047085187652939,0.020128000527620316,0.044096000492572784,0.024606400076299904,0.022304000332951546,0.00622857166823714,0.0,0.0,0.0,0.0,0.0,2.0678720474243164,2.1297600269317627,2.0837119817733765,2.0779199600219727,0.017880044357986735,1024,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,8192,0.375,64.0,64.0,1.0,0.625,0.12169081635504074,1.3125,6.9892029662356325,0.06879425048828125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9524274993623046,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02831999957561493,0.043487999588251114,0.031744000129401685,0.02991999965161085,0.003973415836428909,0.02070399932563305,0.029343999922275543,0.022886400017887353,0.021743999794125557,0.0026873065045088873,0.0,0.0,0.0,0.0,0.0,2.916032075881958,3.0819520950317383,2.9805248022079467,2.9656319618225098,0.05482195799019572,2048,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,16384,0.375,128.0,128.0,1.0,0.796875,0.07935434147688751,1.1796875,6.9954297964750305,0.044734954833984375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8971818172100244,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.03830400109291077,0.06268800050020218,0.043337599746882914,0.040511999279260635,0.005823016247946116,0.023135999217629433,0.03747199848294258,0.025206399988383053,0.02393599972128868,0.003271381256231161,0.0,0.0,0.0,0.0,0.0,4.421599864959717,4.535359859466553,4.486294317245483,4.497056007385254,0.036990243787549344,4096,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,32768,0.375,256.0,256.0,1.0,0.8203125,0.060849326483103046,1.17578125,6.9973188375859685,0.033740997314453125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8113207890716184,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.05660799890756607,0.07932800054550171,0.06249920018017292,0.06039999984204769,0.005636461691480845,0.02956799976527691,0.03747199848294258,0.031126399897038935,0.030287999659776688,0.002157571006631541,0.0,0.0,0.0,0.0,0.0,7.302591800689697,7.402751922607422,7.354758310317993,7.3464319705963135,0.032142662400335566,8192,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,65536,0.375,512.0,512.0,1.0,0.890625,0.0412323087266341,1.08984375,6.998772433185578,0.02334284782409668,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.883909666885606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02502400055527687,0.06092799827456474,0.033839999698102474,0.028672000393271446,0.010315255343709818,0.019360000267624855,0.0352960005402565,0.023424000293016434,0.022064000368118286,0.0038898102289194572,0.0,0.0,0.0,0.0,0.0,0.35280001163482666,0.39692801237106323,0.37662720382213594,0.37196800112724304,0.013401318050665304,1,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,8,0.375,0.0625,0.0625,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0233364439829928,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.025280000641942024,0.05132799968123436,0.030641599837690593,0.027343999594449997,0.007562409232763304,0.020160000771284103,0.052960000932216644,0.024145600199699403,0.021424000151455402,0.007450198645599985,0.0,0.0,0.0,0.0,0.0,0.4692479968070984,0.5523840188980103,0.5134752035140991,0.5100640058517456,0.02291433464135784,8,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,64,0.375,0.5,0.5,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9806430689981738,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024831999093294144,0.046560000628232956,0.030371200107038022,0.02763199992477894,0.005878487205028602,0.020320000126957893,0.04560000076889992,0.02601920012384653,0.02270400058478117,0.00751129965906799,0.0,0.0,0.0,0.0,0.0,0.34652799367904663,0.4119040071964264,0.3789471983909607,0.38550400733947754,0.02073945105803335,16,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,128,0.375,1.0,1.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9103623678483975,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024288000538945198,0.049375999718904495,0.03086080001667142,0.0267359996214509,0.0070864531384997225,0.020479999482631683,0.030912000685930252,0.02274719988927245,0.021359999664127827,0.0029966813650672505,0.0,0.0,0.0,0.0,0.0,0.31462401151657104,0.7456960082054138,0.38617280423641204,0.34545600414276123,0.12230201266253077,32,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,256,0.375,2.0,2.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9610778571819444,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02393599972128868,0.04342399910092354,0.029380799923092126,0.026688000187277794,0.0057374782000526574,0.020031999796628952,0.036607999354600906,0.022487999964505435,0.020911999978125095,0.0038135203062103235,0.0,0.0,0.0,0.0,0.0,0.32521599531173706,0.419871985912323,0.36325119733810424,0.34968000650405884,0.03161798848223672,64,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,512,0.375,4.0,4.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9952941013961014,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.025407999753952026,0.05510399863123894,0.031430399790406224,0.02798399981111288,0.007335050273982725,0.020479999482631683,0.03561599925160408,0.02275839988142252,0.021551999263465405,0.003545718365165811,0.0,0.0,0.0,0.0,0.0,0.289792001247406,0.4663360118865967,0.4091839998960495,0.41655999422073364,0.0446001986506615,128,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,1024,0.375,8.0,8.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9273256282883522,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.0244159996509552,0.05395200103521347,0.03188959984108806,0.026031999848783016,0.00943995927387483,0.02051199972629547,0.036607999354600906,0.023247999791055917,0.02147199958562851,0.003910623837069648,0.0,0.0,0.0,0.0,0.0,0.3761279881000519,0.4416320025920868,0.40686399936676027,0.40540799498558044,0.02257778769899645,256,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,2048,0.375,16.0,16.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0380599882396606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024512000381946564,0.04620800167322159,0.02884640023112297,0.026320000179111958,0.005516557583355925,0.02054399996995926,0.03846399858593941,0.023401600029319524,0.021263999864459038,0.0044742944198265374,0.0,0.0,0.0,0.0,0.0,0.7172480225563049,0.8663039803504944,0.7723807990550995,0.7591840028762817,0.04164772379451242,512,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,4096,0.375,32.0,32.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9569603278386984,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02412799932062626,0.06940799951553345,0.030817600432783365,0.026800000108778477,0.010047085187652939,0.020128000527620316,0.044096000492572784,0.024606400076299904,0.022304000332951546,0.00622857166823714,0.0,0.0,0.0,0.0,0.0,1.0195519924163818,1.2216639518737793,1.1253888130187988,1.1453600525856018,0.06548005322243594,1024,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,8192,0.375,64.0,64.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9524274993623046,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02831999957561493,0.043487999588251114,0.031744000129401685,0.02991999965161085,0.003973415836428909,0.02070399932563305,0.029343999922275543,0.022886400017887353,0.021743999794125557,0.0026873065045088873,0.0,0.0,0.0,0.0,0.0,1.7490559816360474,1.9644800424575806,1.8529024004936219,1.814303994178772,0.08042565617327288,2048,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,16384,0.375,128.0,128.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8971818172100244,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.03830400109291077,0.06268800050020218,0.043337599746882914,0.040511999279260635,0.005823016247946116,0.023135999217629433,0.03747199848294258,0.025206399988383053,0.02393599972128868,0.003271381256231161,0.0,0.0,0.0,0.0,0.0,3.2479360103607178,3.385279893875122,3.296070408821106,3.2800960540771484,0.04525529026442046,4096,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,32768,0.375,256.0,256.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8113207890716184,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.05660799890756607,0.07932800054550171,0.06249920018017292,0.06039999984204769,0.005636461691480845,0.02956799976527691,0.03747199848294258,0.031126399897038935,0.030287999659776688,0.002157571006631541,0.0,0.0,0.0,0.0,0.0,6.344799995422363,6.517856121063232,6.464438438415527,6.478623867034912,0.05116674443145098,8192,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,65536,0.375,512.0,512.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.883909666885606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02502400055527687,0.06092799827456474,0.033839999698102474,0.028672000393271446,0.010315255343709818,0.019360000267624855,0.0352960005402565,0.023424000293016434,0.022064000368118286,0.0038898102289194572,0.0,0.0,0.0,0.0,0.0,0.2648000121116638,0.325439989566803,0.28852800130844114,0.28390398621559143,0.01933778377635077,1,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,8,0.375,0.0625,0.0625,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0233364439829928,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.025280000641942024,0.05132799968123436,0.030641599837690593,0.027343999594449997,0.007562409232763304,0.020160000771284103,0.052960000932216644,0.024145600199699403,0.021424000151455402,0.007450198645599985,0.0,0.0,0.0,0.0,0.0,0.7347840070724487,0.862496018409729,0.7769344031810761,0.769216001033783,0.03290485328796285,8,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,64,0.375,0.5,0.5,0.421875,0.0,1.346291201783626,6.0,5.652114648336087,0.636962890625,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9806430689981738,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024831999093294144,0.046560000628232956,0.030371200107038022,0.02763199992477894,0.005878487205028602,0.020320000126957893,0.04560000076889992,0.02601920012384653,0.02270400058478117,0.00751129965906799,0.0,0.0,0.0,0.0,0.0,0.9198399782180786,0.9646080136299133,0.9412063956260681,0.9411839842796326,0.014939365085478117,16,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,128,0.375,1.0,1.0,0.5703125,0.0,1.118033988749895,5.0,6.008641773518898,0.580810546875,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9103623678483975,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024288000538945198,0.049375999718904495,0.03086080001667142,0.0267359996214509,0.0070864531384997225,0.020479999482631683,0.030912000685930252,0.02274719988927245,0.021359999664127827,0.0029966813650672505,0.0,0.0,0.0,0.0,0.0,1.2796800136566162,1.3484159708023071,1.3006976008415223,1.2929120063781738,0.020807998177176254,32,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,256,0.375,2.0,2.0,0.8828125,0.0,0.6959705453537527,3.0,6.60872850615583,0.38055419921875,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9610778571819444,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02393599972128868,0.04342399910092354,0.029380799923092126,0.026688000187277794,0.0057374782000526574,0.020031999796628952,0.036607999354600906,0.022487999964505435,0.020911999978125095,0.0038135203062103235,0.0,0.0,0.0,0.0,0.0,1.3630399703979492,1.4430400133132935,1.3909215927124023,1.3892319798469543,0.022335744492366926,64,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,512,0.375,4.0,4.0,0.984375,0.0,0.5201036555341637,3.0,6.798826509158851,0.28302001953125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9952941013961014,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.025407999753952026,0.05510399863123894,0.031430399790406224,0.02798399981111288,0.007335050273982725,0.020479999482631683,0.03561599925160408,0.02275839988142252,0.021551999263465405,0.003545718365165811,0.0,0.0,0.0,0.0,0.0,1.27948796749115,1.3904000520706177,1.3176063895225525,1.309440016746521,0.038060887827312775,128,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,1024,0.375,8.0,8.0,1.0,0.25,0.3511282039725661,1.875,6.91002266305238,0.1970977783203125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9273256282883522,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.0244159996509552,0.05395200103521347,0.03188959984108806,0.026031999848783016,0.00943995927387483,0.02051199972629547,0.036607999354600906,0.023247999791055917,0.02147199958562851,0.003910623837069648,0.0,0.0,0.0,0.0,0.0,1.264415979385376,1.3145920038223267,1.2791999936103822,1.2753440141677856,0.014130605249568332,256,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,2048,0.375,16.0,16.0,1.0,0.375,0.24692938483248605,1.6875,6.955481130775285,0.13909912109375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0380599882396606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024512000381946564,0.04620800167322159,0.02884640023112297,0.026320000179111958,0.005516557583355925,0.02054399996995926,0.03846399858593941,0.023401600029319524,0.021263999864459038,0.0044742944198265374,0.0,0.0,0.0,0.0,0.0,1.3081920146942139,1.347648024559021,1.3292255997657776,1.329967975616455,0.014558863679016933,512,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,4096,0.375,32.0,32.0,1.0,0.625,0.17143053326165383,1.5625,6.9786675275754035,0.09520339965820312,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9569603278386984,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02412799932062626,0.06940799951553345,0.030817600432783365,0.026800000108778477,0.010047085187652939,0.020128000527620316,0.044096000492572784,0.024606400076299904,0.022304000332951546,0.00622857166823714,0.0,0.0,0.0,0.0,0.0,1.242751955986023,1.3112000226974487,1.2747935891151427,1.266207993030548,0.021093073517695057,1024,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,8192,0.375,64.0,64.0,1.0,0.78125,0.11000099875256815,1.296875,6.991308871213679,0.062183380126953125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9524274993623046,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02831999957561493,0.043487999588251114,0.031744000129401685,0.02991999965161085,0.003973415836428909,0.02070399932563305,0.029343999922275543,0.022886400017887353,0.021743999794125557,0.0026873065045088873,0.0,0.0,0.0,0.0,0.0,1.7388160228729248,1.8077759742736816,1.772764801979065,1.772704005241394,0.021056644077284283,2048,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,16384,0.375,128.0,128.0,1.0,0.78125,0.0864630150197678,1.1796875,6.994552526394139,0.048796653747558594,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8971818172100244,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.03830400109291077,0.06268800050020218,0.043337599746882914,0.040511999279260635,0.005823016247946116,0.023135999217629433,0.03747199848294258,0.025206399988383053,0.02393599972128868,0.003271381256231161,0.0,0.0,0.0,0.0,0.0,2.6563520431518555,2.7063679695129395,2.6785055875778196,2.6791679859161377,0.01639963463052944,4096,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,32768,0.375,256.0,256.0,1.0,0.8671875,0.06127686514721937,1.16015625,6.997291583027146,0.03497934341430664,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8113207890716184,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.05660799890756607,0.07932800054550171,0.06249920018017292,0.06039999984204769,0.005636461691480845,0.02956799976527691,0.03747199848294258,0.031126399897038935,0.030287999659776688,0.002157571006631541,0.0,0.0,0.0,0.0,0.0,4.386879920959473,4.452256202697754,4.4108480453491214,4.406303882598877,0.019768161791937636,8192,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,65536,0.375,512.0,512.0,1.0,0.884765625,0.041723768525324195,1.1171875,6.998746434318934,0.02298593521118164,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.883909666885606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02502400055527687,0.06092799827456474,0.033839999698102474,0.028672000393271446,0.010315255343709818,0.019360000267624855,0.0352960005402565,0.023424000293016434,0.022064000368118286,0.0038898102289194572,0.0,0.0,0.0,0.0,0.0,0.24208000302314758,0.4028480052947998,0.3041536003351212,0.277103990316391,0.05660721484881584,1,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,8,0.375,0.0625,0.0625,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0233364439829928,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.025280000641942024,0.05132799968123436,0.030641599837690593,0.027343999594449997,0.007562409232763304,0.020160000771284103,0.052960000932216644,0.024145600199699403,0.021424000151455402,0.007450198645599985,0.0,0.0,0.0,0.0,0.0,0.2447039932012558,0.30502399802207947,0.26446720361709597,0.26265600323677063,0.016744548364435372,8,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,64,0.375,0.5,0.5,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9806430689981738,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024831999093294144,0.046560000628232956,0.030371200107038022,0.02763199992477894,0.005878487205028602,0.020320000126957893,0.04560000076889992,0.02601920012384653,0.02270400058478117,0.00751129965906799,0.0,0.0,0.0,0.0,0.0,0.2337920069694519,0.2881599962711334,0.26074880361557007,0.264384001493454,0.016469850143940968,16,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,128,0.375,1.0,1.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9103623678483975,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024288000538945198,0.049375999718904495,0.03086080001667142,0.0267359996214509,0.0070864531384997225,0.020479999482631683,0.030912000685930252,0.02274719988927245,0.021359999664127827,0.0029966813650672505,0.0,0.0,0.0,0.0,0.0,0.23369599878787994,0.28591999411582947,0.25465920120477675,0.25385600328445435,0.01593204652619194,32,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,256,0.375,2.0,2.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9610778571819444,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02393599972128868,0.04342399910092354,0.029380799923092126,0.026688000187277794,0.0057374782000526574,0.020031999796628952,0.036607999354600906,0.022487999964505435,0.020911999978125095,0.0038135203062103235,0.0,0.0,0.0,0.0,0.0,0.2295999974012375,0.26556798815727234,0.24674240052700042,0.2497600018978119,0.010345732066199003,64,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,512,0.375,4.0,4.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9952941013961014,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.025407999753952026,0.05510399863123894,0.031430399790406224,0.02798399981111288,0.007335050273982725,0.020479999482631683,0.03561599925160408,0.02275839988142252,0.021551999263465405,0.003545718365165811,0.0,0.0,0.0,0.0,0.0,0.21721599996089935,0.29020801186561584,0.2394208014011383,0.2346400022506714,0.018747330357768585,128,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,1024,0.375,8.0,8.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9273256282883522,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.0244159996509552,0.05395200103521347,0.03188959984108806,0.026031999848783016,0.00943995927387483,0.02051199972629547,0.036607999354600906,0.023247999791055917,0.02147199958562851,0.003910623837069648,0.0,0.0,0.0,0.0,0.0,0.2717759907245636,0.305184006690979,0.28813759982585907,0.28809599578380585,0.01183422600545854,256,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,2048,0.375,16.0,16.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0380599882396606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024512000381946564,0.04620800167322159,0.02884640023112297,0.026320000179111958,0.005516557583355925,0.02054399996995926,0.03846399858593941,0.023401600029319524,0.021263999864459038,0.0044742944198265374,0.0,0.0,0.0,0.0,0.0,0.3917759954929352,0.43772798776626587,0.41130879521369934,0.4131519943475723,0.012992473640805227,512,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,4096,0.375,32.0,32.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9569603278386984,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02412799932062626,0.06940799951553345,0.030817600432783365,0.026800000108778477,0.010047085187652939,0.020128000527620316,0.044096000492572784,0.024606400076299904,0.022304000332951546,0.00622857166823714,0.0,0.0,0.0,0.0,0.0,0.6176639795303345,0.7009919881820679,0.642767995595932,0.6330719888210297,0.024074084919938756,1024,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,8192,0.375,64.0,64.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9524274993623046,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02831999957561493,0.043487999588251114,0.031744000129401685,0.02991999965161085,0.003973415836428909,0.02070399932563305,0.029343999922275543,0.022886400017887353,0.021743999794125557,0.0026873065045088873,0.0,0.0,0.0,0.0,0.0,1.0820800065994263,1.1674879789352417,1.1034304022789,1.0977439880371094,0.0233067292981566,2048,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,16384,0.375,128.0,128.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8971818172100244,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.03830400109291077,0.06268800050020218,0.043337599746882914,0.040511999279260635,0.005823016247946116,0.023135999217629433,0.03747199848294258,0.025206399988383053,0.02393599972128868,0.003271381256231161,0.0,0.0,0.0,0.0,0.0,1.9809919595718384,2.0415360927581787,2.003715181350708,1.992751955986023,0.022066076434645737,4096,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,32768,0.375,256.0,256.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8113207890716184,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.05660799890756607,0.07932800054550171,0.06249920018017292,0.06039999984204769,0.005636461691480845,0.02956799976527691,0.03747199848294258,0.031126399897038935,0.030287999659776688,0.002157571006631541,0.0,0.0,0.0,0.0,0.0,3.790112018585205,3.8651199340820312,3.829139161109924,3.8230879306793213,0.025177464160110564,8192,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,65536,0.375,512.0,512.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.883909666885606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02502400055527687,0.06092799827456474,0.033839999698102474,0.028672000393271446,0.010315255343709818,0.019360000267624855,0.0352960005402565,0.023424000293016434,0.022064000368118286,0.0038898102289194572,0.0,0.0,0.0,0.0,0.0,0.212351992726326,0.24383999407291412,0.22760000079870224,0.22723200172185898,0.01050568575837594,1,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,8,0.375,0.0625,0.0625,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0233364439829928,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.025280000641942024,0.05132799968123436,0.030641599837690593,0.027343999594449997,0.007562409232763304,0.020160000771284103,0.052960000932216644,0.024145600199699403,0.021424000151455402,0.007450198645599985,0.0,0.0,0.0,0.0,0.0,0.47494399547576904,0.5184000134468079,0.4920704007148743,0.49169600009918213,0.011991064701471855,8,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,64,0.375,0.5,0.5,0.3984375,0.0,1.3919410907075054,6.0,5.570159765557392,0.667236328125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9806430689981738,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024831999093294144,0.046560000628232956,0.030371200107038022,0.02763199992477894,0.005878487205028602,0.020320000126957893,0.04560000076889992,0.02601920012384653,0.02270400058478117,0.00751129965906799,0.0,0.0,0.0,0.0,0.0,0.6360960006713867,0.7004479765892029,0.6608384013175964,0.6572319865226746,0.020416877242438597,16,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,128,0.375,1.0,1.0,0.625,0.0,1.0307764064044151,4.0,6.138251855282827,0.5382080078125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9103623678483975,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024288000538945198,0.049375999718904495,0.03086080001667142,0.0267359996214509,0.0070864531384997225,0.020479999482631683,0.030912000685930252,0.02274719988927245,0.021359999664127827,0.0029966813650672505,0.0,0.0,0.0,0.0,0.0,0.780896008014679,0.8301439881324768,0.80346559882164,0.8030399978160858,0.016801230312128875,32,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,256,0.375,2.0,2.0,0.859375,0.0,0.6903350635742038,3.0,6.5943747091218174,0.38067626953125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9610778571819444,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02393599972128868,0.04342399910092354,0.029380799923092126,0.026688000187277794,0.0057374782000526574,0.020031999796628952,0.036607999354600906,0.022487999964505435,0.020911999978125095,0.0038135203062103235,0.0,0.0,0.0,0.0,0.0,0.8607040047645569,0.9195200204849243,0.8783008038997651,0.8751039803028107,0.01719059253115595,64,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,512,0.375,4.0,4.0,0.9765625,0.0,0.49410588440130926,2.75,6.814452474347134,0.271270751953125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9952941013961014,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.025407999753952026,0.05510399863123894,0.031430399790406224,0.02798399981111288,0.007335050273982725,0.020479999482631683,0.03561599925160408,0.02275839988142252,0.021551999263465405,0.003545718365165811,0.0,0.0,0.0,0.0,0.0,0.833952009677887,0.894752025604248,0.8619967997074127,0.863215982913971,0.018716378851797198,128,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,1024,0.375,8.0,8.0,1.0,0.25,0.33693529145074724,2.125,6.9186075263155535,0.1867218017578125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9273256282883522,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.0244159996509552,0.05395200103521347,0.03188959984108806,0.026031999848783016,0.00943995927387483,0.02051199972629547,0.036607999354600906,0.023247999791055917,0.02147199958562851,0.003910623837069648,0.0,0.0,0.0,0.0,0.0,0.834879994392395,0.8871039748191833,0.8651552021503448,0.8638879954814911,0.015262430894400969,256,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,2048,0.375,16.0,16.0,1.0,0.375,0.25567294018677456,1.8125,6.952441049154937,0.14349365234375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0380599882396606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024512000381946564,0.04620800167322159,0.02884640023112297,0.026320000179111958,0.005516557583355925,0.02054399996995926,0.03846399858593941,0.023401600029319524,0.021263999864459038,0.0044742944198265374,0.0,0.0,0.0,0.0,0.0,0.8518080115318298,0.9097599983215332,0.8810272097587586,0.8751040101051331,0.017005819633271906,512,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,4096,0.375,32.0,32.0,1.0,0.625,0.1747801353218523,1.53125,6.978069554482723,0.09820938110351562,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9569603278386984,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02412799932062626,0.06940799951553345,0.030817600432783365,0.026800000108778477,0.010047085187652939,0.020128000527620316,0.044096000492572784,0.024606400076299904,0.022304000332951546,0.00622857166823714,0.0,0.0,0.0,0.0,0.0,0.8470079898834229,0.9010239839553833,0.8694015920162201,0.8716959953308105,0.016138881171190216,1024,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,8192,0.375,64.0,64.0,1.0,0.65625,0.1158122428154187,1.3125,6.9901908183358845,0.06445503234863281,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9524274993623046,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02831999957561493,0.043487999588251114,0.031744000129401685,0.02991999965161085,0.003973415836428909,0.02070399932563305,0.029343999922275543,0.022886400017887353,0.021743999794125557,0.0026873065045088873,0.0,0.0,0.0,0.0,0.0,1.1698240041732788,1.2311359643936157,1.1888479948043824,1.1890720129013062,0.017978797794492758,2048,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,16384,0.375,128.0,128.0,1.0,0.78125,0.08347181893108634,1.1796875,6.994921772573154,0.046871185302734375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8971818172100244,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.03830400109291077,0.06268800050020218,0.043337599746882914,0.040511999279260635,0.005823016247946116,0.023135999217629433,0.03747199848294258,0.025206399988383053,0.02393599972128868,0.003271381256231161,0.0,0.0,0.0,0.0,0.0,1.7702720165252686,1.8097599744796753,1.7919103980064393,1.7956640124320984,0.012641295676021557,4096,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,32768,0.375,256.0,256.0,1.0,0.78125,0.06866734477822484,1.20703125,6.996602562938728,0.03801727294921875,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8113207890716184,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.05660799890756607,0.07932800054550171,0.06249920018017292,0.06039999984204769,0.005636461691480845,0.02956799976527691,0.03747199848294258,0.031126399897038935,0.030287999659776688,0.002157571006631541,0.0,0.0,0.0,0.0,0.0,2.968672037124634,3.0278079509735107,2.9899007797241213,2.9824799299240112,0.01816414122631667,8192,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,65536,0.375,512.0,512.0,1.0,0.8984375,0.04399546833977376,1.126953125,6.998607314922362,0.024699926376342773,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.883909666885606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02502400055527687,0.06092799827456474,0.033839999698102474,0.028672000393271446,0.010315255343709818,0.019360000267624855,0.0352960005402565,0.023424000293016434,0.022064000368118286,0.0038898102289194572,0.0,0.0,0.0,0.0,0.0,0.19574399292469025,0.2512960135936737,0.21939200013875962,0.2199999988079071,0.017532156418212565,1,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,8,0.375,0.0625,0.0625,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0233364439829928,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.025280000641942024,0.05132799968123436,0.030641599837690593,0.027343999594449997,0.007562409232763304,0.020160000771284103,0.052960000932216644,0.024145600199699403,0.021424000151455402,0.007450198645599985,0.0,0.0,0.0,0.0,0.0,0.20483200252056122,0.24006399512290955,0.22215040028095245,0.22433599829673767,0.00969639786132892,8,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,64,0.375,0.5,0.5,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9806430689981738,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024831999093294144,0.046560000628232956,0.030371200107038022,0.02763199992477894,0.005878487205028602,0.020320000126957893,0.04560000076889992,0.02601920012384653,0.02270400058478117,0.00751129965906799,0.0,0.0,0.0,0.0,0.0,0.20559999346733093,0.24726399779319763,0.22126719802618028,0.22207999974489212,0.01328093478485499,16,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,128,0.375,1.0,1.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9103623678483975,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024288000538945198,0.049375999718904495,0.03086080001667142,0.0267359996214509,0.0070864531384997225,0.020479999482631683,0.030912000685930252,0.02274719988927245,0.021359999664127827,0.0029966813650672505,0.0,0.0,0.0,0.0,0.0,0.20003199577331543,0.2301120012998581,0.21453119963407516,0.21598400175571442,0.010402239855151332,32,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,256,0.375,2.0,2.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9610778571819444,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02393599972128868,0.04342399910092354,0.029380799923092126,0.026688000187277794,0.0057374782000526574,0.020031999796628952,0.036607999354600906,0.022487999964505435,0.020911999978125095,0.0038135203062103235,0.0,0.0,0.0,0.0,0.0,0.19551999866962433,0.22972799837589264,0.21238719969987868,0.21488000452518463,0.010706095835489097,64,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,512,0.375,4.0,4.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9952941013961014,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.025407999753952026,0.05510399863123894,0.031430399790406224,0.02798399981111288,0.007335050273982725,0.020479999482631683,0.03561599925160408,0.02275839988142252,0.021551999263465405,0.003545718365165811,0.0,0.0,0.0,0.0,0.0,0.19420799612998962,0.2903999984264374,0.2211231991648674,0.21476799994707108,0.025955008886196004,128,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,1024,0.375,8.0,8.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9273256282883522,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.0244159996509552,0.05395200103521347,0.03188959984108806,0.026031999848783016,0.00943995927387483,0.02051199972629547,0.036607999354600906,0.023247999791055917,0.02147199958562851,0.003910623837069648,0.0,0.0,0.0,0.0,0.0,0.2290560007095337,0.289247989654541,0.2543327987194061,0.24939200282096863,0.01663236753503115,256,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,2048,0.375,16.0,16.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0380599882396606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.024512000381946564,0.04620800167322159,0.02884640023112297,0.026320000179111958,0.005516557583355925,0.02054399996995926,0.03846399858593941,0.023401600029319524,0.021263999864459038,0.0044742944198265374,0.0,0.0,0.0,0.0,0.0,0.30831998586654663,0.36953601241111755,0.3324000000953674,0.3288639932870865,0.018617999572156707,512,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,4096,0.375,32.0,32.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9569603278386984,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02412799932062626,0.06940799951553345,0.030817600432783365,0.026800000108778477,0.010047085187652939,0.020128000527620316,0.044096000492572784,0.024606400076299904,0.022304000332951546,0.00622857166823714,0.0,0.0,0.0,0.0,0.0,0.462911993265152,0.5497919917106628,0.4893856018781662,0.4816960096359253,0.023348152887178286,1024,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,8192,0.375,64.0,64.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9524274993623046,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.02831999957561493,0.043487999588251114,0.031744000129401685,0.02991999965161085,0.003973415836428909,0.02070399932563305,0.029343999922275543,0.022886400017887353,0.021743999794125557,0.0026873065045088873,0.0,0.0,0.0,0.0,0.0,0.7662079930305481,0.8717759847640991,0.788454395532608,0.7744799852371216,0.03174746482604812,2048,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,16384,0.375,128.0,128.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8971818172100244,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.03830400109291077,0.06268800050020218,0.043337599746882914,0.040511999279260635,0.005823016247946116,0.023135999217629433,0.03747199848294258,0.025206399988383053,0.02393599972128868,0.003271381256231161,0.0,0.0,0.0,0.0,0.0,1.363935947418213,1.4143040180206299,1.3812703967094422,1.3798720240592957,0.014770450075530007,4096,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,32768,0.375,256.0,256.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8113207890716184,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
0.05660799890756607,0.07932800054550171,0.06249920018017292,0.06039999984204769,0.005636461691480845,0.02956799976527691,0.03747199848294258,0.031126399897038935,0.030287999659776688,0.002157571006631541,0.0,0.0,0.0,0.0,0.0,2.5507519245147705,2.680704116821289,2.579859209060669,2.566223978996277,0.03673283558365955,8192,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,65536,0.375,512.0,512.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.883909666885606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
1 time_stats.moe_gating_linear.min time_stats.moe_gating_linear.max time_stats.moe_gating_linear.mean time_stats.moe_gating_linear.median time_stats.moe_gating_linear.std time_stats.moe_gating_routing_topk.min time_stats.moe_gating_routing_topk.max time_stats.moe_gating_routing_topk.mean time_stats.moe_gating_routing_topk.median time_stats.moe_gating_routing_topk.std time_stats.moe_shuffling.min time_stats.moe_shuffling.max time_stats.moe_shuffling.mean time_stats.moe_shuffling.median time_stats.moe_shuffling.std time_stats.moe_grouped_gemm.min time_stats.moe_grouped_gemm.max time_stats.moe_grouped_gemm.mean time_stats.moe_grouped_gemm.median time_stats.moe_grouped_gemm.std num_tokens num_experts num_experts_per_device expert_parallel_size routing_runtime_path routing_assignment_policy routing_weight_policy routing_uses_router_logits gating_runtime_context gating_runtime_context_impl router_topk hidden_dim expert_hidden_dim use_gated num_tensor_parallel_workers total_routed_tokens model_expansion_ratio tokens_per_expert_avg tokens_to_experts_ratio expert_utilization min_load_ratio load_imbalance_cv max_load_ratio load_entropy load_gini_coefficient load_distribution seed moe_grouped_gemm_backend measurement_type profiling_precision model_arch quant_signature router_median_nonadditivity_ratio projection_policy
2 0.02502400055527687 0.06092799827456474 0.033839999698102474 0.028672000393271446 0.010315255343709818 0.019360000267624855 0.0352960005402565 0.023424000293016434 0.022064000368118286 0.0038898102289194572 0.0 0.0 0.0 0.0 0.0 0.33926400542259216 0.405023992061615 0.36780479848384856 0.36507199704647064 0.01690507644474779 1 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 8 0.375 0.0625 0.0625 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 1.0233364439829928 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
3 0.025280000641942024 0.05132799968123436 0.030641599837690593 0.027343999594449997 0.007562409232763304 0.020160000771284103 0.052960000932216644 0.024145600199699403 0.021424000151455402 0.007450198645599985 0.0 0.0 0.0 0.0 0.0 1.1943039894104004 1.286784052848816 1.228384006023407 1.2273280024528503 0.02832547242381263 8 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 64 0.375 0.5 0.5 0.3984375 0.0 1.346291201783626 4.0 5.59375 0.661865234375 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9806430689981738 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
4 0.024831999093294144 0.046560000628232956 0.030371200107038022 0.02763199992477894 0.005878487205028602 0.020320000126957893 0.04560000076889992 0.02601920012384653 0.02270400058478117 0.00751129965906799 0.0 0.0 0.0 0.0 0.0 1.679744005203247 1.766144037246704 1.7095808148384095 1.7015680074691772 0.02438921262998535 16 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 128 0.375 1.0 1.0 0.625 0.0 1.015504800579495 5.0 6.15516433212955 0.529052734375 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9103623678483975 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
5 0.024288000538945198 0.049375999718904495 0.03086080001667142 0.0267359996214509 0.0070864531384997225 0.020479999482631683 0.030912000685930252 0.02274719988927245 0.021359999664127827 0.0029966813650672505 0.0 0.0 0.0 0.0 0.0 2.1576640605926514 2.2921600341796875 2.2097824096679686 2.188944101333618 0.045572321842012986 32 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 256 0.375 2.0 2.0 0.875 0.0 0.6343057228182637 2.5 6.64370748444639 0.35369873046875 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9610778571819444 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
6 0.02393599972128868 0.04342399910092354 0.029380799923092126 0.026688000187277794 0.0057374782000526574 0.020031999796628952 0.036607999354600906 0.022487999964505435 0.020911999978125095 0.0038135203062103235 0.0 0.0 0.0 0.0 0.0 2.422368049621582 2.5130879878997803 2.4516672134399413 2.434159994125366 0.03278900287381846 64 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 512 0.375 4.0 4.0 0.984375 0.0 0.4921254921257382 2.25 6.817190042344769 0.272369384765625 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9952941013961014 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
7 0.025407999753952026 0.05510399863123894 0.031430399790406224 0.02798399981111288 0.007335050273982725 0.020479999482631683 0.03561599925160408 0.02275839988142252 0.021551999263465405 0.003545718365165811 0.0 0.0 0.0 0.0 0.0 2.2217600345611572 2.289599895477295 2.2571327924728393 2.263375997543335 0.021660416089449488 128 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 1024 0.375 8.0 8.0 1.0 0.125 0.3486861500690843 1.875 6.908192310183997 0.197662353515625 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9273256282883522 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
8 0.0244159996509552 0.05395200103521347 0.03188959984108806 0.026031999848783016 0.00943995927387483 0.02051199972629547 0.036607999354600906 0.023247999791055917 0.02147199958562851 0.003910623837069648 0.0 0.0 0.0 0.0 0.0 2.18668794631958 2.318079948425293 2.2218016147613526 2.211087942123413 0.035380897213135316 256 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 2048 0.375 16.0 16.0 1.0 0.4375 0.2525504668006971 1.875 6.953347743053017 0.1410369873046875 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 1.0380599882396606 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
9 0.024512000381946564 0.04620800167322159 0.02884640023112297 0.026320000179111958 0.005516557583355925 0.02054399996995926 0.03846399858593941 0.023401600029319524 0.021263999864459038 0.0044742944198265374 0.0 0.0 0.0 0.0 0.0 2.2291839122772217 2.3929600715637207 2.2908096313476562 2.2804640531539917 0.04479348924786221 512 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 4096 0.375 32.0 32.0 1.0 0.65625 0.15765965680164504 1.5625 6.98229848728205 0.08779525756835938 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9569603278386984 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
10 0.02412799932062626 0.06940799951553345 0.030817600432783365 0.026800000108778477 0.010047085187652939 0.020128000527620316 0.044096000492572784 0.024606400076299904 0.022304000332951546 0.00622857166823714 0.0 0.0 0.0 0.0 0.0 2.0678720474243164 2.1297600269317627 2.0837119817733765 2.0779199600219727 0.017880044357986735 1024 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 8192 0.375 64.0 64.0 1.0 0.625 0.12169081635504074 1.3125 6.9892029662356325 0.06879425048828125 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9524274993623046 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
11 0.02831999957561493 0.043487999588251114 0.031744000129401685 0.02991999965161085 0.003973415836428909 0.02070399932563305 0.029343999922275543 0.022886400017887353 0.021743999794125557 0.0026873065045088873 0.0 0.0 0.0 0.0 0.0 2.916032075881958 3.0819520950317383 2.9805248022079467 2.9656319618225098 0.05482195799019572 2048 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 16384 0.375 128.0 128.0 1.0 0.796875 0.07935434147688751 1.1796875 6.9954297964750305 0.044734954833984375 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.8971818172100244 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
12 0.03830400109291077 0.06268800050020218 0.043337599746882914 0.040511999279260635 0.005823016247946116 0.023135999217629433 0.03747199848294258 0.025206399988383053 0.02393599972128868 0.003271381256231161 0.0 0.0 0.0 0.0 0.0 4.421599864959717 4.535359859466553 4.486294317245483 4.497056007385254 0.036990243787549344 4096 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 32768 0.375 256.0 256.0 1.0 0.8203125 0.060849326483103046 1.17578125 6.9973188375859685 0.033740997314453125 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.8113207890716184 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
13 0.05660799890756607 0.07932800054550171 0.06249920018017292 0.06039999984204769 0.005636461691480845 0.02956799976527691 0.03747199848294258 0.031126399897038935 0.030287999659776688 0.002157571006631541 0.0 0.0 0.0 0.0 0.0 7.302591800689697 7.402751922607422 7.354758310317993 7.3464319705963135 0.032142662400335566 8192 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 65536 0.375 512.0 512.0 1.0 0.890625 0.0412323087266341 1.08984375 6.998772433185578 0.02334284782409668 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.883909666885606 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
14 0.02502400055527687 0.06092799827456474 0.033839999698102474 0.028672000393271446 0.010315255343709818 0.019360000267624855 0.0352960005402565 0.023424000293016434 0.022064000368118286 0.0038898102289194572 0.0 0.0 0.0 0.0 0.0 0.35280001163482666 0.39692801237106323 0.37662720382213594 0.37196800112724304 0.013401318050665304 1 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 8 0.375 0.0625 0.0625 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 1.0233364439829928 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
15 0.025280000641942024 0.05132799968123436 0.030641599837690593 0.027343999594449997 0.007562409232763304 0.020160000771284103 0.052960000932216644 0.024145600199699403 0.021424000151455402 0.007450198645599985 0.0 0.0 0.0 0.0 0.0 0.4692479968070984 0.5523840188980103 0.5134752035140991 0.5100640058517456 0.02291433464135784 8 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 64 0.375 0.5 0.5 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9806430689981738 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
16 0.024831999093294144 0.046560000628232956 0.030371200107038022 0.02763199992477894 0.005878487205028602 0.020320000126957893 0.04560000076889992 0.02601920012384653 0.02270400058478117 0.00751129965906799 0.0 0.0 0.0 0.0 0.0 0.34652799367904663 0.4119040071964264 0.3789471983909607 0.38550400733947754 0.02073945105803335 16 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 128 0.375 1.0 1.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9103623678483975 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
17 0.024288000538945198 0.049375999718904495 0.03086080001667142 0.0267359996214509 0.0070864531384997225 0.020479999482631683 0.030912000685930252 0.02274719988927245 0.021359999664127827 0.0029966813650672505 0.0 0.0 0.0 0.0 0.0 0.31462401151657104 0.7456960082054138 0.38617280423641204 0.34545600414276123 0.12230201266253077 32 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 256 0.375 2.0 2.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9610778571819444 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
18 0.02393599972128868 0.04342399910092354 0.029380799923092126 0.026688000187277794 0.0057374782000526574 0.020031999796628952 0.036607999354600906 0.022487999964505435 0.020911999978125095 0.0038135203062103235 0.0 0.0 0.0 0.0 0.0 0.32521599531173706 0.419871985912323 0.36325119733810424 0.34968000650405884 0.03161798848223672 64 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 512 0.375 4.0 4.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9952941013961014 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
19 0.025407999753952026 0.05510399863123894 0.031430399790406224 0.02798399981111288 0.007335050273982725 0.020479999482631683 0.03561599925160408 0.02275839988142252 0.021551999263465405 0.003545718365165811 0.0 0.0 0.0 0.0 0.0 0.289792001247406 0.4663360118865967 0.4091839998960495 0.41655999422073364 0.0446001986506615 128 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 1024 0.375 8.0 8.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9273256282883522 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
20 0.0244159996509552 0.05395200103521347 0.03188959984108806 0.026031999848783016 0.00943995927387483 0.02051199972629547 0.036607999354600906 0.023247999791055917 0.02147199958562851 0.003910623837069648 0.0 0.0 0.0 0.0 0.0 0.3761279881000519 0.4416320025920868 0.40686399936676027 0.40540799498558044 0.02257778769899645 256 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 2048 0.375 16.0 16.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 1.0380599882396606 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
21 0.024512000381946564 0.04620800167322159 0.02884640023112297 0.026320000179111958 0.005516557583355925 0.02054399996995926 0.03846399858593941 0.023401600029319524 0.021263999864459038 0.0044742944198265374 0.0 0.0 0.0 0.0 0.0 0.7172480225563049 0.8663039803504944 0.7723807990550995 0.7591840028762817 0.04164772379451242 512 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 4096 0.375 32.0 32.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9569603278386984 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
22 0.02412799932062626 0.06940799951553345 0.030817600432783365 0.026800000108778477 0.010047085187652939 0.020128000527620316 0.044096000492572784 0.024606400076299904 0.022304000332951546 0.00622857166823714 0.0 0.0 0.0 0.0 0.0 1.0195519924163818 1.2216639518737793 1.1253888130187988 1.1453600525856018 0.06548005322243594 1024 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 8192 0.375 64.0 64.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9524274993623046 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
23 0.02831999957561493 0.043487999588251114 0.031744000129401685 0.02991999965161085 0.003973415836428909 0.02070399932563305 0.029343999922275543 0.022886400017887353 0.021743999794125557 0.0026873065045088873 0.0 0.0 0.0 0.0 0.0 1.7490559816360474 1.9644800424575806 1.8529024004936219 1.814303994178772 0.08042565617327288 2048 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 16384 0.375 128.0 128.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.8971818172100244 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
24 0.03830400109291077 0.06268800050020218 0.043337599746882914 0.040511999279260635 0.005823016247946116 0.023135999217629433 0.03747199848294258 0.025206399988383053 0.02393599972128868 0.003271381256231161 0.0 0.0 0.0 0.0 0.0 3.2479360103607178 3.385279893875122 3.296070408821106 3.2800960540771484 0.04525529026442046 4096 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 32768 0.375 256.0 256.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.8113207890716184 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
25 0.05660799890756607 0.07932800054550171 0.06249920018017292 0.06039999984204769 0.005636461691480845 0.02956799976527691 0.03747199848294258 0.031126399897038935 0.030287999659776688 0.002157571006631541 0.0 0.0 0.0 0.0 0.0 6.344799995422363 6.517856121063232 6.464438438415527 6.478623867034912 0.05116674443145098 8192 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 1 65536 0.375 512.0 512.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.883909666885606 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
26 0.02502400055527687 0.06092799827456474 0.033839999698102474 0.028672000393271446 0.010315255343709818 0.019360000267624855 0.0352960005402565 0.023424000293016434 0.022064000368118286 0.0038898102289194572 0.0 0.0 0.0 0.0 0.0 0.2648000121116638 0.325439989566803 0.28852800130844114 0.28390398621559143 0.01933778377635077 1 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 8 0.375 0.0625 0.0625 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 1.0233364439829928 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
27 0.025280000641942024 0.05132799968123436 0.030641599837690593 0.027343999594449997 0.007562409232763304 0.020160000771284103 0.052960000932216644 0.024145600199699403 0.021424000151455402 0.007450198645599985 0.0 0.0 0.0 0.0 0.0 0.7347840070724487 0.862496018409729 0.7769344031810761 0.769216001033783 0.03290485328796285 8 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 64 0.375 0.5 0.5 0.421875 0.0 1.346291201783626 6.0 5.652114648336087 0.636962890625 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9806430689981738 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
28 0.024831999093294144 0.046560000628232956 0.030371200107038022 0.02763199992477894 0.005878487205028602 0.020320000126957893 0.04560000076889992 0.02601920012384653 0.02270400058478117 0.00751129965906799 0.0 0.0 0.0 0.0 0.0 0.9198399782180786 0.9646080136299133 0.9412063956260681 0.9411839842796326 0.014939365085478117 16 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 128 0.375 1.0 1.0 0.5703125 0.0 1.118033988749895 5.0 6.008641773518898 0.580810546875 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9103623678483975 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
29 0.024288000538945198 0.049375999718904495 0.03086080001667142 0.0267359996214509 0.0070864531384997225 0.020479999482631683 0.030912000685930252 0.02274719988927245 0.021359999664127827 0.0029966813650672505 0.0 0.0 0.0 0.0 0.0 1.2796800136566162 1.3484159708023071 1.3006976008415223 1.2929120063781738 0.020807998177176254 32 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 256 0.375 2.0 2.0 0.8828125 0.0 0.6959705453537527 3.0 6.60872850615583 0.38055419921875 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9610778571819444 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
30 0.02393599972128868 0.04342399910092354 0.029380799923092126 0.026688000187277794 0.0057374782000526574 0.020031999796628952 0.036607999354600906 0.022487999964505435 0.020911999978125095 0.0038135203062103235 0.0 0.0 0.0 0.0 0.0 1.3630399703979492 1.4430400133132935 1.3909215927124023 1.3892319798469543 0.022335744492366926 64 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 512 0.375 4.0 4.0 0.984375 0.0 0.5201036555341637 3.0 6.798826509158851 0.28302001953125 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9952941013961014 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
31 0.025407999753952026 0.05510399863123894 0.031430399790406224 0.02798399981111288 0.007335050273982725 0.020479999482631683 0.03561599925160408 0.02275839988142252 0.021551999263465405 0.003545718365165811 0.0 0.0 0.0 0.0 0.0 1.27948796749115 1.3904000520706177 1.3176063895225525 1.309440016746521 0.038060887827312775 128 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 1024 0.375 8.0 8.0 1.0 0.25 0.3511282039725661 1.875 6.91002266305238 0.1970977783203125 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9273256282883522 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
32 0.0244159996509552 0.05395200103521347 0.03188959984108806 0.026031999848783016 0.00943995927387483 0.02051199972629547 0.036607999354600906 0.023247999791055917 0.02147199958562851 0.003910623837069648 0.0 0.0 0.0 0.0 0.0 1.264415979385376 1.3145920038223267 1.2791999936103822 1.2753440141677856 0.014130605249568332 256 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 2048 0.375 16.0 16.0 1.0 0.375 0.24692938483248605 1.6875 6.955481130775285 0.13909912109375 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 1.0380599882396606 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
33 0.024512000381946564 0.04620800167322159 0.02884640023112297 0.026320000179111958 0.005516557583355925 0.02054399996995926 0.03846399858593941 0.023401600029319524 0.021263999864459038 0.0044742944198265374 0.0 0.0 0.0 0.0 0.0 1.3081920146942139 1.347648024559021 1.3292255997657776 1.329967975616455 0.014558863679016933 512 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 4096 0.375 32.0 32.0 1.0 0.625 0.17143053326165383 1.5625 6.9786675275754035 0.09520339965820312 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9569603278386984 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
34 0.02412799932062626 0.06940799951553345 0.030817600432783365 0.026800000108778477 0.010047085187652939 0.020128000527620316 0.044096000492572784 0.024606400076299904 0.022304000332951546 0.00622857166823714 0.0 0.0 0.0 0.0 0.0 1.242751955986023 1.3112000226974487 1.2747935891151427 1.266207993030548 0.021093073517695057 1024 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 8192 0.375 64.0 64.0 1.0 0.78125 0.11000099875256815 1.296875 6.991308871213679 0.062183380126953125 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9524274993623046 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
35 0.02831999957561493 0.043487999588251114 0.031744000129401685 0.02991999965161085 0.003973415836428909 0.02070399932563305 0.029343999922275543 0.022886400017887353 0.021743999794125557 0.0026873065045088873 0.0 0.0 0.0 0.0 0.0 1.7388160228729248 1.8077759742736816 1.772764801979065 1.772704005241394 0.021056644077284283 2048 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 16384 0.375 128.0 128.0 1.0 0.78125 0.0864630150197678 1.1796875 6.994552526394139 0.048796653747558594 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.8971818172100244 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
36 0.03830400109291077 0.06268800050020218 0.043337599746882914 0.040511999279260635 0.005823016247946116 0.023135999217629433 0.03747199848294258 0.025206399988383053 0.02393599972128868 0.003271381256231161 0.0 0.0 0.0 0.0 0.0 2.6563520431518555 2.7063679695129395 2.6785055875778196 2.6791679859161377 0.01639963463052944 4096 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 32768 0.375 256.0 256.0 1.0 0.8671875 0.06127686514721937 1.16015625 6.997291583027146 0.03497934341430664 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.8113207890716184 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
37 0.05660799890756607 0.07932800054550171 0.06249920018017292 0.06039999984204769 0.005636461691480845 0.02956799976527691 0.03747199848294258 0.031126399897038935 0.030287999659776688 0.002157571006631541 0.0 0.0 0.0 0.0 0.0 4.386879920959473 4.452256202697754 4.4108480453491214 4.406303882598877 0.019768161791937636 8192 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 65536 0.375 512.0 512.0 1.0 0.884765625 0.041723768525324195 1.1171875 6.998746434318934 0.02298593521118164 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.883909666885606 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
38 0.02502400055527687 0.06092799827456474 0.033839999698102474 0.028672000393271446 0.010315255343709818 0.019360000267624855 0.0352960005402565 0.023424000293016434 0.022064000368118286 0.0038898102289194572 0.0 0.0 0.0 0.0 0.0 0.24208000302314758 0.4028480052947998 0.3041536003351212 0.277103990316391 0.05660721484881584 1 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 8 0.375 0.0625 0.0625 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 1.0233364439829928 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
39 0.025280000641942024 0.05132799968123436 0.030641599837690593 0.027343999594449997 0.007562409232763304 0.020160000771284103 0.052960000932216644 0.024145600199699403 0.021424000151455402 0.007450198645599985 0.0 0.0 0.0 0.0 0.0 0.2447039932012558 0.30502399802207947 0.26446720361709597 0.26265600323677063 0.016744548364435372 8 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 64 0.375 0.5 0.5 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9806430689981738 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
40 0.024831999093294144 0.046560000628232956 0.030371200107038022 0.02763199992477894 0.005878487205028602 0.020320000126957893 0.04560000076889992 0.02601920012384653 0.02270400058478117 0.00751129965906799 0.0 0.0 0.0 0.0 0.0 0.2337920069694519 0.2881599962711334 0.26074880361557007 0.264384001493454 0.016469850143940968 16 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 128 0.375 1.0 1.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9103623678483975 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
41 0.024288000538945198 0.049375999718904495 0.03086080001667142 0.0267359996214509 0.0070864531384997225 0.020479999482631683 0.030912000685930252 0.02274719988927245 0.021359999664127827 0.0029966813650672505 0.0 0.0 0.0 0.0 0.0 0.23369599878787994 0.28591999411582947 0.25465920120477675 0.25385600328445435 0.01593204652619194 32 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 256 0.375 2.0 2.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9610778571819444 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
42 0.02393599972128868 0.04342399910092354 0.029380799923092126 0.026688000187277794 0.0057374782000526574 0.020031999796628952 0.036607999354600906 0.022487999964505435 0.020911999978125095 0.0038135203062103235 0.0 0.0 0.0 0.0 0.0 0.2295999974012375 0.26556798815727234 0.24674240052700042 0.2497600018978119 0.010345732066199003 64 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 512 0.375 4.0 4.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9952941013961014 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
43 0.025407999753952026 0.05510399863123894 0.031430399790406224 0.02798399981111288 0.007335050273982725 0.020479999482631683 0.03561599925160408 0.02275839988142252 0.021551999263465405 0.003545718365165811 0.0 0.0 0.0 0.0 0.0 0.21721599996089935 0.29020801186561584 0.2394208014011383 0.2346400022506714 0.018747330357768585 128 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 1024 0.375 8.0 8.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9273256282883522 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
44 0.0244159996509552 0.05395200103521347 0.03188959984108806 0.026031999848783016 0.00943995927387483 0.02051199972629547 0.036607999354600906 0.023247999791055917 0.02147199958562851 0.003910623837069648 0.0 0.0 0.0 0.0 0.0 0.2717759907245636 0.305184006690979 0.28813759982585907 0.28809599578380585 0.01183422600545854 256 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 2048 0.375 16.0 16.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 1.0380599882396606 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
45 0.024512000381946564 0.04620800167322159 0.02884640023112297 0.026320000179111958 0.005516557583355925 0.02054399996995926 0.03846399858593941 0.023401600029319524 0.021263999864459038 0.0044742944198265374 0.0 0.0 0.0 0.0 0.0 0.3917759954929352 0.43772798776626587 0.41130879521369934 0.4131519943475723 0.012992473640805227 512 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 4096 0.375 32.0 32.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9569603278386984 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
46 0.02412799932062626 0.06940799951553345 0.030817600432783365 0.026800000108778477 0.010047085187652939 0.020128000527620316 0.044096000492572784 0.024606400076299904 0.022304000332951546 0.00622857166823714 0.0 0.0 0.0 0.0 0.0 0.6176639795303345 0.7009919881820679 0.642767995595932 0.6330719888210297 0.024074084919938756 1024 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 8192 0.375 64.0 64.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9524274993623046 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
47 0.02831999957561493 0.043487999588251114 0.031744000129401685 0.02991999965161085 0.003973415836428909 0.02070399932563305 0.029343999922275543 0.022886400017887353 0.021743999794125557 0.0026873065045088873 0.0 0.0 0.0 0.0 0.0 1.0820800065994263 1.1674879789352417 1.1034304022789 1.0977439880371094 0.0233067292981566 2048 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 16384 0.375 128.0 128.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.8971818172100244 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
48 0.03830400109291077 0.06268800050020218 0.043337599746882914 0.040511999279260635 0.005823016247946116 0.023135999217629433 0.03747199848294258 0.025206399988383053 0.02393599972128868 0.003271381256231161 0.0 0.0 0.0 0.0 0.0 1.9809919595718384 2.0415360927581787 2.003715181350708 1.992751955986023 0.022066076434645737 4096 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 32768 0.375 256.0 256.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.8113207890716184 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
49 0.05660799890756607 0.07932800054550171 0.06249920018017292 0.06039999984204769 0.005636461691480845 0.02956799976527691 0.03747199848294258 0.031126399897038935 0.030287999659776688 0.002157571006631541 0.0 0.0 0.0 0.0 0.0 3.790112018585205 3.8651199340820312 3.829139161109924 3.8230879306793213 0.025177464160110564 8192 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 2 65536 0.375 512.0 512.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.883909666885606 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
50 0.02502400055527687 0.06092799827456474 0.033839999698102474 0.028672000393271446 0.010315255343709818 0.019360000267624855 0.0352960005402565 0.023424000293016434 0.022064000368118286 0.0038898102289194572 0.0 0.0 0.0 0.0 0.0 0.212351992726326 0.24383999407291412 0.22760000079870224 0.22723200172185898 0.01050568575837594 1 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 8 0.375 0.0625 0.0625 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 1.0233364439829928 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
51 0.025280000641942024 0.05132799968123436 0.030641599837690593 0.027343999594449997 0.007562409232763304 0.020160000771284103 0.052960000932216644 0.024145600199699403 0.021424000151455402 0.007450198645599985 0.0 0.0 0.0 0.0 0.0 0.47494399547576904 0.5184000134468079 0.4920704007148743 0.49169600009918213 0.011991064701471855 8 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 64 0.375 0.5 0.5 0.3984375 0.0 1.3919410907075054 6.0 5.570159765557392 0.667236328125 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9806430689981738 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
52 0.024831999093294144 0.046560000628232956 0.030371200107038022 0.02763199992477894 0.005878487205028602 0.020320000126957893 0.04560000076889992 0.02601920012384653 0.02270400058478117 0.00751129965906799 0.0 0.0 0.0 0.0 0.0 0.6360960006713867 0.7004479765892029 0.6608384013175964 0.6572319865226746 0.020416877242438597 16 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 128 0.375 1.0 1.0 0.625 0.0 1.0307764064044151 4.0 6.138251855282827 0.5382080078125 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9103623678483975 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
53 0.024288000538945198 0.049375999718904495 0.03086080001667142 0.0267359996214509 0.0070864531384997225 0.020479999482631683 0.030912000685930252 0.02274719988927245 0.021359999664127827 0.0029966813650672505 0.0 0.0 0.0 0.0 0.0 0.780896008014679 0.8301439881324768 0.80346559882164 0.8030399978160858 0.016801230312128875 32 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 256 0.375 2.0 2.0 0.859375 0.0 0.6903350635742038 3.0 6.5943747091218174 0.38067626953125 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9610778571819444 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
54 0.02393599972128868 0.04342399910092354 0.029380799923092126 0.026688000187277794 0.0057374782000526574 0.020031999796628952 0.036607999354600906 0.022487999964505435 0.020911999978125095 0.0038135203062103235 0.0 0.0 0.0 0.0 0.0 0.8607040047645569 0.9195200204849243 0.8783008038997651 0.8751039803028107 0.01719059253115595 64 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 512 0.375 4.0 4.0 0.9765625 0.0 0.49410588440130926 2.75 6.814452474347134 0.271270751953125 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9952941013961014 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
55 0.025407999753952026 0.05510399863123894 0.031430399790406224 0.02798399981111288 0.007335050273982725 0.020479999482631683 0.03561599925160408 0.02275839988142252 0.021551999263465405 0.003545718365165811 0.0 0.0 0.0 0.0 0.0 0.833952009677887 0.894752025604248 0.8619967997074127 0.863215982913971 0.018716378851797198 128 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 1024 0.375 8.0 8.0 1.0 0.25 0.33693529145074724 2.125 6.9186075263155535 0.1867218017578125 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9273256282883522 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
56 0.0244159996509552 0.05395200103521347 0.03188959984108806 0.026031999848783016 0.00943995927387483 0.02051199972629547 0.036607999354600906 0.023247999791055917 0.02147199958562851 0.003910623837069648 0.0 0.0 0.0 0.0 0.0 0.834879994392395 0.8871039748191833 0.8651552021503448 0.8638879954814911 0.015262430894400969 256 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 2048 0.375 16.0 16.0 1.0 0.375 0.25567294018677456 1.8125 6.952441049154937 0.14349365234375 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 1.0380599882396606 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
57 0.024512000381946564 0.04620800167322159 0.02884640023112297 0.026320000179111958 0.005516557583355925 0.02054399996995926 0.03846399858593941 0.023401600029319524 0.021263999864459038 0.0044742944198265374 0.0 0.0 0.0 0.0 0.0 0.8518080115318298 0.9097599983215332 0.8810272097587586 0.8751040101051331 0.017005819633271906 512 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 4096 0.375 32.0 32.0 1.0 0.625 0.1747801353218523 1.53125 6.978069554482723 0.09820938110351562 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9569603278386984 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
58 0.02412799932062626 0.06940799951553345 0.030817600432783365 0.026800000108778477 0.010047085187652939 0.020128000527620316 0.044096000492572784 0.024606400076299904 0.022304000332951546 0.00622857166823714 0.0 0.0 0.0 0.0 0.0 0.8470079898834229 0.9010239839553833 0.8694015920162201 0.8716959953308105 0.016138881171190216 1024 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 8192 0.375 64.0 64.0 1.0 0.65625 0.1158122428154187 1.3125 6.9901908183358845 0.06445503234863281 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9524274993623046 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
59 0.02831999957561493 0.043487999588251114 0.031744000129401685 0.02991999965161085 0.003973415836428909 0.02070399932563305 0.029343999922275543 0.022886400017887353 0.021743999794125557 0.0026873065045088873 0.0 0.0 0.0 0.0 0.0 1.1698240041732788 1.2311359643936157 1.1888479948043824 1.1890720129013062 0.017978797794492758 2048 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 16384 0.375 128.0 128.0 1.0 0.78125 0.08347181893108634 1.1796875 6.994921772573154 0.046871185302734375 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.8971818172100244 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
60 0.03830400109291077 0.06268800050020218 0.043337599746882914 0.040511999279260635 0.005823016247946116 0.023135999217629433 0.03747199848294258 0.025206399988383053 0.02393599972128868 0.003271381256231161 0.0 0.0 0.0 0.0 0.0 1.7702720165252686 1.8097599744796753 1.7919103980064393 1.7956640124320984 0.012641295676021557 4096 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 32768 0.375 256.0 256.0 1.0 0.78125 0.06866734477822484 1.20703125 6.996602562938728 0.03801727294921875 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.8113207890716184 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
61 0.05660799890756607 0.07932800054550171 0.06249920018017292 0.06039999984204769 0.005636461691480845 0.02956799976527691 0.03747199848294258 0.031126399897038935 0.030287999659776688 0.002157571006631541 0.0 0.0 0.0 0.0 0.0 2.968672037124634 3.0278079509735107 2.9899007797241213 2.9824799299240112 0.01816414122631667 8192 128 128 1 standard_fused_topk logit_topk softmax_renorm True standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 65536 0.375 512.0 512.0 1.0 0.8984375 0.04399546833977376 1.126953125 6.998607314922362 0.024699926376342773 uniform_random_logits 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.883909666885606 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
62 0.02502400055527687 0.06092799827456474 0.033839999698102474 0.028672000393271446 0.010315255343709818 0.019360000267624855 0.0352960005402565 0.023424000293016434 0.022064000368118286 0.0038898102289194572 0.0 0.0 0.0 0.0 0.0 0.19574399292469025 0.2512960135936737 0.21939200013875962 0.2199999988079071 0.017532156418212565 1 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 8 0.375 0.0625 0.0625 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 1.0233364439829928 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
63 0.025280000641942024 0.05132799968123436 0.030641599837690593 0.027343999594449997 0.007562409232763304 0.020160000771284103 0.052960000932216644 0.024145600199699403 0.021424000151455402 0.007450198645599985 0.0 0.0 0.0 0.0 0.0 0.20483200252056122 0.24006399512290955 0.22215040028095245 0.22433599829673767 0.00969639786132892 8 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 64 0.375 0.5 0.5 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9806430689981738 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
64 0.024831999093294144 0.046560000628232956 0.030371200107038022 0.02763199992477894 0.005878487205028602 0.020320000126957893 0.04560000076889992 0.02601920012384653 0.02270400058478117 0.00751129965906799 0.0 0.0 0.0 0.0 0.0 0.20559999346733093 0.24726399779319763 0.22126719802618028 0.22207999974489212 0.01328093478485499 16 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 128 0.375 1.0 1.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9103623678483975 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
65 0.024288000538945198 0.049375999718904495 0.03086080001667142 0.0267359996214509 0.0070864531384997225 0.020479999482631683 0.030912000685930252 0.02274719988927245 0.021359999664127827 0.0029966813650672505 0.0 0.0 0.0 0.0 0.0 0.20003199577331543 0.2301120012998581 0.21453119963407516 0.21598400175571442 0.010402239855151332 32 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 256 0.375 2.0 2.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9610778571819444 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
66 0.02393599972128868 0.04342399910092354 0.029380799923092126 0.026688000187277794 0.0057374782000526574 0.020031999796628952 0.036607999354600906 0.022487999964505435 0.020911999978125095 0.0038135203062103235 0.0 0.0 0.0 0.0 0.0 0.19551999866962433 0.22972799837589264 0.21238719969987868 0.21488000452518463 0.010706095835489097 64 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 512 0.375 4.0 4.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9952941013961014 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
67 0.025407999753952026 0.05510399863123894 0.031430399790406224 0.02798399981111288 0.007335050273982725 0.020479999482631683 0.03561599925160408 0.02275839988142252 0.021551999263465405 0.003545718365165811 0.0 0.0 0.0 0.0 0.0 0.19420799612998962 0.2903999984264374 0.2211231991648674 0.21476799994707108 0.025955008886196004 128 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 1024 0.375 8.0 8.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9273256282883522 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
68 0.0244159996509552 0.05395200103521347 0.03188959984108806 0.026031999848783016 0.00943995927387483 0.02051199972629547 0.036607999354600906 0.023247999791055917 0.02147199958562851 0.003910623837069648 0.0 0.0 0.0 0.0 0.0 0.2290560007095337 0.289247989654541 0.2543327987194061 0.24939200282096863 0.01663236753503115 256 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 2048 0.375 16.0 16.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 1.0380599882396606 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
69 0.024512000381946564 0.04620800167322159 0.02884640023112297 0.026320000179111958 0.005516557583355925 0.02054399996995926 0.03846399858593941 0.023401600029319524 0.021263999864459038 0.0044742944198265374 0.0 0.0 0.0 0.0 0.0 0.30831998586654663 0.36953601241111755 0.3324000000953674 0.3288639932870865 0.018617999572156707 512 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 4096 0.375 32.0 32.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9569603278386984 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
70 0.02412799932062626 0.06940799951553345 0.030817600432783365 0.026800000108778477 0.010047085187652939 0.020128000527620316 0.044096000492572784 0.024606400076299904 0.022304000332951546 0.00622857166823714 0.0 0.0 0.0 0.0 0.0 0.462911993265152 0.5497919917106628 0.4893856018781662 0.4816960096359253 0.023348152887178286 1024 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 8192 0.375 64.0 64.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.9524274993623046 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
71 0.02831999957561493 0.043487999588251114 0.031744000129401685 0.02991999965161085 0.003973415836428909 0.02070399932563305 0.029343999922275543 0.022886400017887353 0.021743999794125557 0.0026873065045088873 0.0 0.0 0.0 0.0 0.0 0.7662079930305481 0.8717759847640991 0.788454395532608 0.7744799852371216 0.03174746482604812 2048 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 16384 0.375 128.0 128.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.8971818172100244 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
72 0.03830400109291077 0.06268800050020218 0.043337599746882914 0.040511999279260635 0.005823016247946116 0.023135999217629433 0.03747199848294258 0.025206399988383053 0.02393599972128868 0.003271381256231161 0.0 0.0 0.0 0.0 0.0 1.363935947418213 1.4143040180206299 1.3812703967094422 1.3798720240592957 0.014770450075530007 4096 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 32768 0.375 256.0 256.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.8113207890716184 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
73 0.05660799890756607 0.07932800054550171 0.06249920018017292 0.06039999984204769 0.005636461691480845 0.02956799976527691 0.03747199848294258 0.031126399897038935 0.030287999659776688 0.002157571006631541 0.0 0.0 0.0 0.0 0.0 2.5507519245147705 2.680704116821289 2.579859209060669 2.566223978996277 0.03673283558365955 8192 128 128 1 standard_fused_topk fixed_hotset8 softmax_renorm False standalone_legacy vllm020_replicated_linear 8 2048 768 True 4 65536 0.375 512.0 512.0 0.0625 0.0 3.872983346207417 16.0 3.0 0.9375 hotset8 20260716 FlashInfer CUTLASS CUDA_EVENT BF16 generic none 0.883909666885606 measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize

View File

@@ -0,0 +1,8 @@
3937c623adcea3d379b82bb7ab63d3b293f917fa59fa6fd6a147581d6a252cb6 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/raw/flashattn-code-longctx-tp1.json
4e6d6002db037bdc71d40f76d6de979ef47960612c15dd985c04048c25bd1d41 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/aituner.commit
420e4a613b1bcdb44b873eaa911f4a23129713ff9d486fc783b31a009a55cc85 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/batch-specs.txt
5dd612a7806900e9d12afc27ccef196cc5518aea831aa4ed3b0a53bcfb5746cf runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/max-model-len.txt
484da176e7355ab5532d5129a8e78bea0ae84b3aa0a42c7f0bea9689667762ff runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/pip-freeze.txt
595b25965f76bf7ac25aa3b60126a437a083619303303aaf63346e993862253a runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/source.sha256
4355a46b19d348dc2f57c046f8ef63d4538ebb936000f3c9ee954a27460dd865 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/vllm-allow-long-max-model-len.txt
11e8f5af440e3db4c01b519a7b0e30bbcecc3a57bfb53dd86d32e601beb95a32 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/vllm-source.commit

View File

@@ -0,0 +1 @@
e251046c306a8e03214fe1b2f40caabb98f76b7b

View File

@@ -0,0 +1,13 @@
q8ks48k
q8ks64k
q8ks80k
q8ks96k
q8ks112k
q8ks128k
q8ks136k
q512s128k
q2ks66k
q4ks100k
q6ks134k
q1ks8k
q512s4k

View File

@@ -0,0 +1,179 @@
aiohappyeyeballs==2.7.1
aiohttp==3.14.1
aiosignal==1.4.0
annotated-doc==0.0.4
annotated-types==0.7.0
anthropic==0.117.0
anyio==4.14.2
apache-tvm-ffi==0.1.9
astor==0.8.1
attrs==26.1.0
blake3==1.0.9
cachetools==7.1.4
cbor2==6.1.3
certifi==2026.6.17
cffi==2.1.0
charset-normalizer==3.4.9
click==8.4.2
cloudpickle==3.1.2
compressed-tensors==0.15.0.1
cryptography==49.0.0
cuda-bindings==12.9.7
cuda-pathfinder==1.5.6
cuda-python==12.9.7
cuda-tile==1.5.0
cuda-toolkit==12.9.1
depyf==0.20.0
detect-installer==0.1.0
dill==0.4.1
diskcache==5.6.3
distro==1.9.0
dnspython==2.8.0
docstring-parser==0.18.0
einops==0.8.2
email-validator==2.3.0
fastapi==0.139.2
fastapi-cli==0.0.32
fastapi-cloud-cli==0.22.2
fastar==0.11.0
fastsafetensors==0.3.3
filelock==3.31.1
flashinfer-cubin==0.6.8.post1
flashinfer-python==0.6.8.post1
frozenlist==1.8.0
fsspec==2026.6.0
gguf==0.19.0
googleapis-common-protos==1.75.0
grpcio==1.82.1
h11==0.16.0
hf-xet==1.5.2
httpcore==1.0.9
httptools==0.8.0
httpx==0.28.1
httpx-sse==0.4.3
huggingface-hub==1.24.0
idna==3.18
ijson==3.5.1
interegular==0.3.3
jinja2==3.1.6
jiter==0.16.0
jmespath==1.1.0
jsonschema==4.26.0
jsonschema-specifications==2025.9.1
lark==1.2.2
llguidance==1.3.0
llvmlite==0.47.0
lm-format-enforcer==0.11.3
loguru==0.7.3
markdown-it-py==4.2.0
markupsafe==3.0.3
mcp==1.28.1
mdurl==0.1.2
mistral-common==1.11.6
ml-dtypes==0.5.4
model-hosting-container-standards==0.1.16
mpmath==1.3.0
msgspec==0.21.1
multidict==6.7.1
networkx==3.6.1
ninja==1.13.0
numba==0.65.0
numpy==2.3.5
nvidia-cublas-cu12==12.9.1.4
nvidia-cuda-cupti-cu12==12.9.79
nvidia-cuda-nvrtc-cu12==12.9.86
nvidia-cuda-runtime-cu12==12.9.79
nvidia-cudnn-cu12==9.17.1.4
nvidia-cudnn-frontend==1.18.0
nvidia-cufft-cu12==11.4.1.4
nvidia-cufile-cu12==1.14.1.1
nvidia-curand-cu12==10.3.10.19
nvidia-cusolver-cu12==11.7.5.82
nvidia-cusparse-cu12==12.5.10.65
nvidia-cusparselt-cu12==0.7.1
nvidia-cutlass-dsl==4.5.3
nvidia-cutlass-dsl-libs-base==4.5.3
nvidia-ml-py==13.610.43
nvidia-nccl-cu12==2.28.9
nvidia-nvjitlink-cu12==12.9.86
nvidia-nvshmem-cu12==3.4.5
nvidia-nvtx-cu12==12.9.79
openai==2.46.0
openai-harmony==0.0.8
opencv-python-headless==5.0.0.93
opentelemetry-api==1.44.0
opentelemetry-exporter-otlp==1.44.0
opentelemetry-exporter-otlp-proto-common==1.44.0
opentelemetry-exporter-otlp-proto-grpc==1.44.0
opentelemetry-exporter-otlp-proto-http==1.44.0
opentelemetry-proto==1.44.0
opentelemetry-sdk==1.44.0
opentelemetry-semantic-conventions==0.65b0
opentelemetry-semantic-conventions-ai==0.5.1
outlines-core==0.2.14
packaging==26.2
partial-json-parser==0.2.1.1.post7
pillow==12.3.0
prometheus-client==0.25.0
prometheus-fastapi-instrumentator==8.0.2
propcache==0.5.2
protobuf==7.35.1
psutil==7.2.2
py-cpuinfo==9.0.0
pybase64==1.4.3
pycountry==26.2.16
pycparser==3.0
pydantic==2.13.4
pydantic-core==2.46.4
pydantic-extra-types==2.11.1
pydantic-settings==2.14.2
pygments==2.20.0
pyjwt==2.13.0
python-dotenv==1.2.2
python-json-logger==4.1.0
python-multipart==0.0.32
pyyaml==6.0.3
pyzmq==27.1.0
quack-kernels==0.5.0
referencing==0.37.0
regex==2026.7.19
requests==2.34.2
rich==15.0.0
rich-toolkit==0.20.3
rignore==0.8.0
rpds-py==2026.6.3
safetensors==0.8.0
sentencepiece==0.2.2
sentry-sdk==2.66.0
setproctitle==1.3.7
setuptools==80.10.2
shellingham==1.5.4
six==1.17.0
sniffio==1.3.1
sse-starlette==3.4.5
starlette==1.3.1
supervisor==4.3.0
sympy==1.14.0
tabulate==0.10.0
tiktoken==0.13.0
tilelang==0.1.9
tokenizers==0.22.2
torch==2.11.0+cu129
torch-c-dlpack-ext==0.1.5
torchaudio==2.11.0+cu129
torchvision==0.26.0+cu129
tqdm==4.69.0
transformers==5.14.1
triton==3.6.0
typer==0.27.0
typing-extensions==4.16.0
typing-inspection==0.4.2
urllib3==2.7.0
uvicorn==0.51.0
uvloop==0.22.1
vllm @ https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0%2Bcu129-cp38-abi3-manylinux_2_31_x86_64.whl
watchfiles==1.2.0
websockets==16.1.1
xgrammar==0.2.3
yarl==1.24.5
z3-solver==4.15.4.0

View File

@@ -0,0 +1,2 @@
42381f4bd1b67710d4068eb8993240930326a804c6e5167b0b78df3420e47c74 /home/admin/cpfs/wjh/aituner/aituner-code-trace-fbaa909/runs/frontier-code-trace-v0/../frontier-qwen30-vllm020-profile-v1/profile_vllm020_flashattn.py
8a7a72b3adb25db6a61a86f9b04d2d19ebbfe3f2d20480852ef2f45376ac3b85 runs/frontier-code-trace-v0/run_flashattn_code_longctx.sh

View File

@@ -0,0 +1 @@
88d34c6409e9fb3c7b8ca0c04756f061d2099eb1

View File

@@ -0,0 +1,564 @@
{
"environment": {
"attention_backend": "FLASH_ATTN",
"block_size": 16,
"dtype": "bfloat16",
"gpu": "NVIDIA H20",
"model": "/home/admin/cpfs/wjh/models/Qwen/Qwen3-30B-A3B",
"profile_kv_update": true,
"profile_method": "cuda_event",
"torch_cuda": "12.9",
"torch_version": "2.11.0+cu129",
"vllm_source_commit": "88d34c6409e9fb3c7b8ca0c04756f061d2099eb1",
"vllm_version": "0.20.0"
},
"rows": [
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks48k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.0851840004324913,
"mean_ms": 0.07839040011167527,
"median_ms": 0.07703999802470207,
"min_ms": 0.07552000135183334,
"std_ms": 0.0030805904904383677
},
"max_time": 0.043892990112304686,
"mean_time": 0.04368390693664551,
"memory_allocated_mb": 240.0849609375,
"memory_reserved_mb": 246.0,
"min_time": 0.043620254516601564,
"std_time": 8.458163192147439e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 187529.01410308387
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks64k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.0854400023818016,
"mean_ms": 0.07895359992980958,
"median_ms": 0.07791999727487564,
"min_ms": 0.07718399912118912,
"std_ms": 0.0024425064759109735
},
"max_time": 0.059730175018310544,
"mean_time": 0.0595021312713623,
"memory_allocated_mb": 272.0888671875,
"memory_reserved_mb": 374.0,
"min_time": 0.05942179107666016,
"std_time": 0.00010367381308510448,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 137675.74076699864
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks80k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.09055999666452408,
"mean_ms": 0.08208959847688675,
"median_ms": 0.08087999746203423,
"min_ms": 0.07875200361013412,
"std_ms": 0.0033614630055883482
},
"max_time": 0.07555481719970703,
"mean_time": 0.07538758392333984,
"memory_allocated_mb": 304.0927734375,
"memory_reserved_mb": 534.0,
"min_time": 0.0752852783203125,
"std_time": 8.870109119325343e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 108665.10867797918
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks96k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.09888000041246414,
"mean_ms": 0.0819871999323368,
"median_ms": 0.08008000254631042,
"min_ms": 0.07788799703121185,
"std_ms": 0.005835618489111142
},
"max_time": 0.09145164489746094,
"mean_time": 0.09124225158691406,
"memory_allocated_mb": 336.0966796875,
"memory_reserved_mb": 726.0,
"min_time": 0.09113442993164063,
"std_time": 0.0001009018902177725,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 89782.96630697016
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks112k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.08790399879217148,
"mean_ms": 0.07967040091753005,
"median_ms": 0.07808000221848488,
"min_ms": 0.07596799731254578,
"std_ms": 0.0036209252740976605
},
"max_time": 0.10734063720703126,
"mean_time": 0.1069560287475586,
"memory_allocated_mb": 368.1005859375,
"memory_reserved_mb": 950.0,
"min_time": 0.10689055633544922,
"std_time": 0.00013032874360676438,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 76592.22295299545
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.08534400165081024,
"mean_ms": 0.07928640022873878,
"median_ms": 0.07787200063467026,
"min_ms": 0.07648000121116638,
"std_ms": 0.0028305104107121735
},
"max_time": 0.1229840316772461,
"mean_time": 0.12283342437744141,
"memory_allocated_mb": 400.1044921875,
"memory_reserved_mb": 1206.0,
"min_time": 0.12271724700927734,
"std_time": 9.625274868603512e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 66691.94514049937
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks136k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.08774399757385254,
"mean_ms": 0.08081279993057251,
"median_ms": 0.07972799986600876,
"min_ms": 0.07664000242948532,
"std_ms": 0.003487270970977775
},
"max_time": 0.13109359741210938,
"mean_time": 0.13079229431152345,
"memory_allocated_mb": 416.1064453125,
"memory_reserved_mb": 1478.0,
"min_time": 0.1306565399169922,
"std_time": 0.00013138911961272913,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 62633.65929255852
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.040063999593257904,
"mean_ms": 0.019459199998527764,
"median_ms": 0.01601599995046854,
"min_ms": 0.01532800029963255,
"std_ms": 0.007496165774486326
},
"max_time": 0.009649087905883789,
"mean_time": 0.009489968109130859,
"memory_allocated_mb": 265.0458984375,
"memory_reserved_mb": 1478.0,
"min_time": 0.009443936347961425,
"std_time": 5.961646816296781e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 53951.70922728124
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q2ks66k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.040832001715898514,
"mean_ms": 0.02942080032080412,
"median_ms": 0.027888000011444092,
"min_ms": 0.02691200003027916,
"std_ms": 0.004040048279091587
},
"max_time": 0.016906623840332032,
"mean_time": 0.01678766403198242,
"memory_allocated_mb": 168.04248046875,
"memory_reserved_mb": 1478.0,
"min_time": 0.016753759384155274,
"std_time": 4.236470233451937e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 121994.34037387963
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q4ks100k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.04867200180888176,
"mean_ms": 0.04565120078623295,
"median_ms": 0.04468800127506256,
"min_ms": 0.04416000097990036,
"std_ms": 0.0018079971851529544
},
"max_time": 0.05050300979614258,
"mean_time": 0.05035054740905761,
"memory_allocated_mb": 272.06640625,
"memory_reserved_mb": 1478.0,
"min_time": 0.05029715347290039,
"std_time": 7.24480602714254e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 81349.66173700758
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q6ks134k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.06684800237417221,
"mean_ms": 0.06276160031557083,
"median_ms": 0.061824001371860504,
"min_ms": 0.060447998344898224,
"std_ms": 0.0024045636532566625
},
"max_time": 0.09653209686279297,
"mean_time": 0.0961898666381836,
"memory_allocated_mb": 376.09033203125,
"memory_reserved_mb": 1478.0,
"min_time": 0.09607142639160156,
"std_time": 0.0001365794879996899,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 63873.67208970714
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q1ks8k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.0326399989426136,
"mean_ms": 0.021955199912190436,
"median_ms": 0.020560000091791153,
"min_ms": 0.018559999763965607,
"std_ms": 0.00393975597463501
},
"max_time": 0.0011526399850845337,
"mean_time": 0.0011425888061523438,
"memory_allocated_mb": 34.0205078125,
"memory_reserved_mb": 1480.0,
"min_time": 0.0011403199434280396,
"std_time": 3.498447348453724e-06,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 896210.4253833097
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s4k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.02783999964594841,
"mean_ms": 0.017366399988532066,
"median_ms": 0.01611199975013733,
"min_ms": 0.015424000099301338,
"std_ms": 0.003547472261377056
},
"max_time": 0.00034601598978042605,
"mean_time": 0.0003366623997688293,
"memory_allocated_mb": 17.015625,
"memory_reserved_mb": 1480.0,
"min_time": 0.0003308480083942413,
"std_time": 4.813544996936389e-06,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 1520811.3539010207
}
],
"schema_version": "qwen30_vllm020_flashattn_raw.v1"
}

View File

@@ -0,0 +1,8 @@
f113ab5a05167f8ac4c1938c4a04fadc11d5dc2bb8c536a69db23638e6ef73d4 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/raw/flashattn-code-longctx-tp1.json
4e6d6002db037bdc71d40f76d6de979ef47960612c15dd985c04048c25bd1d41 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/aituner.commit
420e4a613b1bcdb44b873eaa911f4a23129713ff9d486fc783b31a009a55cc85 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/batch-specs.txt
5dd612a7806900e9d12afc27ccef196cc5518aea831aa4ed3b0a53bcfb5746cf runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/max-model-len.txt
484da176e7355ab5532d5129a8e78bea0ae84b3aa0a42c7f0bea9689667762ff runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/pip-freeze.txt
595b25965f76bf7ac25aa3b60126a437a083619303303aaf63346e993862253a runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/source.sha256
4355a46b19d348dc2f57c046f8ef63d4538ebb936000f3c9ee954a27460dd865 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/vllm-allow-long-max-model-len.txt
11e8f5af440e3db4c01b519a7b0e30bbcecc3a57bfb53dd86d32e601beb95a32 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/vllm-source.commit

View File

@@ -0,0 +1 @@
e251046c306a8e03214fe1b2f40caabb98f76b7b

Some files were not shown because too many files have changed in this diff Show More