Compare commits
227 Commits
93daf291f6
...
feature/si
| Author | SHA1 | Date | |
|---|---|---|---|
| e0811e95b4 | |||
| e1f2557a0c | |||
| 14ed991059 | |||
| d5bb9745f8 | |||
| da480a8761 | |||
| 766f09d3ed | |||
| 7aed90f9e6 | |||
| b80d3f03de | |||
| 89d5ebbbc0 | |||
| 8462afa56f | |||
| 5c45944388 | |||
| 5a4011ca12 | |||
| ab952b47e7 | |||
| 029c8991b6 | |||
| a9e88f14de | |||
| e251046c30 | |||
| 288f7b239f | |||
| 1d9182f305 | |||
| fbaa909723 | |||
| cd7665d882 | |||
| 08921193a1 | |||
| 4f22688bfd | |||
| cf610003ed | |||
| ecc5599381 | |||
| 1126d9be7d | |||
| c1c200b7cd | |||
| cb67ac8621 | |||
| fd859bc52c | |||
| be523b1c07 | |||
| 2c3220c2be | |||
| 9c1175a434 | |||
| 788270183d | |||
| e651ecc923 | |||
| 809ad9ffef | |||
| a033a72195 | |||
| dfe3f345d8 | |||
| 157bf3668d | |||
| f727cbcf76 | |||
| f38e260639 | |||
| 3453fbe522 | |||
| b302954dcb | |||
| ca999c4e49 | |||
| 75946d9d73 | |||
| 39766141fb | |||
| 18c0b25ae7 | |||
| 2970f74d67 | |||
| a6c9beaa89 | |||
| 409d83a876 | |||
| e602d41c80 | |||
| 27b2daf7ce | |||
| 3be6cd04ad | |||
| 51f2072d30 | |||
| 6b80266aa7 | |||
| 9c9479c313 | |||
| 3349b23290 | |||
| d5c3c93577 | |||
| b49c072502 | |||
| 670eca176d | |||
| b6dfbfcad7 | |||
| 5927b6bfc3 | |||
| 10567da523 | |||
| 4ca295af0b | |||
| 80ab724608 | |||
| 506e63a633 | |||
| f4a75aa8e4 | |||
| 9c34650746 | |||
| d17a2d4ab0 | |||
| 3d3878c5aa | |||
| 4c8d581a5b | |||
| e0ea7e9961 | |||
| fbf0f7c50b | |||
| 81f3a5c76d | |||
| 71dbf80ed4 | |||
| 5f48f7ec8b | |||
| 42e4ae3422 | |||
| b72de0fd15 | |||
| b2f97927de | |||
| 979f179a47 | |||
| dfd9646d2c | |||
| 922d66c5c1 | |||
| ac5061fa5d | |||
| d563c30b42 | |||
| a21382c8aa | |||
| 5067bc2cb1 | |||
| 39e4719b28 | |||
| 79e9870975 | |||
| 80a067e3a5 | |||
| 3e32ea609f | |||
| 40bac6dbf4 | |||
| e631f6a269 | |||
| a3c8cb5808 | |||
| f4813cf537 | |||
| c8383c9c4c | |||
| 7ea9635878 | |||
| 33b73afe9b | |||
| 909a80f0a6 | |||
| 9837fa5133 | |||
| 6726318792 | |||
| e98608911e | |||
| f11daa6776 | |||
| b9523cef5c | |||
| b7f9cef9c5 | |||
| ecb45cf762 | |||
| 686b050517 | |||
| 28ffb22bee | |||
| a01c1206ab | |||
| 5fc0b48fa1 | |||
| 7a437b4d91 | |||
| 12705411f9 | |||
| ddacc5f7a6 | |||
| a7e3c0fdf3 | |||
| 44104bd96e | |||
| 65fee8450a | |||
| e6f3e4a690 | |||
| e2cd43808d | |||
| 27a2a3468a | |||
| 2deb53cb72 | |||
| bdc357dc6c | |||
| 47355a9411 | |||
| d3be91dd58 | |||
| e7b482658e | |||
| 8b054ffcb1 | |||
| 903cbe682d | |||
| 03aa794448 | |||
| 2aa713f5d6 | |||
| e9ae04d852 | |||
| ec775cdafd | |||
| 10afe4e2e8 | |||
| b6ef6eeae7 | |||
| 202ae718b3 | |||
| a2cf361ffe | |||
| 6e8704d525 | |||
| 0c747448b6 | |||
| 1fa203384f | |||
| 95f4af3d99 | |||
| 3a59d5df96 | |||
| 97c2f34700 | |||
| 76107d3e87 | |||
| 008324e70c | |||
| 630de9f573 | |||
| 07b1eb4b75 | |||
| 414838a799 | |||
| fcca259475 | |||
| b8f5f1dcf3 | |||
| 885db8527a | |||
| 9b54075e75 | |||
| 0e9b64893f | |||
| 48d2c2fb80 | |||
| 32ab584789 | |||
| 3121e35b0e | |||
| 4bf9bdf28f | |||
| 1fb777fc11 | |||
| 4fed4329cf | |||
| 9bbfb87a85 | |||
| 9bc38d8851 | |||
| a9aed87518 | |||
| 5a958077a7 | |||
| adaf68badc | |||
| a70a1fac71 | |||
| b3e437acb0 | |||
| 2f04971c90 | |||
| fb12e3b502 | |||
| e9a4be6153 | |||
| 13f0713bf2 | |||
| 344af3a428 | |||
| def9e24c4d | |||
| 3db632552b | |||
| 501ceb8171 | |||
| 163fb3d9e8 | |||
| 993cce6143 | |||
| f7052edd14 | |||
| 678388084c | |||
| 84584c719c | |||
| e47bbf3e76 | |||
| ef5c17e6ec | |||
| 91869c531c | |||
| 33c91109d7 | |||
| 1fabea26ef | |||
| dadcdbd351 | |||
| 5dbcd9f51c | |||
| ae933f2256 | |||
| 5fadc86edd | |||
| ed500fa016 | |||
| ca88c06d06 | |||
| 6307620e20 | |||
| 2f730417e0 | |||
| d64f12eb5b | |||
| 76e0de33ee | |||
| 9275619984 | |||
| 6e619b75d2 | |||
| 584af7b253 | |||
| a6f101b09e | |||
| a54f69352d | |||
| c71f379110 | |||
| 97e66ae276 | |||
| e6e6fef41a | |||
| c9a5f0a0c3 | |||
| d9df3003dd | |||
| 684a2de413 | |||
| 9c8570f36b | |||
| 39b767e384 | |||
| e5fd463f05 | |||
| d229f2a85e | |||
| 0d16838097 | |||
| 8c930ba3a1 | |||
| 2af22dbce4 | |||
| 3facb18bcf | |||
| c5ab073af5 | |||
| 1db737e641 | |||
| 823c550e53 | |||
| 26c2cdab2b | |||
| 7fd9563550 | |||
| c0b40af24f | |||
| 2afc6eeb8d | |||
| 52a9dc13dd | |||
| 650f54b35e | |||
| 0515ad8ecc | |||
| 791f7a8889 | |||
| 7a3631b528 | |||
| f01819680d | |||
| 24a9c27b10 | |||
| 2261818994 | |||
| 1f32ae217e | |||
| 4ad699ef97 | |||
| 12d1d4ad02 | |||
| a3b25f4a92 | |||
| 23142aa359 |
22
.gitignore
vendored
22
.gitignore
vendored
@@ -19,3 +19,25 @@ runs/**/*.jsonl
|
||||
.ruff_cache/
|
||||
# Recovered dash1 interaction-run stores (100 MB raw tune logs, kept on disk only)
|
||||
recovered-stores/
|
||||
|
||||
# Local reference material and accidental shell output.
|
||||
/AITuner系统优化与挑战.pdf
|
||||
/16
|
||||
/docs/assets/simulator-fidelity/*.svg
|
||||
|
||||
# Generated experiment state. Protocols, analysis code, compact result tables,
|
||||
# and frozen manifests remain tracked next to these directories.
|
||||
/runs/frontier-phase-factorial-v0/fleet-artifacts*/
|
||||
/runs/frontier-phase-factorial-v0/fleet-state*/
|
||||
/runs/frontier-phase-factorial-v0/invalid-overlap-*/
|
||||
/runs/frontier-phase-factorial-v0/simulator-smoke/
|
||||
/runs/frontier-phase-factorial-v0/simulator-*/cache
|
||||
/runs/frontier-phase-factorial-v0/simulator-*/runs/
|
||||
/runs/frontier-phase-factorial-v0/simulator-*/traces/
|
||||
/runs/frontier-phase-factorial-v0/results/final/qwen30-prefill-ranking.png
|
||||
/runs/frontier-qwen30-vllm020-profile-v1/comparison/
|
||||
/runs/frontier-qwen30-vllm020-profile-v1/fleet-artifacts/
|
||||
/runs/frontier-qwen30-vllm020-profile-v1/fleet-state/
|
||||
/runs/frontier-multicase-sufficiency-v1/fleet-artifacts/
|
||||
/runs/frontier-multicase-sufficiency-v1/fleet-state/
|
||||
/runs/frontier-multicase-sufficiency-v1/frontier-smoke-failure/
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
# 实验 S0:good/bad case 分裂的统一分解(margin vs differential residual)
|
||||
|
||||
> **状态:** 已完成(2026-07-20;含 S0b 方向化修正与一轮 strict review 修复)
|
||||
>
|
||||
> 用户指令:核心要务是分析为什么部分 case 下 Frontier work、部分不 work 的 system 根因;本 card 是该诊断 campaign 的第一个 slice,仅使用 frozen artifacts,零 GPU 成本。人工 review 由用户的直接指令("只有做好这个分析我们才能推进下一步")满足。
|
||||
|
||||
## Claim 与决策
|
||||
|
||||
- **Parent claim:** ongoing.md H2——误差机制是 action-conditioned residual;本实验把它细化为"分裂从哪来"。
|
||||
- **现象(已冻结):** 同一 best-effort Frontier 栈上,Q30/Q235 的 Trace-PD 与多数 PO 面 selection 近优,而 Fixed-PD 面 14–58% regret;失败 objective 随负载档切换(Q30 低压 TPOT/E2E 全反、高压 TTFT 56–58%);A1 collective profile 修复了 Q235 Trace-PO p90 21.2%→0.3% 但对 Fixed-PD 33% 完全无效(本 card frozen-inputs/q235-ablation-a1)。
|
||||
- **Competing hypotheses:**
|
||||
- **H-SCALE:** 分裂完全由「config-differential residual vs 真机 decision margin」的关系解释:good case 的 sim/real 比值跨 config 近似均匀(乘性偏移,argmin 不变),bad case 的比值跨 config 分散且超过 margin。workload shape 本身不需要出现在解释里。
|
||||
- **H-THRESH:** 绝对 service-time 高估近似均匀,但与离散机制(MNS admission cap、MoE token-bucket、graph bucket)交互后被转换为 config-differential 误差;fixed uniform workload 把所有请求同步到同一 state 轨迹,使阈值交叉对整个 cell 相干生效;trace 的长度/到达异质性把阈值效应摊平。
|
||||
- **H-STATE:** 失败由 simulator 闭环 batch state 分布漂移主导(Q235:sim decode batch 13.5 vs real 3.9 + B4→B5 profile cliff 正反馈);即使打破 workload 同步性,闭环漂移仍可翻转排序。
|
||||
- 三者关系:H-SCALE 是现象层(必要条件),H-THRESH/H-STATE 是 differential residual 的两种产生机制,可共存但可判别(见事前预测)。
|
||||
- **事前预测:**
|
||||
- H-SCALE 成立 ⟺ 对每个 case×objective,failure 恰好发生在「top 邻域 log-ratio spread > log1p(真机相对 margin)」处(两侧同为 log-space 尺度),无反例。
|
||||
- H-THRESH 独有:bad case 的 differential 误差集中于阈值语义分量(first-scheduling wait、bucket 跳变段),且 sim-only 反事实(去阈值/加 jitter)恢复排序——Q30 高压 TTFT 的 admission 反事实已支持一例。
|
||||
- H-STATE 独有:差异化误差在去掉阈值分量后仍在 execution 项内(Q235 Fixed-PD 的 own-composition −20.07 vs exact-state +10.90 已支持一例)。
|
||||
- **判定规则:** S0 只裁决 H-SCALE 与「分量定位」(queue vs execution);H-THRESH/H-STATE 的干预判别属 S1+(sim-only 反事实)与 GPU 实验(需另行 review)。若 H-SCALE 出现反例(good case 有 spread>margin 仍选对,或 bad case spread<margin),必须原样报告,不得平滑。
|
||||
|
||||
## Setup
|
||||
|
||||
- **输入(全部 frozen,runs/frontier-split-rootcause-v0/frozen-inputs/):** q30-trace-pd(graph-piecewise comparison)、q30-fixed-hi(fixed-pd/fixed-po 高压面)、q30-expansion-lo(低压 fixed-pd/fixed-po/trace-po)、q235-fourcase-a0、q235-ablation-a1(+provenance)、q235-state-diag、q30-admission-diag。来源 cpfs 路径与 SHA 见各目录内 manifest/launch 记录。
|
||||
- **计算(每 case×objective):** per-config 比值 r_c=sim_c/real_c;config-uniform scale=geomean(r_c);differential residual=log-ratio spread(全 surface 与真机 top-3 邻域各一);真机 relative margin(best 与 2nd-best、best 与 sim-winner 的真机值差);failure flag=regret>5%;H-SCALE 检验=failure ⟺ 邻域 spread>log1p(margin)(review 修正:初版直接以 ln 差比较普通 relative margin,尺度不一致;修正后 70 行 verdict 不变)。
|
||||
- **分量定位:** q30-admission-diag 提供 TTFT=first-scheduling wait+prefill execution 分解;q235-state-diag 提供 own-composition vs exact-state contrast;把这些已知分量证据合并进统一表。
|
||||
- **交叉核对:** 重算的 regret 必须与各 frozen comparison.md 表一致(抽查 58.0%、33.0%、0.0%);A0 vs A1 的 Q235 对比必须复现 trace-po p90 21.2%→0.3%、fixed-pd 四项不变。
|
||||
|
||||
## 预期产物与 review
|
||||
|
||||
- runs/frontier-split-rootcause-v0/analyze_split_decomposition.py(只读 frozen-inputs,确定性输出)
|
||||
- runs/frontier-split-rootcause-v0/results/decomposition.{json,md}:统一表,每行 case×objective,列出 winner、regret、scale、spread(全/邻域)、margin、H-SCALE verdict、已知分量归因
|
||||
- runs/frontier-split-rootcause-v0/results/margin-vs-residual.png:x=真机 margin,y=邻域 differential residual,点色=selection 对错;H-SCALE 成立则对错点被对角线分离
|
||||
- 人工验收:编排者亲自重跑脚本、抽查交叉核对数字、亲自查看渲染图
|
||||
|
||||
## 复现信息
|
||||
|
||||
- **Code:** AITuner branch feature/sim,自 HEAD 18c0b25 起;worker/reviewer job-id 见下方「Review 与 provenance 补记」;脚本与产物随本 card 同一 commit 入库(含 frozen-inputs 本地拷贝)。
|
||||
- **Environment:** 本地 workstation,CPU-only,python3+matplotlib;不访问远端。
|
||||
- **已知 deviation:** frozen-inputs 是 cpfs 原件的本地拷贝(scp,2026-07-20);q30-expansion-lo 的低压 Fixed-PD 面已被高压面取代为 primary,本分析将两档并列为独立观测,不混合。
|
||||
|
||||
## 结果
|
||||
|
||||
- **观察事实:**
|
||||
- 70 行(14 个 case surface)全部算出,无数据缺口;四组硬性交叉核对通过;连续运行产物 SHA 一致。
|
||||
- **H-SCALE 判为必要非充分**:23 个 material failure(regret>5%)全部满足「top-3 邻域 log spread > log1p(margin)」,无一例失败发生在 residual 小于 margin 处;但另有 40/70 行同样满足该条件却均非 material failure(其中仅 16 行 exact winner match,其余 24 行是小 regret 的 winner 错位——14 个 MNS 精确 tie 与 10 个 strict reversal)——无方向 spread 不携带决策信息。
|
||||
- **S0b 方向化后的机制普查**:23 个 material failure 的 winner-deciding pair 分布为 tp-axis 11、mixed 10、mns-axis 2(Q235 A0/A1 Fixed-PD 的 8 个 TPOT/E2E failure 全为 tp-axis;Q30 Fixed-PD 高低压为 tp/mixed);仅有的 2 个 mns-axis failure 是 Q235 A0/A1 Trace-PD E2E p90(regret 6.2%,勉强越过 5% 门槛)。trace 面严格反序中 tp-axis 为 0(q30 trace-pd 三轴全 0)。
|
||||
- **A1 对照的轴分解**:serving-matched collective profile 把 Q235 两个 PO 面的 tp-axis 反序从 4/4 清零(trace-po p90 regret 21.2%→0.3%),但 Fixed-PD 仅 5→4、四项 regret 一位小数不动——prefill 路径的 TP-differential 误差源=collective profile(可修),decode 耦合的 TP-differential 误差另有来源。
|
||||
- **MNS 不敏感缺陷**:14 个 winner-label mismatch 是 simulator 逐位相等的 tie,全部 mns-axis(如 q30 fixed-po 的 MNS16↔32、q235 fixed-pd 的 MNS64↔128);tie 计入后 MNS 边界误差 31 与 TP 严格反序 32 相当,但 MNS 侧 regret 小。
|
||||
- **成功的鲁棒性**:23 个 exact-winner success 中 17 个 margin-robust(margin≥1%),6 个 fragile(含 q30 trace-pd E2E p90 的 0.1% margin 与 q235 A1 fixed-po 四项)。
|
||||
- **面级 scale 对照**:prefill-only 面 geomean scale 0.96–1.37×(绝对预测基本准确),含 decode 的面 4.3–130×——绝对误差灾难集中于 decode。
|
||||
- **异常:** 无数据异常。strict review(FAIL:3 Major/1 Minor)指出 H-SCALE 尺度混用(log spread vs relative margin)、tie 轴普查缺失、card 状态过期、Q235 一致性表述过强;全部修复,修复后 70 行 verdict 逐行不变。
|
||||
- **含义:** 分裂的现象层解释是「margin 保护 + config-differential 误差」。机制层上,21/23 个 material failure 由 TP/mixed pair 决定,且所有大 regret(≥13%)failure 都发生在含 decode 的面上:其中 Q235 Fixed-PD 有 state-drift 直接证据、Q30 高压 TTFT 有 admission 反事实证据,而 **Q30 低压 TPOT/E2E 反转的机制尚未诊断**(S1 目标)。「decode 耦合的 TP-differential 误差是主要载体」是当前最强归纳,不是对全部 failure 的已证机制归因;2 个 mns-axis 边缘 failure(6.2%)在该归纳之外。trace 面成功伴随「TP 反序为零 + TP margin 宽」,但「误差小」与「margin 宽」谁是主因仍未判——这正是 H-THRESH vs H-STATE 的判别缺口。轴标签与机制不一一对应(Q30 admission 是 MNS 阈值机制但 deciding pair 为 tp/mixed,因 TP 改变到达压力)。
|
||||
- **Claim update:** H2(action-conditioned residual)supported 且被细化:residual 的决策相关分量集中在 TP 轴、由 decode 状态耦合产生;H-SCALE 降级为必要条件;H-THRESH/H-STATE 保持 competing,待 S1 判别。
|
||||
- **下一步:** S1(sim-only 反事实:Q30 低压 Fixed-PD TPOT 反转的分量定位——这是唯一无机制解释的 material failure;fixed workload jitter 判别 H-THRESH vs H-STATE);GPU 判别实验(加压 Trace-PD、jittered Fixed-PD 真机面,dash1–4)另行出 card 供 review。
|
||||
|
||||
## Review 与 provenance 补记
|
||||
|
||||
- S0 worker:codex `task-mrsn1s9c-z6ha0w`;S0b:`task-mrsnjzn6-k18x4n`(resume);strict reviewer(fresh 只读):`task-mrsocmlb-386btc`(verdict FAIL);修复轮:`task-mrsop06n-pwxo0b`(fresh writable)。编排者独立验收:脚本重跑、SHA 比对、两图目视检查。
|
||||
- 产物 SHA(修复后):decomposition.json `81ea56b2…`、decomposition.md `4d19af54…`、margin-vs-residual.png `df2110c4…`、decision-pair-axis.png `2e787301…`。
|
||||
@@ -0,0 +1,183 @@
|
||||
# Frontier workload-regime taxonomy
|
||||
|
||||
- Date: 2026-07-20
|
||||
- Status: proposed; awaiting review before workload generation or GPU runs
|
||||
- Scope: explain when Frontier preserves the real-system config ranking, rather than merely comparing Fixed with Trace
|
||||
|
||||
## Claim under test
|
||||
|
||||
Frontier reliability is controlled by three quantities:
|
||||
|
||||
1. the latency-model residual between simulator and real execution;
|
||||
2. the closed-loop gain from timing to scheduler state (batch, MoE routing, CUDA-graph bucket, MNS occupancy, admission/KV pressure);
|
||||
3. the real decision margin between configurations.
|
||||
|
||||
For a config pair `a,b`, define
|
||||
|
||||
```text
|
||||
D_real(a,b) = log L_real(a) - log L_real(b)
|
||||
delta(a,b) = [log L_sim(a)-log L_real(a)]
|
||||
- [log L_sim(b)-log L_real(b)]
|
||||
slack(a,b) = sign(D_real) * [D_real + delta]
|
||||
```
|
||||
|
||||
`slack < 0` means the simulator reverses the real pairwise ordering. The primary hypothesis is that reversals occur when simulator and real execution land on different sides of a scheduler-state knee, or when the real decision margin is too small to absorb the differential residual. `Fixed` and `Trace` are not themselves the causal classes.
|
||||
|
||||
## Existing evidence motivating the experiment
|
||||
|
||||
- Q30 Trace-PD preserves all six objective winners, but many pairwise residuals oppose the real winner. Its success is therefore often margin protection, not zero residual.
|
||||
- Q235 Trace-PD preserves TTFT/TPOT winners but misses E2E p90 by 6.2%; Trace is not universally safe.
|
||||
- Q30/Q235 Fixed-PD decode objectives show negative minimum signed slack and 13--37% regret.
|
||||
- In Q30 low-load Fixed-PD, Frontier's batch-1 TP ordering is correct, while the closed-loop simulator increases TP4's effective batch and changes the MoE cost enough to reverse the ordering. This identifies a concrete state knee, but does not yet establish a general rule.
|
||||
|
||||
## Workload families
|
||||
|
||||
All comparisons use the same request multiset where applicable, the same total observation window, and the same normalized offered decode load
|
||||
|
||||
```text
|
||||
rho = request_rate * E[output_tokens] / measured_reference_decode_capacity.
|
||||
```
|
||||
|
||||
This avoids equating equal request rates with equal load.
|
||||
|
||||
| ID | Shape / request lengths | Arrival process | Prefix/session state | Isolated effect |
|
||||
|---|---|---|---|---|
|
||||
| W0 | short fixed `2048 -> 128` | uniform | off | known low-residence failure anchor |
|
||||
| W1 | trace-mean fixed ISL/OSL | uniform | off | homogeneous baseline |
|
||||
| W2 | trace-mean fixed ISL/OSL | trace timestamps | off | arrival burst only |
|
||||
| W3 | exact trace ISL/OSL multiset | uniform | off | length heterogeneity only |
|
||||
| W4 | exact trace ISL/OSL multiset | trace timestamps | off | length + burst |
|
||||
| W5 | exact trace prompts/ISL/OSL | uniform | exact prefix/session identity | prefix state without burst |
|
||||
| W6 | exact trace prompts/ISL/OSL | trace timestamps | exact prefix/session identity | full production trace |
|
||||
|
||||
Prefix is intentionally a nested factor: enabling a synthetic prefix graph on fixed identical requests would introduce a different workload rather than isolate production prefix reuse. Therefore this is not presented as a full `2^3` factorial.
|
||||
|
||||
## Load sweep and expected patterns
|
||||
|
||||
Simulator discovery sweep: `rho in {0.05, 0.25, 0.50, 0.90, 1.20}`. The points mean deep low load, light batching, moderate batching, capacity knee, and overload; their request rates are derived independently for every workload family.
|
||||
|
||||
| Pattern | Observable state | Prediction for Frontier |
|
||||
|---|---|---|
|
||||
| P1 singleton-linear | real and sim stay below the first batch/graph knee | works if the batch-1 operator ordering is correct |
|
||||
| P2 knee-straddling | real and sim occupy opposite sides of a batch/MoE/graph/MNS knee | fails systematically; Fixed-PD is the current example |
|
||||
| P3 same-side batched | both systems cross the same knee and remain below admission pressure | works if batch-conditioned operator ordering is correct |
|
||||
| P4 capacity/admission aligned | both systems are governed by the same capacity bottleneck | TTFT/config winner may work despite large absolute error; E2E/MNS can remain fragile |
|
||||
| P5 heterogeneity-smoothed | broad lengths reduce coherent threshold occupancy at matched `rho` | may work; this is a hypothesis, not an established explanation |
|
||||
| P6 burst-sensitive | same request multiset, but transient queue/MNS occupancy differs | mean ranking may work while TTFT/E2E tail ranking fails |
|
||||
| P7 prefix-state-sensitive | hit/eviction and reused-token distributions differ | TTFT ranking fails unless prefix-state transitions are modeled; decode TPOT may remain stable |
|
||||
| P8 decision-boundary | real config margin is comparable to run variance/residual | fragile; an exact winner match is not reliable evidence |
|
||||
|
||||
## Hypotheses and distinguishing tests
|
||||
|
||||
### H1: state-regime hypothesis (primary)
|
||||
|
||||
I believe config-ranking failures occur when the latency residual moves a workload across a scheduler-state knee, because the residual is then amplified into a different batch/resource trajectory. I will verify this by checking whether signed-slack zero crossings co-locate with measured real/simulator state-knee crossings.
|
||||
|
||||
### H2: heterogeneity-smoothing hypothesis
|
||||
|
||||
I believe length heterogeneity can reduce coherent threshold amplification, because requests reach scheduler boundaries at dispersed times. I will verify it with W1 vs W3 and W2 vs W4 at matched `rho`, requiring a smaller real/sim state-distribution gap rather than merely a correct winner.
|
||||
|
||||
### H3: bottleneck/margin-protection alternative
|
||||
|
||||
Trace success may instead be explained entirely by a large real decision margin or a shared capacity bottleneck. This hypothesis wins over H2 if W3/W4 do not reduce state-distribution error after matching load and margin, while ranking correctness remains predicted by margin alone.
|
||||
|
||||
### H4: burst and prefix are independent failure channels
|
||||
|
||||
I believe arrival bursts primarily affect waiting/admission and tail TTFT/E2E, whereas prefix mismatch primarily affects prefill/TTFT state. I will verify this with W1/W2, W3/W4, and W3/W5 paired comparisons.
|
||||
|
||||
## Configuration and model scope
|
||||
|
||||
Discovery uses Qwen30B because its 12-cell `TP x MNS` surface already has simulator and real anchors:
|
||||
|
||||
- TP: `{1, 2, 4}`
|
||||
- MNS: `{8, 16, 32, 64}`
|
||||
- objectives: mean/p90 TTFT, TPOT, E2E
|
||||
|
||||
Qwen235B is a held-out confirmation, not pooled into discovery:
|
||||
|
||||
- existing four feasible TP/MNS configurations;
|
||||
- only the workload/load patterns that discriminate H1--H4 after Q30 converges.
|
||||
|
||||
## Measurements
|
||||
|
||||
End-to-end:
|
||||
|
||||
- completed/failed requests and achieved request/token rate;
|
||||
- TTFT, TPOT, E2E mean/p50/p90/p95;
|
||||
- config regret, pairwise agreement, signed decision slack;
|
||||
- run-to-run winner stability.
|
||||
|
||||
Closed-loop state:
|
||||
|
||||
- prefill/decode batch-size histograms and time-weighted batch;
|
||||
- Running/Waiting distributions and admission delay;
|
||||
- MNS active-token occupancy and KV/context pressure;
|
||||
- CUDA-graph bucket residency and fallback frequency;
|
||||
- prefix hit/reused-token/eviction distributions for W5/W6.
|
||||
|
||||
## Decision rules
|
||||
|
||||
A workload/load region is:
|
||||
|
||||
- **reliable** if regret is at most 5%, pairwise agreement is at least 0.8 at two adjacent load points, and the winner is stable across confirmation trials;
|
||||
- **fragile** if regret is at most 5% but the real margin overlaps run uncertainty, or a small rate/timing perturbation changes the winner;
|
||||
- **failed** if regret exceeds 5% or a decision-critical pair has negative signed slack;
|
||||
- **mechanistically explained by H1** only if the ranking transition co-locates with an observed state-regime transition. Correlation with the Fixed/Trace label is insufficient.
|
||||
|
||||
H2 is supported only if the heterogeneous member of a matched pair reduces state-distribution error and shifts the failure boundary in repeated trials. A correct winner alone does not support smoothing.
|
||||
|
||||
## Execution plan after review
|
||||
|
||||
1. Materialize W0--W6 with one manifest recording request multiset, arrival timestamps, prefix identity, rate contract, and hashes.
|
||||
2. Run the simulator sweep across `rho` and the Q30 surface; emit a per-stage state ledger.
|
||||
3. Select real-machine pilot points only around the predicted knees plus one safe-side control. Use guard configs `TP1/MNS64`, `TP4/MNS8`, and `TP4/MNS64`; add `TP2/MNS32` only if the transition is not bracketed.
|
||||
4. Use only `dash1`, `dash2`, `dash3`, and `dash4`, each verified as an 8×H20 host. `dash0` is excluded from probing, synchronization, and execution. Pin one independent experiment group to each host so at most four groups run in parallel; do not split one trial across hosts.
|
||||
5. Run one pilot trial per selected point. Confirm only hypothesis-discriminating points with three fresh-server trials and rotated order.
|
||||
6. Apply the resulting classifier unchanged to the Q235 held-out cases.
|
||||
|
||||
Provisional four-way allocation after the simulator identifies the discriminating points:
|
||||
|
||||
| Host | Experiment group | Primary contrast |
|
||||
|---|---|---|
|
||||
| dash1 | homogeneous controls | W0/W1 across safe side and first knee |
|
||||
| dash2 | arrival effect | W1 vs W2 and W3 vs W4 |
|
||||
| dash3 | length heterogeneity | W1 vs W3 and W2 vs W4 |
|
||||
| dash4 | prefix/full trace | W3 vs W5 and W4 vs W6 |
|
||||
|
||||
The groups are logical queues, not permanent ownership: if a host probe fails, that host is excluded and its group waits or moves to another permitted idle host. Cross-host latency values are not pooled until a common canary config verifies that host effects are within run uncertainty.
|
||||
|
||||
No GPU run is authorized by this card yet. The review decision is whether the workload decomposition and decision rules are sufficient to implement the materializer and launch Phase 1.
|
||||
|
||||
## Expected figure
|
||||
|
||||
The accompanying mock figure is schematic, not data. Panel A shows the state knee that real and simulator trajectories may cross at different loads. Panel B shows the corresponding minimum signed decision slack; a negative value denotes a ranking reversal. The claim is supported only if measured zero crossings and state knees align across workload families.
|
||||
|
||||
## Risks and controls
|
||||
|
||||
- Equal `rho` does not guarantee equal prefill pressure; report both prefill and decode offered work and stratify if necessary.
|
||||
- Full-trace overload can collapse all configs to similarly poor latency. Such points identify a capacity-limited region but cannot validate fine-grained ranking.
|
||||
- MNS ties and censored/failed requests can create false winners; exclude invalid cells before calculating regret and report the exclusion.
|
||||
- One trace cannot establish generality. The initial result is a mechanism boundary for this trace/model/hardware, followed by held-out Q235 validation.
|
||||
|
||||
## Execution log
|
||||
|
||||
### 2026-07-20: materialization and simulator launch
|
||||
|
||||
- Code baseline: `feature/sim@157bf36` for the valid v4 sweep.
|
||||
- Hosts probed: `dash1`, `dash2`, `dash3`, `dash4`; each exposed 8 NVIDIA H20 GPUs with 0 MiB used at probe time. `dash0` was not probed or used.
|
||||
- Source cohort: 129 Q30 Trace-PD requests. The private artifact supplies exact prompts, lengths, outputs, timestamps, sessions, and runtime block identities; the simulator projection retains only the first `floor(ISL/16)` complete block identities.
|
||||
- Materialized: 35 cases = W0--W6 × `rho {0.05,0.25,0.50,0.90,1.20}`. Audit passed request count, exact decode offered load, empirical arrival rate, prefix block count, and prefix-off empty identity vectors.
|
||||
- Simulator smoke: W0 / `rho=0.05` / TP4-MNS64 completed 129/129. Simulator TTFT mean/p90 was 109.81/124.16 ms and TPOT mean/p90 was 36.26/36.79 ms. This is a harness check, not real-system fidelity evidence.
|
||||
- Invalid attempts retained for audit: v1 had a Bash argument-expansion error; v2 mixed multiple workload families into a runner that requires strictly increasing anchors from one family; v3 exposed a scikit-learn cache-version mismatch. None is used as scientific evidence.
|
||||
- Valid v4 controls: isolated output/predictor cache per TP/prefix group; scikit-learn 1.9.0 matching the predictor cache format; per-family five-point runner invocations; stage batch ledger enabled; TP1 exempted from the collective fallback gate because a single rank has no all-reduce.
|
||||
- Active v4 allocation: dash1=TP1 prefix off/on, dash2=TP2 prefix off/on, dash3=TP4 prefix off, dash4=TP4 prefix on. The four fleet jobs are running from fresh `sim-v4` output roots. First-process audit found the explicit isolated `--metrics_config_cache_dir` on all hosts and zero cross-version warnings.
|
||||
- First valid v4 tranche: 16/16 observed cells completed, each with 129 requests, request metrics, and a stage-batch ledger; no traceback, fallback, or version warning was found. The tranche covers all five W0 load points at TP1/TP2/TP4-MNS8 plus the first W5 prefix points at TP4-MNS8.
|
||||
- Early load-boundary observation: W0 at `rho=0.05` is low-latency for TP4-MNS8 (simulator TTFT mean 109.25 ms) but already queues for TP1-MNS8 (25.70 s); at `rho=0.25`, even TP4-MNS8 reaches 27.13 s mean TTFT. Because `rho` normalizes decode tokens only, high-rate short-output W0 also raises prefill and active-sequence pressure. These points map the overload boundary and are not eligible as reasonable-latency real pilots.
|
||||
- Real-runtime gate: a stock vLLM 0.20.0 environment passed import/H20 checks but used CUDA 13.0, so it is excluded from comparison with the historical CUDA 12.9 baseline. The replacement environment `vllm-0.20.0-cu129-workload-regime-v2` passes `vllm CLI=0.20.0+cu129`, torch `2.11.0+cu129`, CUDA runtime 12.9, H20 visibility, and all 179 package dependency checks. The first CPFS install used file copies and was stopped after download because it was still copying roughly 7 GB after 12 minutes; its incomplete directory is retained with an `invalid-copy-incomplete` suffix, while v2 uses same-filesystem hardlinks from the validated cache.
|
||||
- Load-contract correction: the original Fixed-PD surface held request rate per GPU constant, so global arrival rate scaled with TP. The v4 sweep holds global arrival rate constant and is retained as the control that isolates service-topology changes. A matched per-GPU sweep is now required to reproduce the original closed-loop intervention: TP1/TP2/TP4 receive `1x/2x/4x` global arrival rate at the same per-GPU `rho`.
|
||||
- Per-GPU low-load materialization: 105 cases = W0--W6 × `rho {0.0025,0.005,0.01,0.02,0.05}` × TP `{1,2,4}` were generated under `traces-per-gpu-low`. Audit passed 105 unique paths, 129 public/private rows per case, digests, arrival alignment, and exact `global_rate / TP = per_gpu_rate`. W0 `rho=0.01` is 0.239375 req/s/GPU, bracketing the original 0.215 req/s/GPU Fixed-PD point with `rho=0.005`.
|
||||
- The per-GPU sweep writes to a separate `sim-per-gpu-v1` result root but reuses the completed v4 predictor cache for the same TP/prefix/config. Predictor cache provenance is explicit in every surface manifest; workload results and state ledgers are never shared.
|
||||
- `wait_and_dispatch_per_gpu.sh` is active locally as a serial gate. It requires all four exact v4 run directories to contain `finished_at` and exit code zero before probing dash1--dash4 and dispatching the four per-GPU jobs; it does not launch a second sweep while v4 is still consuming CPU.
|
||||
- A first materialization attempt rounded both `rho=0.005` and `rho=0.01` to the same `rho0p01` directory. Digest validation stopped before simulator launch; the invalid directories were retained with an `invalid-rho-label-collision` suffix. The label function now preserves up to 12 significant digits and has a regression test.
|
||||
|
||||
Current decision: finish the v4 fixed-global-rate control, then reuse its trained predictors for the low-load per-GPU sweep before selecting discriminating real-machine pilot points. No real latency result from vLLM 0.20.2 will be compared with the historical vLLM 0.20.0 baseline until the runtime-version gate is resolved.
|
||||
@@ -0,0 +1,45 @@
|
||||
# 实验:Qwen235 Fixed-PD state-matched decode diagnosis
|
||||
|
||||
> **状态:** 已批准,执行中
|
||||
>
|
||||
> 本 card 记录 Qwen235 Fixed-PD 在真实 collective profile 后仍保留 30%+ selection regret 的下一层判因实验。
|
||||
|
||||
## Claim 与决策
|
||||
|
||||
- **Parent claim:** Qwen235 Fixed-PD 的错误排序来自 action-conditioned decode residual,而不是缺失的 TP8 all-reduce profile。
|
||||
- **目的:** 区分 simulator 错在 state distribution,还是相同 state 下的 conditional execution-time composition。
|
||||
- **Competing hypotheses:** H1:Frontier 生成的 decode batch/context/graph state 与真机不同;H2:state 对齐后 Frontier 仍把 TP8/EP8 预测得更快,误差位于 MoE/EP、graph 或 attention 的 conditional stage model。
|
||||
- **事前预测:** 真机 config contrast 为 `TP8-TP4=+6.95 ms/token`,A1 simulator 为 `-20.07 ms/token`。若 H1 成立,用真实 state 重加权后 contrast 应翻正;若 H2 成立,matched-state contrast 仍为负。
|
||||
- **判定规则:** 先比较 frozen-real coarse state 与 full-ledger Frontier state。state 明显不匹配则补 iteration telemetry;state 支持重叠且 matched-state predictor 仍反序,才进入 stage breakdown。任何 stage 只有在 measured substitution 能使 winner 翻转时才称为 decision-bearing root cause。
|
||||
|
||||
## Setup
|
||||
|
||||
- **自变量:** state source(frozen real / Frontier);后续 matched-state replay 中固定 decode batch、context-length、graph bucket 与 routing load。
|
||||
- **控制变量:** Qwen235 FP8、vLLM 0.20.0、H20、Fixed-PD 4096→256、0.2 req/s/GPU、MBT8192、MNS64、TP4/EP1 与 TP8/EP8、Frontier commit、r2 operator profiles、A1 measured collective CSV 全部冻结。
|
||||
- **选择 MNS64:** A1 中 MNS64/128 的 TTFT/TPOT/E2E 完全相同;先去掉不提供判别力的重复维度。
|
||||
- **第一阶段:** CPU-only 重放原 A1 commands,只打开 `frontier_stage_batch_ledger` 与 individual batch metrics;验证 request metrics 与原 A1 bitwise/score 等价。真机先复用 3 次 frozen server logs 的 10 秒 Running/Waiting/KV samples,明确标为 coarse proxy,不冒充 per-iteration batch。
|
||||
- **第二阶段触发条件:** coarse proxy 不足以判断或 state mismatch 显著时,短窗口重跑真机并采集 per-iteration `decode_batch_size/context_length_hist/cudagraph bucket`;否则进入 matched-state whole-decode-step。
|
||||
- **Metrics:** decode batch/token distribution、prefill fraction、scheduler steps/s、graph bucket/padding、queue/KV proxy;configuration contrast `TP8-TP4`;stage measured-substitution 后的 winner。
|
||||
|
||||
## 预期产物与 review
|
||||
|
||||
- **预期数据:** 两个 Frontier full-ledger replays;三次真机日志的 coarse state summary;state overlap/reweighting verdict;必要时的 short-window iteration telemetry。
|
||||
- **Figure prototype:** `../../runs/frontier-fidelity-envelope-v1/qwen235-state-matched-diagnosis-mock.png`。左图对比 real/sim state;右图展示 H1 与 H2 下 matched-state contrast 的可区分方向。全部数值标为 schematic/mock。
|
||||
- **人工 review:** 已批准(用户在分析方案后要求“推进”)。
|
||||
- **Review 意见:** 先做最便宜的 state audit,不直接启动完整 Nsight sweep;每一步只在能改变下一决策时升级证据成本。
|
||||
|
||||
## 复现信息
|
||||
|
||||
- **Code:** AITuner `feature/sim`;运行 commit 待冻结。Frontier `6e8e0d845bceff11b0b62cb29df3a1a93411fdd4`。
|
||||
- **Environment:** dash0;Frontier replay CPU-only;后续真机才使用 4/8×H20。
|
||||
- **输入:** `/home/admin/cpfs/wjh/aituner/qwen235-collective-profile-ablation-20260719-r1/sim/fixed-pd` 与 frozen real campaign `/home/admin/cpfs/wjh/aituner/qwen235-v020-fourcase-20260719-r1/real/fixed-pd`。
|
||||
- **产物路径:** `/home/admin/cpfs/wjh/aituner/qwen235-fixed-pd-state-diagnosis-20260719-r1`。
|
||||
- **已知 deviation:** frozen real logs 的 Running 指标是 10 秒采样的 active-request proxy,不是 scheduler iteration ledger;不能单独支持 matched-state causal claim。
|
||||
|
||||
## 结果
|
||||
|
||||
- **观察事实:** 待运行。
|
||||
- **异常:** 待运行。
|
||||
- **含义:** 待运行。
|
||||
- **Claim update:** unchanged
|
||||
- **下一步:** 运行两项 CPU state replay 与 coarse real-state analysis。
|
||||
@@ -0,0 +1,65 @@
|
||||
# 实验 EXP-SIMFID-Q235-CC-TP8:真实 TP4/TP8 collective profile 消融
|
||||
|
||||
> **状态:** review 通过,执行中
|
||||
>
|
||||
> 本 card 是 SHA、command、config、log 等 provenance 的唯一归宿;本轮只重跑 simulator,不重跑已经冻结的 48 个真机 trial。
|
||||
|
||||
## Claim 与决策
|
||||
|
||||
- **Parent claim:** Qwen235 Fixed-PD 的 30%+ TPOT/E2E selection regret,是否主要由 TP8 collective profile 缺失及 TP4 profile 与真实 serving backend 不匹配造成。
|
||||
- **目的:** 支持或反驳 mechanism hypothesis;不是用同一 workload 的 E2E calibration 修正 simulator。
|
||||
- **Competing hypotheses:**
|
||||
- H1:collective profile coverage/backend mismatch 是排序反转的必要主因。换成与真机 serving 一致的 TP4/TP8 实测 profile 后,Frontier 的 Fixed-PD TPOT winner 从 TP8 翻到 TP4,mean/p90 TPOT 与 E2E selection regret 降到 10% 以内。
|
||||
- H2:collective mismatch 只解释部分误差。换 profile 后 TP8 仍是 Frontier winner,Fixed-PD TPOT/E2E regret 仍超过 10%;下一主因应定位 decode batch/state-conditioned MoE composition。
|
||||
- **事前预测:** 当前 Frontier 在 Fixed-PD 上预测 TP4/TP8 mean TPOT 为 87.77/61.59 ms,TP8 有 26.19 ms 优势;真实 TP4/TP8 为 21.04/27.99 ms。若新的 TP4/TP8 collective profile 使这个 26.19 ms 的 simulator margin 反转,则支持 H1;若不能,则支持 H2。
|
||||
- **判定规则:** 只以 frozen simulator rerun 的 winner 与真实 frozen surface 计算 selection regret。绝对 latency ratio 作为 secondary metric,不用它替代 selection verdict。
|
||||
|
||||
## Setup
|
||||
|
||||
- **自变量:**
|
||||
- A0:当前 `measured-allreduce.csv`;TP4 是 Qwen30 hidden=2048 的旧实测,且 profiler 只检查 FlashInfer 可用、没有证明每个 payload 的实际 dispatch;TP8 无行并静默 analytical fallback。
|
||||
- A1:Qwen235 serving-matched piecewise collective profile;TP4/TP8 都在 dash0 H20、vLLM 0.20.0 commit `88d34c640...` 上实测。Frozen server logs 证明真机同时使用 `disable_custom_all_reduce=true` 与 FlashInfer-TRTLLM `allreduce_rms` fusion;profile 对 fusion-eligible payload 测同一 FlashInfer communicator,对阈值外 payload 测真实 PyNCCL/symmetric fallback。
|
||||
- **控制变量:** Frontier commit、Qwen235 operator profiles、runtime contract、四类 frozen traces、候选配置、MNS/MBT、prefix policy、real results 与分析脚本全部不变。
|
||||
- **System context:** Qwen3-235B-A22B-FP8;vLLM 0.20.0+cu129;dash0 8×H20;`{TP4/EP1, TP8/EP8} × MNS{64,128}`;MBT=8192;Frontier piecewise graph path。
|
||||
- **Workload 或 trace:** 重跑四类 simulator surface:Fixed-PD 4096→256 @ 0.2 req/s/GPU、Fixed-PO 4096→1、Trace-PD、Trace-PO;每 cell 沿用原 129-request trace。Fixed-PD 是 primary,另外三类检查 profile 替换是否引入新的 selection regression。
|
||||
- **Profile protocol:** payload 覆盖所有 Qwen235 decode graph buckets(1--256,含真实 capture sizes)、fusion 阈值两侧 `{63,64,65}` / `{255,256,257}`,以及 512--8192 prefill sizes;每个 TP 与 payload 先 warmup,再保留 3×20 个 per-rank CUDA-event samples。raw JSON 记录实际 backend dispatch、fusion byte limit、world size、dtype、payload bytes、GPU/runtime/commit 与 source hashes。TP4/TP8 使用相同 payload grid;不把 microbenchmark 直接当作 E2E 结论。
|
||||
- **Profile contract:** H20/SM90 上 vLLM 0.20 的 fusion limit 是 TP4 2 MiB、TP8 0.5 MiB,即 Q235 BF16 hidden=4096 时分别为 256/64 tokens。simulator runner 启动前解析 CSV,要求所选 configs 的每个 `TP>1` 都有有限、正值的 measured rows;缺覆盖立即失败。结果 manifest 写入 CSV SHA-256、TP coverage、row counts 与 piecewise backend 集合。决策实验禁止 analytical fallback。
|
||||
- **Baselines:** A0 current Frontier、A1 measured-profile Frontier、frozen real hardware surface。
|
||||
- **Metrics:** profile latency median/p90 与跨 rank spread;simulated mean/p90 TTFT/TPOT/E2E;winner、selection regret、tau-b(可定义时);每个 TP 的 measured-profile hit/fallback counters。
|
||||
|
||||
## 预期产物与 review
|
||||
|
||||
- **预期数据:** TP4/TP8 raw collective JSON;materialized Frontier CSV + manifest;四类 A1 simulator surface;A0/A1/real comparison JSON/Markdown;profile cost ledger。
|
||||
- **Figure prototype:** `../../runs/frontier-fidelity-envelope-v1/qwen235-collective-ablation-mock.png`;左图固定真实与 A0 TPOT,并为 A1 留待测 series;右图明确“winner flip→0% regret / unchanged→33% regret”的判定。它回答 profile 修复是否足以改变配置选择。
|
||||
- **人工 review:** 通过(2026-07-19,用户明确要求“推进实验”)。
|
||||
- **Review 意见:** 保留 frozen real surface,只补真实 TP4/TP8 profile 后重跑 simulator;每个 simulator 实验必须使用真实 profile,缺失 coverage 或运行时 analytical fallback 立即失败。
|
||||
|
||||
## Benchmark design audit(experiment-design-review)
|
||||
|
||||
| Crime | Verdict | Severity | Evidence | Fix / gate |
|
||||
|---|---|---|---|---|
|
||||
| 用 microbenchmark 代替 E2E | PASS | — | collective profile 只作为自变量;结论来自完整 simulator surface 对 frozen real surface 的 selection regret | 保留 A0/A1/real 三方结果 |
|
||||
| calibration set 等于 evaluation set | PASS | — | A1 只测 collective operator,不使用 real E2E latency 拟合参数 | 禁止 E2E scale/calibration |
|
||||
| selective benchmarking | PASS | — | primary Fixed-PD 外,同时重跑另外三类 workload | 报告所有 16 个 simulator cells |
|
||||
| 缺失平台/版本 | PASS | — | raw/manifest 绑定 H20、vLLM commit、model、backend 与 hashes | 任一 provenance 缺失则 profile 不可采纳 |
|
||||
| 缺失方差 | NEEDS EVIDENCE | Major | 尚未执行 profile repeats | raw artifact 必须保留 per-rank repeated samples并报告 spread |
|
||||
| profile 覆盖静默降级 | FAIL(A0) | Blocking | TP8 无 measured rows,Frontier 使用 analytical fallback | A1 runner fail-fast;fallback count 必须为 0 |
|
||||
| backend/fusion 阈值未对齐 | FAIL(A0) | Blocking | 真机日志启用 FlashInfer `allreduce_rms`;vLLM 源码规定 H20 TP4/TP8 fusion limit 为 2/0.5 MiB,旧 CSV 未记录这条 piecewise contract | A1 在阈值两侧实测并记录每行 dispatch |
|
||||
|
||||
**总体建议:** 已批准执行;coverage gate、backend match 与 provenance gate 任一不通过则 Block。
|
||||
|
||||
## 复现信息
|
||||
|
||||
- **Code:** AITuner branch `feature/sim`;本 card 创建时 HEAD `f4a75aa8e400ead4eb6d305178192e85940c6de6`,后续运行 commit 待填。vLLM source commit `88d34c6409e9fb3c7b8ca0c04756f061d2099eb1`。
|
||||
- **Environment:** dash0 8×H20;`/tmp/wjh/venvs/vllm-0.20.0-cu129-profiler-v1`;model `/home/admin/cpfs/wjh/models/Qwen/Qwen3-235B-A22B-FP8`。
|
||||
- **产物路径:** 待 review 后冻结;不得覆盖旧 campaign `/home/admin/cpfs/wjh/aituner/qwen235-v020-fourcase-20260719-r1`。
|
||||
- **已知 deviation:** `--disable-custom-all-reduce` 只关闭 vLLM custom AR,不关闭编译器的 FlashInfer `allreduce_rms` fusion。真机 TP4/TP8 日志都显示自动选择 `trtllm` workspace;旧 TP4 profiler 没有记录/执行真实 fusion-limit piecewise dispatch。A1 因此必须同时重测 TP4 与 TP8,不能只追加 TP8 行。
|
||||
- **执行异常:** 首次 simulator launch 误把 frozen operator profile root 写成 r1;原 A0 campaign 实际使用 r2。四个 Fixed-PD cells 因缺少 `attn_decode_in_mixed` predictor 均 fail-fast,未产生可用 metric。command diff 确认后停止后续运行,恢复 r2 并从 failed cells 重新执行;这些失败不计入 A1 surface。
|
||||
|
||||
## 结果
|
||||
|
||||
- **观察事实:** 待运行。
|
||||
- **异常:** 待运行。
|
||||
- **含义:** 待运行。
|
||||
- **Claim update:** unchanged
|
||||
- **下一步:** 依次完成 collective profiling、profile materialization、CPU simulator rerun 与 analysis。
|
||||
@@ -0,0 +1,46 @@
|
||||
# 实验:Qwen30 Fixed-PD TTFT admission diagnosis
|
||||
|
||||
> **状态:** 已完成
|
||||
>
|
||||
> 用户要求分析 Frontier 在 Qwen30 Fixed-PD 高压 case 上 56--58% TTFT
|
||||
> selection regret 的根因,并给出简洁结论。
|
||||
|
||||
## Claim 与决策
|
||||
|
||||
- **Parent claim:** Frontier 在 capacity knee 附近的配置排序是否会因 state transition error 失效。
|
||||
- **目的:** 区分 TP2/TP4 conditional prefill-time 错误、mixed-step composition 错误与 admission queue feedback。
|
||||
- **Competing hypotheses:** H1:Frontier 把 TP4 prefill execution 相对 TP2 算慢;H2:decode service time 的绝对误差使 `arrival_rate × residence_time` 越过 MNS cap,首次调度等待被阈值放大;H3:即使固定 admission state,mixed prefill/decode composition 仍反序。
|
||||
- **事前预测:** H1 下去掉 queue 后 TP2 仍有更低 prefill time;H2 下去掉 queue 后 TP4 恢复更快,且只有 simulator 的 required concurrency 超过 MNS;H3 下 state-matched stage contrast 仍支持 TP2。
|
||||
- **判定规则:** 只有 state ledger/scorer 等价、queue counterfactual 和真机 Running/Waiting 同时支持时才归因 H2;否则保留 H1/H3 并补最小 telemetry。
|
||||
|
||||
## Setup
|
||||
|
||||
- **自变量:** config 为 Frontier winner `TP2/MNS64`、real mean winner `TP4/MNS32` 与 real p90 winner `TP4/MNS64`。
|
||||
- **控制变量:** Qwen3-30B-A3B BF16、community vLLM 0.20、H20、Fixed-PD 4096→256、1.125 req/s/GPU、MBT8192、piecewise graph、原 measured profiles/collectives 和原 257-request traces全部冻结。
|
||||
- **Workload:** uniform open-loop arrival;global rate 随 TP 为 2.25/4.5 req/s;prefix cache off;每个真机 cell 三次 fresh-server。
|
||||
- **Baselines:** 完整 12-cell frozen real/sim surface;真机三轮 pooled metrics。
|
||||
- **Metrics:** TTFT=`first scheduling delay + prefill execution`;request execution/residence time;`arrival_rate × service_time` 相对 MNS;Running/Waiting;stage ledger composition。
|
||||
|
||||
## 预期产物与 review
|
||||
|
||||
- **预期数据:** 三个 scorer-equivalent Frontier state replays;service/admission decomposition;H1--H3 verdict。
|
||||
- **Figure prototype:** `../../runs/frontier-fidelity-envelope-v1/qwen30-fixed-pd-ttft-admission-mock.png`;左图区分 TTFT execution 与 queue,右图显示 required concurrency 是否跨越 MNS。
|
||||
- **人工 review:** 已批准(用户要求直接分析清楚该 case)。
|
||||
- **Review 意见:** 先复用 existing artifacts 和 CPU replay;只有现有真机 periodic queue proxy 不足时才增加 GPU telemetry。
|
||||
|
||||
## 复现信息
|
||||
|
||||
- **Code:** AITuner `2970f74d`;Frontier `deadc4a321f0baaa534c6ebd17f974123733cdc2`。
|
||||
- **Environment:** dash0;Frontier replay CPU-only,GPU visibility disabled。
|
||||
- **输入:** `/home/admin/cpfs/wjh/aituner/qwen30-fixed-pressure-surface-20260719-r1`。
|
||||
- **产物路径:** `/home/admin/cpfs/wjh/aituner/qwen30-fixed-pd-ttft-diagnosis-20260719-r1`。
|
||||
- **已知 deviation:** 原真机日志只有 10 秒 periodic Running/Waiting,没有 per-iteration ledger;它可验证 steady queue 是否积压,但不用于细粒度 stage timing。
|
||||
|
||||
## 结果
|
||||
|
||||
- **观察事实:** Frontier 把 TP2/MNS64 与 TP4/MNS64 的 TPOT 分别高估 8.09× 与 5.61×。按 frozen request execution time 计算,TP2 需要 58.0 个并发槽,未超过 MNS64;TP4 需要 79.8 个,超过 MNS64。真机 TP4 的 E2E-based required-slot upper bound 只有 14.7,MNS16/32/64 三组 periodic logs 的 Waiting max 均为 0。
|
||||
- **关键反事实:** observed Frontier 中 `TP4/MNS32 - TP2/MNS64` TTFT 为 `+27087.0 ms`;减去每请求的 first-scheduling delay 后变为 `-55.4 ms`,与真机 `-71.4 ms` 同方向。TP4/MNS32 的 27.3 秒 simulated TTFT 中,27.18 秒来自首次调度前等待,而不是 prefill execution。
|
||||
- **异常:** 计划中的 state-ledger replay 会重新训练 frozen no-cache predictors;8 分钟后仍停留在 predictor training。由于原 request metrics 已精确提供 TTFT=first-scheduling wait+prefill execution,且 MNS sweep 已构成 controlled intervention,继续 ledger 不改变判定,故主动停止;partial output 保留在产物根目录但不进入结果。
|
||||
- **含义:** H2 supported;H1/H3 对“TP topology 排序从哪里被反转”均 rejected。Frontier 的无排队 prefill 仍预测 TP4 比 TP2 快,错误由 decode service-time 绝对高估使 TP4 独自跨过 admission cap,再经 queue feedback 放大产生。该反事实恢复的是 TP4 topology 方向,不声称恢复 exact MNS winner;现有真机日志也不能把最初的 service-time overprediction 继续归因到某个单独 operator。
|
||||
- **Claim update:** supported
|
||||
- **下一步:** 若研究问题升级为“为什么 TPOT 绝对值高估 4--8×”,需增加真机 stage timing/overlap 证据;它不是解释本次 TTFT selection reversal 所必需。
|
||||
79
.research/ongoing.md
Normal file
79
.research/ongoing.md
Normal file
@@ -0,0 +1,79 @@
|
||||
# AITuner 研究当前状态
|
||||
|
||||
> 2026-07-17|写给未参与项目的读者,可直接作为 presentation 讲稿。历史过程与复现信息见 `../runs/*/` 各 experiment card、`../docs/` 各 campaign 文档。
|
||||
>
|
||||
> **2026-07-19 update:** Qwen235 Fixed-PD 的错误排序在 exact real state composition 下已经翻正,主因是 simulator closed-loop batch state,而不是 collective。Qwen30 Fixed-PD 的 56--58% TTFT regret 也已定位:Frontier 将 decode service time 高估 4--8×,使 TP4 的 modeled concurrency 越过 MNS admission cap并产生虚假排队;去掉该等待后 Frontier 与真机都判定 TP4 topology 更快。详见 [`experiments/qwen30-fixed-pd-ttft-admission-diagnosis-20260719.md`](experiments/qwen30-fixed-pd-ttft-admission-diagnosis-20260719.md)。
|
||||
>
|
||||
> **2026-07-20 update:** 对全部 14 个 frozen case surface(70 个 case×objective)做了统一的 margin-vs-residual 分解与方向化机制普查([`experiments/frontier-split-rootcause-s0-20260720.md`](experiments/frontier-split-rootcause-s0-20260720.md))。三个要点:(1) 「residual 超过 margin」是失败的必要条件但远非充分——good/bad 分裂不能用无方向误差量解释;(2) 23 个 material failure 的 winner-deciding pair 中 21 个落在 TP 轴或 mixed(其余 2 个是 6.2% regret 的边缘 mns-axis case),trace 面的 TP 反序为零;A1 measured collective 把 Qwen235 两个 prefill-only 面的 TP 反序清零(trace-PO p90 regret 21.2%→0.3%)却对 Fixed-PD 完全无效——prefill 路径的 TP 差异化误差源是 collective profile(可修),decode 耦合的 TP 差异化误差是当前所有 material failure 的载体;(3) 「Fixed-PD 失败因为高压」被否证:失败 Fixed-PD 的真机 in-flight(14.05)低于全对的 Trace-PD(38.69),且低压 Fixed-PD 同样失败、失败 objective 随负载切换。另有次要缺陷:14 个 winner 错位来自 simulator 对 MNS 逐位不敏感的精确 tie。
|
||||
>
|
||||
> **2026-07-20 root-cause update:** Q30 低压 Fixed-PD 的 exact stage ledger 关闭了最后一个未解释的 material failure。相同 batch=1 state 下 Frontier full predictor 给 TP4 `18.3515 ms/step`、TP1 `19.3506 ms/step`,方向正确;但 per-GPU 固定到达率使 cluster arrival 随 TP 增长,叠加 decode residence 高估后,TP4 在 simulator 内自激到 time-weighted batch `3.0437`(96.13% decode 时间 batch≥3),own-state step 变为 `28.1712 ms`。其中相对 batch=1 的 `+9.8197 ms` 有 `+8.9297 ms` 来自 batch-conditioned MoE,collective 仅 `+0.0121 ms`。因此 Fixed-PD 的根因不是“固定 workload”或“高压力”本身,而是 **execution-time residual 进入离散事件时钟后改变 future scheduler state;该 state 再通过 MoE/profile/graph 或 MNS admission 非线性放大,形成 action-dependent signed residual 并穿过 decision margin**。Q30 低压是平滑 state-feedback,Q30 高压是跨 MNS cap 的 threshold amplification,Q235 是 composition drift;三者为同一闭环机制族。
|
||||
>
|
||||
> **2026-07-20 load-audit update:** Trace-PD overload 不是 Fixed/Trace good-bad 分裂的统一解释。旧 Q30 Trace-PD decode offered/observed-peak throughput≈`1.00×`、peak Running/Waiting=`47/0`;降到 `0.10 req/s/GPU` 后 TTFT `245.95/685.51 → 228.14/835.38 ms`(mean/p90,不出现 tail collapse),TPOT `13.18/15.39 → 7.91/8.90 ms`。旧 Q235 则是 `3.44×` 明确过载、peak=`116/3`;降到 `0.035 req/s/GPU` 后 TTFT `1141.54/2616.69 → 478.14/1347.75 ms`,TPOT `61.89/78.62 → 24.00/28.49 ms`。旧 surface 仍有 `417×/32.6×` mean-TTFT spread,否定“所有配置一样差”。八 case baseline 与 claim boundary 见 [`experiments/frontier-eightcase-load-audit-20260720.md`](experiments/frontier-eightcase-load-audit-20260720.md)。
|
||||
|
||||
## 一眼看懂
|
||||
|
||||
- **Topic / problem:** LLM serving 的自动、低成本配置调优(AITuner)。当前主线问题:用 simulator 给部署配置(并行度、批量上限等)排序,什么时候可信?需要补多少真机证据?算上这些成本还划算吗?
|
||||
- **Central claim:** simulator 要能帮助配置调优,必须先满足 scheduler transition 的 liveness/coverage,再满足「配置相关残差小于真机 decision margin」;前者决定 capacity 是否有定义,后者决定排序是否正确。(ID: C0)
|
||||
- **当前结论:** 早先 35 个 trace stall 不是 Frontier scheduler liveness failure:adapter 为不满 16-token 的 prefix block 错误生成了 cache identity,Frontier 又没有 fail-fast。修正为完整 block、使用真实 graph buckets/KV blocks 和 `piecewise`/`KERNEL_ONLY` profile 后,Qwen30 Trace-PD 的全部 12 个 cell 完成 129/129 request,Frontier 对 TTFT/TPOT/E2E 的 6 个 argmin 均与三次 fresh-server 真机一致;但绝对 latency 仍高估 4--511×。这只证明一个 MoE Trace-PD surface 的 selection fidelity,不能外推到 prefill-only、fixed workload 或 235B。
|
||||
- **最大 uncertainty / risk:** 根因已收敛,且 overload 已被排除为统一解释,但可信域边界仍未画清:trace 面的 heterogeneity 是否让 closed-loop state residual 变小,还是当前 success 主要由 capacity/MNS margin 保护?两个降载点只建立 reference-config latency baseline,不能证明新负载下全 surface 仍选对。
|
||||
- **下一项 critical action:** 不再做无锚点的 jitter 猜测;保持 request shape 不变,在预测的 MoE/MNS knee 两侧做小规模 rate sweep,并用少量真机 state/batch anchor 验证 `λR(B)` fixed point。成功标准是同时预测 state-regime、排名与 knee,而不只是某个点的 regret。
|
||||
- **停止条件:** T1 出 verdict 且成本账本建立后:pass 且摊销论证成立 → 转向「sim 剪枝 + 真机终选」的 hybrid 机制设计;fail → 转入失败机制归因;两条路都无 insight 增量 → 收敛写作。
|
||||
|
||||
## 核心概念
|
||||
|
||||
- **Frontier:** 本项目使用的 simulator,属 Vidur 系(直接使用 vidur backend),加自研 FP8/MoE/EP/decode-profile 兼容补丁。
|
||||
- **Regret:** 按 simulator 排序选配置,相对真机最优配置的性能损失百分比(以每 GPU capacity 计)。primary metric:排序选对则 regret=0。
|
||||
- **τ-b(Kendall tau-b):** simulator 排序与真机排序的秩相关,1 = 完全一致,−1 = 完全反序,0 = 无关;tie-aware。
|
||||
- **Decision margin:** 真机上头部配置之间的性能差距,即 simulator 误差的容忍带。
|
||||
- **Action-differential residual:** simulator 误差中随配置(action)不同而不同的部分。Why needed:所有配置统一偏移不影响排序,只有差异化残差才可能穿过 margin 改变选择——这解释了「绝对误差 33%」与「排序全对」为何可以同时成立。
|
||||
- **Capacity bracket:** 真机 anchor 为候选配置的 capacity 划出的上下界;「bracket 不反转」指未测的负载点不可能推翻 top 选择。
|
||||
- **Decision-valid coverage:** simulator 能从初始状态推进到所有请求完成,并为 config×workload cell 产生合法 SLO metric 的比例。若 reachable nonterminal state 没有 enabled transition/future event,capacity 与 rank 都没有定义,不能把该 cell 当作 infeasible。
|
||||
- **Workload realism 阶梯:** prefill-only(无 decode)→ fixed-shape mixed(固定输入输出长度的混合负载)→ trace-faithful mixed(生产 trace 忠实回放)。fidelity 结论不能向更高一级外推。
|
||||
|
||||
## Claim 层级
|
||||
|
||||
- **Central claim:** 见「一眼看懂」。(ID: C0)
|
||||
- **Subclaim:** zero-shot 排序失败是真实现象。(ID: C1;supported)
|
||||
- 30B 纯 profile 驱动的 regret 为 25.63%(τ-b=0);另一 throughput-proxy 评测口径下为 30.46%。Boundary:均发生在 capacity-point + SLO-gated selection——恰是 Vidur 论文自己声明预测误差会爆炸、评测刻意回避的 regime(见 claim map)。
|
||||
- **Subclaim:** 少量结构化的真机证据可以恢复低 regret 排序。(ID: C2)
|
||||
- **Hypothesis(decision-bearing):** trace-faithful 回放下,同栈 profile + 真机 KV capacity + 兼容补丁、且不做逐案例端到端校准的 Frontier,能满足 gate:regret ≤5% ∧ τ-b ≥0.8 ∧ bracket 不反转。(ID: H1;weakened)
|
||||
- **Supporting:** 235B prefill-only regret=0;235B fixed-shape mixed 的 top set 全中;30B 加 per-TP 校准后 regret 0.76%(但这是外部端到端 scale 给出的上界,不是原生 profile 保真度)。
|
||||
- **Counterevidence:** 修正 prefix trace contract 后的 TP2/MNS16 `none`-graph run 完成但 p50 TPOT 约 96 ms,真机为约 14 ms;然而该比较尚未对齐 real vLLM 的 `FULL_AND_PIECEWISE` graph path。
|
||||
- **下一项 discriminative experiment:** 补齐 `KERNEL_ONLY` graph family并以 `piecewise` 重跑相同 trace;若 full surface 仍错,graph omission 不再是可用解释。
|
||||
- **Hypothesis(机制,active):** 误差机制是 action-conditioned residual——执行状态的转移(并行拓扑、kernel family、graph mode、batch 组成)使按算子 profile 的组合预测跨配置不可复合;残差大于 margin 时排序失败。(ID: H2;supported,已细化)
|
||||
- **Supporting:** 三个 TP 档的端到端校准系数为 0.72/0.47/0.35,残差确实随配置剧烈变化;235B 的批量上限交互预测错误但被 2× margin 容忍;30B prefill-only 在低负载近似对齐、饱和后按 TP 反向放大,最终 τ-b=−1。
|
||||
- **细化(2026-07-20 统一普查):** 决策相关的残差分量集中在 TP 轴且由 decode 状态耦合产生——prefill-only 面的绝对 scale 仅 0.96–1.37× 且 measured collective 即可清除其 TP 反序,而含 decode 的面 scale 4.3–130×、全部 material failure 都由 TP/mixed pair 决定。「residual>margin」只是必要条件;失败还需要残差对准 winner-deciding pair。
|
||||
- **机制 verdict(2026-07-20):** closed-loop state drift 是根因,离散阈值是其放大器而非 competing explanation。Q30 低压 exact ledger 显示同 state 的 TP 方向正确,但 TP4 被模拟 residence 反馈推到 batch 3--4,MoE step 增长后反序;Q30 高压进一步跨过 MNS admission cap;Q235 换成 exact real composition 后排序翻正。下一步从“找根因”转为测量 state-regime/knee 的可信边界。
|
||||
- **Subclaim:** 成本论证只有在摊销前提下成立。(ID: C3)
|
||||
- **Hypothesis(active):** 每个 model×硬件×runtime 的一次性对齐成本,摊销到大配置面、频繁重调(引擎版本 churn 的频率证据见 claim map)或禁止在线实验的场景后,低于重复真机调优。(ID: H3;untested——分母已实测,分子未入账)
|
||||
- **下一步:** 建 cost ledger(见「下一步」)。
|
||||
|
||||
## 当前 critical experiment
|
||||
|
||||
- **Question:** 生产 trace 忠实回放(prefix 打开、原始到达时间与会话结构)下,best-effort Frontier 能否满足 low-regret gate?
|
||||
- **为什么现在做:** 这是 H1 的判决实验;所有已完成的机制分解都在人工 workload 上,不能替代这个 verdict。
|
||||
- **当前状态:** Trace-PD 的 graph-aligned surface 已通过原负载 selection gate,但绝对 latency 不通过 calibration;Fixed-PD 的 failure 已定位为 closed-loop state drift。两个降载 Trace-PD anchor 已通过完成率/admission/backlog gate,下一步需要 full surface rate sweep 才能检验 ranking 是否跨 load regime 保持。
|
||||
- **Result → decision:** 若其它 surface 排序失败,保留 Trace-PD success 为条件化 envelope,并按 fixed/trace/prefill/decode 的差异定位 state composition;若都通过,才扩大到 Q235 或寻找 simulator 已解决范围之外的新问题。
|
||||
- **Experiment card:** [`../runs/frontier-fidelity-envelope-v1/experiment-card.md`](../runs/frontier-fidelity-envelope-v1/experiment-card.md)
|
||||
|
||||
## Key evidence(最多 3 条)
|
||||
|
||||
- **E1(否证「prefill-only 是充分 easy condition」,支持 H2):** 30B BF16、去掉 decode/prefix/混合 batch 后,真机最优是 TP4(8 vs 7 req/s/GPU),simulator 却把 TP4 排最差(6 vs 8):top set 无交集,regret 12.5%,τ-b=−1。产物:`../runs/frontier-phase-factorial-v0/results/final/`(dash0,12.07 H20-GPUh)。
|
||||
- **E2(统一机制普查:material failure 全部由 decode 耦合的 TP 差异化误差决定,支持 H2 细化):** 对 14 个 frozen surface、70 个 case×objective 的方向化分解显示:23 个 material failure 中 21 个由 TP/mixed pair 决定(仅 2 个 6.2% 边缘 mns-axis case)、trace 面 TP 反序为零;measured collective(A1)把 Qwen235 两个 prefill-only 面的 TP 反序清零(trace-PO p90 regret 21.2%→0.3%)但对 Fixed-PD 的 33% 无效;「residual>margin」仅为失败的必要条件。产物:[`../runs/frontier-split-rootcause-v0/results/`](../runs/frontier-split-rootcause-v0/results/decomposition.md)(实验 card:[`experiments/frontier-split-rootcause-s0-20260720.md`](experiments/frontier-split-rootcause-s0-20260720.md))。
|
||||
- **E3(closed-loop state 是 Fixed-PD 根因,而非同 state predictor 反序):** Q30 低压相同 batch=1 state 下 TP4 比 TP1 快约 1.00 ms/step,但 TP4 own state 的 time-weighted batch=3.0437,使 step 增加 9.8197 ms(其中 MoE +8.9297 ms)并反序;Q235 用 exact real composition 重放也把 TP8−TP4 从错向 −20.07 ms 翻为正确 +10.90 ms。Q30 高压再由 MNS cap 将同族 state/residence 误差放大成约 27 s 排队。产物:[`experiments/frontier-split-rootcause-s1-20260720.md`](experiments/frontier-split-rootcause-s1-20260720.md)。
|
||||
|
||||
## 下一步(最多 3 项)
|
||||
|
||||
- [ ] **画可信域边界:** 固定 request shape,在预测的 MoE/MNS knee 两侧做最小 rate sweep;只在判别点补真机 batch/state anchor,验证 `B≈min(MNS, λR(B))` 是否同时解释 state 与 ranking。
|
||||
- [ ] **Q235 portability gate:** 先验证 vLLM0.20 TP4/TP8 FP8 runtime 和 deadc4a profile provenance,再决定是否允许其 Fixed-P sweep。
|
||||
- [ ] **建 cost ledger:** parent H3;完成标准 = 每 case 一行(profiling GPU-h、补丁工时、校准探测、sim CPU-h),与已实测的真机调优成本同表,随每个 case 更新。
|
||||
|
||||
## Blocker 或 anomaly
|
||||
|
||||
- **当前运行状态:** 八 case load audit 的新增真机 run 已完成;未启动 full-surface rate sweep,避免把两个 single-config anchor 外推成 ranking claim。自 2026-07-20 起,本任务只允许使用 `dash1`--`dash4`(每台 8×H20、最多四组并行);`dash0` 保留给其他同事,不做 probe、同步或运行。
|
||||
- **Anomaly(保留):** 235B pilot 中 simulator 把 10/34 个 anchor 误判为不可行——false-infeasible 是 H1 的主要威胁模式,T1 分析时须单独报告。
|
||||
- **平台边界(已更新):** 历史结果仍来自其各自 card 记录的平台,不改写 provenance;后续实验平台切换为 `dash1`--`dash4`。跨主机比较前必须跑相同 canary 并量化 host effect。fixed-shape pilot 的主 SLO(TPOT 40ms)无判别力,150ms 是事后明示的敏感性分析,不得写成盲选的 primary。
|
||||
|
||||
## Related work
|
||||
|
||||
- Claim map:[`../docs/simulator-claim-map-20260716.md`](../docs/simulator-claim-map-20260716.md)。核心缺口:capacity-point + SLO-gated selection 的 regret 无人用真机 ground-truth 面验证过;alignment 成本无人与真机调优成本放进同一张表比较。
|
||||
14
AGENTS.md
14
AGENTS.md
@@ -1,10 +1,11 @@
|
||||
# Project Operating Notes
|
||||
|
||||
## Remote experiment host
|
||||
## Remote experiment hosts
|
||||
|
||||
- Default experiment machine: `dash0`.
|
||||
- Hardware expectation: 8 NVIDIA H20 GPUs.
|
||||
- SSH check: use `ssh dash0` before scheduling or debugging remote runs.
|
||||
- Experiment machines: `dash1`, `dash2`, `dash3`, and `dash4`.
|
||||
- Do not use or probe `dash0`; it is reserved for other users.
|
||||
- Hardware expectation: 8 NVIDIA H20 GPUs per host.
|
||||
- Before scheduling, probe only `dash1`--`dash4` and confirm all eight GPUs are idle and healthy.
|
||||
- Remote project path: `/home/admin/cpfs/wjh/aituner/aituner`.
|
||||
- If remote downloads are slow or fail, start the proxy from the remote `wjh`
|
||||
home directory with `./auto_proxy.sh`, then run downloads in a shell where
|
||||
@@ -13,7 +14,8 @@
|
||||
|
||||
## Local/remote sync workflow
|
||||
|
||||
- Treat this local repository and the `dash0` repository as the same project checkout.
|
||||
- Treat this local repository and the `dash1`--`dash4` repositories as the same project checkout.
|
||||
- Synchronize code through Git using `commit`, `push`, and `pull`.
|
||||
- For remote experiments, commit local changes, push to `origin`, then pull on `dash0` in `/home/admin/cpfs/wjh/aituner/aituner` before running.
|
||||
- For remote experiments, commit local changes, push to `origin`, then pull on each assigned host in `/home/admin/cpfs/wjh/aituner/aituner` before running.
|
||||
- Up to four independent 8-GPU experiment groups may run in parallel, one group per host; pin every job explicitly to one of `dash1`--`dash4`.
|
||||
- Do not ask for the remote host or project path again unless the user explicitly changes them.
|
||||
|
||||
179
docs/action-aware-constraint-pilot-v0-protocol-20260714.md
Normal file
179
docs/action-aware-constraint-pilot-v0-protocol-20260714.md
Normal file
@@ -0,0 +1,179 @@
|
||||
# Action-aware constraint pilot v0 protocol
|
||||
|
||||
Status: **FROZEN BEFORE NEW GPU RUNS**.
|
||||
|
||||
Date: 2026-07-14 (Asia/Singapore).
|
||||
|
||||
## Headline question
|
||||
|
||||
Can telemetry from one complete initial-config benchmark identify which of two
|
||||
competing knob families should be changed, before either target configuration
|
||||
is evaluated?
|
||||
|
||||
This pilot tests a narrow prerequisite, not an end-to-end tuner claim. It
|
||||
uses fields already present in the per-step OpProf stream to reconstruct exact
|
||||
zero-slack conditions for `max_num_seqs` (MNS) and
|
||||
`max_num_batched_tokens` (MBBT). No new vLLM instrumentation is justified
|
||||
unless those action-conditioned conditions predict crossed real-system
|
||||
intervention responses.
|
||||
|
||||
## Hypothesis
|
||||
|
||||
I believe config-normalized scheduler constraints provide a stronger tuning
|
||||
signal than an aggregate queue symptom because the same waiting queue can be
|
||||
blocked by different admission limits.
|
||||
|
||||
I will verify it by holding model, hardware, TP, request bands, arrival times,
|
||||
and offered load fixed while constructing two source configurations with
|
||||
different binding constraints. From each source run alone, the larger
|
||||
exclusive binding fraction predicts the action family. Both candidate
|
||||
actions are then measured on the same requests for the full 300-second replay.
|
||||
|
||||
## Frozen platform and workload
|
||||
|
||||
- Host: `dash0`, solo placement on GPUs 0-3, four NVIDIA H20 GPUs.
|
||||
- Model: Qwen3-30B-A3B BF16.
|
||||
- Engine: patched vLLM `0.24.1.dev3+opprof`, TP=4.
|
||||
- Workload: the three disjoint `mid` bands from
|
||||
`chat_w20260312_1000`, 2.125 requests/s/GPU, 300-second arrival window,
|
||||
exactly 128 output tokens.
|
||||
- SLO: the unchanged study TTFT/TPOT thresholds and 0.95 pass-rate target.
|
||||
- Every config starts one fresh server, performs the accepted 16-request
|
||||
warm-up and the existing burn-in, then runs all three disjoint measured
|
||||
bands in its frozen order.
|
||||
- SLO early stopping is disabled. A measured run must drain all selected
|
||||
requests and finish within the 450-second client deadline.
|
||||
|
||||
## Frozen configuration and action matrix
|
||||
|
||||
| ID | MNS | MBBT | Role |
|
||||
|---|---:|---:|---|
|
||||
| `b_base` | 64 | 256 | token-budget-bound source; operational gate runs first |
|
||||
| `a_base` | 16 | 8192 | MNS-bound source |
|
||||
| `shared` | 64 | 8192 | MNS action from A; MBBT action from B |
|
||||
| `b_mns` | 128 | 256 | competing MNS action from B |
|
||||
| `a_mbbt` | 16 | 16384 | competing MBBT action from A |
|
||||
|
||||
The two decisions are therefore:
|
||||
|
||||
```text
|
||||
Regime A: a_base -> {shared (increase MNS), a_mbbt (increase MBBT)}
|
||||
Regime B: b_base -> {b_mns (increase MNS), shared (increase MBBT)}
|
||||
```
|
||||
|
||||
The candidate magnitudes are intentionally large in this feasibility pilot so
|
||||
that a missing crossed response is not explained by an imperceptibly small
|
||||
intervention. This does not establish that these are production step sizes.
|
||||
|
||||
Frozen config order is `b_base`, `a_base`, `shared`, `b_mns`, `a_mbbt`.
|
||||
Frozen repetition orders are respectively `123`, `231`, `312`, `132`, and
|
||||
`213`, reducing band/time alignment without reusing a server across configs.
|
||||
|
||||
## Pre-action signal
|
||||
|
||||
For each source run, let `waiting` include the normal and deferred waiting
|
||||
queues, and let `scheduled_tokens = prefill_tokens + decode_tokens`.
|
||||
|
||||
```text
|
||||
mns_exclusive = waiting > 0
|
||||
and running == configured MNS
|
||||
and scheduled_tokens < configured MBBT
|
||||
|
||||
mbbt_exclusive = waiting > 0
|
||||
and scheduled_tokens == configured MBBT
|
||||
and running < configured MNS
|
||||
|
||||
both = waiting > 0
|
||||
and running == configured MNS
|
||||
and scheduled_tokens == configured MBBT
|
||||
```
|
||||
|
||||
Each score is the fraction of all scheduler records in the measured interval
|
||||
that satisfies the condition. The predicted action is the family with the
|
||||
larger exclusive fraction. This uses no target telemetry or target outcome.
|
||||
KV usage and preemptions are reported as possible alternative constraints but
|
||||
are not silently reassigned to either score.
|
||||
|
||||
These conditions reproduce two scheduler loop boundaries, but they are still
|
||||
a Level-0 proxy: they do not expose the exact request rejected at the boundary
|
||||
or run a shadow schedule. The pilot explicitly tests whether that additional
|
||||
engine patch is warranted.
|
||||
|
||||
## Outcomes and baselines
|
||||
|
||||
Primary intervention outcome:
|
||||
|
||||
```text
|
||||
SLO-goodput = full-run SLO pass count / 300-second arrival window
|
||||
```
|
||||
|
||||
Also report pass rate, TTFT p50/p95/p99, TPOT p50/p95/p99, drain elapsed time,
|
||||
KV usage, preemptions, queue area, and CUDA-graph padding.
|
||||
|
||||
Required decision baselines:
|
||||
|
||||
1. always choose the MNS family;
|
||||
2. always choose the MBBT family;
|
||||
3. queue-pressure-only, which has no candidate-specific score and therefore
|
||||
must use one frozen family for both regimes;
|
||||
4. the pre-action exclusive-binding prediction.
|
||||
|
||||
This is a mechanism ablation. It does not compare against a trained black-box
|
||||
tuner because two regimes are not a valid training surface.
|
||||
|
||||
## Gates and failure meanings
|
||||
|
||||
Data validity requires 15 uncensored measured runs, exact request/arrival/input
|
||||
hashes across each repetition, full request accounting, one continuous OpProf
|
||||
stream per config, zero dropped records, monotonic timestamps and step indices,
|
||||
nonnegative counters, bounded ratios, clean GPU placement, and config values in
|
||||
the result matching the server command.
|
||||
|
||||
The crossed-response gate passes only if, in all three repetitions:
|
||||
|
||||
- the MNS target has higher SLO-goodput than the MBBT target in Regime A;
|
||||
- the MBBT target has higher SLO-goodput than the MNS target in Regime B;
|
||||
- each winning target exceeds its competing target by at least 10% of the
|
||||
source SLO-goodput. A source with zero goodput makes the run invalid for
|
||||
this relative gate rather than changing the denominator.
|
||||
|
||||
The binding gate passes only if, in both regimes:
|
||||
|
||||
- the predicted family matches the measured winning family in all three
|
||||
repetitions;
|
||||
- the median winning-family exclusive fraction is at least 0.10;
|
||||
- it is at least 5x the median competing-family exclusive fraction;
|
||||
- the direction is unchanged under cumulative 25%, 50%, 75%, and 100%
|
||||
checkpoints after the 25% checkpoint.
|
||||
|
||||
Decision meanings:
|
||||
|
||||
- `STOP_WORKLOAD_NOT_CROSSED`: candidate outcomes do not have different
|
||||
winners; the experiment cannot test action selection.
|
||||
- `STOP_BINDING_NOT_PREDICTIVE`: outcomes cross but source-only constraint
|
||||
scores do not select them; do not implement shadow scheduling from this
|
||||
hypothesis.
|
||||
- `STOP_NO_NEW_INSTRUMENTATION_NEEDED`: the signal works but every required
|
||||
field was already present; keep it as an analysis/tuner feature and do not
|
||||
claim a new engine-instrumentation contribution.
|
||||
- `OPEN_EXACT_ATTRIBUTION_ABLATION`: the signal works but unresolved/both/KV
|
||||
cases are material enough that exact rejection reasons could change a
|
||||
decision. Only this result authorizes a minimal vLLM attribution patch.
|
||||
|
||||
Ambiguity is material only when, in either regime, the median
|
||||
`both + waiting_unresolved` fraction is at least the median absolute gap
|
||||
between the two exclusive fractions, or when any source run records a
|
||||
preemption or median source KV maximum is at least 0.90. Otherwise all fields
|
||||
needed for the observed decision were already present and the result is
|
||||
`STOP_NO_NEW_INSTRUMENTATION_NEEDED`.
|
||||
|
||||
No result from this development pilot is a paper-level E2E tuning claim.
|
||||
|
||||
## Cost and stopping discipline
|
||||
|
||||
- Hard cap: 8.0 H20-hours, including failed sessions.
|
||||
- Expected: 6.0-7.2 H20-hours and 90-110 minutes wall time.
|
||||
- `b_base` runs first. If its first measured band cannot drain by 450 seconds,
|
||||
the controller stops before any comparative analysis; MBBT=256 is then an
|
||||
operationally invalid source, not negative evidence.
|
||||
- Any data red flag stops analysis before computing a tuning conclusion.
|
||||
44
docs/action-aware-constraint-pilot-v1-protocol-20260714.md
Normal file
44
docs/action-aware-constraint-pilot-v1-protocol-20260714.md
Normal file
@@ -0,0 +1,44 @@
|
||||
# Action-aware constraint pilot v1 amendment
|
||||
|
||||
Status: **FROZEN AFTER V0 OPERATIONAL STOP AND BEFORE V1 GPU RUNS**.
|
||||
|
||||
Date: 2026-07-14 (Asia/Singapore).
|
||||
|
||||
The complete claim, workload, baselines, metrics, action matrix, analysis gates,
|
||||
and data-validity requirements remain those in the v0 protocol. This amendment
|
||||
changes only the token-bound source severity and adds an operational burn-in
|
||||
gate.
|
||||
|
||||
## Why v0 produced no comparative evidence
|
||||
|
||||
The first v0 session used MNS=64 and MBBT=256. During the 510-request,
|
||||
60-second burn-in, the client had run for 197 seconds and the engine still held
|
||||
64 requests: 13 running and 51 waiting. The last step scheduled exactly 256
|
||||
tokens, KV usage was 0.01151, and there were zero preemptions. No measured run
|
||||
or target configuration had started.
|
||||
|
||||
The session was stopped and cleanly released all GPUs after consuming
|
||||
0.3859868995 H20-hours. This is evidence that MBBT=256 is a real token-budget
|
||||
bottleneck, but it is not an admissible source for the 2.125 requests/s/GPU
|
||||
comparison because it cannot sustain the offered load. V0 contributes no
|
||||
tuning label and none of its runtime data is reused by V1.
|
||||
|
||||
Authoritative failure artifact:
|
||||
`/home/admin/cpfs/wjh/action-aware-constraint-v0-20260714/operational-stop-v0.json`.
|
||||
|
||||
## V1 changes
|
||||
|
||||
- `b_base`: MNS=64, MBBT **2048** instead of 256.
|
||||
- `b_mns`: MNS=128, MBBT **2048** instead of 256.
|
||||
- The B-family MBBT action is therefore 2048 -> 8192.
|
||||
- All five configurations and all three repetitions run fresh under a new V1
|
||||
run root.
|
||||
- Before any measured run, every config's 510-request/60-second burn-in must
|
||||
drain in at most **90 seconds**. A slower config is an operational failure;
|
||||
the controller stops before comparative analysis.
|
||||
- V1 incremental hard cap is 7.6140131005 H20-hours so that V0 plus V1 remains
|
||||
within the original global 8.0 H20-hour cap.
|
||||
|
||||
The V0 protocol's crossed-response and source-only binding gates are unchanged.
|
||||
In particular, V1 still requires three-of-three action-family predictions in
|
||||
both regimes and a different real winner in Regime A versus Regime B.
|
||||
35
docs/action-aware-constraint-pilot-v2-protocol-20260714.md
Normal file
35
docs/action-aware-constraint-pilot-v2-protocol-20260714.md
Normal file
@@ -0,0 +1,35 @@
|
||||
# Action-aware constraint pilot v2 amendment
|
||||
|
||||
Status: **FROZEN AFTER V1 CONTROLLER STOP AND BEFORE V2 GPU RUNS**.
|
||||
|
||||
Date: 2026-07-14 (Asia/Singapore).
|
||||
|
||||
The V1 configuration matrix and every scientific gate remain unchanged. V2
|
||||
fixes one controller bug and reruns every configuration and repetition fresh.
|
||||
|
||||
## V1 controller failure
|
||||
|
||||
The MNS64/MBBT2048 burn-in completed all 510 requests in 61.259 seconds, below
|
||||
the frozen 90-second operational limit. However, the controller accidentally
|
||||
assigned the preceding 16-request warm-up result to `burnin_result`; its state
|
||||
therefore recorded 4.376 seconds and evaluated the wrong object.
|
||||
|
||||
The first measured replay was terminated after 75 seconds, before it produced
|
||||
a result. No target configuration had started. V1 consumed
|
||||
0.3184109431 H20-hours and contributes no action label or telemetry to V2.
|
||||
|
||||
## V2 correction and regression gate
|
||||
|
||||
- The completed burn-in `run_client()` return value is assigned to
|
||||
`burnin_result`.
|
||||
- A dedicated `burnin_gate()` rejects any non-anchor object, any request count
|
||||
other than 510, and elapsed time above 90 seconds.
|
||||
- Unit tests explicitly pass a warm-up object and require rejection, then test
|
||||
accepted and over-limit burn-ins.
|
||||
- All five configs and 15 measured runs use a new run root; no V0/V1 runtime
|
||||
artifact is reused.
|
||||
|
||||
V0 and V1 together consumed 0.7043978426 H20-hours. V2's incremental hard cap
|
||||
is therefore 7.2956021574 H20-hours, preserving the original global 8.0
|
||||
H20-hour cap. The authoritative accounting file is
|
||||
`runs/action-aware-v0/prior-attempts-v2.json`.
|
||||
260
docs/action-aware-constraint-pilot-v2-results-20260714.md
Normal file
260
docs/action-aware-constraint-pilot-v2-results-20260714.md
Normal file
@@ -0,0 +1,260 @@
|
||||
# Action-aware constraint pilot v2 results
|
||||
|
||||
Date: 2026-07-14 (Asia/Singapore).
|
||||
|
||||
Decision: **`STOP_WORKLOAD_NOT_CROSSED`**.
|
||||
|
||||
The pilot produced one valid positive regime and one invalid-for-effect-size
|
||||
regime. It supports continuing a narrower action-response investigation, but
|
||||
it does not justify an end-to-end telemetry-guided tuner claim or new engine
|
||||
instrumentation yet.
|
||||
|
||||
## Question tested
|
||||
|
||||
Given only a completed source run, can existing engine telemetry distinguish
|
||||
which of two one-knob interventions will improve SLO-goodput more?
|
||||
|
||||
The frozen score counted scheduler steps with backlog where either MNS or MBBT
|
||||
was exclusively at its configured limit. It made two pre-intervention
|
||||
predictions on the same workload and offered load:
|
||||
|
||||
- Regime A: source `(MNS=16, MBBT=8192)` predicts increasing MNS to 64 over
|
||||
increasing MBBT to 16384.
|
||||
- Regime B: source `(MNS=64, MBBT=2048)` predicts increasing MBBT to 8192 over
|
||||
increasing MNS to 128.
|
||||
|
||||
The primary outcome was 300-second SLO-goodput. A predicted action had to beat
|
||||
the alternative on every paired request band by at least 10% of that band's
|
||||
source goodput. Telemetry direction also had to remain stable at 25%, 50%,
|
||||
75%, and 100% of the replay.
|
||||
|
||||
## Setup
|
||||
|
||||
- Host: `dash0`, GPU 0-3 used exclusively; four NVIDIA H20 GPUs; TP=4. GPU
|
||||
4-7 remained idle to avoid co-location effects.
|
||||
- Model: Qwen3-30B-A3B BF16 at
|
||||
`/home/admin/cpfs/wjh/models/Qwen/Qwen3-30B-A3B`.
|
||||
- Runtime: patched vLLM `0.24.1.dev3+g668cfb7e2`, source commit `4b253fd`, with
|
||||
OpProf Layer-1 telemetry.
|
||||
- Workload: `chat_w20260312_1000`, 2.125 requests/s/GPU, 300-second arrival
|
||||
window, 128 output tokens, three disjoint request bands.
|
||||
- Five fresh-server configurations, three measured runs each, counter-rotated
|
||||
repetition order, 16-request warm-up, and a 510-request/60-second burn-in.
|
||||
SLO early stopping was disabled.
|
||||
- Exact request-id, arrival-order, and input-length hashes matched for every
|
||||
paired comparison.
|
||||
|
||||
The authoritative run root is
|
||||
`/home/admin/cpfs/wjh/action-aware-constraint-v2-20260714`. The final audit is
|
||||
`pilot-audit-final.json`, SHA256
|
||||
`7ebe080fcc4970bef423bc587253d157e75aed1ea8b410bd37770c17708135ab`.
|
||||
|
||||
## End-to-end result
|
||||
|
||||
### Regime A: strong MNS constraint
|
||||
|
||||
| Rep | Source 16/8192 | MBBT action 16/16384 | MNS action 64/8192 | MNS-only source steps | MBBT-only source steps | `(MNS action - MBBT action) / source` |
|
||||
|---:|---:|---:|---:|---:|---:|---:|
|
||||
| 1 | 4.710 | 7.687 | 8.500 | 79.85% | 0.038% | 17.27% |
|
||||
| 2 | 2.803 | 4.150 | 8.500 | 94.61% | 0.005% | 155.17% |
|
||||
| 3 | 2.227 | 3.943 | 8.500 | 79.87% | 0.005% | 204.64% |
|
||||
|
||||
Units are SLO-goodput requests/s except the percentage columns. The source
|
||||
prediction was stable at every phase checkpoint and correct in all three
|
||||
paired bands. The predicted MNS action cleared the frozen 10% material-margin
|
||||
gate in all three bands.
|
||||
|
||||
This is a real but limited positive result. The source was an extreme case:
|
||||
MNS was full on nearly every scheduler step that retained backlog, TTFT p50 was
|
||||
1.64-5.61 seconds, KV usage remained below 2.45%, and no preemption occurred.
|
||||
An expert or a simple rule could identify this case without a learned tuner.
|
||||
|
||||
### Regime B: MBBT direction at an outcome ceiling
|
||||
|
||||
| Rep | Source 64/2048 | MNS action 128/2048 | MBBT action 64/8192 | MBBT-only source steps | `(MBBT action - MNS action) / source` |
|
||||
|---:|---:|---:|---:|---:|
|
||||
| 1 | 8.423 | 8.420 | 8.500 | 10.90% | 0.950% |
|
||||
| 2 | 8.447 | 8.500 | 8.500 | 12.03% | 0% |
|
||||
| 3 | 8.500 | 8.497 | 8.500 | 8.79% | 0.039% |
|
||||
|
||||
The telemetry direction was stable in all phase checkpoints. The predicted
|
||||
action won twice and tied once, while the wrong MNS action left the MBBT-only
|
||||
state intact. However, the source already delivered 99.10-100% of the offered
|
||||
8.5 requests/s. Even a perfect action could not reach the preregistered 10%
|
||||
margin. Regime B therefore does not test material weak-signal value; it is a
|
||||
workload-selection failure, not evidence that telemetry does or does not help
|
||||
near a decision boundary.
|
||||
|
||||
The missing preflight condition is mathematical. For an effect threshold
|
||||
`delta` and offered goodput ceiling `G`, the source must satisfy
|
||||
`source <= G / (1 + delta)`. Here `G=8.5` and `delta=0.10`, so any source above
|
||||
7.727 requests/s cannot possibly pass before either target is measured.
|
||||
|
||||
## Why the exclusive-limit rule is incomplete
|
||||
|
||||
The alternative MBBT action improved Regime A by 48.0%, 63.2%, and 77.1% over
|
||||
the source even though MBBT was almost never the exclusive backlog constraint.
|
||||
This rules out the binary interpretation "a knob that is not exclusively at
|
||||
its cap cannot help."
|
||||
|
||||
Existing richer telemetry provides a plausible mechanism:
|
||||
|
||||
| Rep | Split-prefill requests, source -> MBBT action | Prefill steps, source -> MBBT action | Prefill requests/step, source -> MBBT action | Prefix-hit rate, source -> MBBT action |
|
||||
|---:|---:|---:|---:|---:|
|
||||
| 1 | 41 -> 2 | 2324 -> 2022 | 1.071 -> 1.249 | 13.851% -> 13.747% |
|
||||
| 2 | 7 -> 0 | 2410 -> 2291 | 1.012 -> 1.076 | 13.078% -> 12.988% |
|
||||
| 3 | 12 -> 1 | 2377 -> 2270 | 1.018 -> 1.096 | 13.613% -> 13.604% |
|
||||
|
||||
Increasing MBBT allows more prefill work to be packed into one iteration and
|
||||
nearly eliminates split prefills. Under MNS=16, this can reduce the number of
|
||||
iterations for which long prompts occupy scarce running slots. Prefix-cache
|
||||
hit rates differ by at most 0.104 percentage points, and exact workload hashes
|
||||
match, so neither explains the gain. Step-duration p99 also remains similar;
|
||||
one 1.127-second decode-step outlier appears in `a_mbbt/rep2`, but the same
|
||||
action direction occurs in all three bands.
|
||||
|
||||
This is a mechanism-consistent explanation, not a completed causal
|
||||
decomposition. MBBT simultaneously changes total per-iteration token budget,
|
||||
per-request chunk size, and multi-request packing. Instrumentation observes
|
||||
their joint response but cannot separate those effects without another
|
||||
intervention.
|
||||
|
||||
## Instrumentation decision
|
||||
|
||||
Do **not** add a new engine patch for this mechanism yet. The existing OpProf
|
||||
stream already records submit/complete timestamps, prefill/decode composition,
|
||||
chunked-prefill categories, prefix hits, queues, KV usage, and CUDA graph mode.
|
||||
Those fields are sufficient to identify the interaction missed by the initial
|
||||
exclusive-limit rule.
|
||||
|
||||
The next narrow mechanism ablation is available in the current vLLM runtime:
|
||||
|
||||
1. `(MNS=16, MBBT=8192, long-prefill-threshold=0)` is the current source.
|
||||
2. `(MNS=16, MBBT=16384, long-prefill-threshold=8192)` keeps individual long
|
||||
chunks at 8192 while increasing total packing budget.
|
||||
3. `(MNS=16, MBBT=16384, long-prefill-threshold=0)` is the current MBBT action.
|
||||
|
||||
The runtime exposes `--long-prefill-token-threshold`; with a threshold of 8192,
|
||||
the second arm separates total packing headroom from the larger per-request
|
||||
chunk allowed by the third arm. A formal test must rerun all three arms fresh
|
||||
with counter-rotated order rather than reuse today's endpoints.
|
||||
|
||||
## Correct tuning-research route
|
||||
|
||||
The pilot does not support turning the frozen equality checks into a larger
|
||||
rule tree. The supported route is **intervention-calibrated, action-conditioned
|
||||
system identification**:
|
||||
|
||||
```text
|
||||
source event sequence + normalized config delta
|
||||
-> predicted distribution of Delta SLO-goodput and evaluation cost
|
||||
-> uncertainty-aware next-config selection
|
||||
```
|
||||
|
||||
The policy input should retain continuous distributions and phase evolution:
|
||||
queue/running residency, MNS and token slack, prefill/decode composition,
|
||||
partial-prefill occupancy, step time, KV state, and graph behavior. Human
|
||||
bottleneck labels and hand-authored `if queue then increase MNS` mappings are
|
||||
not policy inputs. Mechanism summaries remain audit and interpretation tools;
|
||||
the action response is learned from paired real interventions.
|
||||
|
||||
The harness has a narrower, non-heuristic role:
|
||||
|
||||
- define legal configurations and exact paired workloads;
|
||||
- reject source points without outcome headroom before a full sweep;
|
||||
- randomize/counter-rotate execution order and preserve failures/cost;
|
||||
- validate stream coverage, hashes, request accounting, and censoring;
|
||||
- expose target outcomes only after a source-only prediction is frozen;
|
||||
- evaluate fixed-budget regret and H20-hours, not explanation quality alone.
|
||||
|
||||
The next tuning experiment should use non-extreme, non-ceiling source points
|
||||
and a local two-dimensional MNS/MBBT neighborhood. A short run may screen load
|
||||
only; every inferential telemetry and outcome result remains a 300-second run.
|
||||
At least one held-out workload must be reserved before choosing features or
|
||||
thresholds.
|
||||
|
||||
Primary evaluation is H20-hours/trials to reach 95% of the real local oracle
|
||||
and cost-normalized regret AUC. Required baselines are random search,
|
||||
config/outcome-only sequential search, the current rule heuristic, and the
|
||||
same action-response model with telemetry removed. Action-ranking accuracy is
|
||||
supporting evidence only.
|
||||
|
||||
## What this pilot establishes and does not establish
|
||||
|
||||
Established:
|
||||
|
||||
- Long-window engine state can make a correct, phase-stable action-family
|
||||
prediction in an extreme MNS-constrained regime.
|
||||
- A naive `queue > 0 -> increase MNS` rule would choose an ineffective action
|
||||
in Regime B; action-conditioned state distinguishes the mechanism direction,
|
||||
although the measured effect is immaterial at the selected load.
|
||||
- Binary exclusive-cap attribution misses a substantial MNS/MBBT interaction;
|
||||
existing chunk/step telemetry reveals a plausible explanation.
|
||||
- Source outcome headroom must be an explicit experiment admission gate.
|
||||
|
||||
Not established:
|
||||
|
||||
- telemetry improves an end-to-end tuner over an outcome-only baseline;
|
||||
- weak or mixed constraints can be ranked with material gain;
|
||||
- the response transfers across workloads, models, TP, or hardware;
|
||||
- new engine instrumentation is necessary;
|
||||
- the chunking/packing breakdown is causal rather than mechanism-consistent.
|
||||
|
||||
## Change and verification
|
||||
|
||||
Reproduction:
|
||||
|
||||
```bash
|
||||
python3 runs/action-aware-v0/test_pilot.py
|
||||
|
||||
python3 runs/action-aware-v0/analyze_pilot.py \
|
||||
--run-root /home/admin/cpfs/wjh/action-aware-constraint-v2-20260714/runs/pilot \
|
||||
--manifest runs/action-aware-v0/pilot-manifest-v2.json \
|
||||
--output /home/admin/cpfs/wjh/action-aware-constraint-v2-20260714/pilot-audit-final.json
|
||||
```
|
||||
|
||||
The fresh GPU run used AITuner commit `c5ab073`; asynchronous coverage was
|
||||
corrected in `3facb18`; reproducible mechanism summaries were added in
|
||||
`2af22db`. The raw run is unchanged across those analyzer-only commits.
|
||||
|
||||
Change: added a crossed real-intervention controller and audit, fixed the
|
||||
burn-in result gate, corrected asynchronous per-step coverage accounting, and
|
||||
added reproducible step/chunk/prefix mechanism summaries.
|
||||
|
||||
Expected effect: distinguish descriptive telemetry from source-only action
|
||||
predictions that survive paired real interventions.
|
||||
|
||||
Verification: local and remote action-aware test suites pass; all five sessions
|
||||
completed; all stream/footer and request-accounting invariants pass; the final
|
||||
analyzer was run twice and produced byte-identical output.
|
||||
|
||||
Result: Regime A passes; Regime B is invalid for the frozen effect-size test;
|
||||
the global decision is `STOP_WORKLOAD_NOT_CROSSED`.
|
||||
|
||||
Remaining risk: one model, one TP, one trace family, three bands, two action
|
||||
families, and deliberately constructed endpoints are development evidence
|
||||
only. The strong positive regime is too obvious to support a paper claim.
|
||||
|
||||
## Data sanity
|
||||
|
||||
- Measured runs: n=15; elapsed 300.610-317.350 seconds; 15 distinct. Pass
|
||||
rate min/max 0.2620/1.0 with 11 distinct values; SLO-goodput min/max
|
||||
2.2267/8.5 requests/s with 11 distinct values.
|
||||
- Telemetry intervals: n=15; records min/max 13,621/23,711; 15 distinct.
|
||||
Start gaps min/max 0.0412/0.1227 seconds; end gaps 0.00053/0.0649; uncovered
|
||||
internal gaps 0/0.3231. One submit gap reached 1.1190 seconds but was fully
|
||||
covered by a 1.1269-second recorded execution; contiguous indices and zero
|
||||
drops were preserved.
|
||||
- Sessions: n=5; cost min/max 1.1691/1.2730 H20-hours; 5 distinct. V2 cost
|
||||
was 6.0862 H20-hours; V0/V1/V2 total was 6.7906, below the 8.0 cap.
|
||||
- Regime-A split-prefill observations: n=6; min/max 0/41 requests; 6 distinct.
|
||||
Prefix-hit rates: n=6; min/max 0.12988/0.13851; 6 distinct.
|
||||
- Checked invariants: non-negative counters and durations; ratios in `[0,1]`;
|
||||
exact request, arrival, and length hashes; 2550/2550 request accounting per
|
||||
measured run; uncensored outcomes; outcomes across configurations not all
|
||||
identical;
|
||||
five complete streams; monotonic timestamps; contiguous step indices; zero
|
||||
drops; footer/sidecar agreement; chunk-token accounting; bounded prefix hits;
|
||||
no OOM, controller error, or residual GPU allocation. No unresolved red
|
||||
flag remains. The three identical shared goodputs are reported as the
|
||||
offered-load ceiling, not treated as independent performance variation.
|
||||
136
docs/active-intervention-v0-protocol-20260715.md
Normal file
136
docs/active-intervention-v0-protocol-20260715.md
Normal file
@@ -0,0 +1,136 @@
|
||||
# Active intervention + measurement v0 protocol
|
||||
|
||||
Date: 2026-07-15 (Asia/Singapore)
|
||||
|
||||
Status: **FROZEN BEFORE THE `chat_w20260313_1000` GPU RUN**.
|
||||
|
||||
## Research question
|
||||
|
||||
This experiment asks whether a tuner conditioned on direct engine-state
|
||||
trajectories can choose both a measurement horizon and a coupled configuration
|
||||
intervention with lower real-GPU cost than the same tuner using only external
|
||||
prefix outcomes.
|
||||
|
||||
The contribution is not the controller, legality checks, telemetry collection,
|
||||
or the ridge model. The route remains open only if engine state changes an
|
||||
actual decision and reduces cost-to-near-oracle on unseen workloads.
|
||||
|
||||
## Development result that motivates, but does not pass, the route
|
||||
|
||||
The frozen trace-12 dataset contains 72 examples: six source decisions, four
|
||||
measurement checkpoints, and `noop/MNS/MBBT` actions. Features are direct
|
||||
continuous Layer-1 state summaries; cap-exclusive and bottleneck labels are
|
||||
excluded. Leave-one-repetition-out sequential replay uses the same model,
|
||||
candidate set, confidence rule, and checkpoint set for both modes.
|
||||
|
||||
The external-outcome policy and telemetry policy both put all six decisions
|
||||
within 2% regret. Outcome-only selected a mean 262.5-second source measurement
|
||||
and cost 3.750 replay H20-hours across the six replayed decisions; telemetry
|
||||
selected 275 seconds and cost 3.833 H20-hours. Telemetry therefore increased
|
||||
the replay lower-bound cost by 2.22%, with no regret reduction. This is a
|
||||
negative result. It does not settle the question because the dataset has only
|
||||
two source regimes, one source is at the offered ceiling, and there is no joint
|
||||
MNS+MBBT action.
|
||||
|
||||
Sanity: n=6 decisions; regret min=0, max=0.009412, distinct=3; source cutoff
|
||||
min=150s, max=300s, distinct=3 across the two policies; all costs are
|
||||
non-negative, regrets are in `[0,1]`, target results are not all identical, and
|
||||
the six decisions are complete exact-workload pairs.
|
||||
|
||||
## Frozen prospective setup
|
||||
|
||||
- Host: `dash0`, 8 NVIDIA H20 GPUs available; each TP4 server runs alone on
|
||||
GPUs 0-3. Co-location is prohibited for SLO verdicts.
|
||||
- Engine: patched vLLM `0.24.1.dev3+g668cfb7e2` from clean source commit
|
||||
`4b253fd8619764b6971a7f2e3a3aa7545f6ace05` at
|
||||
`/home/admin/cpfs/wjh/opprof-phase2-dash0-20260711/vllm-v0.24.0`, using
|
||||
`/tmp/wjh-opprof-phase2-dash0-20260711/.venv`.
|
||||
- Model: `/home/admin/cpfs/wjh/models/Qwen/Qwen3-30B-A3B`, BF16.
|
||||
- Workload: unseen `chat_w20260313_1000`; input 0-8192; output exactly 128;
|
||||
replay scale 0.5; 300-second arrival window.
|
||||
- Three disjoint repetitions: source rows are assigned by a deterministic
|
||||
SHA-256 modulo-3 partition before input filtering. Each repetition selects
|
||||
approximately 3300 requests, or 2.75 requests/s/GPU at TP4.
|
||||
- SLO: at least 95% pass; stepped TTFT 2/4/6 seconds; TPOT at most 50 ms.
|
||||
- Checkpoints: 75, 150, 225, and 300 seconds.
|
||||
- Full 2x2 surface:
|
||||
- source: `MNS=32, MBBT=4096`;
|
||||
- MNS action: `64,4096`;
|
||||
- MBBT action: `32,8192`;
|
||||
- joint action: `64,8192`;
|
||||
- `noop` retains the source.
|
||||
- Four config sessions are serialized. Each session uses a fresh server,
|
||||
warm-up, burn-in, and counter-rotated repetition order.
|
||||
- Expected campaign cost: 4.6-5.5 H20-hours; hard cap: 6.0 H20-hours;
|
||||
expected wall time: 75-100 minutes.
|
||||
|
||||
The source is executed first. The frozen telemetry policy selects the next
|
||||
real config session; all remaining cells are then measured only to construct
|
||||
the exact finite-surface oracle. Oracle annotation after the selected action
|
||||
is reported separately from tuner cost.
|
||||
|
||||
## Frozen policies
|
||||
|
||||
Both policies fit the paired treatment effect
|
||||
|
||||
```text
|
||||
target normalized SLO-goodput - source normalized SLO-goodput
|
||||
```
|
||||
|
||||
from source config, full config delta, offered load, and external prefix
|
||||
outcomes. The telemetry policy additionally receives fixed direct Layer-1
|
||||
summaries and their interactions with `delta_log2(MNS)` and
|
||||
`delta_log2(MBBT)`. It does not receive a bottleneck label or a
|
||||
diagnosis-to-knob rule.
|
||||
|
||||
At each checkpoint, jackknife models produce an effect distribution for
|
||||
`noop`, MNS, MBBT, and joint actions. Measurement stops at the earliest second
|
||||
consecutive checkpoint with the same confident best action; otherwise it uses
|
||||
the full 300 seconds. Confidence requires a predicted margin of at least 0.02
|
||||
and the best lower bound to exceed the second-best upper bound. If the final
|
||||
choice is not confident, the next run is the positive-UCB action, explicitly
|
||||
marked as a diagnostic intervention. The exact same rule is used for the
|
||||
outcome-only baseline.
|
||||
|
||||
## Hypotheses and gates
|
||||
|
||||
### H1: action value
|
||||
|
||||
Engine state must change the selected intervention or its ranking and reduce
|
||||
real action regret. Prediction error or bottleneck-label accuracy is not a
|
||||
success metric.
|
||||
|
||||
### H2: measurement value
|
||||
|
||||
Engine state must select a shorter stable source measurement without increasing
|
||||
action regret. A shorter reconstructed prefix is only a trigger; it is not an
|
||||
actual GPU-cost claim until an early-terminated confirmation run measures
|
||||
startup, warm-up, drain, and cleanup.
|
||||
|
||||
### H3: end-to-end cost
|
||||
|
||||
Primary development metric is H20-hours to first reach a configuration within
|
||||
2% of the exact median-goodput oracle. The outcome-only and telemetry policies
|
||||
use the same measured config costs and differ only in source information.
|
||||
|
||||
- At least 10% prospective replay cost reduction, telemetry regret at most 2%,
|
||||
and no outcome-only-to-telemetry harm triggers an actual early-stop
|
||||
confirmation.
|
||||
- At least 20% measured all-in H20-hour reduction is required for a contribution
|
||||
claim. This one task can only establish development feasibility; a paper
|
||||
claim additionally requires task-held-out replication.
|
||||
- Source median normalized goodput at or above 0.98 stops the surface before
|
||||
target runs because the workload has no material improvement headroom.
|
||||
- Any hash mismatch, missing/censored result, telemetry drop, non-monotonic
|
||||
phase, negative cost, ratio outside `[0,1]`, or all-identical config outcomes
|
||||
is a red flag and stops analysis.
|
||||
|
||||
If the 10% trigger fails, this route is closed for the current engine-state
|
||||
representation. The experimental control plane is not retained as a fallback
|
||||
research contribution.
|
||||
|
||||
Pre-run provenance amendment: the first controller dry-run on 2026-07-15
|
||||
rejected two stale engine paths before starting a server. The paths and exact
|
||||
runtime version above were recovered from the accepted trace-12 campaign and
|
||||
corrected before any trace-13 GPU work. No scientific treatment or gate was
|
||||
changed.
|
||||
106
docs/active-intervention-v0-results-20260715.md
Normal file
106
docs/active-intervention-v0-results-20260715.md
Normal file
@@ -0,0 +1,106 @@
|
||||
# Active intervention v0: held-out trace-13 result
|
||||
|
||||
Date: 2026-07-15 (Asia/Singapore)
|
||||
|
||||
Decision: **close the passive-telemetry treatment-effect route**. The held-out
|
||||
campaign produced no telemetry-induced action change, measurement reduction, or
|
||||
GPU-cost reduction. It did show that the engine state contained the correct
|
||||
action-specific mechanism; the current feature model failed to use it.
|
||||
|
||||
## Headline result
|
||||
|
||||
The outcome-only and telemetry policies both measured the source for 300
|
||||
seconds, selected `joint=(MNS64,MBBT8192)`, and produced the same complete
|
||||
acquisition order. Both reached the exact finite-surface oracle after the
|
||||
first intervention at a reconstructed all-in lower-bound cost of 2.4284
|
||||
H20-hours. Telemetry GPU-cost reduction was therefore exactly 0%, below the
|
||||
10% confirmation trigger and 20% contribution gate. No actual early-stop
|
||||
confirmation was launched.
|
||||
|
||||
The complete annotation campaign cost 5.0379 H20-hours, below the 6.0 H20-hour
|
||||
hard cap. It ran 12 uncensored real-GPU outcomes: four configs, three disjoint
|
||||
request partitions, and a fresh server per config.
|
||||
|
||||
## Exact response surface
|
||||
|
||||
Median normalized SLO-goodput was:
|
||||
|
||||
| Config | Rep values | Median |
|
||||
|---|---|---:|
|
||||
| `MNS32,MBBT4096` source | 0.40091 / 0.39788 / 0.42061 | 0.40091 |
|
||||
| `MNS64,MBBT4096` | 1.00000 / 0.99970 / 1.00000 | 1.00000 |
|
||||
| `MNS32,MBBT8192` | 0.44394 / 0.41515 / 0.42606 | 0.42606 |
|
||||
| `MNS64,MBBT8192` joint | 1.00000 / 1.00000 / 1.00000 | 1.00000 |
|
||||
|
||||
Increasing MNS alone was sufficient and joint was redundant. Increasing MBBT
|
||||
alone improved the median by only 0.02515, versus 0.59909 for MNS. This is a
|
||||
strong non-additive action response, not a setting where independently tuning
|
||||
the knobs and merging their improvements is valid.
|
||||
|
||||
## What the telemetry actually said
|
||||
|
||||
Across 41,086 source scheduler records, 93.12% of steps had waiting work,
|
||||
85.36% were MNS-exclusive binding, 1.11% were MBBT-exclusive, mean running-slot
|
||||
utilization was 97.39%, mean token-budget utilization was 15.69%, mean KV usage
|
||||
was 2.75%, and there were no preemptions.
|
||||
|
||||
The intervention transition agreed with that state:
|
||||
|
||||
| Config | Waiting | MNS-exclusive | MBBT-exclusive | Median goodput |
|
||||
|---|---:|---:|---:|---:|
|
||||
| source | 93.12% | 85.36% | 1.11% | 0.40091 |
|
||||
| MNS only | 5.38% | 0% | 5.38% | 1.00000 |
|
||||
| MBBT only | 91.19% | 91.09% | 0.04% | 0.42606 |
|
||||
| joint | 0.89% | 0% | 0.89% | 1.00000 |
|
||||
|
||||
Thus this experiment does **not** support the claim that engine telemetry lacks
|
||||
tuning information. It rejects the narrower claim that adding passive state
|
||||
summaries to the current small-data ridge policy converts that information into
|
||||
lower tuning cost.
|
||||
|
||||
## Why the learned policy failed
|
||||
|
||||
At 300 seconds, the telemetry model predicted joint, MNS, and MBBT effects of
|
||||
0.35190, 0.26118, and 0.09686. The actual median effects were 0.59909,
|
||||
0.59909, and 0.02515. Telemetry therefore made the nonexistent joint-over-MNS
|
||||
gap larger: 0.09072 predicted versus 0 actual; the outcome-only model predicted
|
||||
0.03188.
|
||||
|
||||
The failure has three concrete causes:
|
||||
|
||||
1. The six training decisions contain no joint intervention. The
|
||||
`delta_product` feature has no support, so joint ranking is extrapolation.
|
||||
2. Passive raw summaries do not represent the counterfactual scheduler work
|
||||
unlocked by each action. Capacity-normalized MNS pressure was visible, but
|
||||
the model was not structurally required to map it to MNS marginal value.
|
||||
3. The policy maximizes predicted effect. It does not identify the smallest
|
||||
epsilon-optimal intervention or price unsupported action complexity.
|
||||
|
||||
## Research implication
|
||||
|
||||
Do not retain the harness or the passive telemetry model as a contribution.
|
||||
The next defensible route is engine-native, action-conditional counterfactual
|
||||
instrumentation: at a real scheduling state, shadow-replay the exact scheduler
|
||||
decision under an MNS relaxation, MBBT relaxation, and their joint relaxation,
|
||||
then expose the incremental queued work admitted by each action. Real paired
|
||||
interventions calibrate how those one-step shadow effects map to E2E SLO
|
||||
goodput. This is distinct from a hand-written cap-to-knob rule and from a
|
||||
full-system simulator: it reuses the exact live queue, scheduler, and cache
|
||||
state while simulating only the local decision boundary.
|
||||
|
||||
That route should be evaluated against outcome-only search, the present passive
|
||||
telemetry model, a cap-hit expert rule, and a full simulator. The paper-level
|
||||
gate remains at least 20% measured H20-hour reduction to a 2%-oracle config on
|
||||
task-held-out workloads with at most 2% regret.
|
||||
|
||||
## Sanity
|
||||
|
||||
Surface outcomes: n=12, min=0.39788, max=1.0, distinct=8. Session costs: n=4,
|
||||
min=1.1702, max=1.3566 H20-hours, distinct=4. Scheduler-record counts: n=4,
|
||||
min=37,001, max=41,348, distinct=4. All counters and costs were non-negative;
|
||||
all ratios were in `[0,1]`; request hashes matched; all 12 runs were uncensored;
|
||||
the controller and four sessions completed; and config outcomes were not all
|
||||
identical. No red flags were found.
|
||||
|
||||
Machine-readable summary: `runs/active-intervention-v0/trace13-results.json`.
|
||||
Raw immutable root: `/home/admin/cpfs/wjh/active-intervention-prospective-20260715`.
|
||||
60
docs/assets/simulator-fidelity/data.json
Normal file
60
docs/assets/simulator-fidelity/data.json
Normal file
@@ -0,0 +1,60 @@
|
||||
{
|
||||
"schema": "simulator-fidelity-figure-data-v1",
|
||||
"objective": "maximum_tested_slo_feasible_offered_request_rate_per_gpu",
|
||||
"qwen30_mixed": {
|
||||
"sources": {
|
||||
"real": "recovered-stores/aituner-interaction-runs-dash1-20260710/interaction-mixed-qwen30b-tp-mns-surface-high1-dash1-d8899c5-20260701T095858Z",
|
||||
"comparison": "/home/gahow/phd/replayserve/runs/simfid_s2rb/results/metrics.json"
|
||||
},
|
||||
"configs": [
|
||||
{"name": "tp1_mns8", "tp": 1, "mns": 8, "real": 2.1, "frontier_profile_only": 1.1, "frontier_calibrated": 1.7166666666666666},
|
||||
{"name": "tp1_mns16", "tp": 1, "mns": 16, "real": 2.35, "frontier_profile_only": 1.1, "frontier_calibrated": 2.3833333333333333},
|
||||
{"name": "tp1_mns32", "tp": 1, "mns": 32, "real": 2.283333333333333, "frontier_profile_only": 1.1, "frontier_calibrated": 2.3833333333333333},
|
||||
{"name": "tp1_mns64", "tp": 1, "mns": 64, "real": 2.283333333333333, "frontier_profile_only": 1.1, "frontier_calibrated": 2.3833333333333333},
|
||||
{"name": "tp2_mns8", "tp": 2, "mns": 8, "real": 2.275, "frontier_profile_only": 0.0, "frontier_calibrated": 1.7416666666666667},
|
||||
{"name": "tp2_mns16", "tp": 2, "mns": 16, "real": 2.275, "frontier_profile_only": 1.1916666666666667, "frontier_calibrated": 2.3},
|
||||
{"name": "tp2_mns32", "tp": 2, "mns": 32, "real": 3.283333333333333, "frontier_profile_only": 0.0, "frontier_calibrated": 3.75},
|
||||
{"name": "tp2_mns64", "tp": 2, "mns": 64, "real": 3.2583333333333333, "frontier_profile_only": 0.0, "frontier_calibrated": 3.75},
|
||||
{"name": "tp4_mns8", "tp": 4, "mns": 8, "real": 1.2833333333333334, "frontier_profile_only": 0.0, "frontier_calibrated": 1.3208333333333333},
|
||||
{"name": "tp4_mns16", "tp": 4, "mns": 16, "real": 2.441666666666667, "frontier_profile_only": 0.0, "frontier_calibrated": 2.5},
|
||||
{"name": "tp4_mns32", "tp": 4, "mns": 32, "real": 2.441666666666667, "frontier_profile_only": 1.3208333333333333, "frontier_calibrated": 2.5},
|
||||
{"name": "tp4_mns64", "tp": 4, "mns": 64, "real": 2.441666666666667, "frontier_profile_only": 1.3208333333333333, "frontier_calibrated": 2.5}
|
||||
],
|
||||
"profile_only_metrics": {
|
||||
"kendall_tau_b": 0.0,
|
||||
"pairwise_exact_sign_accuracy": 0.3787878787878788,
|
||||
"simulator_top_set": ["tp4_mns32", "tp4_mns64"],
|
||||
"real_top_set": ["tp2_mns32"],
|
||||
"top1_regret_worst": 0.25634517766497456
|
||||
},
|
||||
"calibrated_metrics": {
|
||||
"kendall_tau_b": 0.9668009539030813,
|
||||
"pairwise_exact_sign_accuracy": 0.9393939393939394,
|
||||
"simulator_top_set": ["tp2_mns32", "tp2_mns64"],
|
||||
"real_top_set": ["tp2_mns32"],
|
||||
"top1_regret_best": 0.0,
|
||||
"top1_regret_worst": 0.0076142131979695165
|
||||
}
|
||||
},
|
||||
"qwen235_prefill": {
|
||||
"source": "runs/frontier-multicase-sufficiency-v0/best_effort/fixed_cohort_evidence/v2_refined_comparison.json",
|
||||
"configs": [
|
||||
{"name": "tp4_mns64_mbt8192", "tp": 4, "mns": 64, "mbt": 8192, "expert_parallel": false, "real": 0.05, "frontier": 0.0375},
|
||||
{"name": "tp4_mns128_mbt8192", "tp": 4, "mns": 128, "mbt": 8192, "expert_parallel": false, "real": 0.05, "frontier": 0.0375},
|
||||
{"name": "tp4_mns64_mbt16384", "tp": 4, "mns": 64, "mbt": 16384, "expert_parallel": false, "real": 0.075, "frontier": 0.0625},
|
||||
{"name": "tp4_mns128_mbt16384", "tp": 4, "mns": 128, "mbt": 16384, "expert_parallel": false, "real": 0.075, "frontier": 0.0625},
|
||||
{"name": "tp8_mns64_mbt8192", "tp": 8, "mns": 64, "mbt": 8192, "expert_parallel": true, "real": 0.05625, "frontier": 0.05},
|
||||
{"name": "tp8_mns128_mbt8192", "tp": 8, "mns": 128, "mbt": 8192, "expert_parallel": true, "real": 0.05625, "frontier": 0.05},
|
||||
{"name": "tp8_mns64_mbt16384", "tp": 8, "mns": 64, "mbt": 16384, "expert_parallel": true, "real": 0.05625, "frontier": 0.05625},
|
||||
{"name": "tp8_mns128_mbt16384", "tp": 8, "mns": 128, "mbt": 16384, "expert_parallel": true, "real": 0.05625, "frontier": 0.05625}
|
||||
],
|
||||
"metrics": {
|
||||
"spearman_rank_correlation": 0.9486832980505138,
|
||||
"pairwise_non_tied_accuracy": 1.0,
|
||||
"comparable_non_tied_pairs": 20,
|
||||
"simulator_top_set": ["tp4_mns64_mbt16384", "tp4_mns128_mbt16384"],
|
||||
"real_top_set": ["tp4_mns64_mbt16384", "tp4_mns128_mbt16384"],
|
||||
"top1_regret_worst": 0.0
|
||||
}
|
||||
}
|
||||
}
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 133 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 152 KiB |
BIN
docs/assets/simulator-fidelity/qwen30-mixed-config-ranking.png
Normal file
BIN
docs/assets/simulator-fidelity/qwen30-mixed-config-ranking.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 183 KiB |
BIN
docs/assets/simulator-fidelity/qwen30-prefill-ranking.png
Normal file
BIN
docs/assets/simulator-fidelity/qwen30-prefill-ranking.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 106 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 216 KiB |
@@ -1,10 +1,16 @@
|
||||
# Fidelity-aware harness headroom audit
|
||||
|
||||
Status: **PROMISING PREMISE, NO CONTRIBUTION CLAIM**.
|
||||
Status: **HISTORICAL PREMISE DID NOT PASS PROSPECTIVE P1; NO CONTRIBUTION CLAIM**.
|
||||
|
||||
The audit answers whether engine instrumentation has enough incremental signal
|
||||
to justify a prospective experiment. It does not establish generalization.
|
||||
|
||||
Post-run update: the exact-timestamp held-out P1 completed and failed the
|
||||
registered gate. Under the stronger simulator-aware `k=2` end-to-end replay,
|
||||
telemetry preserved zero regret but saved only 1.426% online H20-hours versus
|
||||
sim top-k + real final. The current route is closed; see
|
||||
`docs/fidelity-aware-harness-p1-report-20260714.md`.
|
||||
|
||||
## Simulator shortlist lower bound
|
||||
|
||||
On the frozen 12-cell SimFid task, the strongest calibrated SLO simulator
|
||||
@@ -46,6 +52,35 @@ at 15 seconds it is 88.89% versus 91.67%; at 20 seconds it is 86.11% versus
|
||||
91.67%, but both 0.95 policies make one false reject. Five seconds is therefore
|
||||
a training-selected operating point, not a test result.
|
||||
|
||||
## Strong simulator-aware calibration baseline
|
||||
|
||||
The original nested comparison used the same simulator shortlist but did not
|
||||
put Frontier's per-anchor prediction in either model. A stronger retrospective
|
||||
audit now gives both models frozen-calibrated simulated throughput, simulated
|
||||
SLO pass rate, and simulated feasibility. Under the same leave-one-cell-out
|
||||
folds, 5-second cutoff, L2 logistic family, regularization 1.0, and threshold
|
||||
0.95:
|
||||
|
||||
| Metric | Sim + outcome | Sim + outcome + instrumentation | Delta |
|
||||
|---|---:|---:|---:|
|
||||
| Accuracy | 81.08% | 89.19% | +8.11 pp |
|
||||
| Balanced accuracy | 72.42% | 81.55% | +9.13 pp |
|
||||
| Brier score | 0.1058 | 0.0957 | -0.0101 |
|
||||
| Safe early decisions | 20/37 | 25/37 | +5 |
|
||||
| Valid full-trial cost reduction | 50.89% | 68.98% | +18.09 pp |
|
||||
| Residual verification H20-hours | 0.5240 | 0.3310 | -36.84% |
|
||||
|
||||
Both 0.95 policies have zero false accept and zero false reject on this
|
||||
retrospective task. Only three 0.5-threshold classifications differ in favor
|
||||
of instrumentation and none in favor of the strong baseline; McNemar's exact
|
||||
two-sided p-value is 0.25. The cell-bootstrap accuracy-delta interval is
|
||||
`[0.00,+18.18]` percentage points. The result is not robust to regularization:
|
||||
at 0.1 the strong baseline is more accurate and the instrumentation policy
|
||||
makes two unsafe decisions; at 10.0 the strong baseline is also more accurate.
|
||||
Thus the stronger comparison still has enough point-estimate headroom for a
|
||||
held-out test, but it materially weakens the evidence and makes a prospective
|
||||
task-level result mandatory.
|
||||
|
||||
## Interpretation
|
||||
|
||||
There is enough headroom to run a held-out pilot, but not enough evidence to
|
||||
@@ -73,6 +108,9 @@ with three full repetitions. The registered protocol is
|
||||
- `runs/fidelity-headroom/prefix-metrics.json`
|
||||
- `runs/fidelity-headroom/test_analysis.py`
|
||||
- `runs/fidelity-headroom/test_prefix_analysis.py`
|
||||
- `runs/fidelity-headroom/analyze_strong_baseline.py`
|
||||
- `runs/fidelity-headroom/strong-baseline-metrics.json`
|
||||
- `runs/fidelity-headroom/test_strong_baseline.py`
|
||||
|
||||
## Sanity block
|
||||
|
||||
@@ -86,6 +124,8 @@ with three full repetitions. The registered protocol is
|
||||
| Outcome probability | 37 | in `[0,1]` | in `[0,1]` | >1 | Checked before metrics |
|
||||
| Instrumentation probability | 37 | in `[0,1]` | in `[0,1]` | >1 | Checked before metrics |
|
||||
| Layer-1 streams | 12 | 14,174 records | 58,725 records | 12 | Contiguous, zero drops |
|
||||
| Matched frozen simulator anchors | 37 | pass rate 0.0688 | pass rate 1.0 | 12 pass-rate values | Every prefix matched exactly once |
|
||||
| Frozen simulator anchor corpus | 92 | positive throughput | positive throughput | >1 | No duplicate cell/anchor run |
|
||||
|
||||
Checked invariants: same folds/model family and cutoff; no full verdict in a
|
||||
feature; prefix-only Layer-1 slicing; non-negative costs/counters; bounded
|
||||
|
||||
242
docs/fidelity-aware-harness-p1-report-20260714.md
Normal file
242
docs/fidelity-aware-harness-p1-report-20260714.md
Normal file
@@ -0,0 +1,242 @@
|
||||
# Fidelity-aware harness P1 result
|
||||
|
||||
Status: **REGISTERED ROUTE REJECTED; DO NOT OPEN P2/P3 FOR THE CURRENT METHOD**.
|
||||
|
||||
Date: 2026-07-14 (Asia/Singapore).
|
||||
|
||||
## Outcome
|
||||
|
||||
The registered five-second instrumentation-aware verifier did not pass P1.
|
||||
The stronger simulator-aware comparison also failed the independent
|
||||
contribution bar. On the frozen `k=2` end-to-end replay:
|
||||
|
||||
- `sim top-k + real final` selected the real oracle with zero regret;
|
||||
- instrumentation-aware also selected the oracle, but reduced online H20-hours
|
||||
by only **1.426%** (1.329% when the prior failed attempt is added to both);
|
||||
- the required reduction was 30% versus full real final and 20% versus a safe
|
||||
outcome-only calibrator;
|
||||
- the outcome-only calibrator was not safe: it rejected the true best cell, so
|
||||
its apparent cost saving is not a deployable comparison.
|
||||
|
||||
This rejects the claim that the **current joint logistic verifier**, trained on
|
||||
one historical workload, gives the harness an independent tuning contribution.
|
||||
It does not prove that engine telemetry contains no useful signal. Telemetry
|
||||
improved held-out classification and removed unsafe decisions, but did not turn
|
||||
that signal into meaningful end-to-end tuning-cost reduction.
|
||||
|
||||
## Frozen setup
|
||||
|
||||
- Host: `dash0`, 8 NVIDIA H20 GPUs; cells were serialized and used TP1, TP2,
|
||||
or TP4 without co-resident serving jobs.
|
||||
- Engine/model: patched vLLM 0.24.1.dev3, Qwen3-30B-A3B BF16.
|
||||
- Workload: held-out `chat_w20260312_1000`, seven disjoint repeat bands,
|
||||
60-second replay after 0.1 time scaling, input `[0,8192]`, exactly 128 output
|
||||
tokens.
|
||||
- SLO: stepped TTFT 2/4/6 seconds, TPOT 50 ms, request pass rate at least 0.95.
|
||||
- Cells: TP1/MNS8, TP1/MNS64, TP2/MNS8, TP2/MNS64, TP4/MNS16, TP4/MNS64.
|
||||
- Per cell: burn-in, three low-rate repeats, and three high-rate repeats. The
|
||||
first repeat supplied the five-second prefix; 2-of-3 supplied its label.
|
||||
- Models: the registered pair used config/workload/outcome versus the same
|
||||
vector plus Layer-1 engine telemetry. The strengthened pair additionally
|
||||
gave both models identical frozen Frontier throughput, SLO pass-rate, and
|
||||
feasibility predictions.
|
||||
- Policy: accept at `p>=0.95`, reject at `p<=0.05`, otherwise continue the same
|
||||
trial. Model, cutoff, threshold, role order, request hashes, and cap were
|
||||
frozen before their applicable evaluation.
|
||||
|
||||
The first launch failed its warm-up input-count validation before a measured
|
||||
anchor. It cost 0.020552 H20-hours. The corrected primary attempt cost
|
||||
1.722112 H20-hours, so aggregate campaign cost was **1.742664 H20-hours**, below
|
||||
the 3.5 cap. The fix changed only warm-up validation; formal request counts and
|
||||
hash checks were unchanged.
|
||||
|
||||
## P1 labels are not an artificial easy split
|
||||
|
||||
The 12 adjudicated anchor labels contain 7 feasible and 5 infeasible examples.
|
||||
They are not simply “low feasible, high infeasible”:
|
||||
|
||||
- TP2/MNS64 high was feasible in all three repeats;
|
||||
- TP4/MNS64 low and high were feasible in all six repeats;
|
||||
- TP4/MNS16 low and high were infeasible in all six repeats.
|
||||
|
||||
That last pair creates a large real MNS interaction under an otherwise matched
|
||||
TP4 configuration. Frontier correctly predicted TP4/MNS64 high as feasible,
|
||||
but incorrectly predicted TP4/MNS16 low as feasible. It also incorrectly
|
||||
predicted TP1/MNS64 high as feasible. Overall simulator-only feasibility was
|
||||
10/12 correct: 83.33% accuracy, with two false-feasible predictions and no
|
||||
false-infeasible prediction.
|
||||
|
||||
The two false-feasible cases expose the intended latent-state problem. At five
|
||||
seconds, all 26 completed TP4/MNS16-low requests and all 9 completed
|
||||
TP1/MNS64-high requests still passed their SLO, although both full anchors were
|
||||
infeasible. External outcomes had not yet exposed the future failure; queue,
|
||||
running-batch, and scheduler state existed before the tail outcome. This is
|
||||
mechanistic evidence that instrumentation can be useful, not evidence that the
|
||||
current learned policy uses it well enough.
|
||||
|
||||
## Registered and strengthened prefix results
|
||||
|
||||
At the frozen 0.95 policy threshold:
|
||||
|
||||
| Comparison | Accuracy | Balanced acc. | Early decisions | False accept | False reject | Valid primary-trial saving |
|
||||
|---|---:|---:|---:|---:|---:|---:|
|
||||
| Registered outcome-only | 41.67% | 50.00% | 6/12 | 0 | 2 | invalid |
|
||||
| Registered + telemetry | 66.67% | 71.43% | 4/12 | 0 | 0 | 11.44% |
|
||||
| Strong sim + outcome | 66.67% | 68.57% | 5/12 | 0 | 1 | invalid |
|
||||
| Strong sim + outcome + telemetry | 83.33% | 85.71% | 4/12 | 0 | 0 | 11.44% |
|
||||
|
||||
For the strong pair, telemetry was correct on two examples where the baseline
|
||||
was wrong and lost none; McNemar's exact two-sided p-value is 0.5 at `n=12`.
|
||||
This is a safety/classification improvement, not a cost contribution. The
|
||||
registered instrumentation policy made two fewer early decisions than its
|
||||
baseline, so it failed the registered `+3 decisions or +15 percentage points`
|
||||
incremental gate.
|
||||
|
||||
The result is not robust to the frozen regularization sensitivity:
|
||||
|
||||
| L2 lambda | Sim+outcome acc. | +telemetry acc. | Base policy errors | Telemetry policy errors | Base saving | Telemetry saving |
|
||||
|---:|---:|---:|---:|---:|---:|---:|
|
||||
| 0.1 | 41.67% | 75.00% | 4 | 2 | invalid | invalid |
|
||||
| 1.0 | 66.67% | 83.33% | 1 | 0 | invalid | 11.44% |
|
||||
| 10.0 | 83.33% | 83.33% | 0 | 0 | 0.00% | 5.98% |
|
||||
|
||||
Consequently the positive classification delta is neither statistically nor
|
||||
hyperparameter robust.
|
||||
|
||||
## End-to-end shortlist result
|
||||
|
||||
Frontier's simulator-feasible ranking on the tested P1 surface was:
|
||||
|
||||
| Rank | Cell / anchor | Sim throughput/GPU | Real feasible | Real offered goodput/GPU |
|
||||
|---:|---|---:|---:|---:|
|
||||
| 1 | TP4/MNS64 high | 3.0718 | yes | 3.1250 |
|
||||
| 2 | TP1/MNS64 high | 2.8823 | no | 2.9833 |
|
||||
| 3 | TP2/MNS64 high | 2.8096 | yes | 2.8750 |
|
||||
| 4 | TP4/MNS16 low | 2.0866 | no | 2.1250 |
|
||||
| 5 | TP1/MNS8 low | 1.9806 | yes | 2.0333 |
|
||||
| 6 | TP2/MNS8 low | 1.8637 | yes | 1.9083 |
|
||||
|
||||
The real oracle was TP4/MNS64 high at 3.125 req/s/GPU. Cost includes an
|
||||
inferred per-cell startup/warm-up/burn-in/cleanup component and the selected
|
||||
anchor; benchmark-only 2-of-3 annotation intervals are removed. Gaps around
|
||||
annotation runs remain in the shared setup term, making this a conservative
|
||||
method-cost estimate.
|
||||
|
||||
| Frozen policy (`k=2`) | Online H20-hours | + prior failure | Real regret | Safety | Saving vs full |
|
||||
|---|---:|---:|---:|---|---:|
|
||||
| Sim top-2 + real final | 0.281383 | 0.301935 | 0.00% | valid | — |
|
||||
| Sim + outcome prefix | 0.214664 | 0.235216 | no selected cell | 1 false reject | invalid |
|
||||
| Sim + outcome + telemetry | 0.277370 | 0.297922 | 0.00% | valid | **1.426%** |
|
||||
|
||||
For `k=1`, the simulator already picked the oracle and telemetry abstained, so
|
||||
its saving was exactly 0%. For `k=3`, telemetry remained correct but saved
|
||||
1.007%. The result therefore does not depend on an unfavorable choice of `k`.
|
||||
|
||||
Across all six cells, a generous replay that removes benchmark-only repeat
|
||||
intervals estimates 0.944244 H20-hours for the full online workflow. The
|
||||
instrumentation policy saved 0.043467 H20-hours: 11.44% of primary trial time,
|
||||
but only **4.603%** of end-to-end online cost. Even a post-hoc oracle symmetric
|
||||
threshold cannot make the current telemetry model reach the contribution bar;
|
||||
its best zero-error envelope saves at most 5.69% of inferred online cost. A
|
||||
strong outcome-only model at another post-hoc regularization/threshold can save
|
||||
16.13%, which further prevents attributing a unique advantage to telemetry.
|
||||
These oracle-threshold numbers are diagnostics only and are not test evidence.
|
||||
|
||||
## Why the learned verifier did not generalize
|
||||
|
||||
The training corpus has only 37 anchors from one workload/SLO task. P1 shows
|
||||
large covariate shift:
|
||||
|
||||
- sim+outcome: 12/192 feature values exceed 3 training standard deviations and
|
||||
4 exceed 5; maximum absolute z-score is 10.36;
|
||||
- sim+outcome+telemetry: 19/396 exceed 3 and 9 exceed 5;
|
||||
- the largest shifts include admitted input-length mean (10.36), waiting state
|
||||
(7.77), running maximum (6.38), and decode-batch maximum (6.08).
|
||||
|
||||
Coefficient attribution shows that the input-length feature dominates several
|
||||
wrong feasible-anchor logits. Because all training examples share one task,
|
||||
the joint classifier can learn incidental within-task correlation and override
|
||||
a correct simulator prior on TP2/MNS64-high and TP4/MNS64-high. This is a
|
||||
supported diagnosis of model/data insufficiency; it is not a causal proof that
|
||||
one feature alone caused the P1 failure.
|
||||
|
||||
More importantly, retuning lambda, threshold, features, or cutoff on P1 and
|
||||
then calling P1 a held-out result would violate calibration/evaluation
|
||||
separation. P1 may now be used only as development data.
|
||||
|
||||
## Decision and the only defensible reopening condition
|
||||
|
||||
Do not run registered P2/P3 with the current model. It failed the predeclared
|
||||
gate on the favorable primary-trial denominator and is even farther from the
|
||||
bar under end-to-end cost. Spending six-task headline GPU budget on the same
|
||||
method would be metric shopping, not replication.
|
||||
|
||||
A new route may be opened only as a new hypothesis:
|
||||
|
||||
1. Replace the joint classifier with a **simulator-residual verifier**. The
|
||||
simulator prediction remains an explicit prior; nested outcome-only and
|
||||
telemetry models learn when that prior is wrong, rather than freely
|
||||
relearning feasibility and overriding it under workload shift.
|
||||
2. Train on multiple complete workload/SLO tasks. SLO thresholds and target
|
||||
pass rate must be explicit inputs; splits are by complete task.
|
||||
3. Calibrate abstention with task-level risk control. No threshold is selected
|
||||
on a headline task, and “never early decide” is included as the safe
|
||||
outcome-only baseline.
|
||||
4. Treat Phase 6 and P1 as development only, freeze the residual architecture,
|
||||
features, cutoff, threshold, simulator reading, and `k`, then use entirely
|
||||
new trace windows for a new gate.
|
||||
|
||||
This reopening is justified only if development data show both (a) the
|
||||
simulator's errors are predictable from pre-outcome engine state and (b) a
|
||||
simulator-preserving residual model does not corrupt correct simulator
|
||||
predictions. It is a new project decision, not a continuation automatically
|
||||
authorized by P1.
|
||||
|
||||
## Benchmark audit
|
||||
|
||||
| Audit item | Verdict | Severity | Evidence / disposition |
|
||||
|---|---|---|---|
|
||||
| Calibration set separate from P1 | PASS | — | Phase 6/0311 trained; P1/0312 tested |
|
||||
| Strong simulator-aware baseline | PASS | — | Identical Frontier features in both nested models |
|
||||
| Sim top-k + real-final E2E baseline | PASS | — | Frozen `k=2`, tie expansion, measured setup/continuation cost |
|
||||
| Multiple independent headline tasks | NEEDS EVIDENCE | Blocking for a positive claim | P1 gate failed; P2 correctly not opened |
|
||||
| Statistical significance | NEEDS EVIDENCE | Blocking for a positive claim | n=12 anchors from one task; McNemar p=0.5 |
|
||||
| Hyperparameter robustness | FAIL | Blocking | Lambda sensitivity changes safety and relative result |
|
||||
| Full resource accounting | PASS for P1 | — | Failures, startup/warm-up/burn-in, continuation and annotation separated |
|
||||
| Avoid post-test retuning | PASS only if route stops | Blocking if violated | P1 is now development-only |
|
||||
| Selective winning-workload reporting | PASS | — | Negative P1 and TP/MNS losing cases retained |
|
||||
|
||||
Overall recommendation: **Block the current independent harness contribution
|
||||
claim.**
|
||||
|
||||
## Artifacts
|
||||
|
||||
- Registered protocol: `docs/fidelity-aware-harness-protocol-20260714.md`
|
||||
- Historical headroom: `docs/fidelity-aware-harness-headroom-20260714.md`
|
||||
- Registered P1 analysis: `runs/fidelity-headroom/analyze_pilot.py`
|
||||
- Strong P1 analysis: `runs/fidelity-headroom/analyze_strong_pilot.py`
|
||||
- E2E shortlist replay: `runs/fidelity-headroom/analyze_pilot_e2e.py`
|
||||
- External immutable result root:
|
||||
`/home/gahow/phd/replayserve/runs/fidelity_p1_frontier_committed_20260714`
|
||||
|
||||
## Data sanity block
|
||||
|
||||
| Data | n | Min | Max | Distinct | Invariant |
|
||||
|---|---:|---:|---:|---:|---|
|
||||
| P1 labels | 12 | 0 | 1 | 2 | 7 feasible / 5 infeasible |
|
||||
| Primary elapsed seconds | 12 | 19.448 | 61.435 | 12 | Every five-second prefix is in range |
|
||||
| Prefix Layer-1 records | 12 | 332 | 557 | 12 | Contiguous; zero drops |
|
||||
| Exact timestamped outcomes | 12 anchors | 54 | 750 | 11 | Monotonic completion timestamps |
|
||||
| Simulator pass rate | 12 | 0.1548 | 1.0 | 7 | Ratios in `[0,1]` |
|
||||
| Strong nested probabilities | 24 | 0.000208 | 0.809422 | 24 | Ratios in `[0,1]` |
|
||||
| E2E cost components | 36 | 0.001389 | 0.169653 H20-h | 21 | Non-negative |
|
||||
| GPU attempts | 2 | 0.020552 | 1.722112 H20-h | 2 | Aggregate 1.742664 < 3.5 |
|
||||
| Copied raw files | 191 | — | 153,093,348 bytes total | — | Remote/local aggregate SHA identical |
|
||||
|
||||
Checked invariants: six cells and twelve anchors; exact request count and
|
||||
request-ID/arrival/length hashes; all cell validation flags true; both labels
|
||||
present; probabilities bounded; costs and counters non-negative; simulator
|
||||
results not all identical; committed simulator rerun 12/12 numerically
|
||||
identical to the exploratory run; no prompt text in public simulator fixtures;
|
||||
no co-resident serving process; final eight GPUs at 0 MiB and 0% utilization.
|
||||
No red flag remains.
|
||||
@@ -1,9 +1,18 @@
|
||||
# Fidelity-aware real-verification harness protocol
|
||||
|
||||
Status: **PRE-REGISTERED STAGED EVALUATION; CONTRIBUTION NOT YET ESTABLISHED**.
|
||||
Status: **P1 FAILED; P2/P3 CLOSED FOR THE REGISTERED METHOD; CONTRIBUTION NOT ESTABLISHED**.
|
||||
|
||||
Date frozen: 2026-07-14 (Asia/Singapore).
|
||||
|
||||
Post-run disposition (2026-07-14): P1 completed with valid data but failed its
|
||||
registered incremental gate. The strengthened simulator-aware comparison and
|
||||
end-to-end `k=2` replay also failed: instrumentation was safe and retained zero
|
||||
regret, but reduced online H20-hours by only 1.426% versus sim top-k + real
|
||||
final, against the 30% bar. Outcome-only was unsafe. P2/P3 are therefore not
|
||||
opened for this model. Full results and the permitted reopening condition are
|
||||
in `docs/fidelity-aware-harness-p1-report-20260714.md`; the protocol below is
|
||||
retained unchanged as the pre-run record.
|
||||
|
||||
## Research question and contribution bar
|
||||
|
||||
The harness has an independent systems contribution only if engine-internal
|
||||
@@ -52,6 +61,35 @@ difference is Z. The initial family is intentionally simple: a positive result
|
||||
then demonstrates value in the engine signal rather than capacity in a larger
|
||||
learner. A sequence model is admissible only as a later, paired ablation.
|
||||
|
||||
### Amendment A1: strengthen the calibration baseline before P2
|
||||
|
||||
Frozen 2026-07-14 13:08 Asia/Singapore, after P1 launch but before P1
|
||||
completion or analysis. A baseline audit found that the first frozen P1
|
||||
models use the simulator only to define candidate order; their feature vectors
|
||||
do not contain the simulator's per-anchor prediction. This is insufficient
|
||||
for the stronger term **outcome-only calibration**. P1 therefore remains a
|
||||
prospective test of the originally frozen cross-workload predictor, but cannot
|
||||
by itself open a contribution claim.
|
||||
|
||||
For P2/P3, both nested models must additionally receive the identical frozen
|
||||
simulator outputs available at that decision: predicted completed throughput
|
||||
per GPU, predicted SLO pass rate, and predicted feasibility. The comparison
|
||||
is consequently `sim + config + workload + real outcome prefix` versus that
|
||||
exact vector plus real engine state. Simulator features, regularization,
|
||||
cutoff, and thresholds are frozen before any P2 task. If telemetry does not
|
||||
improve this stronger baseline, the harness has no independent contribution.
|
||||
|
||||
The same audit also separates algorithm cost from benchmark-oracle cost.
|
||||
Headline method cost includes every action the method would execute online:
|
||||
simulator profiling/calibration, model onboarding, server startup, warm-up,
|
||||
real prefix, continuation after abstention, method-requested confirmation,
|
||||
logging overhead, failures, and cleanup. Exhaustive real-oracle runs and the
|
||||
extra repetitions used only to construct 2-of-3 evaluation labels are common
|
||||
benchmark annotation cost; they are reported separately and charged to no
|
||||
method. A second, deliberately conservative table adds that common cost to
|
||||
all methods. This prevents both hiding real method cost and making the
|
||||
percentage gate mathematically depend on offline ground-truth annotation.
|
||||
|
||||
The frozen first policy uses a 5-second prefix, L2 regularization 1.0, and a
|
||||
two-sided abstaining threshold of 0.95: accept at `p(feasible)>=0.95`, reject at
|
||||
`p(feasible)<=0.05`, otherwise continue the exact same trial to completion.
|
||||
@@ -64,15 +102,15 @@ therefore not evidence; all claims come from subsequent held-out tasks.
|
||||
|---|---:|---:|---:|---:|---:|
|
||||
| Real-only oracle | no | no | full | optional diagnostic | every candidate/anchor |
|
||||
| Sim top-k + real final | yes | included in full run | full | no decision use | every shortlisted candidate/anchor |
|
||||
| Outcome-only calibration | yes | yes | yes | no | only on abstention |
|
||||
| Instrumentation-aware | yes | yes | yes | yes | only on abstention |
|
||||
| Outcome-only calibration | yes, including its prediction features | yes | yes | no | only on abstention |
|
||||
| Instrumentation-aware | same prediction features | yes | yes | yes | only on abstention |
|
||||
|
||||
Tie buckets are expanded before top-k. `k` is selected on training tasks and
|
||||
is fixed on held-out tasks; an oracle per-task k is forbidden. Outcome-only
|
||||
receives all information available outside the engine, including config and
|
||||
workload features. Instrumentation cannot use any record submitted after the
|
||||
cutoff. The full label, confirmation votes, simulator error, and later
|
||||
requests are never model features.
|
||||
receives all information available outside the engine, including config,
|
||||
workload, and frozen simulator-prediction features. Instrumentation cannot use
|
||||
any record submitted after the cutoff. The full label, confirmation votes,
|
||||
realized simulator error, and later requests are never model features.
|
||||
|
||||
## Staged experiment
|
||||
|
||||
|
||||
71
docs/intervention-response-v0-protocol-20260714.md
Normal file
71
docs/intervention-response-v0-protocol-20260714.md
Normal file
@@ -0,0 +1,71 @@
|
||||
# Telemetry intervention-response v0 protocol
|
||||
|
||||
Status: **FROZEN BEFORE V0 ANALYSIS**.
|
||||
|
||||
Date: 2026-07-14 (Asia/Singapore).
|
||||
|
||||
## Claim boundary
|
||||
|
||||
The closed residual route asked whether one absolute engine-state snapshot can
|
||||
predict unmeasured configurations. V0 asks a different, narrower question:
|
||||
|
||||
> Does an adjacent, controlled MNS intervention produce an early engine-state
|
||||
> response that is distinguishable from same-config repeat noise?
|
||||
|
||||
Passing this gate only authorizes a matched real-GPU pilot. It does not prove
|
||||
that telemetry improves tuning, that any metric is a causal mediator, or that
|
||||
the response transfers to a new workload, topology, or knob family.
|
||||
|
||||
## Data and estimand
|
||||
|
||||
- Source: Phase 6 solo-authoritative Qwen3-30B-A3B/vLLM 0.24 Layer-1 streams.
|
||||
- Action pairs: primary runs at identical study hash, TP, sampling anchor, and
|
||||
request-order hash, with adjacent `MNS={8,16,32,64}` values.
|
||||
- Noise pairs: primary versus confirmation at the same complete config,
|
||||
anchor, and request-order hash. Only primary-to-confirmation pairs are used;
|
||||
confirmations are not combined into pseudo-independent all-pairs.
|
||||
- Fixed early windows: 5 seconds and 10 seconds from the measured interval
|
||||
start. All runs exceed 10 seconds, so early-stop censoring cannot change the
|
||||
telemetry window.
|
||||
- Full-run pass rate and feasibility are descriptive only because an early
|
||||
stop can make full elapsed durations differ.
|
||||
|
||||
The statistical unit is a run pair. Scheduler steps are summarized within a
|
||||
run and are never counted as independent trials.
|
||||
|
||||
## Frozen response gate
|
||||
|
||||
The directly measured gate features are scheduler-step rate, decode-batch
|
||||
mean, prefill-token fraction, waiting/running queue mean, KV-usage mean, and
|
||||
CUDA-graph padding fraction.
|
||||
|
||||
A feature qualifies at one horizon only if:
|
||||
|
||||
1. at least 75% of nonzero action deltas have the same sign;
|
||||
2. median absolute action delta is at least 2x the median absolute repeat
|
||||
delta; and
|
||||
3. at least 50% of action deltas exceed the repeat-noise absolute p95.
|
||||
|
||||
V0 opens a GPU pilot only if:
|
||||
|
||||
- there are exactly 17 frozen adjacent-MNS action pairs;
|
||||
- there are at least 20 primary/confirmation repeat pairs;
|
||||
- all identity, finite-value, counter, and ratio invariants pass; and
|
||||
- at least two gate features qualify at both 5 and 10 seconds.
|
||||
|
||||
Any data red flag stops the analysis before interpreting the response.
|
||||
|
||||
## If V0 passes
|
||||
|
||||
Register a dash0 pilot around a known scaling knee. The pilot must use the
|
||||
same request sequence and arrival times, one serving job at a time, one changed
|
||||
knob, randomized `A/B` versus `B/A` order, common non-censored measurement
|
||||
windows, and trial-level repetitions. It must compare a response-aware next
|
||||
action against an outcome-only policy under complete startup, warm-up, and
|
||||
H20-hour accounting.
|
||||
|
||||
## If V0 fails
|
||||
|
||||
Do not add telemetry fields or train a larger model. The current Layer-1 state
|
||||
does not identify even an MNS intervention above repeat noise on this task, so
|
||||
the telemetry-guided tuning route remains diagnostic only.
|
||||
157
docs/intervention-response-v0-results-20260714.md
Normal file
157
docs/intervention-response-v0-results-20260714.md
Normal file
@@ -0,0 +1,157 @@
|
||||
# Telemetry intervention-response v0/v1 results
|
||||
|
||||
Date: 2026-07-14 (Asia/Singapore).
|
||||
|
||||
## Decision
|
||||
|
||||
**STOP before a new H20 pilot.** The current Layer-1 aggregate telemetry does
|
||||
not identify a sufficiently general early response to an MNS intervention,
|
||||
and it does not improve action-efficacy prediction over exact external prefix
|
||||
outcomes on the available development tasks.
|
||||
|
||||
This is a negative result about the present state representation and
|
||||
experiment design. It does not establish that engine telemetry is useless for
|
||||
tuning, and it is not held-out evidence.
|
||||
|
||||
## Hypothesis and frozen test
|
||||
|
||||
The tested hypothesis was:
|
||||
|
||||
> With the workload and all non-MNS settings held fixed, increasing MNS causes
|
||||
> a 5--10 second engine-state response that is larger than same-config repeat
|
||||
> noise and that predicts whether the action makes the full run feasible.
|
||||
|
||||
A response feature had to satisfy all three frozen conditions at both 5 and
|
||||
10 seconds: at least 0.75 sign consistency, median absolute action effect at
|
||||
least 2x the repeat median, and at least 0.50 of action deltas above the repeat
|
||||
absolute p95. At least two features had to pass. A telemetry feature was
|
||||
decision-relevant only if its leave-one-repeat-out balanced accuracy was at
|
||||
least 0.75 and at least 0.15 above the best exact external prefix outcome.
|
||||
|
||||
## What was implemented
|
||||
|
||||
- A common-window analyzer over the existing per-scheduler-step Layer-1 stream.
|
||||
- Exact action pairing with request-order hash, offered load, TP, load role,
|
||||
and repetition held fixed.
|
||||
- Same-config repeat-noise estimation without treating scheduler steps as
|
||||
independent samples.
|
||||
- Exact 5/10-second request-prefix outcomes using monotonic completion times.
|
||||
- A one-feature leave-one-repeat-out efficacy audit; no multivariate model was
|
||||
fitted to the 12 examples.
|
||||
- Input hashes, stream hashes, frozen thresholds, pair-level deltas, and sanity
|
||||
invariants in machine-readable audit artifacts.
|
||||
- Trial-by-trial validation against the P1 manifest, plus content hashes for
|
||||
every result, request file, and Layer-1 stream.
|
||||
|
||||
## Experiment A: Phase-6 retrospective audit
|
||||
|
||||
Phase 6 supplied 17 adjacent-MNS actions and 29 same-config
|
||||
primary/confirmation pairs. No feature passed at either horizon, producing
|
||||
`STOP_NO_IDENTIFIABLE_RESPONSE`.
|
||||
|
||||
The confirmation sample is not a clean replication distribution: confirmations
|
||||
were selectively run after disputed primary outcomes. Several same-config
|
||||
pairs consequently followed radically different trajectories. This result
|
||||
therefore remains a valid failure of the frozen v0 gate, but it cannot by itself
|
||||
separate normal run variance from confirmation-selection bias.
|
||||
|
||||
## Experiment B: prospective-repeat confirmation
|
||||
|
||||
P1 supplied three pre-arranged, disjoint request bands for every cell/load.
|
||||
Exact matched actions exist for TP1 `MNS 8 -> 64` and TP4 `MNS 16 -> 64`, at
|
||||
low/high load and repetitions 1/2/3. This yields 12 action pairs and 24
|
||||
same-config consecutive-repeat pairs.
|
||||
|
||||
The 24 adjacent repeat differences share their middle repetition within each
|
||||
three-run group. They define a conservative empirical noise reference; they
|
||||
are not used as 24 independent samples in an inferential test.
|
||||
|
||||
The result is `STOP_NO_PROSPECTIVE_RESPONSE`: zero features passed the response
|
||||
gate at either horizon.
|
||||
|
||||
The strongest response was mean waiting-queue occupancy:
|
||||
|
||||
| Horizon | Sign consistency | Action/repeat median | Action above repeat p95 | Gate |
|
||||
|---|---:|---:|---:|---|
|
||||
| 5 s | 1.000 | 1.292x | 0.167 | fail |
|
||||
| 10 s | 1.000 | 2.611x | 0.250 | fail |
|
||||
|
||||
The direction is real enough to merit diagnosis, but the effect is not broad
|
||||
enough to guide a general action. It is large for TP4/high-load trials and
|
||||
small or absent in other regimes.
|
||||
|
||||
Full-run transitions contain six beneficial actions (`false -> true`) and six
|
||||
non-beneficial actions (three `false -> false`, three `true -> true`). The
|
||||
beneficial label is also perfectly confounded with TP4 in this small dataset,
|
||||
so it cannot support a topology-general claim.
|
||||
|
||||
| Horizon | Best telemetry delta | Balanced accuracy | Best external prefix delta | Balanced accuracy | Telemetry advantage |
|
||||
|---|---|---:|---|---:|---:|
|
||||
| 5 s | waiting queue | 0.750 | max TPOT / SLO | 0.833 | -0.083 |
|
||||
| 10 s | waiting queue | 0.750 | outstanding / admitted | 0.750 | 0.000 |
|
||||
|
||||
No telemetry feature reaches the preregistered `+0.15` incremental threshold.
|
||||
|
||||
## What this rules out
|
||||
|
||||
It rules out using the current vector of 5/10-second global means as a solid
|
||||
mechanism for choosing the next config. In particular, adding these aggregates
|
||||
to an LLM prompt or fitting a larger predictor would currently hide, rather
|
||||
than solve, the identifiability problem.
|
||||
|
||||
It does not rule out an instrumentation-aware tuner built around a deliberately
|
||||
excited local system. The existing runs were designed for endpoint/fidelity
|
||||
evaluation, not system identification: the MNS action is large, efficacy is
|
||||
confounded with TP, repeat bands contain different requests, and global means
|
||||
erase when queue buildup or service-rate changes occur.
|
||||
|
||||
## Required redesign before spending H20-hours
|
||||
|
||||
The next admissible experiment is a randomized, local A/B system-identification
|
||||
pilot around one fixed TP and one load knee:
|
||||
|
||||
1. Replay the exact same request sequence and arrival times for both endpoints.
|
||||
2. Use small adjacent actions and randomized `A/B` versus `B/A` order.
|
||||
3. Record event-aligned response curves, including queue growth/drain rate,
|
||||
prefill/decode service rate, and per-step service time, rather than only one
|
||||
global mean.
|
||||
4. Separate a mechanism gate (repeatable response) from the end-to-end gate:
|
||||
fewer trials or H20-hours to select a feasible near-optimal config than an
|
||||
outcome-only tuner.
|
||||
5. Hold out a second load/workload for the final policy comparison.
|
||||
|
||||
Until that design is frozen, a wider sweep would only generate more correlated
|
||||
observations and is not justified by the evidence above.
|
||||
|
||||
## Reproduction
|
||||
|
||||
```bash
|
||||
python3 runs/intervention-response-v0/test_analysis.py
|
||||
python3 runs/intervention-response-v0/test_p1_analysis.py
|
||||
|
||||
python3 runs/intervention-response-v0/analyze_phase6.py \
|
||||
--metrics runs/opprof-phase6/phase6/metrics.json \
|
||||
--raw-root runs/opprof-phase6/phase6/solo-authoritative/cells \
|
||||
--output runs/intervention-response-v0/phase6-audit.json
|
||||
|
||||
python3 runs/intervention-response-v0/analyze_p1.py \
|
||||
--run-root /home/gahow/phd/replayserve/runs/fidelity_p1_frontier_committed_20260714/real/p1b \
|
||||
--manifest /home/gahow/phd/replayserve/runs/fidelity_p1_frontier_committed_20260714/real/p1b/pilot-manifest.json \
|
||||
--output runs/intervention-response-v0/p1-audit.json
|
||||
```
|
||||
|
||||
## Data sanity
|
||||
|
||||
- Phase 6: action pairs `n=17`, repeat pairs `n=29`, trials `n=66`; MNS
|
||||
action size min/max `8/32`, `3` distinct; action-state vectors `n=17`, `17`
|
||||
distinct; streams `n=12`, bytes min/max `12,745,297/52,957,710`, `12`
|
||||
distinct.
|
||||
- P1: action pairs `n=12`, repeat pairs `n=24`, trials `n=36`; MNS action
|
||||
size min/max `48/56`, `2` distinct; efficacy labels `n=12`, min/max `0/1`,
|
||||
`2` distinct; streams `n=6`, bytes min/max `17,449,143/29,431,988`, `6`
|
||||
distinct.
|
||||
- Checked invariants: exact action request hashes and offered loads match;
|
||||
all `36/36` P1 trials match the manifest; expected pair counts hold; all
|
||||
deltas are finite; non-negative counters and bounded ratios hold; per-config
|
||||
state vectors are not all identical; both efficacy classes are present. No
|
||||
red flags were observed.
|
||||
58
docs/intervention-response-v1-p1-protocol-20260714.md
Normal file
58
docs/intervention-response-v1-p1-protocol-20260714.md
Normal file
@@ -0,0 +1,58 @@
|
||||
# Intervention-response v1 prospective-repeat confirmation
|
||||
|
||||
Status: **FROZEN AFTER PHASE-6 V0 FAILURE AND BEFORE P1 RESPONSE ANALYSIS**.
|
||||
|
||||
Date: 2026-07-14 (Asia/Singapore).
|
||||
|
||||
## Why this is a new confirmation, not a relaxed V0
|
||||
|
||||
Phase-6 V0 failed its frozen global response gate. Its 29 same-config
|
||||
confirmations were triggered after disputed outcomes, and the resulting noise
|
||||
sample contains extreme trajectory divergence by construction. V0 remains
|
||||
failed and its thresholds are unchanged.
|
||||
|
||||
The already-completed P1 campaign supplies a distinct test: three
|
||||
prospectively scheduled, disjoint repeat bands for every cell/load. TP1 and
|
||||
TP4 use identical offered loads and exact request-order hashes across their MNS
|
||||
endpoints. V1 asks whether an MNS response is identifiable against this
|
||||
prospective workload-repeat noise, and whether that response predicts action
|
||||
efficacy beyond exact external prefix outcomes.
|
||||
|
||||
P1 is now development data. No result here is held-out or paper-facing.
|
||||
|
||||
## Frozen pairs
|
||||
|
||||
- Action pairs: TP1 `MNS 8 -> 64` and TP4 `MNS 16 -> 64`, at low/high load and
|
||||
repeat 1/2/3. Endpoints must have identical TP, offered rate, repeat role,
|
||||
and request-order hash. Expected `n=12`.
|
||||
- Repeat-noise pairs: consecutive pre-arranged repeat bands within each of six
|
||||
cells and low/high load: `rep1 -> rep2`, `rep2 -> rep3`. Expected `n=24`.
|
||||
Repeat bands intentionally contain different requests and therefore include
|
||||
workload-sampling noise rather than pretending to be identical trials.
|
||||
Adjacent differences share the middle run; the gate uses their empirical
|
||||
magnitude only and does not treat the 24 differences as independent samples
|
||||
for a p-value or confidence interval.
|
||||
- Prefix horizons: 5 and 10 seconds. Exact monotonic request completion times
|
||||
and the same Layer-1 intervals are used.
|
||||
|
||||
## Frozen gates
|
||||
|
||||
The response-identifiability thresholds are exactly the Phase-6 V0 thresholds:
|
||||
75% sign consistency, 2x median effect/repeat noise, and at least 50% of action
|
||||
deltas above repeat absolute p95. At least two response features must qualify
|
||||
at both horizons.
|
||||
|
||||
Action efficacy is one only for an infeasible-to-feasible full-run transition.
|
||||
The 12 action pairs must contain at least four examples of each class.
|
||||
|
||||
For decision relevance, each individual external-outcome response feature and
|
||||
each individual telemetry-response feature is evaluated by leave-one-repeat-
|
||||
band-out threshold fitting. This intentionally avoids a multivariate model on
|
||||
12 examples. At least one telemetry feature must, at both horizons:
|
||||
|
||||
1. reach balanced accuracy at least 0.75; and
|
||||
2. exceed the best external-outcome response feature by at least 0.15.
|
||||
|
||||
Only if data validity, response identifiability, and incremental decision
|
||||
relevance all pass does V1 open a newly registered matched GPU pilot. No
|
||||
threshold or feature is changed after observing V1.
|
||||
@@ -0,0 +1,94 @@
|
||||
# Phase-aware telemetry intervention-response v2 protocol
|
||||
|
||||
Status: **INVALID OPERATIONAL ATTEMPT; SUPERSEDED BY V3**.
|
||||
|
||||
Date: 2026-07-14 (Asia/Singapore).
|
||||
|
||||
The first MNS=16 session timed out while draining the 3.125 requests/s/GPU
|
||||
workload after its 300-second arrival window. It produced no high-load result,
|
||||
and no MNS=64 endpoint was run. No comparative conclusion is drawn from this
|
||||
attempt; see `intervention-response-v3-two-load-protocol-20260714.md`.
|
||||
|
||||
## Correction to v0/v1
|
||||
|
||||
The 5/10-second analyses tested an ultra-early verifier. They did not test
|
||||
whether telemetry observed after the engine has developed queue, batch, and KV
|
||||
state can guide tuning. The P1 replay lasts 60 seconds after time scaling, and
|
||||
the 5/10-second prefixes contain only a small fraction of its requests.
|
||||
|
||||
V2 therefore replaces absolute cutoffs with replay phase. The old audits and
|
||||
their negative decisions remain immutable, but their claim is narrowed to the
|
||||
first 5/10 seconds.
|
||||
|
||||
## Historical corrective audit
|
||||
|
||||
The historical audit is development-only and cannot become confirmatory after
|
||||
the horizon concern was observed.
|
||||
|
||||
- Infer each trial's intended replay duration as selected requests divided by
|
||||
offered requests per second. All trials must agree.
|
||||
- Find every complete 10% replay decile supported by every trial. Analyze all
|
||||
such deciles; selecting only the best horizon is forbidden.
|
||||
- At each decile report both:
|
||||
- cumulative state from replay start to the checkpoint; and
|
||||
- the non-overlapping 10%-wide state block ending at the checkpoint.
|
||||
- Report admitted/completed request coverage, response-versus-repeat statistics,
|
||||
telemetry versus external-outcome efficacy, and per-feature trajectory drift.
|
||||
- Reuse the frozen v1 action and repeat pairs and the frozen response and
|
||||
incremental-efficacy thresholds. These thresholds are descriptive in V2;
|
||||
passing one post-hoc horizon does not open a contribution claim.
|
||||
|
||||
If early stopping prevents complete observation of the replay phases, the
|
||||
historical decision is `REQUIRES_UNCENSORED_PHASE_AWARE_PILOT`, independent of
|
||||
which early decile looks best.
|
||||
|
||||
## Uncensored matched pilot
|
||||
|
||||
The pilot is a mechanism gate, not paper evidence.
|
||||
|
||||
- Hardware/engine/model: solo placement on dash0, 4 NVIDIA H20 GPUs, patched
|
||||
vLLM `0.24.1.dev3+opprof`, Qwen3-30B-A3B, fixed `TP=4`.
|
||||
- Action: `MNS 16 -> 64`; topology, model, engine build, workload, arrival
|
||||
sequence, offered load, and all other settings remain fixed.
|
||||
- Workload: `chat_w20260312_1000` at replay-time scale `0.5`, hence 300 seconds.
|
||||
- Offered loads per GPU: `1.5`, `2.125`, and `3.125` requests/s. These supply a
|
||||
low control and the two already-observed P1 pressure regimes.
|
||||
- Repetitions: three disjoint session bands, exact request sequence matched
|
||||
across action endpoints. Endpoint order alternates `A/B`, `B/A`, `A/B`;
|
||||
load order is counter-rotated across repetitions.
|
||||
- A fresh server receives the accepted long-request warm-up and a bounded
|
||||
burn-in before each measured session.
|
||||
- SLO-unrecoverable early stop is disabled. Every run must observe the full
|
||||
300-second arrival window; a separate 360-second safety deadline may mark a
|
||||
run invalid but cannot manufacture a full-run label.
|
||||
- Cumulative checkpoints: 10%, 25%, 50%, 75%, and 100%, or 30/75/150/225/300
|
||||
seconds. Quarter blocks are analyzed separately from cumulative means.
|
||||
- A measured Layer-1 interval is complete only when its start-boundary,
|
||||
end-boundary, and maximum internal record gaps are each at most one second;
|
||||
timestamps must be monotonic.
|
||||
- Placement is serialized. Co-location remains forbidden because Phase 6
|
||||
observed material co-location-induced outcome shifts.
|
||||
- Hard cap: 8 H20-hours including startup, warm-up, burn-in, invalid attempts,
|
||||
and cleanup.
|
||||
|
||||
## Gates
|
||||
|
||||
Data validity requires complete 300-second Layer-1 coverage, zero dropped
|
||||
records, exact request/arrival/length hashes across action endpoints, monotonic
|
||||
timestamps, full request accounting, idle GPUs before and after each session,
|
||||
and no co-resident GPU process.
|
||||
|
||||
Mechanism evidence requires at least two telemetry features whose matched action
|
||||
response exceeds same-config repeat noise at the same pair of consecutive
|
||||
checkpoints under the unchanged v1 response thresholds. Those features must
|
||||
also have a consistent direction in at least two of the three load regimes.
|
||||
|
||||
Decision evidence additionally requires both action-efficacy classes and at
|
||||
least one of the phase-stable mechanism features to reach
|
||||
leave-one-repetition-out balanced accuracy at least 0.75 and exceed the best
|
||||
external prefix outcome by at least 0.15 at two adjacent predeclared
|
||||
checkpoints from 25% onward. Without label balance the pilot can adjudicate
|
||||
mechanism evidence only.
|
||||
|
||||
No H20 run is launched if the local analyzer/tests, manifest preflight, GPU
|
||||
probe, command dry-run, projected cost, or cleanup plan fails.
|
||||
160
docs/intervention-response-v3-results-20260714.md
Normal file
160
docs/intervention-response-v3-results-20260714.md
Normal file
@@ -0,0 +1,160 @@
|
||||
# Phase-aware telemetry intervention-response v3 results
|
||||
|
||||
Date: 2026-07-14 (Asia/Singapore).
|
||||
|
||||
Decision: **`STOP_NO_INCREMENTAL_TUNING_SIGNAL`**.
|
||||
|
||||
## Claim tested
|
||||
|
||||
After the replay has developed queue, batch, and KV state, does increasing MNS
|
||||
from 16 to 64 create telemetry responses that exceed workload-repeat noise, and
|
||||
does any such response identify whether the action repairs the full-run SLO
|
||||
better than external prefix outcomes alone?
|
||||
|
||||
The first clause passed. The second clause failed. Long-window telemetry is
|
||||
mechanistically informative, but this pilot does not support its necessity for
|
||||
tuning this action.
|
||||
|
||||
## Setup
|
||||
|
||||
- dash0 GPU 0-3: four NVIDIA H20 GPUs; Qwen3-30B-A3B; patched vLLM
|
||||
`0.24.1.dev3+opprof`; TP=4.
|
||||
- Action: MNS `16 -> 64` with exact request, arrival, and input-length hashes.
|
||||
- Trace: `chat_w20260312_1000`, replay-time scale 0.5, 300 seconds.
|
||||
- Loads: 1.5 and 2.125 requests/s/GPU; three disjoint bands; endpoint order
|
||||
A/B, B/A, A/B; load order low/mid, mid/low, low/mid.
|
||||
- Five cumulative checkpoints: 30, 75, 150, 225, and 300 seconds; four
|
||||
non-overlapping quarter blocks.
|
||||
- Six fresh-server sessions, 12 measured runs, six action pairs, and eight
|
||||
same-config repeat pairs.
|
||||
|
||||
The prior three-load attempt is not part of the result. Its MNS=16 workload at
|
||||
3.125 requests/s/GPU could not drain by the 450-second client timeout and
|
||||
produced no high-load result. V3 reran every retained point from scratch.
|
||||
|
||||
## End-to-end outcome
|
||||
|
||||
All three low-load pairs remained feasible (`true->true`, label 0). All three
|
||||
pressure-load pairs changed from infeasible to feasible (`false->true`, label
|
||||
1). MNS=16 pressure pass rates were 0.5604, 0.3145, and 0.2635; all three
|
||||
MNS=64 pressure runs reached 1.0. This yielded a balanced 3/3 action label set.
|
||||
|
||||
## Mechanism result
|
||||
|
||||
No telemetry feature passed the action-versus-repeat gate at 10% or 25%. At
|
||||
50%, graph padding first passed. At both 75% and 100%, graph padding and queue
|
||||
waiting passed, satisfying the requirement for two features at the same pair
|
||||
of adjacent checkpoints and with consistent directions in both load regimes.
|
||||
|
||||
| Feature | Direction for MNS 16->64 | 75% effect/repeat median | 75% above repeat p95 | 100% effect/repeat median | 100% above repeat p95 |
|
||||
|---|---:|---:|---:|---:|---:|
|
||||
| `queue_waiting_mean` | lower | 1346.22 | 3/6 | 898.70 | 3/6 |
|
||||
| `graph_padding_fraction` | higher | 5.00 | 4/6 | 5.74 | 5/6 |
|
||||
|
||||
The very large queue effect/median-repeat ratios should not be read alone: its
|
||||
repeat p95 was much larger than its repeat median, so the independent p95
|
||||
coverage criterion remained binding. Full-window queue-waiting deltas were
|
||||
-0.19 to -0.32 at low load and -20.99 to -32.16 at pressure load. Graph
|
||||
padding increased in every pair, by 0.00133-0.00217 at low load and
|
||||
0.00705-0.00873 at pressure load.
|
||||
|
||||
The mechanism is therefore a real tradeoff: larger MNS reduces queueing,
|
||||
especially under pressure, while increasing CUDA-graph padding.
|
||||
|
||||
## Tuning-signal result
|
||||
|
||||
Leave-one-repetition-out balanced accuracy was evaluated against the best
|
||||
external prefix-outcome feature at every predeclared checkpoint.
|
||||
|
||||
| Replay phase | Best external BA | Best telemetry BA | Incremental telemetry gate |
|
||||
|---:|---:|---:|---|
|
||||
| 10% | 0.833 | 0.833 | fail |
|
||||
| 25% | 1.000 | 1.000 | fail |
|
||||
| 50% | 0.833 | 1.000 | pass at this checkpoint only |
|
||||
| 75% | 1.000 | 1.000 | fail |
|
||||
| 100% | 1.000 | 1.000 | fail |
|
||||
|
||||
At 50%, several telemetry features exceeded the external baseline by at least
|
||||
0.15, including the phase-stable mechanism feature
|
||||
`graph_padding_fraction`. The advantage did not hold at either adjacent
|
||||
checkpoint. Consequently no feature passed the frozen two-adjacent-phase
|
||||
requirement.
|
||||
|
||||
The important ordering is that external TTFT already classified the action
|
||||
perfectly at 25%, whereas the robust two-feature mechanism response did not
|
||||
emerge until 75%-100%. In this setup telemetry explains *why* MNS helps, but it
|
||||
does not provide earlier or more reliable action selection than direct prefix
|
||||
outcomes.
|
||||
|
||||
## Research conclusion
|
||||
|
||||
The 5/10-second negative result was indeed too narrow. It only ruled out an
|
||||
ultra-early telemetry verifier; it did not rule out engine-state information.
|
||||
The 300-second pilot finds a clear and reproducible queueing-versus-padding
|
||||
response.
|
||||
|
||||
However, this does not rescue the direct telemetry-guided tuning claim. For
|
||||
this action and workload, the external signal is already as good or better
|
||||
before the telemetry mechanism becomes stable. The project should therefore
|
||||
not claim that engine instrumentation is necessary for tuning on this evidence,
|
||||
and should not open an E2E policy test from this pilot.
|
||||
|
||||
The narrower simulator-residual route remains logically open: telemetry may
|
||||
explain why a simulator misranks real configurations even when direct online
|
||||
outcomes can guide a tuner. That is a different hypothesis and was not tested
|
||||
here.
|
||||
|
||||
## Change and verification
|
||||
|
||||
Change: absolute 5/10-second prefixes were replaced by phase-aware 30/75/150/
|
||||
225/300-second analysis; SLO early stop was disabled; full Layer-1 coverage,
|
||||
hash, request-accounting, controller, and stream/footer gates were added. The
|
||||
mechanism gate was corrected to require two features at the same adjacent
|
||||
phase pair.
|
||||
|
||||
Expected effect: distinguish “telemetry has not developed yet” from “telemetry
|
||||
does not identify or improve the action decision.”
|
||||
|
||||
Verification: five local analysis/controller test suites passed; remote
|
||||
manifest preflight and command dry-run passed; six serialized sessions passed
|
||||
all stream invariants; the analyzer was rerun and produced byte-identical
|
||||
output.
|
||||
|
||||
Result: mechanism evidence passed; incremental tuning evidence failed. Audit
|
||||
SHA256: `45f6f248712f9cbd3ed72036837ff6dc5b5c14c0f2eb6ba5cd5daceb1aa4ddb7`.
|
||||
|
||||
Remaining risk: this is a development pilot with one model, one TP, one action,
|
||||
two retained loads, three request bands, and six action labels. It is adequate
|
||||
to reject opening the next direct-policy stage, not to establish a universal
|
||||
negative claim about telemetry.
|
||||
|
||||
## Research-validity audit
|
||||
|
||||
| Check | Verdict | Evidence / boundary |
|
||||
|---|---|---|
|
||||
| Real system and E2E outcome | PASS | Real H20/vLLM replay; full SLO outcome accompanies mechanism telemetry. |
|
||||
| Matched action baseline | PASS | Exact request/arrival/length hashes for MNS 16 and 64; external prefix outcome is the decision baseline. |
|
||||
| Repeats and order effects | PASS for pilot | Three disjoint bands; A/B, B/A, A/B endpoint order; counter-rotated load order. |
|
||||
| Selective load removal | PASS with narrowed claim | The 3.125 load produced no result before any action comparison; the failure and cost are retained, and all kept points were freshly rerun. |
|
||||
| Significance/generalization | NEEDS EVIDENCE for a paper claim | Only three bands, one model, one TP, one action, and six labels. This is explicitly a stage gate. |
|
||||
| Calibration versus evaluation | NEEDS EVIDENCE for a positive policy claim | Frozen gates and leave-one-band-out folds reduce leakage, but a new workload/model hold-out is still required. |
|
||||
| Platform/reproducibility | PASS | Commit, commands, manifest, controller state, platform fingerprint, raw remote paths, and audit hashes are recorded. |
|
||||
|
||||
## Data sanity
|
||||
|
||||
- Measured runs: n=12; elapsed 300.346-317.012 seconds; 12 distinct; pass rate
|
||||
0.2635-1.0 with 4 distinct values; selected requests 1800-2550 with 2
|
||||
distinct values.
|
||||
- Sessions: n=6; 0.8413-0.8631 H20-hours; 6 distinct; Layer-1 records
|
||||
58,465-64,776; 6 distinct. V3 cost was 5.0924 H20-hours; total including
|
||||
the invalid attempt was 6.4505, below the 8.0 cap.
|
||||
- Labels: n=6; min/max 0/1; 2 distinct. Action pairs were 6 and repeat pairs
|
||||
were 8 at every checkpoint.
|
||||
- Coverage-gap observations: n=60; start gaps 0.0427-0.1247 seconds; end gaps
|
||||
0.00014-0.1705; maximum internal gaps 0.1695-0.6528, all below one second.
|
||||
- Checked invariants: exact pair hashes and counts, all runs uncensored, full
|
||||
request accounting, monotonic admitted/completed coverage, monotonic Layer-1
|
||||
timestamps, nonnegative counters, bounded ratios, non-identical per-config
|
||||
states, contiguous step indices, zero drops, footer/sidecar agreement, no
|
||||
controller failures, all sessions complete, GPU idle after completion. No
|
||||
red flags were found.
|
||||
73
docs/intervention-response-v3-two-load-protocol-20260714.md
Normal file
73
docs/intervention-response-v3-two-load-protocol-20260714.md
Normal file
@@ -0,0 +1,73 @@
|
||||
# Phase-aware telemetry intervention-response v3 protocol
|
||||
|
||||
Status: **FROZEN AFTER A NON-COMPARATIVE OPERATIONAL FAILURE AND BEFORE V3 RUNS**.
|
||||
|
||||
Date: 2026-07-14 (Asia/Singapore).
|
||||
|
||||
## Why v2 was invalid
|
||||
|
||||
The first v2 session completed the 300-second arrival windows at 1.5 and 2.125
|
||||
requests/s/GPU. At 3.125 requests/s/GPU, MNS=16 could not drain the admitted
|
||||
requests before the 450-second client timeout. The session produced no result
|
||||
and no MNS=64 action endpoint was run. V2 is therefore an invalid operational
|
||||
attempt, not evidence for or against the telemetry hypothesis.
|
||||
|
||||
This failure was observed before any MNS action comparison. V3 excludes only
|
||||
the unmeasurable overload point and reruns every retained point on fresh
|
||||
servers; it does not reuse the completed v2 low/mid results.
|
||||
|
||||
## Question and hypothesis
|
||||
|
||||
Question: after enough replay time for queue, batch, and KV state to develop,
|
||||
does an MNS intervention create telemetry responses that exceed workload-repeat
|
||||
noise, and does any such response predict whether the intervention repairs the
|
||||
full-run SLO outcome better than external prefix outcomes alone?
|
||||
|
||||
Hypothesis: increasing MNS from 16 to 64 has little value at the 1.5
|
||||
requests/s/GPU control load but can repair the 2.125 requests/s/GPU pressure
|
||||
load. Queue, running-set, batch, or KV telemetry should expose the difference
|
||||
at stable replay phases. Label balance is an assumption to test, not a fact.
|
||||
|
||||
## Frozen setup
|
||||
|
||||
- Solo placement on dash0 GPU 0-3: 4 NVIDIA H20 GPUs, Qwen3-30B-A3B, patched
|
||||
vLLM `0.24.1.dev3+opprof`, fixed TP=4.
|
||||
- Action: MNS `16 -> 64`; all other engine and workload parameters fixed.
|
||||
- Workload: `chat_w20260312_1000`, replay-time scale 0.5, hence 300 seconds.
|
||||
- Loads per GPU: 1.5 control and 2.125 pressure requests/s. The failed 3.125
|
||||
overload point is excluded from V3 and retained only as a failure artifact.
|
||||
- Three disjoint request bands. Each MNS action pair has exact request,
|
||||
arrival, and input-length hashes. Endpoint order is A/B, B/A, A/B; load
|
||||
order is low/mid, mid/low, low/mid.
|
||||
- Every session starts a fresh server, then runs the accepted 16-request long
|
||||
warm-up and bounded burn-in before measured runs.
|
||||
- SLO-unrecoverable early stop is disabled. Measured results must cover the
|
||||
full 300-second arrival window and must not be early-stopped.
|
||||
- Cumulative checkpoints are 10%, 25%, 50%, 75%, and 100%; non-overlapping
|
||||
quarter blocks are also reported.
|
||||
- A Layer-1 interval is complete only if timestamps are monotonic and its
|
||||
start, end, and maximum internal record gaps are each at most one second.
|
||||
- Incremental V3 cap is the unused portion of the original 8 H20-hour cap.
|
||||
The exact prior cost and V3 cap are machine-recorded in the manifest.
|
||||
|
||||
## Frozen gates
|
||||
|
||||
Data validity requires six uncensored sessions, six action pairs, eight
|
||||
same-config repeat pairs, exact action-pair hashes, full request accounting,
|
||||
zero Layer-1 drops, continuous coverage, all stream/footer invariants, no
|
||||
co-resident compute process, idle GPUs before and after sessions, nonnegative
|
||||
counters, bounded ratios, non-identical per-config state, and monotonic request
|
||||
coverage across checkpoints. Any red flag stops analysis.
|
||||
|
||||
Mechanism evidence requires at least two telemetry features to exceed the
|
||||
unchanged v1 repeat-noise thresholds at the same pair of adjacent checkpoints.
|
||||
Those features must have a direction consistent in both retained load regimes.
|
||||
|
||||
Decision evidence additionally requires at least two positive and two negative
|
||||
full-run action-efficacy labels, valid leave-one-repetition-out folds, and at
|
||||
least one phase-stable mechanism feature whose balanced accuracy is at least
|
||||
0.75 and at least 0.15 above the best external prefix-outcome feature at two
|
||||
adjacent predeclared checkpoints from 25% onward.
|
||||
|
||||
V3 remains a development mechanism pilot. Even `OPEN_E2E_POLICY_TEST` opens a
|
||||
held-out tuning-policy experiment; it is not itself a paper performance claim.
|
||||
75
docs/simulator-claim-map-20260716.md
Normal file
75
docs/simulator-claim-map-20260716.md
Normal file
@@ -0,0 +1,75 @@
|
||||
# Simulator-for-config-tuning related-work claim map
|
||||
|
||||
日期:2026-07-16。目的:为「Frontier/Vidur-class simulator 能否低成本解决 config tuning」这条主线建立 related-work 边界。Vidur 与 LLMServingSim 的条目基于原文(PDF 全文核读);SimAI 基于论文页与摘要口径。每项按 Context / Claim / Assumption / Mechanism / Evidence / Boundary / 与本 project 的关系提取。
|
||||
|
||||
## Vidur(MLSys 2024,arXiv:2405.05465)
|
||||
|
||||
| 维度 | 内容 |
|
||||
|---|---|
|
||||
| Context | MSR India。首个面向 LLM inference 的大规模模拟器。Motivation 与我们一致:config search 复杂度 O(\|M\|·\|T\|),且 optimal config 是 (model, trace) 的函数——Fig 1b 显示跨 trace misconfiguration 代价最高 2×。 |
|
||||
| Claim | (a) request-level 预测误差 <9%:static trace 上 P95 normalized execution latency 误差 ≤3.33%(4 模型 × 3 trace),dynamic trace 在 **85% capacity** 负载下误差 <5%。(b) Vidur-Search 用约 1 小时 96-core CPU($9.93/h)为 LLaMA2-70B 找到最优 config,对比 deployment-based exploration 估算 42K GPU-hours ≈ $218K。(c) what-if 全量探索 $125 模拟成本 vs 估算 $1.14M 真机成本。 |
|
||||
| Assumption | operator runtime 可由单 GPU profiling + 小型 ML 估计器(random forest)插值;prefill attention 可用等效单序列 sqrt(Σp_i²) 近似;decode attention runtime 只依赖总 KV 读量而非 per-request context 分布;LLM 架构同质(小算子集合跨模型共享)。 |
|
||||
| Mechanism | 声明式 model spec → 算子三分类(token-level / sequence-level / communication)→ 单 GPU CUPTI profiling → RF runtime estimator → event-driven simulator + 三层 hierarchical scheduler(支持 vLLM/Orca+/Sarathi-Serve/FasterTransformer/LightLLM 策略)→ Vidur-Search 对每个 config 二分搜索 max QPS(判据 P99 scheduling delay <5s),目标 QPS/dollar。 |
|
||||
| Evidence | LLaMA2-7B/70B、InternLM-20B、Qwen-72B(全 dense);Azure A100/H100 4-GPU pairwise-NVLink 节点;Chat-1M / Arxiv-4K / BWB-4K trace,总长截断到 4096 tokens。 |
|
||||
| Boundary | **作者明示**:接近 capacity point 时小误差会因排队失控放大,故 fidelity 评测停在 85% capacity。**结构性**:无 MoE、无 FP8/量化、无 prefix-cache reuse(多轮对话按独立请求处理)、无 speculative decoding(列为 future work)、PP 仅同步、长上下文未覆盖(4K 截断)。metric 口径为 normalized execution latency(static 排除 scheduling delay)。**最关键**:sim 选出的 config 在真机 ground-truth surface 上的 selection regret 从未被验证;42K GPU-h/$218K 是反事实估算,分母是穷举式 exploration 而非 strong sequential tuner。 |
|
||||
| 与本 project 的关系 | Frontier 即 Vidur-class(代码直接使用 vidur backend)+ 我们的 FP8/MoE/EP/decode-profile patches。我们的所有实验恰好工作在 Vidur 声明误差爆炸并回避的 regime(capacity point + SLO gate),补的正是它缺的 selection-regret ground truth。我们的 zero-shot 失败(25–30% regret)与其 <9% 不矛盾——不同 metric、不同 load regime、不同 stack alignment,论文必须主动写明这一点。其 Fig 1b 的 workload-conditioned 结论与我们 P4 sign-flip、P6 churn 互为独立佐证 → 支持 retune 频率 / amortization 论证(C3)。 |
|
||||
|
||||
## LLMServingSim(IISWC 2024,arXiv:2408.05499)
|
||||
|
||||
| 维度 | 内容 |
|
||||
|---|---|
|
||||
| Context | KAIST。scale-out LLM serving 的 HW/SW co-simulation,面向 NPU/PIM/异构加速器设计探索,基于 ASTRA-sim。 |
|
||||
| Claim | 对真实 multi-GPU vLLM serving 平均误差 14.7% 且「趋势一致」;比 mNPUsim/GeneSys/NeuPIMs 快 34.7–491×(摘要口径 91.5×)。 |
|
||||
| Assumption | iteration-level 模拟 + decoder-block 冗余复用(编译一个 block 复制展开、attention/非 attention 分离)可在可行时间内保持足够精度;硬件行为可由可插拔 accelerator compiler+simulator 栈表达(GeneSys 原型)。 |
|
||||
| Mechanism | 逐 iteration:scheduler(iteration-level batching、KV paging、operator mapping)→ per-device 硬件模拟 → graph converter(Chakra)→ ASTRA-sim 网络级模拟 → 循环。 |
|
||||
| Evidence | 与 multi-GPU vLLM 真机对照,变量为 LLM 架构、并行方案、NPU 数量、异构度;报告平均误差与趋势一致性。 |
|
||||
| Boundary | 定位是硬件/系统设计空间探索,不是 engine-knob config tuning;validation 口径是 trend-following,无 SLO-gated capacity、无 selection-regret;14.7% 平均误差大于典型 config 间 capacity margin(我们 12-cell 面上 top-2 差距 0.76%),故该精度不足以支撑近邻 config 选择。 |
|
||||
| 与本 project 的关系 | 说明「模拟保 trend」是社区通行 validation 标准;「trend ≠ selection」这一缺口对它同样成立。不构成直接 baseline,但在 related work 中界定我们评测口径(selection regret at capacity point)的必要性。 |
|
||||
|
||||
## SimAI(NSDI 2025,Alibaba,aliyun/SimAI)
|
||||
|
||||
| 维度 | 内容 |
|
||||
|---|---|
|
||||
| Context | 大规模 LLM **training** 的架构设计与参数调优模拟,生产背景(Alibaba Cloud)。 |
|
||||
| Claim | 各测试场景平均 98.1% 与真实结果对齐;为 host 设计与参数设置提供生产可用 guidance。 |
|
||||
| Assumption | training 过程可由 framework + kernel computation + collective communication 的选择性高保真集成复现。 |
|
||||
| Mechanism | 高保真集成三层栈 + 多线程加速 + lock-free global context sharing。 |
|
||||
| Evidence | 与生产 training 场景对齐(论文口径,未逐一核读实验细节)。 |
|
||||
| Boundary | training-only。training iteration 均匀、batch 组成静态——恰是 Vidur 指出 inference 所缺的性质,因此 98.1% 不可外推到 serving capacity point。 |
|
||||
| 与本 project 的关系 | 「simulator 指导 infra 决策」的工业先例与动机背书;不与 serving config tuning claim 竞争。引用价值在 motivation,不在 evaluation 对照。 |
|
||||
|
||||
## Frontier(本 project 被测对象,非 related work)
|
||||
|
||||
内部 Vidur-class 实现(vidur backend)+ 本 project 的 FP8/MoE tuning-key、Qwen MoE serving plan、TP/EP-aware cache key、critical-lane、decode/true-mixed profile 补丁。我们全部 fidelity 结论限定于该实现与已声明的 patch 集;见 `simulator-fidelity.md`。
|
||||
|
||||
## Consensus / disagreement / uncovered regime
|
||||
|
||||
**Consensus(三方一致或与我们互证):**
|
||||
|
||||
1. operator/iteration profile + 调度复合的模拟器,在中低负载下能达到 5–15% latency 误差;模拟成本比真机低数个数量级。
|
||||
2. optimal config 是 (model, workload) 的函数,misconfiguration 代价可达 ~2×(Vidur Fig 1b;我们 P4 pattern sign-flip 与 P6 engine-churn 独立复证)。
|
||||
|
||||
**Disagreement:** 无直接冲突数字。我们的 zero-shot 失败与 Vidur <9% 处于不同 metric/regime,论文需主动解释,防止被误读为矛盾或重复。
|
||||
|
||||
**Uncovered regime(本 project 的空间):**
|
||||
|
||||
1. **capacity-point + SLO-gated selection regret 无人用真机 ground-truth surface 验证。** Vidur 自认该 regime 误差爆炸并把评测停在 85% load;LLMServingSim 只验 trend。而 config tuning 的决策恰好发生在 capacity point。
|
||||
2. MoE、FP8、prefix reuse、speculative decoding、EP topology、长上下文均在已发表 fidelity envelope 之外。
|
||||
3. **alignment/profiling 成本从不与真机 tuning 成本同表比较。** Vidur 的 $218K 对比用穷举做分母;正确分母是 strong sequential tuner(我们实测 0.27–0.45 H20h/task,`runs/tuning-cost/metrics.json`)。
|
||||
4. envelope 失效的低成本检测(workload/runtime/topology 变化后何时还能信 simulator)无人提出。
|
||||
|
||||
## 对本 project claim 的直接影响
|
||||
|
||||
- **C1 定位句**:不是「Vidur 错了」,而是「Vidur-class claim 停在 sub-capacity load 的 prediction fidelity;把它外推到 SLO-gated capacity selection 是社区的隐含用法,我们证明该外推在 zero-shot 下失败(25–30% regret),并给出恢复 ranking 所需的最小真机证据层级」。
|
||||
- **C2**:Vidur 没有 minimum-real-evidence 的概念(要么全模拟、要么全真机);per-TP calibration / 同栈 profile + KV capacity + patches 的证据层级是新贡献面。
|
||||
- **C3**:省钱叙事必须从「数量级」修正为「仅在 amortization 下成立」,分母换成 strong tuner 实测值;Vidur Fig 1b + 我们 P6 churn 共同支撑 retune 频率前提。
|
||||
|
||||
## 待 triage 的相邻工作(未读原文,暂不写 claim)
|
||||
|
||||
APEX(arXiv:2411.17651,并行执行计划模拟)、LLMServingSim 2.0(arXiv:2602.23036,异构+分离式)、Charon(arXiv:2605.17164,training+inference 统一)、inference-fleet-sim(arXiv:2603.16054,排队论容量规划)、AgentServeSim(arXiv:2606.09613,多轮 agent serving)。若审稿风险评估需要,按本表格式各补一行。
|
||||
|
||||
## Sources
|
||||
|
||||
- Vidur: <https://arxiv.org/abs/2405.05465>(全文核读版本:mlsys24 PDF)
|
||||
- LLMServingSim: <https://arxiv.org/pdf/2408.05499>
|
||||
- SimAI: <https://www.usenix.org/conference/nsdi25/presentation/wang-xizheng-simai>、<https://github.com/aliyun/SimAI>
|
||||
14
docs/simulator-tuning/README.md
Normal file
14
docs/simulator-tuning/README.md
Normal file
@@ -0,0 +1,14 @@
|
||||
# Simulator tuning evaluation
|
||||
|
||||
This directory contains decision-level summaries for experiments that compare
|
||||
a serving simulator's selected configuration with the best configuration on
|
||||
real hardware.
|
||||
|
||||
Current report:
|
||||
|
||||
- [Frontier selection regret on Qwen3-30B and Qwen3-235B](frontier-selection-regret-qwen30-qwen235-20260719.md)
|
||||
|
||||
The primary quantity is **real-hardware selection regret**, not simulator
|
||||
absolute-latency error. Raw commands, profiles, traces, and experiment-specific
|
||||
audit records remain under `runs/` or in the immutable remote artifact roots
|
||||
listed by each report.
|
||||
@@ -0,0 +1,75 @@
|
||||
# Frontier selection regret: Qwen3-30B and Qwen3-235B
|
||||
|
||||
> Date: 2026-07-19
|
||||
> Scope: H20, community vLLM 0.20, Frontier piecewise simulation, no SLO gate
|
||||
|
||||
## Question and metric
|
||||
|
||||
For each workload and latency objective, Frontier selects the configuration
|
||||
with the lowest simulated latency. We then look up that configuration on the
|
||||
complete real-hardware surface and compare it with the real-hardware optimum.
|
||||
|
||||
```text
|
||||
selection regret = real_latency(Frontier winner) / real_latency(real winner) - 1
|
||||
```
|
||||
|
||||
Lower is better. `0%` means Frontier selected the real winner. Positive values
|
||||
mean that following Frontier produces slower real serving. Each objective is
|
||||
selected independently; this table does not combine TTFT, TPOT, and E2E into a
|
||||
single score.
|
||||
|
||||
## Qwen3-30B-A3B
|
||||
|
||||
Configuration surface: `TP in {1,2,4} x MNS in {8,16,32,64}`, with
|
||||
`MBT=8192`. Each real cell uses three fresh-server trials.
|
||||
|
||||
| Workload | TTFT mean | TTFT p90 | TPOT mean | TPOT p90 | E2E mean | E2E p90 |
|
||||
|---|---:|---:|---:|---:|---:|---:|
|
||||
| Trace-PD | 0.0% | 0.0% | 0.0% | 0.0% | 0.0% | 0.0% |
|
||||
| Fixed-PD, 4096->256, 1.125 req/s/GPU | **58.0%** | **56.2%** | 0.0% | 0.0% | 1.7% | 5.5% |
|
||||
| Trace-PO, OSL=1 | 3.2% | 0.4% | N/A | N/A | 3.2% | 0.3% |
|
||||
| Fixed-PO, 4096->1, 1.125 req/s/GPU | 0.3% | 0.5% | N/A | N/A | 0.3% | 0.5% |
|
||||
|
||||
Interpretation: Frontier is near-optimal for Trace-PD and both prefill-only
|
||||
cases, but the high-pressure Fixed-PD TTFT choice is materially wrong: its
|
||||
selected configuration is 56--58% slower than the real TTFT optimum.
|
||||
|
||||
## Qwen3-235B-A22B-FP8
|
||||
|
||||
Configuration surface: `{TP4/EP1, TP8/EP8} x MNS in {64,128}`, with
|
||||
`MBT=8192`. Each workload has 129 requests per cell and each real cell uses
|
||||
three fresh-server trials.
|
||||
|
||||
| Workload | TTFT mean | TTFT p90 | TPOT mean | TPOT p90 | E2E mean | E2E p90 |
|
||||
|---|---:|---:|---:|---:|---:|---:|
|
||||
| Trace-PD | 0.0% | 0.0% | 0.0% | 0.0% | 0.6% | 6.2% |
|
||||
| Fixed-PD, 4096->256, 0.2 req/s/GPU | 4.2% | 0.2% | **33.0%** | **37.2%** | **30.7%** | **34.6%** |
|
||||
| Trace-PO, OSL=1 | 7.0% | **21.2%** | N/A | N/A | 7.0% | **21.2%** |
|
||||
| Fixed-PO, 4096->1, 0.2 req/s/GPU | 5.9% | 1.7% | N/A | N/A | 5.9% | 1.7% |
|
||||
|
||||
Interpretation: Trace-PD is mostly near-optimal. Fixed-PD reverses the real
|
||||
decode/E2E preference between the tested parallel configurations and incurs
|
||||
31--37% regret. Trace-PO also has a material p90 failure of 21.2%.
|
||||
|
||||
## Decision
|
||||
|
||||
The tested Frontier stack has **not** solved serving configuration tuning.
|
||||
Its selected configuration can be near-optimal for one workload and materially
|
||||
wrong for another on the same model and hardware. The strongest current
|
||||
counterexamples are Qwen3-30B Fixed-PD TTFT and Qwen3-235B Fixed-PD TPOT/E2E.
|
||||
|
||||
This statement is limited to the two tested MoE models and Frontier. It is not
|
||||
yet evidence about dense models, Vidur/APEX as separately reproduced systems,
|
||||
other hardware, or SLO-constrained tuning.
|
||||
|
||||
## Provenance
|
||||
|
||||
Primary immutable analysis artifacts on `dash0`:
|
||||
|
||||
- Qwen3-30B Trace-PD: `/home/admin/cpfs/wjh/aituner/graph-piecewise-qwen30-20260717/simulator-piecewise-surface-v2/analysis/comparison.json`
|
||||
- Qwen3-30B Fixed-PD/PO: `/home/admin/cpfs/wjh/aituner/qwen30-fixed-pressure-surface-20260719-r1/analysis/`
|
||||
- Qwen3-30B Trace-PO: `/home/admin/cpfs/wjh/aituner/qwen30-latency-expansion-20260718-r2/analysis-r6/trace-po-comparison.json`
|
||||
- Qwen3-235B four-case matrix: `/home/admin/cpfs/wjh/aituner/qwen235-v020-fourcase-20260719-r1/analysis/comparison.json`
|
||||
|
||||
The Qwen3-235B artifact root includes `provenance/artifacts.sha256`; the final
|
||||
matrix contains 48/48 valid real trials and 16/16 complete simulator cells.
|
||||
286
docs/telemetry-residual-tuning-roadmap-20260714.md
Normal file
286
docs/telemetry-residual-tuning-roadmap-20260714.md
Normal file
@@ -0,0 +1,286 @@
|
||||
# Telemetry-conditioned residual tuning roadmap
|
||||
|
||||
Status: **R0 COMPLETE / FAILED; R1 AND R2 CLOSED FOR THIS MODEL**.
|
||||
|
||||
Date: 2026-07-14 (Asia/Singapore).
|
||||
|
||||
## Research question and claim boundary
|
||||
|
||||
The question is whether a small number of real engine observations can correct
|
||||
a simulator's task-specific error over **unmeasured configurations**, and
|
||||
whether that correction reduces the real-GPU cost of finding a high
|
||||
SLO-goodput serving configuration.
|
||||
|
||||
The intended headline claim, if the evidence supports it, is:
|
||||
|
||||
> An engine-state-conditioned residual model turns a simulator prediction into
|
||||
> a task-specific posterior over unmeasured serving configurations, allowing a
|
||||
> sequential tuner to reach near-oracle SLO-goodput with materially fewer
|
||||
> H20-hours than simulator-only and outcome-only tuning.
|
||||
|
||||
Classification accuracy, simulator-error diagnosis, and telemetry overhead are
|
||||
supporting evidence. None is an end-to-end tuning contribution by itself.
|
||||
|
||||
The following method is closed and will not be revived under another name:
|
||||
per-candidate five-second accept/reject as the headline contribution. The P1
|
||||
result showed only 1.426% cost reduction in the frozen `k=2` workflow.
|
||||
|
||||
## Two models, one evaluation
|
||||
|
||||
Both branches use the same legal candidate set, real measurements, task split,
|
||||
cost accounting, and acquisition function.
|
||||
|
||||
### Simulator-residual branch (primary)
|
||||
|
||||
For measured anchor `c_t` and unmeasured candidate `c'`:
|
||||
|
||||
```text
|
||||
y_hat(c') = y_real(c_t)
|
||||
+ [y_sim(c') - y_sim(c_t)]
|
||||
+ f(state_real(c_t) - state_sim(c_t), c' - c_t, workload, SLO)
|
||||
```
|
||||
|
||||
The simulator delta is the prior. The learned model may correct it only with
|
||||
training-supported state/config transitions; uncertainty or distribution shift
|
||||
must shrink the correction back toward the simulator prior.
|
||||
|
||||
### Telemetry-only branch (mandatory)
|
||||
|
||||
```text
|
||||
y_hat(c') = y_real(c_t)
|
||||
+ g(state_real(c_t), c' - c_t, workload, SLO)
|
||||
```
|
||||
|
||||
This branch tests whether the simulator is actually necessary. It does not
|
||||
use a hand-authored bottleneck-to-knob rule.
|
||||
|
||||
### Search policy
|
||||
|
||||
Legal configurations are enumerated independently of telemetry. A generic
|
||||
cost-aware acquisition rule ranks candidates from predicted improvement,
|
||||
uncertainty, and measured H20 cost. The current production harness's
|
||||
bottleneck scores, topology-first ordering, and hand-set relief constants are
|
||||
not consumed by either branch. The validator may enforce legality,
|
||||
full-config no-repeat, failure accounting, and resource caps only.
|
||||
|
||||
## Hypotheses
|
||||
|
||||
| ID | Hypothesis | Direct test | Failure meaning |
|
||||
|---|---|---|---|
|
||||
| H0 | Existing artifacts can express a common, direct-measurement state without heuristic labels. | Engine/simulator extractor coverage and invariants. | Route is not currently implementable. |
|
||||
| H1 | Simulator errors are predictable from engine/simulator state discrepancy at measured anchors. | Task-held-out pairwise inversion correction and new-inversion rate. | Telemetry is diagnostic but cannot correct the surface. |
|
||||
| H2 | Telemetry alone predicts useful config transitions beyond outcome-only history. | Telemetry-only versus real-outcome-only sequential replay. | Direct telemetry-guided tuning has no independent value. |
|
||||
| H3 | Residual correction changes actual tuning decisions and cost. | H20-hours to 95% oracle and regret AUC against the strongest safe baseline. | No system contribution even if H1/H2 prediction metrics improve. |
|
||||
|
||||
## Common-state contract
|
||||
|
||||
Only directly observed or exactly reconstructed quantities are admitted.
|
||||
|
||||
| Quantity | vLLM Layer-1 | Frontier | R0 status |
|
||||
|---|---|---|---|
|
||||
| Scheduled requests / batch size | Per scheduler step | Existing per-batch metric, disabled in P1 output | Common after CPU replay |
|
||||
| Scheduled prefill/decode tokens | Per scheduler step | Existing per-batch metrics | Common after CPU replay |
|
||||
| Scheduler/batch rate | Monotonic step timestamps | Batch count / simulated duration | Common after CPU replay |
|
||||
| Waiting queue area | Time-weighted queue gauge | Sum of request waiting times | Common aggregate |
|
||||
| Running request area | Time-weighted running gauge | Sum of E2E minus waiting time | Common aggregate, semantics audited |
|
||||
| Preemption count | Per step | Per request | Common |
|
||||
| KV usage/headroom | Exact blocks and ratio | Not in committed output | Engine-only until exact reconstruction exists |
|
||||
| CUDA graph mode/padding | Exact per step | Not modeled | Engine-only omitted-mechanism signal |
|
||||
| Request TTFT/TPOT/pass rate | Exact real outcomes | Exact simulated request metrics | Common outcome, not state |
|
||||
|
||||
Unavailable fields remain null. They cannot be imputed from a human
|
||||
`prefill/decode/queueing` label.
|
||||
|
||||
Frontier already contains the required detailed batch and timestamped
|
||||
stage-batch ledger output. P1 disabled it for artifact size. R0 replays the
|
||||
same immutable fixtures with the existing output flags enabled; it does not
|
||||
change the simulator model or calibration.
|
||||
|
||||
## Data separation
|
||||
|
||||
- Phase 6 / `chat_w20260311_1000`: development only.
|
||||
- P1 / `chat_w20260312_1000`: development only.
|
||||
- R1 / `chat_w20260313_1000`: new development surface.
|
||||
- R2: trace windows not used for feature, model, threshold, candidate-space,
|
||||
cutoff, or acquisition decisions.
|
||||
- Splits are by complete workload/SLO task. Anchor- or pair-level random
|
||||
splits are prohibited.
|
||||
- Sequential-policy seeds measure algorithmic variability; they are not
|
||||
counted as independent system tasks.
|
||||
|
||||
The two existing development tasks have an important limitation: the now-
|
||||
available SLO-gated simulator reading already retains the real oracle at its
|
||||
top rank/tie. They therefore cannot establish a positive end-to-end ranking
|
||||
claim. They are used for plumbing, known false-feasible cases, and negative
|
||||
evidence. R1 must be run as an unbiased complete surface, not selected after
|
||||
observing simulator success or failure.
|
||||
|
||||
## Step-by-step roadmap
|
||||
|
||||
### R0.1 — Inventory and roadmap
|
||||
|
||||
Deliverables:
|
||||
|
||||
- this roadmap;
|
||||
- rolling untracked `ONGOING.md`;
|
||||
- exact engine/simulator field and artifact inventory.
|
||||
|
||||
Gate: every claimed input has an authoritative file path and provenance.
|
||||
|
||||
### R0.2 — Common-state plumbing
|
||||
|
||||
Deliverables:
|
||||
|
||||
- `runs/telemetry-residual/common_state.py`;
|
||||
- synthetic correctness tests;
|
||||
- one exact P1 Frontier replay with individual batch metrics and the full
|
||||
stage-batch ledger enabled;
|
||||
- paired engine/simulator state summary for the same fixture.
|
||||
|
||||
Gate:
|
||||
|
||||
- replay request count and SLO scorer exactly agree with the committed replay;
|
||||
- batch/ledger outputs are non-empty;
|
||||
- all counters are non-negative, ratios bounded, times monotonic;
|
||||
- no GPU is visible to Frontier;
|
||||
- output volume is practical before expanding to twelve replays.
|
||||
|
||||
### R0.3 — Development residual/headroom audit
|
||||
|
||||
Use all frozen P1 primary fixtures and corresponding engine intervals. Produce:
|
||||
|
||||
- common-state residuals per anchor;
|
||||
- simulator-error labels and continuous SLO/goodput residuals;
|
||||
- ordered source/target diagnostic that removes both config identities from
|
||||
both roles in every training fold;
|
||||
- oracle upper bound for cross-candidate correction;
|
||||
- explicit comparison with simulator+outcome and telemetry-only features.
|
||||
|
||||
R0 is a feasibility gate, not headline evidence. Proceed to R1 only if:
|
||||
|
||||
1. state features are collected with the measured source anchor, vary across
|
||||
cells, and are available before any target config is evaluated;
|
||||
2. at least one known simulator error has a state discrepancy not exposed by
|
||||
the matched external prefix outcome;
|
||||
3. a prior-preserving model can correct development errors without introducing
|
||||
a larger number of new errors under regularization sensitivity;
|
||||
4. an oracle cross-candidate correction has at least 15% sequential tuning-cost
|
||||
headroom under full startup/warm-up accounting.
|
||||
|
||||
### R0 result and decision
|
||||
|
||||
R0 completed without a data-validity red flag, but failed condition 3. The
|
||||
decision is **STOP_BEFORE_R1**; no H20 job was launched for this route.
|
||||
|
||||
- All 12 detailed Frontier CPU replays exactly reproduced their committed SLO
|
||||
scorers. Runtime was 23.943--54.786 seconds per replay, detailed artifacts
|
||||
were 4.12--13.53 MB, CUDA visibility was empty, and there were zero failures.
|
||||
- The paired surface contains 12 real/sim anchors, two known simulator
|
||||
false-feasible anchors, and 120 legal cross-config ordered transitions. A
|
||||
fold removes both the source and target TP/MNS identity from source and
|
||||
target roles; the two offered-load anchors remain part of the same task.
|
||||
- Raw Frontier feasibility is 83.33% on the repeated transition view. The
|
||||
structurally correct hybrid model uses
|
||||
`r_target = r_source + delta_r`; the direct model uses
|
||||
`y_target = y_source + delta_y` and never reads simulator fields.
|
||||
- Direct telemetry is not robust relative to real-outcome-only: its accuracy
|
||||
delta over L2 `{0.1,1,10,100}` is `{-0.83,+1.67,0,-4.17}` percentage points,
|
||||
and its best absolute accuracy is 54.17%, below the raw simulator's 83.33%.
|
||||
- Hybrid telemetry raises classification accuracy over the corresponding
|
||||
simulator+outcome transition regression by 1.67--4.17 percentage points,
|
||||
but worsens pass-rate RMSE by 0.141--0.201 and MAE by 0.084--0.125. Its full
|
||||
correction reaches only 46.67--53.33% absolute accuracy.
|
||||
- Across 24 nonzero `(L2, raw-simulator-prior weight)` combinations, no model
|
||||
both corrects an existing simulator error without more new errors and avoids
|
||||
worsening RMSE/MAE. Whenever a correction fixes at least one error, it
|
||||
corrupts at least 11 previously correct transitions.
|
||||
- A perfect correction could skip the frozen simulator rank-2 real final and
|
||||
save 0.043469 H20-hours: 15.45% of the prospective online `k=2` cost, or
|
||||
14.40% when the prior failed launch is charged. On this development task the
|
||||
simulator top-1 already is the real oracle with zero regret, so headroom
|
||||
versus the observed-safe top-1 baseline is 0%.
|
||||
|
||||
The result does not prove that engine telemetry is useless. It shows that the
|
||||
current one-task anchor-transition evidence cannot support either a safe
|
||||
simulator-residual tuner or a simulator-free telemetry tuner. A larger model
|
||||
or an R1 run would add capacity/data after a failed gate and is therefore not
|
||||
authorized under this roadmap.
|
||||
|
||||
### R1 — New development surface
|
||||
|
||||
Status: **NOT LAUNCHED; CLOSED BY R0**.
|
||||
|
||||
Frozen starting setup:
|
||||
|
||||
- host: dash0, eight NVIDIA H20 GPUs;
|
||||
- cells run solo; no co-location for SLO verdicts;
|
||||
- patched vLLM 0.24.1.dev3, Qwen3-30B-A3B BF16;
|
||||
- trace: `chat_w20260313_1000`;
|
||||
- output tokens: exactly 128;
|
||||
- SLO: stepped TTFT 2/4/6 seconds, TPOT 50 ms, pass rate at least 0.95;
|
||||
- config surface: TP `{1,2,4}` × MNS `{8,16,32,64}`;
|
||||
- hard campaign cap: 4 H20-hours.
|
||||
|
||||
The load ladder, repetitions, randomized order, exact commands, expected wall
|
||||
time, and artifact paths are frozen only after R0. A resolved echo is required
|
||||
before launch.
|
||||
|
||||
R1 passes only if a frozen sequential replay shows at least 15% E2E H20-hour
|
||||
headroom over the strongest safe baseline with final regret at most 5%. R1 is
|
||||
development evidence and cannot be reported as the held-out result.
|
||||
|
||||
### R2 — Held-out sequential tuning
|
||||
|
||||
Status: **NOT LAUNCHED; CLOSED BY R0**.
|
||||
|
||||
Required baselines:
|
||||
|
||||
1. random search;
|
||||
2. real-outcome-only Bayesian/sequential search;
|
||||
3. Frontier ranking plus real top-k final;
|
||||
4. simulator plus real-outcome residual;
|
||||
5. telemetry-only transition tuner;
|
||||
6. simulator plus telemetry residual tuner;
|
||||
7. complete real surface as oracle, not as a cost competitor.
|
||||
|
||||
Primary metric: end-to-end H20-hours to first reach 95% of the real full-surface
|
||||
SLO-goodput oracle. Secondary metrics are cost-normalized regret AUC, final
|
||||
regret at fixed budgets, oracle false-prune, wall time, and per-task regressions.
|
||||
|
||||
The route is successful only if the winning telemetry method reduces the
|
||||
primary cost by at least 20% versus the strongest safe baseline and ends within
|
||||
5% regret on every headline task. If hybrid beats telemetry-only by at least
|
||||
10%, simulator residual correction is the primary method. If telemetry-only
|
||||
is within 5% or better, the simulator dependency is removed. If neither clears
|
||||
the contribution bar, the route is closed and telemetry remains a diagnostic
|
||||
facility only.
|
||||
|
||||
## Cost discipline
|
||||
|
||||
- R0 simulator work is CPU-only and must set empty CUDA visibility.
|
||||
- R1 cannot exceed 4 H20-hours.
|
||||
- R2 receives no budget until R1 passes.
|
||||
- Startup, warm-up, burn-in, failed launches, real probes, continuation, and
|
||||
final validation are charged. Benchmark-only annotation repeats are
|
||||
reported separately and cannot disappear from campaign accounting.
|
||||
|
||||
## Final R0 sanity block
|
||||
|
||||
| Data | n | Min | Max | Distinct | Checked invariant |
|
||||
|---|---:|---:|---:|---:|---|
|
||||
| Phase 6 cells | 12 | TP1/MNS8 | TP4/MNS64 | 12 | Surface not identical; solo SLO tier authoritative |
|
||||
| Phase 6 Layer-1 primary steps | 37 streams | 343 | 12,103 | 37 | Contiguous; zero drops |
|
||||
| P1 primary anchors | 12 | infeasible | feasible | 2 labels | 7 feasible / 5 infeasible |
|
||||
| P1 Frontier runtime | 12 | 24.093 s | 54.575 s | 12 | CPU-only; zero failures |
|
||||
| Detailed Frontier replay runtime | 12 | 23.943 s | 54.786 s | 12 | Exact committed scorers; CUDA hidden |
|
||||
| Detailed artifact bytes | 12 | 4,123,724 | 13,527,776 | 12 | Non-negative; practical CPU replay size |
|
||||
| Cross-config transitions | 120 | real pass 0.1067 | real pass 1.0 | 6 outcomes | Both endpoint config identities held out |
|
||||
| State residual vectors | 12 | 16 fields | 16 fields | 12 vectors | Finite; no missing common field |
|
||||
| R0 E2E cost values | 4 | 0.237914 | 0.301935 H20-h | 4 | Non-negative; `k=1/2`, online/conservative |
|
||||
|
||||
Checked invariants: non-negative counts and costs; pass rates in `[0,1]`;
|
||||
simulator results not all identical; exact request count/hash agreement; Layer-1
|
||||
step continuity and zero drops; no co-resident SLO measurements; no calibration
|
||||
or evaluation split reuse for a future headline claim. No current red flag
|
||||
invalidates R0 plumbing. The R0 tuning gate itself failed because safe
|
||||
prior-preserving correction was absent.
|
||||
391
docs/tuning-core-challenges-cost-audit-20260715.md
Normal file
391
docs/tuning-core-challenges-cost-audit-20260715.md
Normal file
@@ -0,0 +1,391 @@
|
||||
# AITuner tuning:核心挑战、统一成本口径与研究路线
|
||||
|
||||
日期:2026-07-15(Asia/Singapore)
|
||||
|
||||
状态:**问题定义与历史成本审计完成;新的 tuner 贡献尚未建立。**
|
||||
|
||||
## 结论先行
|
||||
|
||||
我们不应该把 tuning 定义成“根据当前 telemetry 判断哪个 cap 满了,再调对应 knob”。这个定义同时遗漏了 knob interaction、反事实识别、实验成本和跨任务失配。更准确的问题是:
|
||||
|
||||
> 给定模型、engine version、hardware、workload、SLO 和一个声明好的合法配置空间,tuner 如何用最少的真实 GPU 成本,依次选择可能包含多个 knob 的 intervention,找到 SLO-goodput regret 不超过 `epsilon` 的配置?
|
||||
|
||||
AITuner 可以形成的系统贡献应当是:
|
||||
|
||||
> **一个 intervention-calibrated、action-conditioned、cost-aware 的 tuner:它从真实 engine trajectory 和已测 intervention 中学习联合 config action 的反事实收益分布,并以 cost-to-oracle 而非规则命中率作为目标。Harness 只负责实验语义、合法性、配对、记账和可复现性,不负责用人工 bottleneck rule 决定 action。**
|
||||
|
||||
现有结果支持这个问题值得做,但不支持宣称它已经解决:
|
||||
|
||||
- 在真实 `TP x MNS` surface 上,one-knob-at-a-time 会停在比 oracle 低 **25.6%** 的 coordinate-wise local optimum。
|
||||
- 在 action-aware pilot 中,增加 MBBT 在“几乎从未独占打满 MBBT cap”的情况下仍把 source goodput 提高 **48.0%--77.1%**;因此 `cap -> knob` 不是完整模型。
|
||||
- 同一 dash0 任务上,当前 guided harness 到 5% empirical regret 只比纯 LLM 少 **5.85%** H20-hours;到 2% regret 则少 **61.09%**。这说明必须比较完整 cost--regret curve,不能只比较最终最好值。
|
||||
- Frontier 的 decision-bearing throughput top-1 在 12-cell surface 上有 **30.46%** real regret。Simulator 本身的边际 GPU cost 是 0,但通过 real-final 恢复 oracle 需要 tie-expanded 4 个真实 cell,即 **0.7828 reconstructed H20-hours**。
|
||||
|
||||
## 1. Tuning 问题和成功标准
|
||||
|
||||
固定 task context:
|
||||
|
||||
```text
|
||||
T = {model, engine build, hardware, workload/trace, SLO, legal config space C}
|
||||
```
|
||||
|
||||
每个完整配置 `c in C` 的目标为:
|
||||
|
||||
```text
|
||||
f_T(c) = max request_rate_per_gpu
|
||||
subject to request SLO pass rate >= target
|
||||
```
|
||||
|
||||
有限空间 oracle 为:
|
||||
|
||||
```text
|
||||
f*_T = max_{c in C} f_T(c)
|
||||
regret(c) = 1 - f_T(c) / f*_T
|
||||
```
|
||||
|
||||
顺序 tuner 在第 `t` 步基于历史 `D_t` 选择一个完整 config intervention:
|
||||
|
||||
```text
|
||||
a_t = c_t -> c_{t+1}
|
||||
```
|
||||
|
||||
成功不是“最后找到一个不错的值”,而是同时满足:
|
||||
|
||||
1. `regret(best_t) <= epsilon`;
|
||||
2. 达到该点之前的 all-in H20-hours 最小;
|
||||
3. launch、correctness、SLO 和失败率约束不退化;
|
||||
4. 结论在 held-out task 上成立,而不是在用于设计规则的 task 上成立。
|
||||
|
||||
### 1.1 GPU cost 的统一定义
|
||||
|
||||
未来实验的 task-marginal cost 应定义为:
|
||||
|
||||
```text
|
||||
C_task = sum_j allocated_GPU_count_j
|
||||
* (GPU_idle_or_release_time_j - allocation_start_time_j)
|
||||
```
|
||||
|
||||
它包括 method 实际触发的 startup、warm-up、prefix/full replay、confirmation、failure、cleanup;如果 LLM 思考期间 GPU 仍被占用,也计入。Simulator/模型的一次性 onboarding 成本单独报告:
|
||||
|
||||
```text
|
||||
C_e2e(N tasks) = C_profile_or_training / N + C_task
|
||||
```
|
||||
|
||||
另外报告 CPU-hours、LLM API latency/cost,但不把它们伪装成 GPU-hours。构建 benchmark oracle 的 exhaustive annotation cost 是公共评测成本,单独报告,不计入任何方法;同时可给一个将其等量加回所有方法的 conservative view。
|
||||
|
||||
历史记录没有 allocation start/release timestamp。本次只能从每个 `engine.log` 的首末时间戳重建:
|
||||
|
||||
```text
|
||||
C_engine_lower_bound = parallel_size * engine_log_span / 3600
|
||||
```
|
||||
|
||||
因此下面所有历史 H20-hour 数字都是 **engine-lifetime lower bound**,不是 all-in cost。尤其 simulator 的一次性 H20 operator profiling 成本没有记录,不能称为完全免费。
|
||||
|
||||
### 1.2 两种 oracle 必须分开
|
||||
|
||||
- **Exact finite-surface oracle**:声明好的 12-cell `TP x MNS` 空间全部真实测量,oracle 是 `TP2/MNS32 = 3.2833 req/s/GPU`。
|
||||
- **Broader empirical reference**:dash0 两个 sequential run 中观察到的最好值 `3.35 req/s/GPU`。它包含 surface 外的 MBBT/chunk/GMU action,但只是 best observed,不是全局 oracle。
|
||||
|
||||
不能把 empirical best 写成 global oracle,也不能让每个方法使用不同的 oracle 定义。
|
||||
|
||||
## 2. 现有方案的 cost-to-oracle 审计
|
||||
|
||||
可复算输入和完整结果在:
|
||||
|
||||
- `runs/tuning-cost/manifest.json`
|
||||
- `runs/tuning-cost/analyze.py`
|
||||
- `runs/tuning-cost/metrics.json`
|
||||
|
||||
### 2.1 严格同任务对照:纯 LLM vs 当前 guided harness
|
||||
|
||||
两组均为 dash0、Qwen3-30B-A3B、community-vLLM 0.20.0、8xH20 可见、`chat_w20260311_1000`、input 0--8k、output 128、replay scale 0.1、TTFT 2/4/6s、TPOT 50ms、pass rate 0.95。除 tuner method 和服务端口外,固定 task spec 相同。
|
||||
|
||||
Reference 是两组中 best observed `3.35 req/s/GPU`:
|
||||
|
||||
| Method | 到 <=5% regret | 到 <=2% regret | 到 <=1% regret | 完整 run 成本 | 最终 best |
|
||||
|---|---:|---:|---:|---:|---:|
|
||||
| Pure LLM, no harness | 0.2847 H20h,trial 2,regret 2.736% | 1.1458,trial 6,regret 1.493% | 1.3719,trial 7,regret 0% | 2.2825 | 3.35 |
|
||||
| Guided harness v2 | 0.2681 H20h,trial 2,regret 2.736% | 0.4458,trial 3,regret 1.990% | 未达到 | 0.6231 | 3.30,regret 1.493% |
|
||||
|
||||
直接结论:
|
||||
|
||||
- 5% endpoint:guided 比 pure LLM 少 **5.85%**,不是 material contribution。
|
||||
- 2% endpoint:guided 比 pure LLM 少 **61.09%**,有明显 headroom signal,但只有一个 task,不能外推。
|
||||
- Pure LLM 在 trial 7 已找到 best observed,之后又花了 `2.2825 - 1.3719 = 0.9106 H20h` 而没有改进,说明 trustworthy stopping 本身就是成本来源。
|
||||
- Pure LLM 的 trial 3 使用当前 binary 不支持的 `--expert-parallel-size` 并在 launch 前失败。当前 harness 的 legality/version contract 有实际价值,但它仍不是性能 action-ranking 贡献。
|
||||
|
||||
### 2.2 Simulator:零边际 GPU cost 不等于零 tuning cost
|
||||
|
||||
Frontier fidelity suite 在 CPU 上执行 184 个 simulation,耗时 **2.055 CPU-hours**,simulation 本身为 0 marginal H20-hours。其对应的 exact dash1 12-cell real surface annotation lower bound 为 **3.5953 H20-hours**。
|
||||
|
||||
Decision-bearing `frozen-calibrated/throughput-proxy`:
|
||||
|
||||
| Policy | Real cells evaluated | Real-final H20h lower bound | Selected real regret |
|
||||
|---|---:|---:|---:|
|
||||
| Simulator-only top-1 | 0 | 0 | **30.46%**,选 TP1/MNS64 |
|
||||
| Throughput top-1 + real final | 1 | 0.1353 | **30.46%** |
|
||||
| Throughput top-2 + real final | 2 | 0.2672 | **30.46%** |
|
||||
| Throughput nominal top-3 + real final | tie-expanded 4 | 0.7828 | 0%,找到 TP2/MNS32 |
|
||||
|
||||
Post-hoc `SLO-gated` reading 把 `{TP2/MNS32, TP2/MNS64}` 放在 top tie bucket;测两个 cell 需 **0.5156 H20h** 并能找到 oracle。但它不是 preregistered decision-bearing policy,而且 anchor verdict 中有 21 个 false-feasible、7 个 false-infeasible,只能作为诊断上界,不能反写成 prospective simulator 结果。
|
||||
|
||||
Pure LLM/harness 数据来自 dash0,simulator exact surface 来自 dash1。模型、engine、trace、GPU type 匹配,但 host 和 campaign 不同。因此两块内部可以直接比较,跨块只能做 development-level 指示;paper 结论必须在同 host、同 task execution protocol 下重跑。
|
||||
|
||||
### 2.3 我们要达到的成本目标
|
||||
|
||||
在当前 reconstructed lower-bound 口径下,一个有意义的单任务 development bar 是:
|
||||
|
||||
| Endpoint | 当前最强同任务 baseline | 20% reduction bar | 兼顾 post-hoc sim+real 的 30% bar | 暂定目标 |
|
||||
|---|---:|---:|---:|---:|
|
||||
| <=5% empirical regret | guided 0.2681 | 0.2144 | 0.3609 | **<=0.2144 H20h** |
|
||||
| <=2% empirical regret | guided 0.4458 | 0.3567 | 0.3609 | **<=0.3567 H20h** |
|
||||
|
||||
这两个数字不是 paper result,只用于检查 proposed method 是否有足够 headroom:
|
||||
|
||||
- 5% endpoint 已经由 baseline + TP2 两个完整 trial 达到。任何必须先跑 source 再跑 target 的 telemetry tuner 都不能靠减少 trial count 获得 20% 优势;它必须能够 one-shot warm-start、跳过 baseline,或安全地缩短其中一次测量。
|
||||
- 2% endpoint 有更合理的结构性空间:从一个 source 直接选择 joint `TP2 + MBBT/chunk` target,可能跳过当前中间 trial;如果仍按当前三次完整 trial 顺序执行,就不会达到 bar。
|
||||
|
||||
Paper-facing gate 不使用这些跨 campaign 绝对数,而使用 prospective same-host all-in cost:在每个 held-out task 上 regret <=5%,相对最强 safe outcome-only/current harness 至少省 20%,相对 frozen simulator+real 至少省 30%,并报告 task-level paired confidence interval。
|
||||
|
||||
## 3. 四个最核心的 tuning challenge
|
||||
|
||||
### Challenge 1:响应面是联合、条件化且 regime-dependent 的
|
||||
|
||||
#### 问题本质
|
||||
|
||||
一般情况下:
|
||||
|
||||
```text
|
||||
f(c) != base + sum_k effect_k(c_k)
|
||||
```
|
||||
|
||||
一个 knob 的 effect 是当前完整 context 的函数:
|
||||
|
||||
```text
|
||||
Delta_x(c, workload, engine state)
|
||||
```
|
||||
|
||||
它可能随 topology、另一个 runtime knob、load、SLO 或 engine version 改变大小甚至改变符号。因此不能先分别求每个 knob 的最优值再 merge,也不能固定一个低质量 context 去判断另一个 knob。
|
||||
|
||||
#### 已有真实证据
|
||||
|
||||
在 C1 12-cell real surface:
|
||||
|
||||
- `MNS 8 -> 32` 在 TP1/TP2/TP4 下分别提升约 **8.7% / 44.3% / 90.3%**。
|
||||
- 从同一 `TP1/MNS8` 起点,先 tune MNS 再 TP 会停在 `TP4/MNS16 = 2.4417`;该点沿任一单维都没有 strictly improving move,但 joint/global surface oracle `TP2/MNS32 = 3.2833` 高 **34.5% relative to the local point**,即 local point 对 oracle 有 **25.6% regret**。
|
||||
- C3 中 `MBT 256 -> 384` 的 effect 根据 topology/MNS 从 0 到约 -9.2%;`MNS 64 -> 128` 从 0 到约 +10.1%。
|
||||
- Action-aware Regime A 中 MBBT 几乎从不作为 exclusive cap,但 MBBT action 仍把 source goodput 提高 48.0%--77.1%。它通过 chunk size、prefill packing 和 scarce MNS slot residency 的联合变化获得收益。
|
||||
|
||||
这直接否定两类通用策略:OAT/coordinate greedy,以及 `which cap is full -> tune that knob`。
|
||||
|
||||
#### Tuner 必须具备的能力
|
||||
|
||||
- Action 的基本单位是完整 `config delta`,允许 sparse joint action,而不是孤立 knob/value。
|
||||
- 对 topology/runtime family 使用 crossed anchors 或信息增益设计,主动测 interaction;不是默认所有 interaction 都强。
|
||||
- 能从数据判断 task 是 topology-dominant、runtime-interaction-dominant 还是 flat/noisy,并据此分配实验,而不是把固定 search order 写进规则。
|
||||
|
||||
### Challenge 2:当前状态是 observational signal,tuning 需要 counterfactual identification
|
||||
|
||||
#### 问题本质
|
||||
|
||||
一次 telemetry trace 只能告诉我们:
|
||||
|
||||
```text
|
||||
P(engine trajectory | current config, workload)
|
||||
```
|
||||
|
||||
Tuning 真正需要的是:
|
||||
|
||||
```text
|
||||
P(Delta SLO-goodput, failure, cost
|
||||
| source trajectory, proposed full-config action)
|
||||
```
|
||||
|
||||
Queue、KV、padding、split prefill 等状态既可能是原因,也可能是 workload/config 的结果。看见某种状态,不等于知道哪个 action 能修复它。一个 action 也可能同时改变多条机制;例如 MBBT 同时改变总 token budget、per-request chunk 和 multi-request packing,现有 telemetry 的解释是 mechanism-consistent,不是已完成的 causal decomposition。
|
||||
|
||||
#### 已有真实证据
|
||||
|
||||
- 5/10 秒 telemetry 确实太短;300 秒 phase-aware experiment 中,MNS action 的 queue/padding 机制直到 replay 75%--100% 才稳定出现。
|
||||
- 但 external TTFT outcome 在 25% 已完美区分该 action 是否修复 SLO。Telemetry 解释了 why,却没有比 outcome 更早或更可靠地指导 tuning。
|
||||
- 3.125 req/s/GPU 的 source 无法在 timeout 内 drain;另一组 source 已达 offered ceiling 的 99.1%--100%,数学上不可能通过 10% improvement gate。没有 exposure/headroom 和 censoring control,模型学到的不是 action response。
|
||||
- Same-config repeats 与 matched intervention 的波动不可忽略;只比较两个未经配对的 run 会混入 arrival/order/warm-state noise。
|
||||
|
||||
#### Tuner 必须具备的能力
|
||||
|
||||
- 训练样本必须是 exact-workload paired intervention:`(source trajectory, action) -> target delta`,保留失败和 censoring。
|
||||
- 使用 phase-binned continuous trajectory,而不是人工 bottleneck label 或 threshold rule。
|
||||
- 输出 response distribution 和 uncertainty;证据不足时 abstain,而不是强行给 diagnosis。
|
||||
- Telemetry 的价值必须通过同 cutoff、同 model capacity 的 outcome-only ablation 证明。若不能降低 end-to-end H20-hours,instrumentation 只保留为 debugging/解释工具。
|
||||
|
||||
### Challenge 3:这是异构成本下的 sequential experimental design,不是静态 ranking
|
||||
|
||||
#### 问题本质
|
||||
|
||||
每个 trial 的成本不同:TP4 是 TP1 的四倍 GPU multiplier,startup/warm-up 可能主导短 probe,失败也有成本;同时 tuner 不知道 oracle,只能在 exploitation、information gain 和 cost 之间权衡。选对 top-1 的 accuracy 不能代表 tuning 效果。
|
||||
|
||||
必须回答三个连续问题:
|
||||
|
||||
1. 下一次测哪个联合 action?
|
||||
2. 测多久,何时 continuation/confirmation?
|
||||
3. 什么证据允许停止,并声称 best 已在 `epsilon` 内?
|
||||
|
||||
#### 已有真实证据
|
||||
|
||||
- Pure LLM 达到 best observed 后仍浪费 0.9106 reconstructed H20h。
|
||||
- Simulator top-1 虽然 0 marginal GPUh,却因 rank error 损失 30.46%;real-final 的 k 增大又迅速增加 H20h。
|
||||
- 5% endpoint 上两个方法都只需两个 trial,selection-count headroom 很小;2% endpoint 才暴露 action quality 和 stopping 的巨大差异。
|
||||
- Prefix 不是天然便宜:如果 startup、warm-up 和稳定状态形成占主要成本,缩短 replay window 未必带来等比例 H20h reduction。
|
||||
|
||||
#### Tuner 必须具备的能力
|
||||
|
||||
- Acquisition 直接优化 expected regret reduction / predicted H20 cost,并把 failure probability 纳入约束。
|
||||
- 在 run 前做与 tuning policy 分离的 workload admissibility check:避免 outcome ceiling、无法 drain、无请求或 measurement cap。
|
||||
- 使用 uncertainty-aware continuation 和 stop;stop criterion 针对声明的 candidate set 中“仍存在 >epsilon improvement 的概率”,而不是连续几次没提升。
|
||||
- 主结果报告 H20-hours-to-5%/2%/1%、fixed-budget regret 和 cost-normalized regret AUC,不 metric shopping。
|
||||
|
||||
### Challenge 4:任何 mechanism model 都有 fidelity 和 transfer boundary
|
||||
|
||||
#### 问题本质
|
||||
|
||||
Simulator、learned surrogate、LLM prior 都是近似。Workload、SLO、model、hardware、engine version 改变后,operator cost、scheduler state transition、合法 flag 和 response surface 都可能变化。模型在 calibration task 上解释得好,不表示能在 held-out task 上排序正确。
|
||||
|
||||
#### 已有真实证据
|
||||
|
||||
- Frontier throughput reading 在完全匹配的 12-cell task 上仍把 real oracle 排错,top-1 regret 30.46%。这说明预测绝对 throughput 还不够,局部 rank fidelity 才是 tuning 关键。
|
||||
- Post-hoc SLO reading 的 top bucket 正确,但有大量 anchor feasibility error,也没有 prospective policy status。
|
||||
- Pure LLM 提出了当前 community-vLLM binary 不支持的 flag;engine/API version knowledge 本身会漂移。
|
||||
- 已有 cross-version experiment 中 vLLM 0.20 的强配置在 0.24 上出现大幅退化,说明 response prior 不能无条件迁移。
|
||||
|
||||
#### Tuner 必须具备的能力
|
||||
|
||||
- Simulator 只能作为 prior mean 或 candidate prior;真实 outcome 是 authoritative update。
|
||||
- 学习 simulator residual:把 `sim prediction + source state + action` 映射到 real response,而不是用 telemetry 重新实现另一个无校准 simulator。
|
||||
- 对 task-level OOD 显式提高 uncertainty/abstain;train/test 按完整 task 分割,不能按 request、anchor 或同一 surface cell 随机分割。
|
||||
- 分开报告 cold-start profile/training cost 与 per-task marginal cost,并在 N=1/10/100 等 amortization horizon 下展示。
|
||||
|
||||
## 4. 对应的系统设计
|
||||
|
||||
### 4.1 Harness:从 rule-based tuner 收缩成 experimental control plane
|
||||
|
||||
Harness 保留以下确定性职责:
|
||||
|
||||
- engine-version-aware config schema、合法性和资源约束;
|
||||
- 完整 config/action canonicalization,禁止隐式 merge 和重复试验;
|
||||
- exact trace/request/arrival/length hash,配对、随机化和 counter-rotation;
|
||||
- engine trajectory、external outcome、failure/censoring 的统一时间轴;
|
||||
- all-in GPU cost ledger、oracle annotation 分账、budget enforcement;
|
||||
- data sanity、coverage、SLO/correctness 和 stop-proof audit。
|
||||
|
||||
Harness **不**包含 `queue > N -> increase MNS`、`cap full -> tune knob` 或人工 diagnosis-to-action mapping。这里的规则是实验语义和安全 invariant,不是性能决策 heuristic。
|
||||
|
||||
### 4.2 Action-conditioned response model
|
||||
|
||||
每条学习记录为:
|
||||
|
||||
```text
|
||||
x = {source full config,
|
||||
workload/SLO context,
|
||||
source external outcome,
|
||||
phase-binned engine trajectory}
|
||||
a = normalized full-config delta
|
||||
y = {Delta SLO-goodput, target feasibility/failure, measured H20 cost}
|
||||
```
|
||||
|
||||
学习:
|
||||
|
||||
```text
|
||||
p_theta(y | x, a, optional simulator prediction)
|
||||
```
|
||||
|
||||
第一版应使用适合小数据且有 uncertainty 的 action-conditioned Gaussian-process/bootstrapped surrogate;kernel/feature ablation包括:
|
||||
|
||||
1. config + external outcome;
|
||||
2. 同样输入 + telemetry trajectory;
|
||||
3. simulator + config + outcome;
|
||||
4. 同样输入 + telemetry residual features。
|
||||
|
||||
Telemetry 保留 continuous phase distributions:queue/running residency、MNS/token slack、prefill/decode composition、partial/split prefill、step duration、KV、graph/padding。模型学习它们与 action 的 interaction;不先压成 bottleneck label。
|
||||
|
||||
### 4.3 Cost-aware policy
|
||||
|
||||
在合法的 single/joint candidate set 上选择:
|
||||
|
||||
```text
|
||||
a* = argmax_a
|
||||
expected constrained improvement(a)
|
||||
/ expected all-in H20 cost(a)
|
||||
```
|
||||
|
||||
探索项来自 posterior uncertainty/information gain;launch/SLO failure 有显式 penalty。Simulator 可提供 prior mean,但 simulator 与 real discrepancy 会被 posterior residual 更新。一次 target measurement 后更新 response model,并重新计算下一步 action 或停止概率。
|
||||
|
||||
LLM 在这个 tuning core 中不是 telemetry classifier。它最多作为可移除的 candidate/prior source,提出 schema 内的 sparse joint actions 或检索 engine mechanism;每个 proposal 都由同一个 response model、cost acquisition 和 real validator 评分。只有 `with LLM` 相对 `same tuner without LLM` 在 held-out tasks 上继续降低 cost-to-oracle,才能讨论 LLM 必要性。
|
||||
|
||||
### 4.4 Stop 条件
|
||||
|
||||
对一个预先声明的有限 candidate set,满足以下条件才 stop:
|
||||
|
||||
```text
|
||||
P(exists c: f(c) > best_observed / (1 - epsilon) | D_t) < alpha
|
||||
```
|
||||
|
||||
并且 best config 通过独立 confirmation、SLO/correctness gate,remaining candidate 的 cost-aware value of information 低于阈值。停止原因、posterior coverage 和未测区域必须写入 audit。
|
||||
|
||||
## 5. 下一阶段如何证明,而不是再次构造 heuristic
|
||||
|
||||
### R0:已有数据 retrospective premise check
|
||||
|
||||
- 用 C1/C3 response surfaces 检查 joint model 是否能避免 OAT trap。
|
||||
- 用 action-aware paired records 比较 outcome-only 与 +telemetry 的 action-delta calibration。
|
||||
- 用 SimFid surface 比较 direct model 与 simulator-residual model 的 rank/regret。
|
||||
- 所有 feature、kernel、candidate encoding 在 held-out task 结果之前冻结。
|
||||
|
||||
R0 只能筛选 model family,不能作为 paper result,因为现有 tasks 已参与路线设计。
|
||||
|
||||
### R1:prospective same-host cost-to-oracle pilot
|
||||
|
||||
- dash0 8xH20,固定 engine build/model;serialized placement,禁止共置干扰。
|
||||
- 至少一个未参与 feature/threshold 选择的新 trace window;选择非 ceiling、可 drain 的 offered load。
|
||||
- 声明一个可穷举的小 surface,至少包含 topology/runtime crossed actions,而不是只有一个 MNS ladder。
|
||||
- Oracle annotation 与 tuner online actions 分开记账;method 只能看到当时可用的数据。
|
||||
- 运行 random/search、OAT、纯 LLM、当前 guided harness、frozen simulator+real、outcome-only response、+telemetry response、sim-residual +telemetry。
|
||||
- 比较完整 H20 cost-to-regret curve,而不是 action classification accuracy。
|
||||
|
||||
Pilot opening gate:
|
||||
|
||||
1. telemetry model 相对相同 response model 去掉 telemetry,确实改变至少一个正确的 prospective action ranking;
|
||||
2. 最终 regret <=5%,无 false-safe accept;
|
||||
3. all-in H20-hours 相对 strongest safe outcome-only 至少下降 20%;
|
||||
4. 如果使用 simulator,需相对 frozen simulator+real 至少下降 30%;
|
||||
5. instrumentation overhead <=1%,所有成本和失败均计入。
|
||||
|
||||
若 1--5 任一失败,就不能把 telemetry/harness 写成 tuning contribution;保留其 debugging/measurement 价值即可。
|
||||
|
||||
### R2:task-held-out replication
|
||||
|
||||
至少 3 个 workload window x 2 个 SLO regime,按完整 task 做 leave-one-task-out 或固定 train/test split。报告每个 task 的 regret、安全和成本,以及 task-level paired bootstrap CI。只有 R2 通过,才能把单 task 的 61.09% lower-bound saving 升级为项目贡献。
|
||||
|
||||
## 6. 当前能说与不能说的贡献
|
||||
|
||||
当前能说:
|
||||
|
||||
- 我们有真实反例证明 OAT 和 cap-to-knob mapping 不是通用 tuning strategy。
|
||||
- Harness 的 legality、exact replay、failure/cost accounting 有必要的实验基础设施价值。
|
||||
- 当前 guided sequence 在一个严格同任务比较中显著减少了达到 2% empirical regret 的 reconstructed engine cost。
|
||||
- Simulator 的边际计算便宜,但 rank error 会转化成显著 real regret 或更多 real-final 成本。
|
||||
|
||||
当前不能说:
|
||||
|
||||
- telemetry 已经对 end-to-end tuning 提供独立增益;现有 direct pilot 对此为 negative。
|
||||
- 当前 harness 的 heuristic action ranking 是系统贡献;5% endpoint 只省 5.85%。
|
||||
- LLM 是必要组件;尚无同 policy 的 with/without LLM held-out ablation。
|
||||
- simulator 总 tuning cost 是 0;profile GPU cost 未审计,real verification 不能忽略。
|
||||
- 3.35 是 global oracle,或 dash0 与 dash1 数字是完全 controlled comparison。
|
||||
|
||||
## Data sanity
|
||||
|
||||
- Dash0 sequential numeric scores:n=9,min/max `1.1042/3.35`,distinct=7;两组 config outcome 不全相同。
|
||||
- Exact surface scores:n=12,min/max `1.2833/3.2833`,distinct=8;12 cells 完整且与 simulator metrics 中的 real scores 一致。
|
||||
- Reconstructed trial/cell attempts 包括 4 个无 engine timestamp 的失败:n=32,min/max `0/0.49778 H20h`,distinct=26;所有可重建成本均非负。
|
||||
- Sequential regret observations:n=16,min/max `0/0.34328`,distinct=6,全部在 `[0,1]`。
|
||||
- Checked invariants:dash0 fixed task contexts 相同(除 method/port);trial counts 与 manifest 相符;engine log timestamps monotonic;surface cell 唯一且 MBT=8192;simulator 无失败且 predictions 不全相同;scores/results 不全相同;cost 非负;regret bounded。
|
||||
- Measurement limitation:primary 12-cell campaign 的 4 个 TP4 pre-ready failure 没有 engine timestamp,随后由 companion campaign 完整重跑;其失败成本在 engine-lifetime reconstruction 中为 0。因此 `3.5953 H20h` 是 completed annotation lower bound,不能作为 all-in annotation cost。这个缺口已显式保留,没有在其上建立 total-cost claim。
|
||||
@@ -182,6 +182,59 @@ def rsync_pull(config: FleetConfig, host: HostSpec, remote_path: str, local_path
|
||||
run_local(argv, cwd=config.project_root, capture_output=True, check=True)
|
||||
|
||||
|
||||
def scp_push(config: FleetConfig, host: HostSpec) -> None:
|
||||
ensure_remote_dir(config, host, host.sync_remote_path)
|
||||
local_src = str(config.sync.local_path.resolve()) + "/."
|
||||
remote_dst = f"{host.ssh_alias}:{host.sync_remote_path.rstrip('/')}/"
|
||||
argv = [
|
||||
"scp",
|
||||
"-o",
|
||||
"BatchMode=yes",
|
||||
"-o",
|
||||
f"ConnectTimeout={config.ssh_timeout_sec}",
|
||||
"-r",
|
||||
"-p",
|
||||
local_src,
|
||||
remote_dst,
|
||||
]
|
||||
run_local(argv, cwd=config.project_root, capture_output=True, check=True)
|
||||
|
||||
|
||||
def scp_pull(config: FleetConfig, host: HostSpec, remote_path: str, local_path: Path) -> None:
|
||||
remote_src = remote_path
|
||||
if remote_path.endswith("/"):
|
||||
ensure_dir(local_path)
|
||||
remote_src = remote_path.rstrip("/") + "/."
|
||||
else:
|
||||
ensure_dir(local_path.parent)
|
||||
argv = [
|
||||
"scp",
|
||||
"-o",
|
||||
"BatchMode=yes",
|
||||
"-o",
|
||||
f"ConnectTimeout={config.ssh_timeout_sec}",
|
||||
"-r",
|
||||
"-p",
|
||||
f"{host.ssh_alias}:{remote_src}",
|
||||
str(local_path),
|
||||
]
|
||||
run_local(argv, cwd=config.project_root, capture_output=True, check=True)
|
||||
|
||||
|
||||
def sync_push(config: FleetConfig, host: HostSpec) -> None:
|
||||
if config.sync.mode == "rsync":
|
||||
rsync_push(config, host)
|
||||
else:
|
||||
scp_push(config, host)
|
||||
|
||||
|
||||
def sync_pull(config: FleetConfig, host: HostSpec, remote_path: str, local_path: Path) -> None:
|
||||
if config.sync.mode == "rsync":
|
||||
rsync_pull(config, host, remote_path, local_path)
|
||||
else:
|
||||
scp_pull(config, host, remote_path, local_path)
|
||||
|
||||
|
||||
def ensure_remote_dir(config: FleetConfig, host: HostSpec, remote_path: str) -> None:
|
||||
run_ssh(config, host, f"mkdir -p {shlex.quote(remote_path)}", capture_output=True, check=True)
|
||||
|
||||
@@ -206,8 +259,10 @@ def load_config(path: Path) -> FleetConfig:
|
||||
local_path=relative_to_root(project_root, sync_raw.get("local_path"), project_root),
|
||||
exclude=[str(item) for item in sync_raw.get("exclude", [])],
|
||||
)
|
||||
if sync.mode != "rsync":
|
||||
if sync.mode not in {"rsync", "scp"}:
|
||||
raise FleetError(f"unsupported sync.mode: {sync.mode}")
|
||||
if sync.mode == "scp" and sync.exclude:
|
||||
raise FleetError("sync.exclude is not supported for sync.mode=scp")
|
||||
|
||||
scheduler_raw = raw.get("scheduler", {})
|
||||
scheduler = SchedulerSpec(
|
||||
@@ -639,7 +694,7 @@ def harvest_run(config: FleetConfig, manifest: dict[str, Any]) -> dict[str, Any]
|
||||
|
||||
local_base = ensure_dir(config.artifacts_dir / refreshed["run_id"])
|
||||
remote_run_dir = refreshed["remote_run_dir"].rstrip("/")
|
||||
rsync_pull(config, host, f"{remote_run_dir}/", local_base / "remote_run")
|
||||
sync_pull(config, host, f"{remote_run_dir}/", local_base / "remote_run")
|
||||
|
||||
for artifact in refreshed.get("artifacts", []):
|
||||
artifact_remote = f"{refreshed['remote_sync_path'].rstrip('/')}/{artifact}"
|
||||
@@ -652,7 +707,7 @@ def harvest_run(config: FleetConfig, manifest: dict[str, Any]) -> dict[str, Any]
|
||||
check=False,
|
||||
)
|
||||
if check.returncode == 0:
|
||||
rsync_pull(config, host, artifact_remote, target)
|
||||
sync_pull(config, host, artifact_remote, target)
|
||||
|
||||
refreshed["harvested_at"] = utc_now()
|
||||
write_run_manifest(config, refreshed)
|
||||
@@ -866,7 +921,7 @@ def dispatch_jobs(
|
||||
)
|
||||
continue
|
||||
if host.name not in synced_hosts:
|
||||
rsync_push(config, host)
|
||||
sync_push(config, host)
|
||||
synced_hosts.add(host.name)
|
||||
manifest = launch_job(config, host, job, gpu_ids)
|
||||
manifests.append(manifest)
|
||||
|
||||
63
runs/action-aware-v0/action_aware_client.py
Normal file
63
runs/action-aware-v0/action_aware_client.py
Normal file
@@ -0,0 +1,63 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Add explicit MBBT/config provenance to the accepted Phase-6 replay client."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
PHASE6 = Path(__file__).resolve().parents[1] / "opprof-phase6"
|
||||
sys.path.insert(0, str(PHASE6))
|
||||
|
||||
import opprof_phase6_client as base # noqa: E402
|
||||
|
||||
|
||||
def parser() -> argparse.ArgumentParser:
|
||||
result = argparse.ArgumentParser()
|
||||
result.add_argument("command", choices=("warmup", "run-anchor"))
|
||||
result.add_argument("--study", required=True)
|
||||
result.add_argument("--cell", required=True)
|
||||
result.add_argument("--anchor", type=float, required=True)
|
||||
result.add_argument("--tp", type=int, required=True)
|
||||
result.add_argument("--mns", type=int, required=True)
|
||||
result.add_argument("--mbbt", type=int, required=True)
|
||||
result.add_argument("--base-url", required=True)
|
||||
result.add_argument("--result-dir", required=True)
|
||||
result.add_argument("--disable-slo-early-stop", action="store_true")
|
||||
return result
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parser().parse_args()
|
||||
result = base.run_replay(args, warmup=args.command == "warmup")
|
||||
result.update(
|
||||
{
|
||||
"schema": "action-aware-pilot-result-v0",
|
||||
"config_id": args.cell,
|
||||
"mbbt": args.mbbt,
|
||||
}
|
||||
)
|
||||
base.atomic_json(Path(args.result_dir) / "result.json", result)
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
key: result[key]
|
||||
for key in (
|
||||
"config_id",
|
||||
"mns",
|
||||
"mbbt",
|
||||
"kind",
|
||||
"pass_rate",
|
||||
"feasible",
|
||||
)
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
697
runs/action-aware-v0/analyze_pilot.py
Normal file
697
runs/action-aware-v0/analyze_pilot.py
Normal file
@@ -0,0 +1,697 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Audit source-only constraint signals against crossed real interventions."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import statistics
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable, Mapping
|
||||
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
COMMON_STATE = HERE.parent / "telemetry-residual"
|
||||
sys.path.insert(0, str(COMMON_STATE))
|
||||
|
||||
from common_state import summarize_engine # noqa: E402
|
||||
|
||||
|
||||
SCHEMA = "action-aware-constraint-pilot-audit-v0"
|
||||
|
||||
|
||||
def sha256_file(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as source:
|
||||
for chunk in iter(lambda: source.read(1 << 20), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def atomic_json(path: Path, payload: Any) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary = path.with_suffix(path.suffix + ".tmp")
|
||||
temporary.write_text(
|
||||
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
||||
)
|
||||
os.replace(temporary, path)
|
||||
|
||||
|
||||
def numeric(values: Iterable[float]) -> dict[str, Any]:
|
||||
finite = [float(value) for value in values]
|
||||
if not finite:
|
||||
raise ValueError("numeric summary requires values")
|
||||
if any(not math.isfinite(value) for value in finite):
|
||||
raise ValueError("numeric summary received non-finite values")
|
||||
return {
|
||||
"n": len(finite),
|
||||
"min": min(finite),
|
||||
"max": max(finite),
|
||||
"distinct_n": len(set(finite)),
|
||||
}
|
||||
|
||||
|
||||
def distribution(values: Iterable[float]) -> dict[str, Any]:
|
||||
finite = [float(value) for value in values]
|
||||
summary = numeric(finite)
|
||||
return {
|
||||
**summary,
|
||||
"mean": statistics.fmean(finite),
|
||||
"p50": quantile(finite, 0.50),
|
||||
"p95": quantile(finite, 0.95),
|
||||
"p99": quantile(finite, 0.99),
|
||||
}
|
||||
|
||||
|
||||
def quantile(values: Iterable[float], probability: float) -> float:
|
||||
ordered = sorted(float(value) for value in values)
|
||||
if not ordered:
|
||||
raise ValueError("quantile requires values")
|
||||
position = probability * (len(ordered) - 1)
|
||||
lower = math.floor(position)
|
||||
upper = math.ceil(position)
|
||||
if lower == upper:
|
||||
return ordered[lower]
|
||||
weight = position - lower
|
||||
return ordered[lower] * (1.0 - weight) + ordered[upper] * weight
|
||||
|
||||
|
||||
def load_jsonl(path: Path) -> list[dict[str, Any]]:
|
||||
records = []
|
||||
with path.open(encoding="utf-8") as source:
|
||||
for line_number, line in enumerate(source, 1):
|
||||
try:
|
||||
records.append(json.loads(line))
|
||||
except json.JSONDecodeError as error:
|
||||
raise ValueError(f"{path}:{line_number}: invalid JSON") from error
|
||||
return records
|
||||
|
||||
|
||||
def binding_summary(
|
||||
records: list[Mapping[str, Any]], *, mns: int, mbbt: int
|
||||
) -> dict[str, Any]:
|
||||
if not records:
|
||||
raise ValueError("binding summary requires scheduler records")
|
||||
counts = {
|
||||
"mns_exclusive": 0,
|
||||
"mbbt_exclusive": 0,
|
||||
"both": 0,
|
||||
"waiting_unresolved": 0,
|
||||
"waiting": 0,
|
||||
}
|
||||
running_utilization = []
|
||||
token_utilization = []
|
||||
kv_usage = []
|
||||
preemptions = 0
|
||||
for record in records:
|
||||
waiting = int(record["queues"]["waiting"]) + int(
|
||||
record["queues"]["deferred"]
|
||||
)
|
||||
running = int(record["queues"]["running"])
|
||||
scheduled_tokens = int(record["prefill_tokens"]) + int(
|
||||
record["decode_tokens"]
|
||||
)
|
||||
if running > mns:
|
||||
raise ValueError("running requests exceed configured MNS")
|
||||
if scheduled_tokens > mbbt:
|
||||
raise ValueError("scheduled tokens exceed configured MBBT")
|
||||
mns_hit = waiting > 0 and running == mns
|
||||
mbbt_hit = waiting > 0 and scheduled_tokens == mbbt
|
||||
if waiting > 0:
|
||||
counts["waiting"] += 1
|
||||
if mns_hit and mbbt_hit:
|
||||
counts["both"] += 1
|
||||
elif mns_hit:
|
||||
counts["mns_exclusive"] += 1
|
||||
elif mbbt_hit:
|
||||
counts["mbbt_exclusive"] += 1
|
||||
else:
|
||||
counts["waiting_unresolved"] += 1
|
||||
running_utilization.append(running / mns)
|
||||
token_utilization.append(scheduled_tokens / mbbt)
|
||||
kv_usage.append(float(record["kv"]["usage"]))
|
||||
preemptions += int(record["preemptions"])
|
||||
count = len(records)
|
||||
return {
|
||||
"records": count,
|
||||
**{f"{name}_count": value for name, value in counts.items()},
|
||||
**{f"{name}_fraction": value / count for name, value in counts.items()},
|
||||
"running_utilization_mean": statistics.fmean(running_utilization),
|
||||
"running_utilization_max": max(running_utilization),
|
||||
"token_utilization_mean": statistics.fmean(token_utilization),
|
||||
"token_utilization_max": max(token_utilization),
|
||||
"kv_usage_mean": statistics.fmean(kv_usage),
|
||||
"kv_usage_max": max(kv_usage),
|
||||
"preemptions": preemptions,
|
||||
}
|
||||
|
||||
|
||||
def telemetry_coverage(
|
||||
records: list[Mapping[str, Any]], *, start_ns: int, end_ns: int
|
||||
) -> tuple[dict[str, float], bool]:
|
||||
if not records:
|
||||
raise ValueError("telemetry coverage requires records")
|
||||
submit_gaps = [
|
||||
(int(right["submit_mono_ns"]) - int(left["submit_mono_ns"])) / 1e9
|
||||
for left, right in zip(records, records[1:], strict=False)
|
||||
]
|
||||
uncovered_gaps = [
|
||||
max(
|
||||
0,
|
||||
int(right["submit_mono_ns"]) - int(left["complete_mono_ns"]),
|
||||
)
|
||||
/ 1e9
|
||||
for left, right in zip(records, records[1:], strict=False)
|
||||
]
|
||||
coverage = {
|
||||
"start_gap_s": (int(records[0]["submit_mono_ns"]) - start_ns) / 1e9,
|
||||
"end_gap_s": (end_ns - int(records[-1]["submit_mono_ns"])) / 1e9,
|
||||
"max_internal_submit_gap_s": max(submit_gaps, default=0.0),
|
||||
"max_uncovered_gap_s": max(uncovered_gaps, default=0.0),
|
||||
}
|
||||
covered = (
|
||||
0.0 <= coverage["start_gap_s"] <= 1.0
|
||||
and 0.0 <= coverage["end_gap_s"] <= 1.0
|
||||
and 0.0 <= coverage["max_uncovered_gap_s"] <= 1.0
|
||||
)
|
||||
return coverage, covered
|
||||
|
||||
|
||||
def mechanism_summary(records: list[Mapping[str, Any]]) -> dict[str, Any]:
|
||||
executed = [record for record in records if bool(record["model_executed"])]
|
||||
if not executed:
|
||||
raise ValueError("mechanism summary requires executed steps")
|
||||
prefill = [record for record in executed if int(record["prefill_tokens"]) > 0]
|
||||
decode_only = [
|
||||
record for record in executed if int(record["prefill_tokens"]) == 0
|
||||
]
|
||||
if not prefill or not decode_only:
|
||||
raise ValueError("mechanism summary requires prefill and decode-only steps")
|
||||
|
||||
def durations_ms(selected: list[Mapping[str, Any]]) -> list[float]:
|
||||
values = [
|
||||
(int(record["complete_mono_ns"]) - int(record["submit_mono_ns"]))
|
||||
/ 1e6
|
||||
for record in selected
|
||||
]
|
||||
if any(value < 0.0 for value in values):
|
||||
raise ValueError("engine step duration must be non-negative")
|
||||
return values
|
||||
|
||||
chunk_keys = ("first", "middle", "final", "unsplit", "tokens")
|
||||
chunks = {
|
||||
key: sum(int(record["chunked_prefill"][key]) for record in executed)
|
||||
for key in chunk_keys
|
||||
}
|
||||
prefill_tokens = [int(record["prefill_tokens"]) for record in prefill]
|
||||
prefill_requests = sum(int(record["prefill_requests"]) for record in prefill)
|
||||
prefix_queries = sum(
|
||||
int(record["prefix"]["local"]["queries"]) for record in executed
|
||||
)
|
||||
prefix_hits = sum(
|
||||
int(record["prefix"]["local"]["hits"]) for record in executed
|
||||
)
|
||||
invariants = {
|
||||
"nonnegative_counts": all(
|
||||
value >= 0
|
||||
for value in (
|
||||
*chunks.values(),
|
||||
prefill_requests,
|
||||
prefix_queries,
|
||||
prefix_hits,
|
||||
)
|
||||
),
|
||||
"chunk_tokens_match_prefill_tokens": chunks["tokens"]
|
||||
== sum(prefill_tokens),
|
||||
"prefix_hits_bounded": 0 <= prefix_hits <= prefix_queries,
|
||||
}
|
||||
return {
|
||||
"executed_steps": len(executed),
|
||||
"step_duration_ms": distribution(durations_ms(executed)),
|
||||
"prefill_steps": len(prefill),
|
||||
"prefill_step_duration_ms": distribution(durations_ms(prefill)),
|
||||
"decode_only_steps": len(decode_only),
|
||||
"decode_only_step_duration_ms": distribution(durations_ms(decode_only)),
|
||||
"prefill": {
|
||||
"requests": prefill_requests,
|
||||
"requests_per_step": prefill_requests / len(prefill),
|
||||
"tokens": sum(prefill_tokens),
|
||||
"tokens_per_step": distribution(prefill_tokens),
|
||||
"chunks": chunks,
|
||||
},
|
||||
"prefix": {
|
||||
"queries": prefix_queries,
|
||||
"hits": prefix_hits,
|
||||
"hit_rate": prefix_hits / prefix_queries if prefix_queries else 0.0,
|
||||
},
|
||||
"sanity": {"invariants": invariants},
|
||||
}
|
||||
|
||||
|
||||
def request_summary(path: Path, expected_count: int) -> dict[str, Any]:
|
||||
rows = load_jsonl(path)
|
||||
if len(rows) != expected_count:
|
||||
raise ValueError(f"request row count mismatch: {path}")
|
||||
ttft = [float(row["ttft_ms"]) for row in rows if row["ttft_ms"] is not None]
|
||||
tpot = [float(row["tpot_ms"]) for row in rows if row["tpot_ms"] is not None]
|
||||
if not ttft or not tpot:
|
||||
raise ValueError(f"missing request latency values: {path}")
|
||||
return {
|
||||
"ttft_ms": {f"p{int(p * 100)}": quantile(ttft, p) for p in (0.5, 0.95, 0.99)},
|
||||
"tpot_ms": {f"p{int(p * 100)}": quantile(tpot, p) for p in (0.5, 0.95, 0.99)},
|
||||
}
|
||||
|
||||
|
||||
def load_stream(session_root: Path) -> tuple[list[dict[str, Any]], dict[str, Any]]:
|
||||
streams = sorted((session_root / "opprof").glob("*.jsonl"))
|
||||
sidecars = sorted((session_root / "opprof").glob("*.jsonl.footer.json"))
|
||||
if len(streams) != 1 or len(sidecars) != 1:
|
||||
raise ValueError(f"expected one OpProf stream and sidecar: {session_root}")
|
||||
decoded = load_jsonl(streams[0])
|
||||
records = [row for row in decoded if "step_index" in row]
|
||||
footers = [row for row in decoded if row.get("record_type") == "footer"]
|
||||
sidecar = json.loads(sidecars[0].read_text(encoding="utf-8"))
|
||||
indexes = [int(row["step_index"]) for row in records]
|
||||
invariants = {
|
||||
"one_footer_last": len(footers) == 1 and decoded[-1] is footers[0],
|
||||
"sidecar_final": sidecar.get("final") is True,
|
||||
"zero_drops": sidecar.get("dropped_records") == 0,
|
||||
"written_matches_records": sidecar.get("written_records") == len(records),
|
||||
"contiguous_step_indexes": indexes == list(range(len(indexes))),
|
||||
"monotonic_timestamps": all(
|
||||
int(right["submit_mono_ns"]) >= int(left["submit_mono_ns"])
|
||||
for left, right in zip(records, records[1:], strict=False)
|
||||
),
|
||||
}
|
||||
return records, {
|
||||
"stream": str(streams[0]),
|
||||
"stream_sha256": sha256_file(streams[0]),
|
||||
"records": len(records),
|
||||
"invariants": invariants,
|
||||
}
|
||||
|
||||
|
||||
def analyze_run(
|
||||
*,
|
||||
run_root: Path,
|
||||
config: Mapping[str, Any],
|
||||
repetition: int,
|
||||
expected: Mapping[str, Any],
|
||||
stream_records: list[Mapping[str, Any]],
|
||||
duration_s: float,
|
||||
phase_fractions: list[float],
|
||||
) -> dict[str, Any]:
|
||||
result_root = run_root / "sessions" / str(config["id"]) / f"rep{repetition}"
|
||||
result_path = result_root / "result.json"
|
||||
result = json.loads(result_path.read_text(encoding="utf-8"))
|
||||
selection = result["selection"]
|
||||
invariants = {
|
||||
"result_schema": result.get("schema") == "action-aware-pilot-result-v0",
|
||||
"config_id": result.get("config_id") == config["id"],
|
||||
"tp": int(result.get("tp", -1)) == 4,
|
||||
"mns": int(result.get("mns", -1)) == int(config["mns"]),
|
||||
"mbbt": int(result.get("mbbt", -1)) == int(config["mbbt"]),
|
||||
"uncensored": not bool(result.get("early_stopped", True)),
|
||||
"slo_early_stop_disabled": result.get("slo_early_stop_disabled") is True,
|
||||
"selection_count": int(selection["count"]) == int(expected["selected_count"]),
|
||||
"request_accounting": int(result["observed_count"])
|
||||
== int(expected["selected_count"]),
|
||||
"request_hash": selection["request_id_order_sha256"]
|
||||
== expected["request_id_order_sha256"],
|
||||
"arrival_hash": selection["arrival_order_sha256"]
|
||||
== expected["arrival_order_sha256"],
|
||||
"length_hash": selection["raw_length_order_sha256"]
|
||||
== expected["input_length_order_sha256"],
|
||||
}
|
||||
start_ns = int(result["interval"]["start_mono_ns"])
|
||||
arrival_end_ns = start_ns + round(duration_s * 1e9)
|
||||
full_records = [
|
||||
record
|
||||
for record in stream_records
|
||||
if start_ns <= int(record["submit_mono_ns"]) <= arrival_end_ns
|
||||
]
|
||||
if not full_records:
|
||||
raise ValueError(f"no telemetry records in measured window: {result_path}")
|
||||
coverage, invariants["telemetry_coverage"] = telemetry_coverage(
|
||||
full_records, start_ns=start_ns, end_ns=arrival_end_ns
|
||||
)
|
||||
binding = binding_summary(
|
||||
full_records, mns=int(config["mns"]), mbbt=int(config["mbbt"])
|
||||
)
|
||||
mechanism = mechanism_summary(full_records)
|
||||
invariants["mechanism_summary"] = all(
|
||||
mechanism["sanity"]["invariants"].values()
|
||||
)
|
||||
phases = {}
|
||||
for fraction in phase_fractions:
|
||||
phase_end = start_ns + round(duration_s * fraction * 1e9)
|
||||
phase_records = [
|
||||
record
|
||||
for record in full_records
|
||||
if int(record["submit_mono_ns"]) <= phase_end
|
||||
]
|
||||
phases[f"{fraction:.2f}"] = binding_summary(
|
||||
phase_records, mns=int(config["mns"]), mbbt=int(config["mbbt"])
|
||||
)
|
||||
state = summarize_engine(
|
||||
full_records,
|
||||
start_ns=start_ns,
|
||||
end_ns=arrival_end_ns,
|
||||
request_count=int(result["observed_count"]),
|
||||
)
|
||||
latency = request_summary(
|
||||
result_root / "requests.jsonl", int(result["observed_count"])
|
||||
)
|
||||
return {
|
||||
"config_id": config["id"],
|
||||
"mns": int(config["mns"]),
|
||||
"mbbt": int(config["mbbt"]),
|
||||
"repetition": repetition,
|
||||
"result_path": str(result_path),
|
||||
"result_sha256": sha256_file(result_path),
|
||||
"selection": {
|
||||
"count": int(selection["count"]),
|
||||
"request_id_order_sha256": selection["request_id_order_sha256"],
|
||||
"arrival_order_sha256": selection["arrival_order_sha256"],
|
||||
"raw_length_order_sha256": selection["raw_length_order_sha256"],
|
||||
},
|
||||
"outcome": {
|
||||
"pass_rate": float(result["pass_rate"]),
|
||||
"feasible": bool(result["feasible"]),
|
||||
"slo_pass_count": int(result["slo_pass_count"]),
|
||||
"slo_goodput_req_s": int(result["slo_pass_count"]) / duration_s,
|
||||
"elapsed_s": float(result["interval"]["elapsed_s"]),
|
||||
**latency,
|
||||
},
|
||||
"binding": binding,
|
||||
"mechanism": mechanism,
|
||||
"phases": phases,
|
||||
"state": state,
|
||||
"coverage": coverage,
|
||||
"invariants": invariants,
|
||||
}
|
||||
|
||||
|
||||
def median(values: Iterable[float]) -> float:
|
||||
return float(statistics.median(float(value) for value in values))
|
||||
|
||||
|
||||
def evaluate_decisions(
|
||||
runs: list[Mapping[str, Any]], manifest: Mapping[str, Any]
|
||||
) -> dict[str, Any]:
|
||||
by_key = {
|
||||
(str(run["config_id"]), int(run["repetition"])): run for run in runs
|
||||
}
|
||||
repetitions = sorted(int(key) for key in manifest["repetitions"])
|
||||
regime_results = {}
|
||||
all_predictions = []
|
||||
crossed_pass = True
|
||||
binding_pass = True
|
||||
material_ambiguity = False
|
||||
for regime_name, regime in manifest["regimes"].items():
|
||||
rows = []
|
||||
source_runs = []
|
||||
for repetition in repetitions:
|
||||
source = by_key[(str(regime["source"]), repetition)]
|
||||
mns_target = by_key[(str(regime["actions"]["mns"]), repetition)]
|
||||
mbbt_target = by_key[(str(regime["actions"]["mbbt"]), repetition)]
|
||||
source_runs.append(source)
|
||||
source_goodput = float(source["outcome"]["slo_goodput_req_s"])
|
||||
mns_goodput = float(mns_target["outcome"]["slo_goodput_req_s"])
|
||||
mbbt_goodput = float(mbbt_target["outcome"]["slo_goodput_req_s"])
|
||||
observed = (
|
||||
"mns"
|
||||
if mns_goodput > mbbt_goodput
|
||||
else "mbbt"
|
||||
if mbbt_goodput > mns_goodput
|
||||
else "tie"
|
||||
)
|
||||
mns_score = float(source["binding"]["mns_exclusive_fraction"])
|
||||
mbbt_score = float(source["binding"]["mbbt_exclusive_fraction"])
|
||||
predicted = (
|
||||
"mns"
|
||||
if mns_score > mbbt_score
|
||||
else "mbbt"
|
||||
if mbbt_score > mns_score
|
||||
else "tie"
|
||||
)
|
||||
phase_predictions = {}
|
||||
for phase, summary in source["phases"].items():
|
||||
left = float(summary["mns_exclusive_fraction"])
|
||||
right = float(summary["mbbt_exclusive_fraction"])
|
||||
phase_predictions[phase] = (
|
||||
"mns" if left > right else "mbbt" if right > left else "tie"
|
||||
)
|
||||
margin = (
|
||||
abs(mns_goodput - mbbt_goodput) / source_goodput
|
||||
if source_goodput > 0
|
||||
else None
|
||||
)
|
||||
row = {
|
||||
"repetition": repetition,
|
||||
"source_goodput_req_s": source_goodput,
|
||||
"mns_target_goodput_req_s": mns_goodput,
|
||||
"mbbt_target_goodput_req_s": mbbt_goodput,
|
||||
"observed_winner": observed,
|
||||
"predicted_winner": predicted,
|
||||
"prediction_correct": predicted == observed,
|
||||
"relative_winner_margin_over_source": margin,
|
||||
"mns_exclusive_fraction": mns_score,
|
||||
"mbbt_exclusive_fraction": mbbt_score,
|
||||
"phase_predictions": phase_predictions,
|
||||
"phase_stable": all(value == predicted for value in phase_predictions.values()),
|
||||
}
|
||||
rows.append(row)
|
||||
all_predictions.append(row)
|
||||
|
||||
expected_winner = "mns" if regime_name == "A" else "mbbt"
|
||||
minimum_margin = float(manifest["gates"]["minimum_relative_winner_margin"])
|
||||
regime_crossed = all(
|
||||
row["observed_winner"] == expected_winner
|
||||
and row["relative_winner_margin_over_source"] is not None
|
||||
and row["relative_winner_margin_over_source"] >= minimum_margin
|
||||
for row in rows
|
||||
)
|
||||
crossed_pass &= regime_crossed
|
||||
winning_key = f"{expected_winner}_exclusive_fraction"
|
||||
losing_key = (
|
||||
"mbbt_exclusive_fraction" if expected_winner == "mns" else "mns_exclusive_fraction"
|
||||
)
|
||||
winning_median = median(row[winning_key] for row in rows)
|
||||
losing_median = median(row[losing_key] for row in rows)
|
||||
ratio_pass = winning_median >= float(
|
||||
manifest["gates"]["minimum_exclusive_ratio"]
|
||||
) * losing_median
|
||||
regime_binding = (
|
||||
all(row["prediction_correct"] and row["phase_stable"] for row in rows)
|
||||
and winning_median
|
||||
>= float(manifest["gates"]["minimum_exclusive_fraction"])
|
||||
and ratio_pass
|
||||
)
|
||||
binding_pass &= regime_binding
|
||||
ambiguity_median = median(
|
||||
float(run["binding"]["both_fraction"])
|
||||
+ float(run["binding"]["waiting_unresolved_fraction"])
|
||||
for run in source_runs
|
||||
)
|
||||
score_gap_median = median(
|
||||
abs(
|
||||
float(run["binding"]["mns_exclusive_fraction"])
|
||||
- float(run["binding"]["mbbt_exclusive_fraction"])
|
||||
)
|
||||
for run in source_runs
|
||||
)
|
||||
kv_max_median = median(
|
||||
float(run["binding"]["kv_usage_max"]) for run in source_runs
|
||||
)
|
||||
any_preemption = any(
|
||||
int(run["binding"]["preemptions"]) > 0 for run in source_runs
|
||||
)
|
||||
regime_material = (
|
||||
ambiguity_median >= score_gap_median
|
||||
or kv_max_median >= float(manifest["gates"]["material_kv_usage"])
|
||||
or any_preemption
|
||||
)
|
||||
material_ambiguity |= regime_material
|
||||
regime_results[regime_name] = {
|
||||
"source": regime["source"],
|
||||
"actions": regime["actions"],
|
||||
"expected_winner": expected_winner,
|
||||
"crossed_response_pass": regime_crossed,
|
||||
"binding_pass": regime_binding,
|
||||
"winning_exclusive_median": winning_median,
|
||||
"losing_exclusive_median": losing_median,
|
||||
"exclusive_ratio_pass": ratio_pass,
|
||||
"ambiguity_median": ambiguity_median,
|
||||
"exclusive_gap_median": score_gap_median,
|
||||
"kv_usage_max_median": kv_max_median,
|
||||
"any_preemption": any_preemption,
|
||||
"material_ambiguity": regime_material,
|
||||
"repetitions": rows,
|
||||
}
|
||||
|
||||
if not crossed_pass:
|
||||
decision = "STOP_WORKLOAD_NOT_CROSSED"
|
||||
elif not binding_pass:
|
||||
decision = "STOP_BINDING_NOT_PREDICTIVE"
|
||||
elif material_ambiguity:
|
||||
decision = "OPEN_EXACT_ATTRIBUTION_ABLATION"
|
||||
else:
|
||||
decision = "STOP_NO_NEW_INSTRUMENTATION_NEEDED"
|
||||
correct = sum(int(row["prediction_correct"]) for row in all_predictions)
|
||||
return {
|
||||
"decision": decision,
|
||||
"crossed_response_pass": crossed_pass,
|
||||
"binding_pass": binding_pass,
|
||||
"material_ambiguity": material_ambiguity,
|
||||
"regimes": regime_results,
|
||||
"baselines": {
|
||||
"always_mns_correct": sum(
|
||||
int(row["observed_winner"] == "mns") for row in all_predictions
|
||||
),
|
||||
"always_mbbt_correct": sum(
|
||||
int(row["observed_winner"] == "mbbt") for row in all_predictions
|
||||
),
|
||||
"binding_correct": correct,
|
||||
"decision_count": len(all_predictions),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def analyze(run_root: Path, manifest_path: Path) -> dict[str, Any]:
|
||||
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
||||
if manifest.get("schema") not in {
|
||||
"action-aware-constraint-pilot-manifest-v0",
|
||||
"action-aware-constraint-pilot-manifest-v1",
|
||||
}:
|
||||
raise ValueError("unexpected manifest schema")
|
||||
duration_s = float(manifest["engine"]["duration_s"])
|
||||
phase_fractions = [float(value) for value in manifest["gates"]["phase_fractions"]]
|
||||
runs = []
|
||||
stream_audits = []
|
||||
for config in manifest["configs"]:
|
||||
session_root = run_root / "sessions" / str(config["id"])
|
||||
stream_records, stream_audit = load_stream(session_root)
|
||||
stream_audit["config_id"] = config["id"]
|
||||
stream_audits.append(stream_audit)
|
||||
for repetition in sorted(int(key) for key in manifest["repetitions"]):
|
||||
runs.append(
|
||||
analyze_run(
|
||||
run_root=run_root,
|
||||
config=config,
|
||||
repetition=repetition,
|
||||
expected=manifest["repetitions"][str(repetition)]["selection"],
|
||||
stream_records=stream_records,
|
||||
duration_s=duration_s,
|
||||
phase_fractions=phase_fractions,
|
||||
)
|
||||
)
|
||||
invariants = {
|
||||
"fifteen_runs": len(runs) == 15,
|
||||
"five_streams": len(stream_audits) == 5,
|
||||
"all_run_invariants": all(
|
||||
all(bool(value) for value in run["invariants"].values()) for run in runs
|
||||
),
|
||||
"all_stream_invariants": all(
|
||||
all(bool(value) for value in stream["invariants"].values())
|
||||
for stream in stream_audits
|
||||
),
|
||||
"nonnegative_counters": all(
|
||||
all(
|
||||
float(run["binding"][key]) >= 0
|
||||
for key in (
|
||||
"mns_exclusive_count",
|
||||
"mbbt_exclusive_count",
|
||||
"both_count",
|
||||
"waiting_unresolved_count",
|
||||
"preemptions",
|
||||
)
|
||||
)
|
||||
for run in runs
|
||||
),
|
||||
"ratios_bounded": all(
|
||||
all(
|
||||
0.0 <= float(run["binding"][key]) <= 1.0
|
||||
for key in (
|
||||
"mns_exclusive_fraction",
|
||||
"mbbt_exclusive_fraction",
|
||||
"both_fraction",
|
||||
"waiting_unresolved_fraction",
|
||||
"kv_usage_mean",
|
||||
"kv_usage_max",
|
||||
)
|
||||
)
|
||||
for run in runs
|
||||
),
|
||||
"per_config_results_not_all_identical": len(
|
||||
{float(run["outcome"]["pass_rate"]) for run in runs}
|
||||
)
|
||||
> 1,
|
||||
}
|
||||
red_flags = [name for name, passed in invariants.items() if not passed]
|
||||
decisions = (
|
||||
evaluate_decisions(runs, manifest)
|
||||
if not red_flags
|
||||
else {
|
||||
"decision": "STOP_DATA_INVALID",
|
||||
"crossed_response_pass": False,
|
||||
"binding_pass": False,
|
||||
"material_ambiguity": False,
|
||||
"regimes": {},
|
||||
"baselines": {},
|
||||
}
|
||||
)
|
||||
payload = {
|
||||
"schema": SCHEMA,
|
||||
"decision": decisions["decision"],
|
||||
"manifest": str(manifest_path),
|
||||
"manifest_sha256": sha256_file(manifest_path),
|
||||
"run_root": str(run_root),
|
||||
"runs": runs,
|
||||
"streams": stream_audits,
|
||||
"decision_audit": decisions,
|
||||
"sanity": {
|
||||
"runs": len(runs),
|
||||
"pass_rate": numeric(run["outcome"]["pass_rate"] for run in runs),
|
||||
"slo_goodput_req_s": numeric(
|
||||
run["outcome"]["slo_goodput_req_s"] for run in runs
|
||||
),
|
||||
"telemetry_records_per_run": numeric(
|
||||
run["binding"]["records"] for run in runs
|
||||
),
|
||||
"mns_values": numeric(run["mns"] for run in runs),
|
||||
"mbbt_values": numeric(run["mbbt"] for run in runs),
|
||||
"invariants": invariants,
|
||||
"red_flags": red_flags,
|
||||
},
|
||||
}
|
||||
return payload
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--run-root", type=Path, required=True)
|
||||
parser.add_argument("--manifest", type=Path, required=True)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
payload = analyze(args.run_root, args.manifest)
|
||||
atomic_json(args.output, payload)
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"decision": payload["decision"],
|
||||
"sanity": payload["sanity"],
|
||||
"decision_audit": payload["decision_audit"],
|
||||
},
|
||||
indent=2,
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
227
runs/action-aware-v0/pilot-manifest-v1.json
Normal file
227
runs/action-aware-v0/pilot-manifest-v1.json
Normal file
@@ -0,0 +1,227 @@
|
||||
{
|
||||
"budget": {
|
||||
"expected_h20_hours": [
|
||||
6.0,
|
||||
7.2
|
||||
],
|
||||
"expected_wall_minutes": [
|
||||
90,
|
||||
110
|
||||
],
|
||||
"global_hard_cap_h20_hours": 8.0,
|
||||
"hard_cap_h20_hours": 7.614013100465138,
|
||||
"prior_attempt_artifact": "/home/admin/cpfs/wjh/action-aware-constraint-v0-20260714/operational-stop-v0.json",
|
||||
"prior_attempt_h20_hours": 0.38598689953486126,
|
||||
"safety_h20_hours": 0.25,
|
||||
"session_estimate_h20_hours": 1.35
|
||||
},
|
||||
"burnin": {
|
||||
"anchor": 0.18919793755240089,
|
||||
"arrival_order_sha256": "6c0ac4cb9a30ef501eeeacc8e6cc631c345e976db5ccf530ea5a1ec706d62a24",
|
||||
"input_length_order_sha256": "7939cc20e1a00d1031d27d71508789f38decbbbb6ea59a1df18b2ec342fd2ef8",
|
||||
"offered_req_s": 8.5,
|
||||
"offered_req_s_per_gpu": 2.125,
|
||||
"request_id_order_sha256": "84f4809acbc8acd3b1d14dfa357134a1dc0b9287341624b33f598dafeef54dc7",
|
||||
"selected_count": 510,
|
||||
"study": "/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/studies/burnin-tp4.json",
|
||||
"study_sha256": "5d6c2098042909a863efd3112818fbee9bafe96f22898ac98b66846dbe1fef0f"
|
||||
},
|
||||
"configs": [
|
||||
{
|
||||
"id": "b_base",
|
||||
"mbbt": 2048,
|
||||
"mns": 64,
|
||||
"repetition_order": [
|
||||
1,
|
||||
2,
|
||||
3
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "a_base",
|
||||
"mbbt": 8192,
|
||||
"mns": 16,
|
||||
"repetition_order": [
|
||||
2,
|
||||
3,
|
||||
1
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "shared",
|
||||
"mbbt": 8192,
|
||||
"mns": 64,
|
||||
"repetition_order": [
|
||||
3,
|
||||
1,
|
||||
2
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "b_mns",
|
||||
"mbbt": 2048,
|
||||
"mns": 128,
|
||||
"repetition_order": [
|
||||
1,
|
||||
3,
|
||||
2
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "a_mbbt",
|
||||
"mbbt": 16384,
|
||||
"mns": 16,
|
||||
"repetition_order": [
|
||||
2,
|
||||
1,
|
||||
3
|
||||
]
|
||||
}
|
||||
],
|
||||
"engine": {
|
||||
"burnin_max_elapsed_s": 90.0,
|
||||
"client_timeout_s": 450.0,
|
||||
"disable_slo_early_stop": true,
|
||||
"duration_s": 300.0,
|
||||
"tp": 4
|
||||
},
|
||||
"gates": {
|
||||
"material_kv_usage": 0.9,
|
||||
"minimum_exclusive_fraction": 0.1,
|
||||
"minimum_exclusive_ratio": 5.0,
|
||||
"minimum_relative_winner_margin": 0.1,
|
||||
"phase_fractions": [
|
||||
0.25,
|
||||
0.5,
|
||||
0.75,
|
||||
1.0
|
||||
]
|
||||
},
|
||||
"regimes": {
|
||||
"A": {
|
||||
"actions": {
|
||||
"mbbt": "a_mbbt",
|
||||
"mns": "shared"
|
||||
},
|
||||
"source": "a_base"
|
||||
},
|
||||
"B": {
|
||||
"actions": {
|
||||
"mbbt": "shared",
|
||||
"mns": "b_mns"
|
||||
},
|
||||
"source": "b_base"
|
||||
}
|
||||
},
|
||||
"repetitions": {
|
||||
"1": {
|
||||
"merged_trace": {
|
||||
"bytes": 337429767,
|
||||
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep1.jsonl",
|
||||
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
|
||||
"rows": 9420,
|
||||
"sha256": "68983266aa0e66aa589562f7c08edbd966f9ba4405e20c105adb43777d2dfbf5",
|
||||
"source_sha256": [
|
||||
"b242d1d9086df3accab57b4c92445d5edd581e12f47e12cea227aa63964c6930",
|
||||
"d23b549f7b69af3647308677bbf76f818a3c226a1c98f9a9f93f09ceee46be87"
|
||||
],
|
||||
"sources": [
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low1.jsonl",
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high1.jsonl"
|
||||
]
|
||||
},
|
||||
"selection": {
|
||||
"anchor": 0.48686986110831465,
|
||||
"arrival_order_sha256": "c2ad99986ce558da5901a9c5ec0a00bd69f198c981d8779235f2773a5c87f1c0",
|
||||
"input_length_order_sha256": "9442bfebdc3fab5062dc1f4d688dc28c02afe3fd806c56dd8159f0ac7e6d0b94",
|
||||
"offered_req_s": 8.5,
|
||||
"offered_req_s_per_gpu": 2.125,
|
||||
"request_id_order_sha256": "0bb61dbc9c26875e991d0d4f984134910d37463e5063f86ee960cf4f8aafb771",
|
||||
"selected_count": 2550,
|
||||
"target_count": 2550,
|
||||
"target_req_s_per_gpu": 2.125
|
||||
},
|
||||
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep1-tp4.json",
|
||||
"study_sha256": "ecfff96e33d458eb1e3b9a6d24386f00cc6f1b19ff926e2ec6320b3f671a7ae3"
|
||||
},
|
||||
"2": {
|
||||
"merged_trace": {
|
||||
"bytes": 337509330,
|
||||
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep2.jsonl",
|
||||
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
|
||||
"rows": 9457,
|
||||
"sha256": "f38e8938f6a481fc6725b71b21aa04ff7eaf79783cdfd6e41aa2f074156f00c2",
|
||||
"source_sha256": [
|
||||
"4cbb0baac082bd54af562ce2f39104c5c23b4671672da365a67b1e8c146adf9f",
|
||||
"bb0bcd2564a88000f435f12feb21c7c902eafc9ea5fe916adfe9d1eae47f3f9a"
|
||||
],
|
||||
"sources": [
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low2.jsonl",
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high2.jsonl"
|
||||
]
|
||||
},
|
||||
"selection": {
|
||||
"anchor": 0.4825698948735577,
|
||||
"arrival_order_sha256": "b9fc12cf3f86bc8a79bee65296e65aa2b8bf2aeca46b2887094c669adcbb9a00",
|
||||
"input_length_order_sha256": "d8d4bd6fc8ba852a45605b673b6b3e4f33b58f459e69f2a032d226ee175b074e",
|
||||
"offered_req_s": 8.5,
|
||||
"offered_req_s_per_gpu": 2.125,
|
||||
"request_id_order_sha256": "56a0616b6b54abafd37875c7cb25f8639afef2706ccc55dfbe568f45859ea382",
|
||||
"selected_count": 2550,
|
||||
"target_count": 2550,
|
||||
"target_req_s_per_gpu": 2.125
|
||||
},
|
||||
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep2-tp4.json",
|
||||
"study_sha256": "d92a576db031db24bb58f354ea725d7f7567cb76699d387117ac5a6c9317bbb9"
|
||||
},
|
||||
"3": {
|
||||
"merged_trace": {
|
||||
"bytes": 337450256,
|
||||
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep3.jsonl",
|
||||
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
|
||||
"rows": 9431,
|
||||
"sha256": "3094084b0bb20cc02eecf465091a5c919b4e5b112f704cdc36a563d1efdcee46",
|
||||
"source_sha256": [
|
||||
"1f7ececb142f9a363d2d1ca25eb7b8488b2cc319a51b55faa384f2a3d51f2142",
|
||||
"6f326234791e1cff4ff866bface0d097d0d6e3844eebb1c97653d8e9c35e9397"
|
||||
],
|
||||
"sources": [
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low3.jsonl",
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high3.jsonl"
|
||||
]
|
||||
},
|
||||
"selection": {
|
||||
"anchor": 0.48664343020532463,
|
||||
"arrival_order_sha256": "efce7339e22d3618cb4d55e6b55bfddb2c563c18faba2a992d5829c13e3f55e9",
|
||||
"input_length_order_sha256": "0792b05fff6729fbd92ab2bb4cb6d31bea7799e232ad42772936bc06efbafb54",
|
||||
"offered_req_s": 8.5,
|
||||
"offered_req_s_per_gpu": 2.125,
|
||||
"request_id_order_sha256": "2a2fabe2c4cf176aeb7e0d32fb8e7dbb1f27429a2e7a0cd18d7d186f23096f19",
|
||||
"selected_count": 2550,
|
||||
"target_count": 2550,
|
||||
"target_req_s_per_gpu": 2.125
|
||||
},
|
||||
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep3-tp4.json",
|
||||
"study_sha256": "fb8ffe256dace32f4ca8a8d49b662d98c3b69b94ecc8fa826e43068b238884ab"
|
||||
}
|
||||
},
|
||||
"sanity": {
|
||||
"invariants": {
|
||||
"all_repetition_orders_are_permutations": true,
|
||||
"five_unique_configs": true,
|
||||
"same_load_all_repetitions": true,
|
||||
"shared_endpoint_reused_by_both_regimes": true,
|
||||
"three_disjoint_repetitions": true
|
||||
},
|
||||
"red_flags": []
|
||||
},
|
||||
"schema": "action-aware-constraint-pilot-manifest-v1",
|
||||
"source": {
|
||||
"base_manifest": "/home/gahow/phd/aituner/runs/intervention-response-v2/pilot-manifest-v3.json",
|
||||
"base_manifest_sha256": "273db1181dcc9d6b64439650d0642ebe553b12e6aa9adebfbe3758a7977e5611",
|
||||
"source_trace": "/home/admin/cpfs/wjh/aituner/aituner/trace_windows/traces/chat_w20260312_1000.jsonl",
|
||||
"source_trace_sha256": "875ba869775deb78086477919f03b322da14e2673c7d070e26528c4190912757",
|
||||
"window_id": "chat_w20260312_1000"
|
||||
},
|
||||
"status": "PASS"
|
||||
}
|
||||
227
runs/action-aware-v0/pilot-manifest-v2.json
Normal file
227
runs/action-aware-v0/pilot-manifest-v2.json
Normal file
@@ -0,0 +1,227 @@
|
||||
{
|
||||
"budget": {
|
||||
"expected_h20_hours": [
|
||||
6.0,
|
||||
7.2
|
||||
],
|
||||
"expected_wall_minutes": [
|
||||
90,
|
||||
110
|
||||
],
|
||||
"global_hard_cap_h20_hours": 8.0,
|
||||
"hard_cap_h20_hours": 7.295602157380846,
|
||||
"prior_attempt_artifact": "/home/admin/cpfs/wjh/aituner/aituner-action-aware-20260714/runs/action-aware-v0/prior-attempts-v2.json",
|
||||
"prior_attempt_h20_hours": 0.7043978426191542,
|
||||
"safety_h20_hours": 0.25,
|
||||
"session_estimate_h20_hours": 1.35
|
||||
},
|
||||
"burnin": {
|
||||
"anchor": 0.18919793755240089,
|
||||
"arrival_order_sha256": "6c0ac4cb9a30ef501eeeacc8e6cc631c345e976db5ccf530ea5a1ec706d62a24",
|
||||
"input_length_order_sha256": "7939cc20e1a00d1031d27d71508789f38decbbbb6ea59a1df18b2ec342fd2ef8",
|
||||
"offered_req_s": 8.5,
|
||||
"offered_req_s_per_gpu": 2.125,
|
||||
"request_id_order_sha256": "84f4809acbc8acd3b1d14dfa357134a1dc0b9287341624b33f598dafeef54dc7",
|
||||
"selected_count": 510,
|
||||
"study": "/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/studies/burnin-tp4.json",
|
||||
"study_sha256": "5d6c2098042909a863efd3112818fbee9bafe96f22898ac98b66846dbe1fef0f"
|
||||
},
|
||||
"configs": [
|
||||
{
|
||||
"id": "b_base",
|
||||
"mbbt": 2048,
|
||||
"mns": 64,
|
||||
"repetition_order": [
|
||||
1,
|
||||
2,
|
||||
3
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "a_base",
|
||||
"mbbt": 8192,
|
||||
"mns": 16,
|
||||
"repetition_order": [
|
||||
2,
|
||||
3,
|
||||
1
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "shared",
|
||||
"mbbt": 8192,
|
||||
"mns": 64,
|
||||
"repetition_order": [
|
||||
3,
|
||||
1,
|
||||
2
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "b_mns",
|
||||
"mbbt": 2048,
|
||||
"mns": 128,
|
||||
"repetition_order": [
|
||||
1,
|
||||
3,
|
||||
2
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "a_mbbt",
|
||||
"mbbt": 16384,
|
||||
"mns": 16,
|
||||
"repetition_order": [
|
||||
2,
|
||||
1,
|
||||
3
|
||||
]
|
||||
}
|
||||
],
|
||||
"engine": {
|
||||
"burnin_max_elapsed_s": 90.0,
|
||||
"client_timeout_s": 450.0,
|
||||
"disable_slo_early_stop": true,
|
||||
"duration_s": 300.0,
|
||||
"tp": 4
|
||||
},
|
||||
"gates": {
|
||||
"material_kv_usage": 0.9,
|
||||
"minimum_exclusive_fraction": 0.1,
|
||||
"minimum_exclusive_ratio": 5.0,
|
||||
"minimum_relative_winner_margin": 0.1,
|
||||
"phase_fractions": [
|
||||
0.25,
|
||||
0.5,
|
||||
0.75,
|
||||
1.0
|
||||
]
|
||||
},
|
||||
"regimes": {
|
||||
"A": {
|
||||
"actions": {
|
||||
"mbbt": "a_mbbt",
|
||||
"mns": "shared"
|
||||
},
|
||||
"source": "a_base"
|
||||
},
|
||||
"B": {
|
||||
"actions": {
|
||||
"mbbt": "shared",
|
||||
"mns": "b_mns"
|
||||
},
|
||||
"source": "b_base"
|
||||
}
|
||||
},
|
||||
"repetitions": {
|
||||
"1": {
|
||||
"merged_trace": {
|
||||
"bytes": 337429767,
|
||||
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep1.jsonl",
|
||||
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
|
||||
"rows": 9420,
|
||||
"sha256": "68983266aa0e66aa589562f7c08edbd966f9ba4405e20c105adb43777d2dfbf5",
|
||||
"source_sha256": [
|
||||
"b242d1d9086df3accab57b4c92445d5edd581e12f47e12cea227aa63964c6930",
|
||||
"d23b549f7b69af3647308677bbf76f818a3c226a1c98f9a9f93f09ceee46be87"
|
||||
],
|
||||
"sources": [
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low1.jsonl",
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high1.jsonl"
|
||||
]
|
||||
},
|
||||
"selection": {
|
||||
"anchor": 0.48686986110831465,
|
||||
"arrival_order_sha256": "c2ad99986ce558da5901a9c5ec0a00bd69f198c981d8779235f2773a5c87f1c0",
|
||||
"input_length_order_sha256": "9442bfebdc3fab5062dc1f4d688dc28c02afe3fd806c56dd8159f0ac7e6d0b94",
|
||||
"offered_req_s": 8.5,
|
||||
"offered_req_s_per_gpu": 2.125,
|
||||
"request_id_order_sha256": "0bb61dbc9c26875e991d0d4f984134910d37463e5063f86ee960cf4f8aafb771",
|
||||
"selected_count": 2550,
|
||||
"target_count": 2550,
|
||||
"target_req_s_per_gpu": 2.125
|
||||
},
|
||||
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep1-tp4.json",
|
||||
"study_sha256": "ecfff96e33d458eb1e3b9a6d24386f00cc6f1b19ff926e2ec6320b3f671a7ae3"
|
||||
},
|
||||
"2": {
|
||||
"merged_trace": {
|
||||
"bytes": 337509330,
|
||||
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep2.jsonl",
|
||||
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
|
||||
"rows": 9457,
|
||||
"sha256": "f38e8938f6a481fc6725b71b21aa04ff7eaf79783cdfd6e41aa2f074156f00c2",
|
||||
"source_sha256": [
|
||||
"4cbb0baac082bd54af562ce2f39104c5c23b4671672da365a67b1e8c146adf9f",
|
||||
"bb0bcd2564a88000f435f12feb21c7c902eafc9ea5fe916adfe9d1eae47f3f9a"
|
||||
],
|
||||
"sources": [
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low2.jsonl",
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high2.jsonl"
|
||||
]
|
||||
},
|
||||
"selection": {
|
||||
"anchor": 0.4825698948735577,
|
||||
"arrival_order_sha256": "b9fc12cf3f86bc8a79bee65296e65aa2b8bf2aeca46b2887094c669adcbb9a00",
|
||||
"input_length_order_sha256": "d8d4bd6fc8ba852a45605b673b6b3e4f33b58f459e69f2a032d226ee175b074e",
|
||||
"offered_req_s": 8.5,
|
||||
"offered_req_s_per_gpu": 2.125,
|
||||
"request_id_order_sha256": "56a0616b6b54abafd37875c7cb25f8639afef2706ccc55dfbe568f45859ea382",
|
||||
"selected_count": 2550,
|
||||
"target_count": 2550,
|
||||
"target_req_s_per_gpu": 2.125
|
||||
},
|
||||
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep2-tp4.json",
|
||||
"study_sha256": "d92a576db031db24bb58f354ea725d7f7567cb76699d387117ac5a6c9317bbb9"
|
||||
},
|
||||
"3": {
|
||||
"merged_trace": {
|
||||
"bytes": 337450256,
|
||||
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep3.jsonl",
|
||||
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
|
||||
"rows": 9431,
|
||||
"sha256": "3094084b0bb20cc02eecf465091a5c919b4e5b112f704cdc36a563d1efdcee46",
|
||||
"source_sha256": [
|
||||
"1f7ececb142f9a363d2d1ca25eb7b8488b2cc319a51b55faa384f2a3d51f2142",
|
||||
"6f326234791e1cff4ff866bface0d097d0d6e3844eebb1c97653d8e9c35e9397"
|
||||
],
|
||||
"sources": [
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low3.jsonl",
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high3.jsonl"
|
||||
]
|
||||
},
|
||||
"selection": {
|
||||
"anchor": 0.48664343020532463,
|
||||
"arrival_order_sha256": "efce7339e22d3618cb4d55e6b55bfddb2c563c18faba2a992d5829c13e3f55e9",
|
||||
"input_length_order_sha256": "0792b05fff6729fbd92ab2bb4cb6d31bea7799e232ad42772936bc06efbafb54",
|
||||
"offered_req_s": 8.5,
|
||||
"offered_req_s_per_gpu": 2.125,
|
||||
"request_id_order_sha256": "2a2fabe2c4cf176aeb7e0d32fb8e7dbb1f27429a2e7a0cd18d7d186f23096f19",
|
||||
"selected_count": 2550,
|
||||
"target_count": 2550,
|
||||
"target_req_s_per_gpu": 2.125
|
||||
},
|
||||
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep3-tp4.json",
|
||||
"study_sha256": "fb8ffe256dace32f4ca8a8d49b662d98c3b69b94ecc8fa826e43068b238884ab"
|
||||
}
|
||||
},
|
||||
"sanity": {
|
||||
"invariants": {
|
||||
"all_repetition_orders_are_permutations": true,
|
||||
"five_unique_configs": true,
|
||||
"same_load_all_repetitions": true,
|
||||
"shared_endpoint_reused_by_both_regimes": true,
|
||||
"three_disjoint_repetitions": true
|
||||
},
|
||||
"red_flags": []
|
||||
},
|
||||
"schema": "action-aware-constraint-pilot-manifest-v1",
|
||||
"source": {
|
||||
"base_manifest": "/home/gahow/phd/aituner/runs/intervention-response-v2/pilot-manifest-v3.json",
|
||||
"base_manifest_sha256": "273db1181dcc9d6b64439650d0642ebe553b12e6aa9adebfbe3758a7977e5611",
|
||||
"source_trace": "/home/admin/cpfs/wjh/aituner/aituner/trace_windows/traces/chat_w20260312_1000.jsonl",
|
||||
"source_trace_sha256": "875ba869775deb78086477919f03b322da14e2673c7d070e26528c4190912757",
|
||||
"window_id": "chat_w20260312_1000"
|
||||
},
|
||||
"status": "PASS"
|
||||
}
|
||||
223
runs/action-aware-v0/pilot-manifest.json
Normal file
223
runs/action-aware-v0/pilot-manifest.json
Normal file
@@ -0,0 +1,223 @@
|
||||
{
|
||||
"budget": {
|
||||
"expected_h20_hours": [
|
||||
6.0,
|
||||
7.2
|
||||
],
|
||||
"expected_wall_minutes": [
|
||||
90,
|
||||
110
|
||||
],
|
||||
"hard_cap_h20_hours": 8.0,
|
||||
"safety_h20_hours": 0.25,
|
||||
"session_estimate_h20_hours": 1.35
|
||||
},
|
||||
"burnin": {
|
||||
"anchor": 0.18919793755240089,
|
||||
"arrival_order_sha256": "6c0ac4cb9a30ef501eeeacc8e6cc631c345e976db5ccf530ea5a1ec706d62a24",
|
||||
"input_length_order_sha256": "7939cc20e1a00d1031d27d71508789f38decbbbb6ea59a1df18b2ec342fd2ef8",
|
||||
"offered_req_s": 8.5,
|
||||
"offered_req_s_per_gpu": 2.125,
|
||||
"request_id_order_sha256": "84f4809acbc8acd3b1d14dfa357134a1dc0b9287341624b33f598dafeef54dc7",
|
||||
"selected_count": 510,
|
||||
"study": "/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/studies/burnin-tp4.json",
|
||||
"study_sha256": "5d6c2098042909a863efd3112818fbee9bafe96f22898ac98b66846dbe1fef0f"
|
||||
},
|
||||
"configs": [
|
||||
{
|
||||
"id": "b_base",
|
||||
"mbbt": 256,
|
||||
"mns": 64,
|
||||
"repetition_order": [
|
||||
1,
|
||||
2,
|
||||
3
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "a_base",
|
||||
"mbbt": 8192,
|
||||
"mns": 16,
|
||||
"repetition_order": [
|
||||
2,
|
||||
3,
|
||||
1
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "shared",
|
||||
"mbbt": 8192,
|
||||
"mns": 64,
|
||||
"repetition_order": [
|
||||
3,
|
||||
1,
|
||||
2
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "b_mns",
|
||||
"mbbt": 256,
|
||||
"mns": 128,
|
||||
"repetition_order": [
|
||||
1,
|
||||
3,
|
||||
2
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "a_mbbt",
|
||||
"mbbt": 16384,
|
||||
"mns": 16,
|
||||
"repetition_order": [
|
||||
2,
|
||||
1,
|
||||
3
|
||||
]
|
||||
}
|
||||
],
|
||||
"engine": {
|
||||
"client_timeout_s": 450.0,
|
||||
"disable_slo_early_stop": true,
|
||||
"duration_s": 300.0,
|
||||
"tp": 4
|
||||
},
|
||||
"gates": {
|
||||
"material_kv_usage": 0.9,
|
||||
"minimum_exclusive_fraction": 0.1,
|
||||
"minimum_exclusive_ratio": 5.0,
|
||||
"minimum_relative_winner_margin": 0.1,
|
||||
"phase_fractions": [
|
||||
0.25,
|
||||
0.5,
|
||||
0.75,
|
||||
1.0
|
||||
]
|
||||
},
|
||||
"regimes": {
|
||||
"A": {
|
||||
"actions": {
|
||||
"mbbt": "a_mbbt",
|
||||
"mns": "shared"
|
||||
},
|
||||
"source": "a_base"
|
||||
},
|
||||
"B": {
|
||||
"actions": {
|
||||
"mbbt": "shared",
|
||||
"mns": "b_mns"
|
||||
},
|
||||
"source": "b_base"
|
||||
}
|
||||
},
|
||||
"repetitions": {
|
||||
"1": {
|
||||
"merged_trace": {
|
||||
"bytes": 337429767,
|
||||
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep1.jsonl",
|
||||
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
|
||||
"rows": 9420,
|
||||
"sha256": "68983266aa0e66aa589562f7c08edbd966f9ba4405e20c105adb43777d2dfbf5",
|
||||
"source_sha256": [
|
||||
"b242d1d9086df3accab57b4c92445d5edd581e12f47e12cea227aa63964c6930",
|
||||
"d23b549f7b69af3647308677bbf76f818a3c226a1c98f9a9f93f09ceee46be87"
|
||||
],
|
||||
"sources": [
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low1.jsonl",
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high1.jsonl"
|
||||
]
|
||||
},
|
||||
"selection": {
|
||||
"anchor": 0.48686986110831465,
|
||||
"arrival_order_sha256": "c2ad99986ce558da5901a9c5ec0a00bd69f198c981d8779235f2773a5c87f1c0",
|
||||
"input_length_order_sha256": "9442bfebdc3fab5062dc1f4d688dc28c02afe3fd806c56dd8159f0ac7e6d0b94",
|
||||
"offered_req_s": 8.5,
|
||||
"offered_req_s_per_gpu": 2.125,
|
||||
"request_id_order_sha256": "0bb61dbc9c26875e991d0d4f984134910d37463e5063f86ee960cf4f8aafb771",
|
||||
"selected_count": 2550,
|
||||
"target_count": 2550,
|
||||
"target_req_s_per_gpu": 2.125
|
||||
},
|
||||
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep1-tp4.json",
|
||||
"study_sha256": "ecfff96e33d458eb1e3b9a6d24386f00cc6f1b19ff926e2ec6320b3f671a7ae3"
|
||||
},
|
||||
"2": {
|
||||
"merged_trace": {
|
||||
"bytes": 337509330,
|
||||
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep2.jsonl",
|
||||
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
|
||||
"rows": 9457,
|
||||
"sha256": "f38e8938f6a481fc6725b71b21aa04ff7eaf79783cdfd6e41aa2f074156f00c2",
|
||||
"source_sha256": [
|
||||
"4cbb0baac082bd54af562ce2f39104c5c23b4671672da365a67b1e8c146adf9f",
|
||||
"bb0bcd2564a88000f435f12feb21c7c902eafc9ea5fe916adfe9d1eae47f3f9a"
|
||||
],
|
||||
"sources": [
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low2.jsonl",
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high2.jsonl"
|
||||
]
|
||||
},
|
||||
"selection": {
|
||||
"anchor": 0.4825698948735577,
|
||||
"arrival_order_sha256": "b9fc12cf3f86bc8a79bee65296e65aa2b8bf2aeca46b2887094c669adcbb9a00",
|
||||
"input_length_order_sha256": "d8d4bd6fc8ba852a45605b673b6b3e4f33b58f459e69f2a032d226ee175b074e",
|
||||
"offered_req_s": 8.5,
|
||||
"offered_req_s_per_gpu": 2.125,
|
||||
"request_id_order_sha256": "56a0616b6b54abafd37875c7cb25f8639afef2706ccc55dfbe568f45859ea382",
|
||||
"selected_count": 2550,
|
||||
"target_count": 2550,
|
||||
"target_req_s_per_gpu": 2.125
|
||||
},
|
||||
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep2-tp4.json",
|
||||
"study_sha256": "d92a576db031db24bb58f354ea725d7f7567cb76699d387117ac5a6c9317bbb9"
|
||||
},
|
||||
"3": {
|
||||
"merged_trace": {
|
||||
"bytes": 337450256,
|
||||
"path": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/traces/rep3.jsonl",
|
||||
"request_id_scheme": "sha256(source_sha256:line_number:original_id)",
|
||||
"rows": 9431,
|
||||
"sha256": "3094084b0bb20cc02eecf465091a5c919b4e5b112f704cdc36a563d1efdcee46",
|
||||
"source_sha256": [
|
||||
"1f7ececb142f9a363d2d1ca25eb7b8488b2cc319a51b55faa384f2a3d51f2142",
|
||||
"6f326234791e1cff4ff866bface0d097d0d6e3844eebb1c97653d8e9c35e9397"
|
||||
],
|
||||
"sources": [
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/low3.jsonl",
|
||||
"/home/admin/cpfs/wjh/fidelity-prefix-pilot-20260714/private/traces/high3.jsonl"
|
||||
]
|
||||
},
|
||||
"selection": {
|
||||
"anchor": 0.48664343020532463,
|
||||
"arrival_order_sha256": "efce7339e22d3618cb4d55e6b55bfddb2c563c18faba2a992d5829c13e3f55e9",
|
||||
"input_length_order_sha256": "0792b05fff6729fbd92ab2bb4cb6d31bea7799e232ad42772936bc06efbafb54",
|
||||
"offered_req_s": 8.5,
|
||||
"offered_req_s_per_gpu": 2.125,
|
||||
"request_id_order_sha256": "2a2fabe2c4cf176aeb7e0d32fb8e7dbb1f27429a2e7a0cd18d7d186f23096f19",
|
||||
"selected_count": 2550,
|
||||
"target_count": 2550,
|
||||
"target_req_s_per_gpu": 2.125
|
||||
},
|
||||
"study": "/home/admin/cpfs/wjh/intervention-response-v3-20260714/private/studies/rep3-tp4.json",
|
||||
"study_sha256": "fb8ffe256dace32f4ca8a8d49b662d98c3b69b94ecc8fa826e43068b238884ab"
|
||||
}
|
||||
},
|
||||
"sanity": {
|
||||
"invariants": {
|
||||
"all_repetition_orders_are_permutations": true,
|
||||
"five_unique_configs": true,
|
||||
"same_load_all_repetitions": true,
|
||||
"shared_endpoint_reused_by_both_regimes": true,
|
||||
"three_disjoint_repetitions": true
|
||||
},
|
||||
"red_flags": []
|
||||
},
|
||||
"schema": "action-aware-constraint-pilot-manifest-v0",
|
||||
"source": {
|
||||
"base_manifest": "/home/gahow/phd/aituner/runs/intervention-response-v2/pilot-manifest-v3.json",
|
||||
"base_manifest_sha256": "273db1181dcc9d6b64439650d0642ebe553b12e6aa9adebfbe3758a7977e5611",
|
||||
"source_trace": "/home/admin/cpfs/wjh/aituner/aituner/trace_windows/traces/chat_w20260312_1000.jsonl",
|
||||
"source_trace_sha256": "875ba869775deb78086477919f03b322da14e2673c7d070e26528c4190912757",
|
||||
"window_id": "chat_w20260312_1000"
|
||||
},
|
||||
"status": "PASS"
|
||||
}
|
||||
638
runs/action-aware-v0/pilot_controller.py
Normal file
638
runs/action-aware-v0/pilot_controller.py
Normal file
@@ -0,0 +1,638 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Serialized controller for the crossed-constraint action-aware pilot."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import shlex
|
||||
import signal
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any, Mapping
|
||||
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
PHASE6 = HERE.parent / "opprof-phase6"
|
||||
sys.path.insert(0, str(PHASE6))
|
||||
|
||||
import opprof_phase6_controller as base # noqa: E402
|
||||
|
||||
|
||||
SCHEMA = "action-aware-constraint-pilot-state-v0"
|
||||
|
||||
|
||||
def atomic_json(path: Path, payload: Any) -> None:
|
||||
base.atomic_json(path, payload)
|
||||
|
||||
|
||||
def wait_all_idle(timeout_s: float = 30.0) -> None:
|
||||
deadline = time.monotonic() + timeout_s
|
||||
last_error: Exception | None = None
|
||||
while time.monotonic() < deadline:
|
||||
try:
|
||||
base.assert_all_idle()
|
||||
return
|
||||
except RuntimeError as error:
|
||||
last_error = error
|
||||
time.sleep(1.0)
|
||||
raise last_error or RuntimeError("GPU idle timeout")
|
||||
|
||||
|
||||
def configure(args: argparse.Namespace, manifest: Mapping[str, Any]) -> None:
|
||||
base.WORKDIR = args.run_root.parent
|
||||
base.RUN_ROOT = args.run_root
|
||||
base.STATE = args.run_root / "controller-state.json"
|
||||
base.SOURCE = args.vllm_source
|
||||
base.VENV = args.venv
|
||||
base.AITUNER = args.aituner_root
|
||||
base.MODEL = args.model
|
||||
base.CLIENT = args.client
|
||||
base.GPU_LIMIT = float(manifest["budget"]["hard_cap_h20_hours"])
|
||||
base.MARKER = "action-aware-constraint-pilot-v0"
|
||||
|
||||
|
||||
def validate_inputs(args: argparse.Namespace, manifest: Mapping[str, Any]) -> None:
|
||||
if manifest.get("schema") not in {
|
||||
"action-aware-constraint-pilot-manifest-v0",
|
||||
"action-aware-constraint-pilot-manifest-v1",
|
||||
}:
|
||||
raise RuntimeError("unexpected action-aware manifest schema")
|
||||
if manifest.get("status") != "PASS":
|
||||
raise RuntimeError("action-aware manifest did not pass preflight")
|
||||
red_flags = manifest.get("sanity", {}).get("red_flags", [])
|
||||
if red_flags:
|
||||
raise RuntimeError(f"manifest red flags: {red_flags}")
|
||||
|
||||
required = {
|
||||
"manifest": args.manifest,
|
||||
"aituner_root": args.aituner_root,
|
||||
"vllm_source": args.vllm_source,
|
||||
"venv_python": args.venv / "bin/python",
|
||||
"venv_vllm": args.venv / "bin/vllm",
|
||||
"model": args.model,
|
||||
"client": args.client,
|
||||
"burnin_study": Path(manifest["burnin"]["study"]),
|
||||
}
|
||||
for repetition, item in manifest["repetitions"].items():
|
||||
required[f"rep{repetition}_study"] = Path(item["study"])
|
||||
required[f"rep{repetition}_trace"] = Path(item["merged_trace"]["path"])
|
||||
missing = {name: str(path) for name, path in required.items() if not path.exists()}
|
||||
if missing:
|
||||
raise RuntimeError(f"action-aware input paths missing: {missing}")
|
||||
|
||||
|
||||
def config_map(manifest: Mapping[str, Any]) -> dict[str, dict[str, Any]]:
|
||||
return {str(item["id"]): dict(item) for item in manifest["configs"]}
|
||||
|
||||
|
||||
def server_command(
|
||||
config: Mapping[str, Any], *, gpus: tuple[int, ...], port: int
|
||||
) -> list[str]:
|
||||
return [
|
||||
"taskset",
|
||||
"-c",
|
||||
base.cpu_mask(gpus),
|
||||
str(base.VENV / "bin/vllm"),
|
||||
"serve",
|
||||
str(base.MODEL),
|
||||
"--host",
|
||||
"127.0.0.1",
|
||||
"--port",
|
||||
str(port),
|
||||
"--served-model-name",
|
||||
"qwen3-30b-a3b-community",
|
||||
"--max-num-batched-tokens",
|
||||
str(config["mbbt"]),
|
||||
"--max-num-seqs",
|
||||
str(config["mns"]),
|
||||
"--tensor-parallel-size",
|
||||
"4",
|
||||
"--shutdown-timeout",
|
||||
"120",
|
||||
]
|
||||
|
||||
|
||||
def client_command(
|
||||
entry: Mapping[str, Any],
|
||||
config: Mapping[str, Any],
|
||||
*,
|
||||
study: str,
|
||||
anchor: float,
|
||||
output: Path,
|
||||
warmup: bool,
|
||||
) -> list[str]:
|
||||
command = [
|
||||
"taskset",
|
||||
"-c",
|
||||
base.cpu_mask(entry["gpus"]),
|
||||
str(base.VENV / "bin/python"),
|
||||
str(base.CLIENT),
|
||||
"warmup" if warmup else "run-anchor",
|
||||
"--study",
|
||||
study,
|
||||
"--cell",
|
||||
str(config["id"]),
|
||||
"--anchor",
|
||||
str(anchor),
|
||||
"--tp",
|
||||
"4",
|
||||
"--mns",
|
||||
str(config["mns"]),
|
||||
"--mbbt",
|
||||
str(config["mbbt"]),
|
||||
"--base-url",
|
||||
f"http://127.0.0.1:{entry['port']}",
|
||||
"--result-dir",
|
||||
str(output),
|
||||
"--disable-slo-early-stop",
|
||||
]
|
||||
return command
|
||||
|
||||
|
||||
def remaining_projection(
|
||||
manifest: Mapping[str, Any], *, completed_sessions: int
|
||||
) -> float:
|
||||
remaining = len(manifest["configs"]) - completed_sessions
|
||||
return (
|
||||
remaining * float(manifest["budget"]["session_estimate_h20_hours"])
|
||||
+ float(manifest["budget"]["safety_h20_hours"])
|
||||
)
|
||||
|
||||
|
||||
def dry_run_plan(
|
||||
args: argparse.Namespace, manifest: Mapping[str, Any]
|
||||
) -> dict[str, Any]:
|
||||
sessions = []
|
||||
for index, config in enumerate(manifest["configs"]):
|
||||
entry = {"gpus": (0, 1, 2, 3), "port": 9050 + index}
|
||||
session_root = args.run_root / "sessions" / str(config["id"])
|
||||
first_repetition = str(config["repetition_order"][0])
|
||||
first = manifest["repetitions"][first_repetition]
|
||||
commands = {
|
||||
"server": server_command(config, gpus=entry["gpus"], port=entry["port"]),
|
||||
"warmup": client_command(
|
||||
entry,
|
||||
config,
|
||||
study=first["study"],
|
||||
anchor=float(first["selection"]["anchor"]),
|
||||
output=session_root / "warmup",
|
||||
warmup=True,
|
||||
),
|
||||
"burnin": client_command(
|
||||
entry,
|
||||
config,
|
||||
study=manifest["burnin"]["study"],
|
||||
anchor=float(manifest["burnin"]["anchor"]),
|
||||
output=session_root / "burnin",
|
||||
warmup=False,
|
||||
),
|
||||
}
|
||||
for repetition in config["repetition_order"]:
|
||||
item = manifest["repetitions"][str(repetition)]
|
||||
commands[f"rep{repetition}"] = client_command(
|
||||
entry,
|
||||
config,
|
||||
study=item["study"],
|
||||
anchor=float(item["selection"]["anchor"]),
|
||||
output=session_root / f"rep{repetition}",
|
||||
warmup=False,
|
||||
)
|
||||
sessions.append(
|
||||
{
|
||||
"config": config["id"],
|
||||
"mns": config["mns"],
|
||||
"mbbt": config["mbbt"],
|
||||
"port": entry["port"],
|
||||
"repetition_order": config["repetition_order"],
|
||||
"commands": {
|
||||
role: shlex.join(command) for role, command in commands.items()
|
||||
},
|
||||
}
|
||||
)
|
||||
return {
|
||||
"schema": "action-aware-constraint-pilot-dry-run-v0",
|
||||
"status": "PASS",
|
||||
"manifest": str(args.manifest),
|
||||
"run_root": str(args.run_root),
|
||||
"projected_h20_hours": remaining_projection(
|
||||
manifest, completed_sessions=0
|
||||
),
|
||||
"hard_cap_h20_hours": manifest["budget"]["hard_cap_h20_hours"],
|
||||
"sessions": sessions,
|
||||
}
|
||||
|
||||
|
||||
def load_state(path: Path, hard_cap: float) -> dict[str, Any]:
|
||||
if path.exists():
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
return {
|
||||
"schema": SCHEMA,
|
||||
"status": "initialized",
|
||||
"hard_cap_h20_hours": hard_cap,
|
||||
"gpu_hours_total": 0.0,
|
||||
"completed_sessions": 0,
|
||||
"sessions": {},
|
||||
"failures": [],
|
||||
"started_at": time.time(),
|
||||
}
|
||||
|
||||
|
||||
def append_echo(run_root: Path, line: str) -> None:
|
||||
run_root.mkdir(parents=True, exist_ok=True)
|
||||
with (run_root / "launch-echo.log").open("a", encoding="utf-8") as target:
|
||||
target.write(line + "\n")
|
||||
print(line, flush=True)
|
||||
|
||||
|
||||
def start_server(
|
||||
*,
|
||||
args: argparse.Namespace,
|
||||
config: Mapping[str, Any],
|
||||
index: int,
|
||||
) -> dict[str, Any]:
|
||||
gpus = (0, 1, 2, 3)
|
||||
session_root = args.run_root / "sessions" / str(config["id"])
|
||||
session_root.mkdir(parents=True, exist_ok=True)
|
||||
port = 9050 + index
|
||||
command = server_command(config, gpus=gpus, port=port)
|
||||
with (session_root / "commands.log").open("a", encoding="utf-8") as log:
|
||||
log.write(f"SERVER {shlex.join(command)}\n")
|
||||
server_log = (session_root / "server.log").open("ab", buffering=0)
|
||||
environment = os.environ.copy()
|
||||
environment.update(
|
||||
{
|
||||
"CUDA_VISIBLE_DEVICES": "0,1,2,3",
|
||||
"VLLM_OPPROF_DIR": str(session_root / "opprof"),
|
||||
"OPPROF_PHASE6_MARKER": base.MARKER,
|
||||
"AITUNER_ROOT": str(base.AITUNER),
|
||||
"HF_HUB_OFFLINE": "1",
|
||||
"TRANSFORMERS_OFFLINE": "1",
|
||||
"PYTHONUNBUFFERED": "1",
|
||||
}
|
||||
)
|
||||
server = subprocess.Popen(
|
||||
command,
|
||||
cwd=base.SOURCE,
|
||||
env=environment,
|
||||
stdout=server_log,
|
||||
stderr=subprocess.STDOUT,
|
||||
start_new_session=True,
|
||||
)
|
||||
base.OWNED_PGIDS.add(server.pid)
|
||||
return {
|
||||
"cell": str(config["id"]),
|
||||
"gpus": gpus,
|
||||
"port": port,
|
||||
"dir": session_root,
|
||||
"server": server,
|
||||
"server_handle": server_log,
|
||||
"spawned_at": time.time(),
|
||||
"results": [],
|
||||
}
|
||||
|
||||
|
||||
def validate_result(
|
||||
result: Mapping[str, Any],
|
||||
*,
|
||||
config: Mapping[str, Any],
|
||||
selection: Mapping[str, Any],
|
||||
role: str,
|
||||
warmup: bool,
|
||||
) -> None:
|
||||
if result.get("schema") != "action-aware-pilot-result-v0":
|
||||
raise RuntimeError(f"unexpected result schema: {role}")
|
||||
if result.get("config_id") != config["id"]:
|
||||
raise RuntimeError(f"config id mismatch: {role}")
|
||||
if int(result["tp"]) != 4:
|
||||
raise RuntimeError(f"TP mismatch: {role}")
|
||||
if int(result["mns"]) != int(config["mns"]):
|
||||
raise RuntimeError(f"MNS mismatch: {role}")
|
||||
if int(result["mbbt"]) != int(config["mbbt"]):
|
||||
raise RuntimeError(f"MBBT mismatch: {role}")
|
||||
if result.get("slo_early_stop_disabled") is not True:
|
||||
raise RuntimeError(f"SLO early stop was not disabled: {role}")
|
||||
if warmup:
|
||||
if result["kind"] != "warmup" or int(result["selection"]["count"]) != 16:
|
||||
raise RuntimeError(f"invalid warmup: {role}")
|
||||
return
|
||||
if bool(result["early_stopped"]):
|
||||
raise RuntimeError(f"uncensored run early-stopped: {role}")
|
||||
if int(result["selection"]["count"]) != int(selection["selected_count"]):
|
||||
raise RuntimeError(f"selection count mismatch: {role}")
|
||||
if int(result["observed_count"]) != int(selection["selected_count"]):
|
||||
raise RuntimeError(f"request accounting mismatch: {role}")
|
||||
for result_key, selection_key in (
|
||||
("request_id_order_sha256", "request_id_order_sha256"),
|
||||
("arrival_order_sha256", "arrival_order_sha256"),
|
||||
("raw_length_order_sha256", "input_length_order_sha256"),
|
||||
):
|
||||
if result["selection"][result_key] != selection[selection_key]:
|
||||
raise RuntimeError(f"selection hash mismatch {result_key}: {role}")
|
||||
|
||||
|
||||
def burnin_gate(
|
||||
result: Mapping[str, Any],
|
||||
*,
|
||||
expected_count: int,
|
||||
maximum_elapsed_s: float,
|
||||
) -> dict[str, Any]:
|
||||
if result.get("kind") != "anchor":
|
||||
raise RuntimeError("burnin gate received a non-anchor result")
|
||||
if int(result["selection"]["count"]) != expected_count:
|
||||
raise RuntimeError("burnin gate received the wrong request set")
|
||||
elapsed_s = float(result["interval"]["elapsed_s"])
|
||||
summary = {
|
||||
"elapsed_s": elapsed_s,
|
||||
"pass_rate": float(result["pass_rate"]),
|
||||
"feasible": bool(result["feasible"]),
|
||||
}
|
||||
if elapsed_s > maximum_elapsed_s:
|
||||
raise RuntimeError(
|
||||
f"burnin throughput gate failed: {elapsed_s:.3f}s > "
|
||||
f"{maximum_elapsed_s:.3f}s"
|
||||
)
|
||||
return summary
|
||||
|
||||
|
||||
def run_client(
|
||||
*,
|
||||
entry: dict[str, Any],
|
||||
config: Mapping[str, Any],
|
||||
role: str,
|
||||
study: str,
|
||||
selection: Mapping[str, Any],
|
||||
output: Path,
|
||||
state: Mapping[str, Any],
|
||||
timeout_s: float,
|
||||
warmup: bool = False,
|
||||
) -> dict[str, Any]:
|
||||
command = client_command(
|
||||
entry,
|
||||
config,
|
||||
study=study,
|
||||
anchor=float(selection["anchor"]),
|
||||
output=output,
|
||||
warmup=warmup,
|
||||
)
|
||||
with (entry["dir"] / "commands.log").open("a", encoding="utf-8") as log:
|
||||
log.write(f"CLIENT role={role} {shlex.join(command)}\n")
|
||||
handle = (output.parent / f"{output.name}.log").open("ab", buffering=0)
|
||||
environment = os.environ.copy()
|
||||
environment.update({"AITUNER_ROOT": str(base.AITUNER), "PYTHONUNBUFFERED": "1"})
|
||||
process = subprocess.Popen(
|
||||
command,
|
||||
cwd=base.WORKDIR,
|
||||
env=environment,
|
||||
stdout=handle,
|
||||
stderr=subprocess.STDOUT,
|
||||
start_new_session=True,
|
||||
)
|
||||
deadline = time.monotonic() + timeout_s
|
||||
try:
|
||||
while process.poll() is None:
|
||||
if time.monotonic() > deadline:
|
||||
raise TimeoutError(f"client timeout: {config['id']} {role}")
|
||||
if entry["server"].poll() is not None:
|
||||
raise RuntimeError(f"server exited during {config['id']} {role}")
|
||||
base.assert_no_other_compute()
|
||||
if state["gpu_hours_total"] + base.live_gpu_hours([entry]) >= base.GPU_LIMIT:
|
||||
raise RuntimeError("action-aware pilot H20-hour hard cap reached")
|
||||
time.sleep(1.0)
|
||||
except Exception:
|
||||
try:
|
||||
os.killpg(process.pid, signal.SIGTERM)
|
||||
except ProcessLookupError:
|
||||
pass
|
||||
try:
|
||||
process.wait(timeout=10.0)
|
||||
except subprocess.TimeoutExpired:
|
||||
try:
|
||||
os.killpg(process.pid, signal.SIGKILL)
|
||||
except ProcessLookupError:
|
||||
pass
|
||||
process.wait(timeout=10.0)
|
||||
raise
|
||||
finally:
|
||||
handle.close()
|
||||
if process.returncode:
|
||||
raise RuntimeError(
|
||||
f"client failed: config={config['id']} role={role} rc={process.returncode}"
|
||||
)
|
||||
result = json.loads((output / "result.json").read_text(encoding="utf-8"))
|
||||
validate_result(
|
||||
result,
|
||||
config=config,
|
||||
selection=selection,
|
||||
role=role,
|
||||
warmup=warmup,
|
||||
)
|
||||
entry["results"].append(
|
||||
{"anchor": float(selection["anchor"]), "dir": str(output), "kind": result["kind"]}
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def execute_session(
|
||||
*,
|
||||
args: argparse.Namespace,
|
||||
manifest: Mapping[str, Any],
|
||||
config: Mapping[str, Any],
|
||||
index: int,
|
||||
state: dict[str, Any],
|
||||
state_path: Path,
|
||||
) -> None:
|
||||
name = str(config["id"])
|
||||
if state["sessions"].get(name, {}).get("status") == "complete":
|
||||
return
|
||||
projection = remaining_projection(
|
||||
manifest, completed_sessions=int(state["completed_sessions"])
|
||||
)
|
||||
if float(state["gpu_hours_total"]) + projection > base.GPU_LIMIT:
|
||||
raise RuntimeError(f"projected cost exceeds cap before {name}")
|
||||
load_values = {
|
||||
float(item["selection"]["offered_req_s_per_gpu"])
|
||||
for item in manifest["repetitions"].values()
|
||||
}
|
||||
load_text = (
|
||||
f"{next(iter(load_values)):.6g}"
|
||||
if len(load_values) == 1
|
||||
else ",".join(f"{value:.6g}" for value in sorted(load_values))
|
||||
)
|
||||
echo = (
|
||||
f"ACTION_AWARE_SESSION_ECHO host=dash0 config={name} tp=4 "
|
||||
f"mns={config['mns']} mbbt={config['mbbt']} gpus=0-3 "
|
||||
f"workload={manifest['source']['window_id']} load_per_gpu={load_text} "
|
||||
f"duration_s={manifest['engine']['duration_s']} "
|
||||
f"repetitions={','.join(map(str, config['repetition_order']))} "
|
||||
f"source={args.manifest} output={args.run_root / 'sessions' / name} "
|
||||
f"spent_h20h={state['gpu_hours_total']:.6f} "
|
||||
f"remaining_projection_h20h={projection:.3f} cap_h20h={base.GPU_LIMIT:.1f}"
|
||||
)
|
||||
append_echo(args.run_root, echo)
|
||||
wait_all_idle()
|
||||
session_state = {
|
||||
"status": "starting",
|
||||
"mns": int(config["mns"]),
|
||||
"mbbt": int(config["mbbt"]),
|
||||
"repetition_order": list(config["repetition_order"]),
|
||||
"started_at": time.time(),
|
||||
"runs": [],
|
||||
}
|
||||
state["status"] = "running"
|
||||
state["sessions"][name] = session_state
|
||||
atomic_json(state_path, state)
|
||||
entry = start_server(args=args, config=config, index=index)
|
||||
failure: Exception | None = None
|
||||
try:
|
||||
base.wait_ready(entry)
|
||||
first = manifest["repetitions"][str(config["repetition_order"][0])]
|
||||
session_state["status"] = "warmup"
|
||||
atomic_json(state_path, state)
|
||||
run_client(
|
||||
entry=entry,
|
||||
config=config,
|
||||
role="warmup",
|
||||
study=first["study"],
|
||||
selection=first["selection"],
|
||||
output=entry["dir"] / "warmup",
|
||||
state=state,
|
||||
timeout_s=180.0,
|
||||
warmup=True,
|
||||
)
|
||||
session_state["status"] = "burnin"
|
||||
atomic_json(state_path, state)
|
||||
burnin = manifest["burnin"]
|
||||
burnin_result = run_client(
|
||||
entry=entry,
|
||||
config=config,
|
||||
role="burnin",
|
||||
study=burnin["study"],
|
||||
selection=burnin,
|
||||
output=entry["dir"] / "burnin",
|
||||
state=state,
|
||||
timeout_s=float(manifest["engine"]["client_timeout_s"]),
|
||||
)
|
||||
session_state["burnin"] = burnin_gate(
|
||||
burnin_result,
|
||||
expected_count=int(burnin["selected_count"]),
|
||||
maximum_elapsed_s=float(manifest["engine"]["burnin_max_elapsed_s"]),
|
||||
)
|
||||
atomic_json(state_path, state)
|
||||
session_state["status"] = "measured"
|
||||
atomic_json(state_path, state)
|
||||
for repetition in config["repetition_order"]:
|
||||
item = manifest["repetitions"][str(repetition)]
|
||||
role = f"rep{repetition}"
|
||||
result = run_client(
|
||||
entry=entry,
|
||||
config=config,
|
||||
role=role,
|
||||
study=item["study"],
|
||||
selection=item["selection"],
|
||||
output=entry["dir"] / role,
|
||||
state=state,
|
||||
timeout_s=float(manifest["engine"]["client_timeout_s"]),
|
||||
)
|
||||
session_state["runs"].append(
|
||||
{
|
||||
"repetition": int(repetition),
|
||||
"pass_rate": result["pass_rate"],
|
||||
"feasible": result["feasible"],
|
||||
"slo_pass_count": result["slo_pass_count"],
|
||||
"elapsed_s": result["interval"]["elapsed_s"],
|
||||
}
|
||||
)
|
||||
atomic_json(state_path, state)
|
||||
session_state["status"] = "stopping"
|
||||
atomic_json(state_path, state)
|
||||
except Exception as error: # noqa: BLE001
|
||||
failure = error
|
||||
finally:
|
||||
try:
|
||||
base.stop_entry(entry)
|
||||
except Exception as error: # noqa: BLE001
|
||||
failure = failure or error
|
||||
time.sleep(2.0)
|
||||
try:
|
||||
wait_all_idle()
|
||||
except Exception as error: # noqa: BLE001
|
||||
failure = failure or error
|
||||
|
||||
session_hours = base.live_gpu_hours([entry])
|
||||
state["gpu_hours_total"] += session_hours
|
||||
session_state["gpu_hours"] = session_hours
|
||||
if failure is not None:
|
||||
session_state["status"] = "failed"
|
||||
session_state["failure"] = repr(failure)
|
||||
state["status"] = "failed"
|
||||
state["failures"].append({"session": name, "failure": repr(failure)})
|
||||
atomic_json(state_path, state)
|
||||
raise failure
|
||||
validation = base.validate_cell(entry)
|
||||
session_state["validation"] = validation
|
||||
session_state["status"] = "complete"
|
||||
session_state["completed_at"] = time.time()
|
||||
state["completed_sessions"] += 1
|
||||
atomic_json(state_path, state)
|
||||
|
||||
|
||||
def parser() -> argparse.ArgumentParser:
|
||||
result = argparse.ArgumentParser()
|
||||
result.add_argument("--manifest", type=Path, required=True)
|
||||
result.add_argument("--run-root", type=Path, required=True)
|
||||
result.add_argument("--aituner-root", type=Path, required=True)
|
||||
result.add_argument("--vllm-source", type=Path, required=True)
|
||||
result.add_argument("--venv", type=Path, required=True)
|
||||
result.add_argument("--model", type=Path, required=True)
|
||||
result.add_argument("--client", type=Path, required=True)
|
||||
result.add_argument("--dry-run", action="store_true")
|
||||
return result
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parser().parse_args()
|
||||
manifest = json.loads(args.manifest.read_text(encoding="utf-8"))
|
||||
validate_inputs(args, manifest)
|
||||
configure(args, manifest)
|
||||
if args.dry_run:
|
||||
print(json.dumps(dry_run_plan(args, manifest), indent=2, sort_keys=True))
|
||||
return
|
||||
args.run_root.mkdir(parents=True, exist_ok=True)
|
||||
copied_manifest = args.run_root / "pilot-manifest.json"
|
||||
if not copied_manifest.exists():
|
||||
atomic_json(copied_manifest, manifest)
|
||||
state_path = args.run_root / "controller-state.json"
|
||||
state = load_state(state_path, base.GPU_LIMIT)
|
||||
state["status"] = "running"
|
||||
atomic_json(state_path, state)
|
||||
for index, config in enumerate(manifest["configs"]):
|
||||
execute_session(
|
||||
args=args,
|
||||
manifest=manifest,
|
||||
config=config,
|
||||
index=index,
|
||||
state=state,
|
||||
state_path=state_path,
|
||||
)
|
||||
state["status"] = "complete"
|
||||
state["completed_at"] = time.time()
|
||||
atomic_json(state_path, state)
|
||||
wait_all_idle()
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"status": state["status"],
|
||||
"completed_sessions": state["completed_sessions"],
|
||||
"gpu_hours_total": state["gpu_hours_total"],
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
195
runs/action-aware-v0/prepare_pilot.py
Normal file
195
runs/action-aware-v0/prepare_pilot.py
Normal file
@@ -0,0 +1,195 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Freeze the crossed-constraint action-aware development pilot."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
SCHEMA_V0 = "action-aware-constraint-pilot-manifest-v0"
|
||||
SCHEMA_V1 = "action-aware-constraint-pilot-manifest-v1"
|
||||
|
||||
|
||||
def configs(token_source_mbbt: int) -> tuple[dict[str, Any], ...]:
|
||||
return (
|
||||
{
|
||||
"id": "b_base",
|
||||
"mns": 64,
|
||||
"mbbt": token_source_mbbt,
|
||||
"repetition_order": [1, 2, 3],
|
||||
},
|
||||
{"id": "a_base", "mns": 16, "mbbt": 8192, "repetition_order": [2, 3, 1]},
|
||||
{"id": "shared", "mns": 64, "mbbt": 8192, "repetition_order": [3, 1, 2]},
|
||||
{
|
||||
"id": "b_mns",
|
||||
"mns": 128,
|
||||
"mbbt": token_source_mbbt,
|
||||
"repetition_order": [1, 3, 2],
|
||||
},
|
||||
{"id": "a_mbbt", "mns": 16, "mbbt": 16384, "repetition_order": [2, 1, 3]},
|
||||
)
|
||||
|
||||
|
||||
def sha256_file(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as source:
|
||||
for chunk in iter(lambda: source.read(1 << 20), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def atomic_json(path: Path, payload: Any) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary = path.with_suffix(path.suffix + ".tmp")
|
||||
temporary.write_text(
|
||||
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
||||
)
|
||||
os.replace(temporary, path)
|
||||
|
||||
|
||||
def build(
|
||||
base_path: Path,
|
||||
*,
|
||||
token_source_mbbt: int = 256,
|
||||
prior_attempt_h20_hours: float = 0.0,
|
||||
prior_attempt_artifact: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
if token_source_mbbt <= 0:
|
||||
raise ValueError("token source MBBT must be positive")
|
||||
if prior_attempt_h20_hours < 0.0 or prior_attempt_h20_hours >= 8.0:
|
||||
raise ValueError("prior attempt cost must be in [0, 8)")
|
||||
base = json.loads(base_path.read_text(encoding="utf-8"))
|
||||
if base.get("schema") != "intervention-response-phase-aware-pilot-manifest-v3":
|
||||
raise ValueError("unexpected base manifest schema")
|
||||
if base.get("status") != "PASS":
|
||||
raise ValueError("base manifest did not pass its preflight")
|
||||
if sorted(int(key) for key in base["repetitions"]) != [1, 2, 3]:
|
||||
raise ValueError("base manifest must contain exactly three repetitions")
|
||||
|
||||
repetitions = {}
|
||||
selection_hashes = []
|
||||
for repetition in (1, 2, 3):
|
||||
source = base["repetitions"][str(repetition)]
|
||||
selection = dict(source["selections"]["mid"])
|
||||
selection_hashes.append(selection["request_id_order_sha256"])
|
||||
repetitions[str(repetition)] = {
|
||||
"study": source["study"],
|
||||
"study_sha256": source["study_sha256"],
|
||||
"selection": selection,
|
||||
"merged_trace": source["merged_trace"],
|
||||
}
|
||||
|
||||
frozen_configs = configs(token_source_mbbt)
|
||||
config_ids = [str(config["id"]) for config in frozen_configs]
|
||||
schema = (
|
||||
SCHEMA_V0
|
||||
if token_source_mbbt == 256 and prior_attempt_h20_hours == 0.0
|
||||
else SCHEMA_V1
|
||||
)
|
||||
payload = {
|
||||
"schema": schema,
|
||||
"status": "PASS",
|
||||
"source": {
|
||||
"base_manifest": str(base_path.resolve()),
|
||||
"base_manifest_sha256": sha256_file(base_path),
|
||||
"window_id": base["source"]["window_id"],
|
||||
"source_trace": base["source"]["source_trace"],
|
||||
"source_trace_sha256": base["source"]["source_trace_sha256"],
|
||||
},
|
||||
"engine": {
|
||||
"tp": 4,
|
||||
"duration_s": 300.0,
|
||||
"disable_slo_early_stop": True,
|
||||
"client_timeout_s": 450.0,
|
||||
"burnin_max_elapsed_s": 90.0,
|
||||
},
|
||||
"burnin": base["burnin"],
|
||||
"repetitions": repetitions,
|
||||
"configs": [dict(config) for config in frozen_configs],
|
||||
"regimes": {
|
||||
"A": {
|
||||
"source": "a_base",
|
||||
"actions": {"mns": "shared", "mbbt": "a_mbbt"},
|
||||
},
|
||||
"B": {
|
||||
"source": "b_base",
|
||||
"actions": {"mns": "b_mns", "mbbt": "shared"},
|
||||
},
|
||||
},
|
||||
"budget": {
|
||||
"global_hard_cap_h20_hours": 8.0,
|
||||
"hard_cap_h20_hours": 8.0 - prior_attempt_h20_hours,
|
||||
"prior_attempt_h20_hours": prior_attempt_h20_hours,
|
||||
"prior_attempt_artifact": prior_attempt_artifact,
|
||||
"session_estimate_h20_hours": 1.35,
|
||||
"safety_h20_hours": 0.25,
|
||||
"expected_h20_hours": [6.0, 7.2],
|
||||
"expected_wall_minutes": [90, 110],
|
||||
},
|
||||
"gates": {
|
||||
"minimum_relative_winner_margin": 0.10,
|
||||
"minimum_exclusive_fraction": 0.10,
|
||||
"minimum_exclusive_ratio": 5.0,
|
||||
"phase_fractions": [0.25, 0.50, 0.75, 1.0],
|
||||
"material_kv_usage": 0.90,
|
||||
},
|
||||
"sanity": {
|
||||
"invariants": {
|
||||
"five_unique_configs": len(config_ids) == len(set(config_ids)) == 5,
|
||||
"three_disjoint_repetitions": len(set(selection_hashes)) == 3,
|
||||
"same_load_all_repetitions": len(
|
||||
{
|
||||
float(item["selection"]["offered_req_s_per_gpu"])
|
||||
for item in repetitions.values()
|
||||
}
|
||||
)
|
||||
== 1,
|
||||
"all_repetition_orders_are_permutations": all(
|
||||
sorted(config["repetition_order"]) == [1, 2, 3]
|
||||
for config in frozen_configs
|
||||
),
|
||||
}
|
||||
},
|
||||
}
|
||||
payload["sanity"]["invariants"]["shared_endpoint_reused_by_both_regimes"] = (
|
||||
payload["regimes"]["A"]["actions"]["mns"]
|
||||
== payload["regimes"]["B"]["actions"]["mbbt"]
|
||||
== "shared"
|
||||
)
|
||||
payload["sanity"]["red_flags"] = [
|
||||
name
|
||||
for name, passed in payload["sanity"]["invariants"].items()
|
||||
if not passed
|
||||
]
|
||||
if payload["sanity"]["red_flags"]:
|
||||
payload["status"] = "FAIL"
|
||||
return payload
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--base-manifest", type=Path, required=True)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
parser.add_argument("--token-source-mbbt", type=int, default=256)
|
||||
parser.add_argument("--prior-attempt-h20-hours", type=float, default=0.0)
|
||||
parser.add_argument("--prior-attempt-artifact")
|
||||
args = parser.parse_args()
|
||||
payload = build(
|
||||
args.base_manifest,
|
||||
token_source_mbbt=args.token_source_mbbt,
|
||||
prior_attempt_h20_hours=args.prior_attempt_h20_hours,
|
||||
prior_attempt_artifact=args.prior_attempt_artifact,
|
||||
)
|
||||
atomic_json(args.output, payload)
|
||||
print(json.dumps(payload["sanity"], sort_keys=True))
|
||||
if payload["status"] != "PASS":
|
||||
raise SystemExit("manifest preflight failed")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
24
runs/action-aware-v0/prior-attempts-v2.json
Normal file
24
runs/action-aware-v0/prior-attempts-v2.json
Normal file
@@ -0,0 +1,24 @@
|
||||
{
|
||||
"global_hard_cap_h20_hours": 8.0,
|
||||
"invariants": {
|
||||
"all_gpus_idle_after_each_stop": true,
|
||||
"no_completed_measured_runs": true,
|
||||
"no_prior_runtime_data_reused": true
|
||||
},
|
||||
"prior_attempt_h20_hours": 0.7043978426191542,
|
||||
"schema": "action-aware-prior-attempts-v2",
|
||||
"stops": [
|
||||
{
|
||||
"artifact": "/home/admin/cpfs/wjh/action-aware-constraint-v0-20260714/operational-stop-v0.json",
|
||||
"h20_hours": 0.38598689953486126,
|
||||
"reason": "MBBT256 burn-in remained throughput-backlogged",
|
||||
"stage": "burnin"
|
||||
},
|
||||
{
|
||||
"artifact": "/home/admin/cpfs/wjh/action-aware-constraint-v1-20260714/operational-stop-v1.json",
|
||||
"h20_hours": 0.31841094308429296,
|
||||
"reason": "controller passed the warmup result to the burn-in gate",
|
||||
"stage": "first measured run in flight; zero measured results completed"
|
||||
}
|
||||
]
|
||||
}
|
||||
292
runs/action-aware-v0/test_pilot.py
Normal file
292
runs/action-aware-v0/test_pilot.py
Normal file
@@ -0,0 +1,292 @@
|
||||
#!/usr/bin/env python3
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
ROOT = HERE.parents[1]
|
||||
|
||||
|
||||
def load(name: str, filename: str):
|
||||
spec = importlib.util.spec_from_file_location(name, HERE / filename)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
assert spec.loader is not None
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
def record(*, waiting: int, running: int, tokens: int) -> dict:
|
||||
return {
|
||||
"queues": {"waiting": waiting, "deferred": 0, "running": running},
|
||||
"prefill_tokens": tokens,
|
||||
"decode_tokens": 0,
|
||||
"kv": {"usage": 0.5},
|
||||
"preemptions": 0,
|
||||
}
|
||||
|
||||
|
||||
def fake_run(
|
||||
config: str,
|
||||
repetition: int,
|
||||
*,
|
||||
goodput: float,
|
||||
mns_score: float = 0.0,
|
||||
mbbt_score: float = 0.0,
|
||||
ambiguous: float = 0.0,
|
||||
) -> dict:
|
||||
binding = {
|
||||
"mns_exclusive_fraction": mns_score,
|
||||
"mbbt_exclusive_fraction": mbbt_score,
|
||||
"both_fraction": ambiguous,
|
||||
"waiting_unresolved_fraction": 0.0,
|
||||
"kv_usage_max": 0.5,
|
||||
"preemptions": 0,
|
||||
}
|
||||
phases = {
|
||||
phase: {
|
||||
"mns_exclusive_fraction": mns_score,
|
||||
"mbbt_exclusive_fraction": mbbt_score,
|
||||
}
|
||||
for phase in ("0.25", "0.50", "0.75", "1.00")
|
||||
}
|
||||
return {
|
||||
"config_id": config,
|
||||
"repetition": repetition,
|
||||
"outcome": {"slo_goodput_req_s": goodput},
|
||||
"binding": binding,
|
||||
"phases": phases,
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
analysis = load("action_aware_analysis", "analyze_pilot.py")
|
||||
summary = analysis.binding_summary(
|
||||
[
|
||||
record(waiting=1, running=16, tokens=8),
|
||||
record(waiting=1, running=8, tokens=32),
|
||||
record(waiting=1, running=16, tokens=32),
|
||||
record(waiting=1, running=8, tokens=8),
|
||||
record(waiting=0, running=8, tokens=8),
|
||||
],
|
||||
mns=16,
|
||||
mbbt=32,
|
||||
)
|
||||
assert summary["mns_exclusive_count"] == 1
|
||||
assert summary["mbbt_exclusive_count"] == 1
|
||||
assert summary["both_count"] == 1
|
||||
assert summary["waiting_unresolved_count"] == 1
|
||||
assert summary["waiting_count"] == 4
|
||||
|
||||
# A per-step stream may have a submit gap above one second when the
|
||||
# preceding model execution itself spans that interval. Such a gap is
|
||||
# covered telemetry, not a dropped-record interval.
|
||||
asynchronous = [
|
||||
{"submit_mono_ns": 0, "complete_mono_ns": 1_200_000_000},
|
||||
{"submit_mono_ns": 1_100_000_000, "complete_mono_ns": 1_300_000_000},
|
||||
]
|
||||
coverage, covered = analysis.telemetry_coverage(
|
||||
asynchronous, start_ns=0, end_ns=1_100_000_000
|
||||
)
|
||||
assert coverage["max_internal_submit_gap_s"] == 1.1
|
||||
assert coverage["max_uncovered_gap_s"] == 0.0
|
||||
assert covered
|
||||
missing = copy.deepcopy(asynchronous)
|
||||
missing[0]["complete_mono_ns"] = 0
|
||||
assert not analysis.telemetry_coverage(
|
||||
missing, start_ns=0, end_ns=1_100_000_000
|
||||
)[1]
|
||||
|
||||
mechanism = analysis.mechanism_summary(
|
||||
[
|
||||
{
|
||||
"model_executed": True,
|
||||
"submit_mono_ns": 0,
|
||||
"complete_mono_ns": 2_000_000,
|
||||
"prefill_tokens": 8,
|
||||
"prefill_requests": 2,
|
||||
"chunked_prefill": {
|
||||
"first": 1,
|
||||
"middle": 0,
|
||||
"final": 0,
|
||||
"unsplit": 1,
|
||||
"tokens": 8,
|
||||
},
|
||||
"prefix": {"local": {"queries": 10, "hits": 2}},
|
||||
},
|
||||
{
|
||||
"model_executed": True,
|
||||
"submit_mono_ns": 2_000_000,
|
||||
"complete_mono_ns": 3_000_000,
|
||||
"prefill_tokens": 0,
|
||||
"prefill_requests": 0,
|
||||
"chunked_prefill": {
|
||||
"first": 0,
|
||||
"middle": 0,
|
||||
"final": 0,
|
||||
"unsplit": 0,
|
||||
"tokens": 0,
|
||||
},
|
||||
"prefix": {"local": {"queries": 0, "hits": 0}},
|
||||
},
|
||||
]
|
||||
)
|
||||
assert mechanism["prefill"]["requests_per_step"] == 2.0
|
||||
assert mechanism["prefill"]["chunks"]["first"] == 1
|
||||
assert mechanism["prefix"]["hit_rate"] == 0.2
|
||||
assert all(mechanism["sanity"]["invariants"].values())
|
||||
|
||||
manifest = {
|
||||
"repetitions": {str(index): {} for index in (1, 2, 3)},
|
||||
"regimes": {
|
||||
"A": {
|
||||
"source": "a_base",
|
||||
"actions": {"mns": "shared", "mbbt": "a_mbbt"},
|
||||
},
|
||||
"B": {
|
||||
"source": "b_base",
|
||||
"actions": {"mns": "b_mns", "mbbt": "shared"},
|
||||
},
|
||||
},
|
||||
"gates": {
|
||||
"minimum_relative_winner_margin": 0.10,
|
||||
"minimum_exclusive_fraction": 0.10,
|
||||
"minimum_exclusive_ratio": 5.0,
|
||||
"material_kv_usage": 0.90,
|
||||
},
|
||||
}
|
||||
runs = []
|
||||
for repetition in (1, 2, 3):
|
||||
runs.extend(
|
||||
[
|
||||
fake_run(
|
||||
"a_base",
|
||||
repetition,
|
||||
goodput=1.0,
|
||||
mns_score=0.8,
|
||||
mbbt_score=0.01,
|
||||
),
|
||||
fake_run(
|
||||
"b_base",
|
||||
repetition,
|
||||
goodput=1.0,
|
||||
mns_score=0.01,
|
||||
mbbt_score=0.7,
|
||||
),
|
||||
fake_run("shared", repetition, goodput=3.0),
|
||||
fake_run("a_mbbt", repetition, goodput=1.5),
|
||||
fake_run("b_mns", repetition, goodput=1.2),
|
||||
]
|
||||
)
|
||||
result = analysis.evaluate_decisions(runs, manifest)
|
||||
assert result["decision"] == "STOP_NO_NEW_INSTRUMENTATION_NEEDED"
|
||||
assert result["baselines"] == {
|
||||
"always_mns_correct": 3,
|
||||
"always_mbbt_correct": 3,
|
||||
"binding_correct": 6,
|
||||
"decision_count": 6,
|
||||
}
|
||||
|
||||
ambiguous = copy.deepcopy(runs)
|
||||
for run in ambiguous:
|
||||
if run["config_id"] == "b_base":
|
||||
run["binding"]["both_fraction"] = 0.8
|
||||
assert (
|
||||
analysis.evaluate_decisions(ambiguous, manifest)["decision"]
|
||||
== "OPEN_EXACT_ATTRIBUTION_ABLATION"
|
||||
)
|
||||
|
||||
wrong = copy.deepcopy(runs)
|
||||
for run in wrong:
|
||||
if run["config_id"] == "b_base":
|
||||
run["binding"]["mns_exclusive_fraction"] = 0.8
|
||||
run["binding"]["mbbt_exclusive_fraction"] = 0.01
|
||||
for phase in run["phases"].values():
|
||||
phase["mns_exclusive_fraction"] = 0.8
|
||||
phase["mbbt_exclusive_fraction"] = 0.01
|
||||
assert (
|
||||
analysis.evaluate_decisions(wrong, manifest)["decision"]
|
||||
== "STOP_BINDING_NOT_PREDICTIVE"
|
||||
)
|
||||
|
||||
prepare = load("action_aware_prepare", "prepare_pilot.py")
|
||||
frozen = prepare.build(
|
||||
ROOT / "runs/intervention-response-v2/pilot-manifest-v3.json"
|
||||
)
|
||||
assert frozen["status"] == "PASS"
|
||||
assert frozen["sanity"]["red_flags"] == []
|
||||
assert [config["id"] for config in frozen["configs"]] == [
|
||||
"b_base",
|
||||
"a_base",
|
||||
"shared",
|
||||
"b_mns",
|
||||
"a_mbbt",
|
||||
]
|
||||
|
||||
controller = load("action_aware_controller", "pilot_controller.py")
|
||||
args = SimpleNamespace(
|
||||
manifest=Path("/tmp/manifest.json"),
|
||||
run_root=Path("/tmp/action-aware"),
|
||||
aituner_root=Path("/tmp/aituner"),
|
||||
vllm_source=Path("/tmp/vllm"),
|
||||
venv=Path("/tmp/venv"),
|
||||
model=Path("/tmp/model"),
|
||||
client=Path("/tmp/client.py"),
|
||||
)
|
||||
controller.configure(args, frozen)
|
||||
plan = controller.dry_run_plan(args, frozen)
|
||||
assert plan["status"] == "PASS"
|
||||
assert len(plan["sessions"]) == 5
|
||||
assert plan["projected_h20_hours"] == 7.0
|
||||
assert "--max-num-batched-tokens 256" in plan["sessions"][0]["commands"]["server"]
|
||||
revised = prepare.build(
|
||||
ROOT / "runs/intervention-response-v2/pilot-manifest-v3.json",
|
||||
token_source_mbbt=2048,
|
||||
prior_attempt_h20_hours=0.38598689953486126,
|
||||
prior_attempt_artifact="/tmp/operational-stop-v0.json",
|
||||
)
|
||||
assert revised["schema"] == "action-aware-constraint-pilot-manifest-v1"
|
||||
assert revised["configs"][0]["mbbt"] == 2048
|
||||
assert revised["configs"][3]["mbbt"] == 2048
|
||||
assert revised["budget"]["hard_cap_h20_hours"] < 8.0
|
||||
controller.configure(args, revised)
|
||||
revised_plan = controller.dry_run_plan(args, revised)
|
||||
assert revised_plan["projected_h20_hours"] < revised_plan["hard_cap_h20_hours"]
|
||||
assert (
|
||||
"--max-num-batched-tokens 2048"
|
||||
in revised_plan["sessions"][0]["commands"]["server"]
|
||||
)
|
||||
accepted_burnin = {
|
||||
"kind": "anchor",
|
||||
"selection": {"count": 510},
|
||||
"interval": {"elapsed_s": 61.25},
|
||||
"pass_rate": 0.5,
|
||||
"feasible": False,
|
||||
}
|
||||
assert controller.burnin_gate(
|
||||
accepted_burnin, expected_count=510, maximum_elapsed_s=90.0
|
||||
)["elapsed_s"] == 61.25
|
||||
warmup = copy.deepcopy(accepted_burnin)
|
||||
warmup["kind"] = "warmup"
|
||||
try:
|
||||
controller.burnin_gate(warmup, expected_count=510, maximum_elapsed_s=90.0)
|
||||
except RuntimeError as error:
|
||||
assert "non-anchor" in str(error)
|
||||
else:
|
||||
raise AssertionError("warmup incorrectly passed the burnin gate")
|
||||
slow = copy.deepcopy(accepted_burnin)
|
||||
slow["interval"]["elapsed_s"] = 91.0
|
||||
try:
|
||||
controller.burnin_gate(slow, expected_count=510, maximum_elapsed_s=90.0)
|
||||
except RuntimeError as error:
|
||||
assert "throughput gate failed" in str(error)
|
||||
else:
|
||||
raise AssertionError("slow burnin incorrectly passed the throughput gate")
|
||||
print("action-aware constraint pilot: PASS")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
325
runs/active-intervention-v0/analyze_prospective.py
Normal file
325
runs/active-intervention-v0/analyze_prospective.py
Normal file
@@ -0,0 +1,325 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Audit held-out action/measurement choices against the exact 2x2 surface."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import statistics
|
||||
from pathlib import Path
|
||||
from typing import Any, Mapping
|
||||
|
||||
|
||||
SCHEMA = "active-intervention-prospective-audit-v0"
|
||||
|
||||
|
||||
def sha256_file(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as source:
|
||||
for chunk in iter(lambda: source.read(1 << 20), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def atomic_json(path: Path, payload: Any) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary = path.with_suffix(path.suffix + ".tmp")
|
||||
temporary.write_text(
|
||||
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
||||
)
|
||||
os.replace(temporary, path)
|
||||
|
||||
|
||||
def numeric(values: list[float]) -> dict[str, Any]:
|
||||
finite = [float(value) for value in values]
|
||||
if not finite or any(not math.isfinite(value) for value in finite):
|
||||
raise ValueError("numeric summary requires finite values")
|
||||
return {
|
||||
"n": len(finite),
|
||||
"min": min(finite),
|
||||
"max": max(finite),
|
||||
"distinct_n": len(set(finite)),
|
||||
}
|
||||
|
||||
|
||||
def load_surface(
|
||||
manifest: Mapping[str, Any], run_root: Path
|
||||
) -> tuple[dict[str, Any], list[dict[str, Any]]]:
|
||||
rows = []
|
||||
aggregate = {}
|
||||
duration_s = float(manifest["engine"]["duration_s"])
|
||||
tp = int(manifest["engine"]["tp"])
|
||||
for config in manifest["configs"]:
|
||||
config_id = str(config["id"])
|
||||
values = []
|
||||
for repetition in sorted(int(key) for key in manifest["repetitions"]):
|
||||
expected = manifest["repetitions"][str(repetition)]["selection"]
|
||||
result_path = (
|
||||
run_root / "sessions" / config_id / f"rep{repetition}" / "result.json"
|
||||
)
|
||||
result = json.loads(result_path.read_text(encoding="utf-8"))
|
||||
if result["selection"]["request_id_order_sha256"] != expected[
|
||||
"request_id_order_sha256"
|
||||
]:
|
||||
raise ValueError(f"request hash mismatch: {config_id} rep{repetition}")
|
||||
offered_total = float(expected["offered_req_s_per_gpu"]) * tp
|
||||
normalized = float(result["slo_pass_count"]) / duration_s / offered_total
|
||||
values.append(normalized)
|
||||
rows.append(
|
||||
{
|
||||
"config_id": config_id,
|
||||
"mns": int(config["mns"]),
|
||||
"mbbt": int(config["mbbt"]),
|
||||
"repetition": repetition,
|
||||
"normalized_slo_goodput": normalized,
|
||||
"slo_goodput_req_s": float(result["slo_pass_count"]) / duration_s,
|
||||
"pass_rate": float(result["pass_rate"]),
|
||||
"elapsed_s": float(result["interval"]["elapsed_s"]),
|
||||
"result": str(result_path),
|
||||
"result_sha256": sha256_file(result_path),
|
||||
}
|
||||
)
|
||||
aggregate[config_id] = {
|
||||
"normalized_slo_goodput_values": values,
|
||||
"median_normalized_slo_goodput": float(statistics.median(values)),
|
||||
"sanity": numeric(values),
|
||||
}
|
||||
return aggregate, rows
|
||||
|
||||
|
||||
def source_cost_estimate(
|
||||
*,
|
||||
source_session: Mapping[str, Any],
|
||||
source_rows: list[Mapping[str, Any]],
|
||||
cutoff_s: float,
|
||||
tp: int,
|
||||
) -> dict[str, float]:
|
||||
actual_h20_hours = float(source_session["gpu_hours"])
|
||||
measured_replay_h20_hours = (
|
||||
tp * sum(float(row["elapsed_s"]) for row in source_rows) / 3600.0
|
||||
)
|
||||
fixed_h20_hours = max(0.0, actual_h20_hours - measured_replay_h20_hours)
|
||||
prefix_replay_h20_hours = tp * len(source_rows) * cutoff_s / 3600.0
|
||||
return {
|
||||
"actual_full_session_h20_hours": actual_h20_hours,
|
||||
"fixed_startup_warmup_burnin_cleanup_h20_hours": fixed_h20_hours,
|
||||
"prefix_replay_h20_hours_lower_bound": prefix_replay_h20_hours,
|
||||
"counterfactual_all_in_h20_hours_lower_bound": fixed_h20_hours
|
||||
+ prefix_replay_h20_hours,
|
||||
}
|
||||
|
||||
|
||||
def replay_policy(
|
||||
*,
|
||||
mode: str,
|
||||
manifest: Mapping[str, Any],
|
||||
decision: Mapping[str, Any],
|
||||
surface: Mapping[str, Any],
|
||||
session_costs: Mapping[str, float],
|
||||
source_cost: Mapping[str, float],
|
||||
) -> dict[str, Any]:
|
||||
acceptable_regret = float(manifest["gates"]["acceptable_regret"])
|
||||
source_id = str(manifest["source_config_id"])
|
||||
oracle = max(
|
||||
float(item["median_normalized_slo_goodput"]) for item in surface.values()
|
||||
)
|
||||
cumulative = float(source_cost["counterfactual_all_in_h20_hours_lower_bound"])
|
||||
source_score = float(surface[source_id]["median_normalized_slo_goodput"])
|
||||
source_regret = 1.0 - source_score / oracle if oracle > 0 else 0.0
|
||||
points = [
|
||||
{
|
||||
"action_id": "noop",
|
||||
"config_id": source_id,
|
||||
"score": source_score,
|
||||
"regret": source_regret,
|
||||
"cumulative_h20_hours_lower_bound": cumulative,
|
||||
}
|
||||
]
|
||||
hit = points[0] if source_regret <= acceptable_regret + 1e-12 else None
|
||||
seen = {source_id}
|
||||
for action_id in decision["decisions"][mode]["intervention_order"]:
|
||||
config_id = str(manifest["actions"][action_id])
|
||||
if config_id in seen:
|
||||
continue
|
||||
seen.add(config_id)
|
||||
cumulative += float(session_costs[config_id])
|
||||
score = float(surface[config_id]["median_normalized_slo_goodput"])
|
||||
regret = 1.0 - score / oracle if oracle > 0 else 0.0
|
||||
point = {
|
||||
"action_id": action_id,
|
||||
"config_id": config_id,
|
||||
"score": score,
|
||||
"regret": regret,
|
||||
"cumulative_h20_hours_lower_bound": cumulative,
|
||||
}
|
||||
points.append(point)
|
||||
if hit is None and regret <= acceptable_regret + 1e-12:
|
||||
hit = point
|
||||
return {
|
||||
"mode": mode,
|
||||
"measurement_cutoff_s": float(
|
||||
decision["decisions"][mode]["selected_cutoff_s"]
|
||||
),
|
||||
"selected_action": decision["decisions"][mode]["selected_action"],
|
||||
"decision_kind": decision["decisions"][mode]["decision_kind"],
|
||||
"intervention_order": decision["decisions"][mode]["intervention_order"],
|
||||
"source_cost": dict(source_cost),
|
||||
"oracle_normalized_slo_goodput": oracle,
|
||||
"cost_to_acceptable": hit,
|
||||
"reached_acceptable": hit is not None,
|
||||
"points": points,
|
||||
}
|
||||
|
||||
|
||||
def build_audit(
|
||||
*, manifest_path: Path, decision_path: Path, run_root: Path
|
||||
) -> dict[str, Any]:
|
||||
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
||||
decision = json.loads(decision_path.read_text(encoding="utf-8"))
|
||||
state_path = run_root / "controller-state.json"
|
||||
state = json.loads(state_path.read_text(encoding="utf-8"))
|
||||
if manifest.get("schema") != "active-intervention-prospective-manifest-v0":
|
||||
raise ValueError("unexpected prospective manifest schema")
|
||||
if decision.get("schema") != "active-intervention-prospective-decision-v0":
|
||||
raise ValueError("unexpected prospective decision schema")
|
||||
if decision["manifest_sha256"] != sha256_file(manifest_path):
|
||||
raise ValueError("decision does not match prospective manifest")
|
||||
surface, rows = load_surface(manifest, run_root)
|
||||
source_id = str(manifest["source_config_id"])
|
||||
sessions = state["sessions"]
|
||||
session_costs = {
|
||||
config_id: float(sessions[config_id]["gpu_hours"])
|
||||
for config_id in surface
|
||||
}
|
||||
source_rows = [row for row in rows if row["config_id"] == source_id]
|
||||
policies = {}
|
||||
for mode in ("outcome_only", "telemetry"):
|
||||
cost = source_cost_estimate(
|
||||
source_session=sessions[source_id],
|
||||
source_rows=source_rows,
|
||||
cutoff_s=float(decision["decisions"][mode]["selected_cutoff_s"]),
|
||||
tp=int(manifest["engine"]["tp"]),
|
||||
)
|
||||
policies[mode] = replay_policy(
|
||||
mode=mode,
|
||||
manifest=manifest,
|
||||
decision=decision,
|
||||
surface=surface,
|
||||
session_costs=session_costs,
|
||||
source_cost=cost,
|
||||
)
|
||||
outcome_hit = policies["outcome_only"]["cost_to_acceptable"]
|
||||
telemetry_hit = policies["telemetry"]["cost_to_acceptable"]
|
||||
if outcome_hit is None or telemetry_hit is None:
|
||||
reduction = None
|
||||
else:
|
||||
outcome_cost = float(outcome_hit["cumulative_h20_hours_lower_bound"])
|
||||
telemetry_cost = float(telemetry_hit["cumulative_h20_hours_lower_bound"])
|
||||
reduction = 1.0 - telemetry_cost / outcome_cost if outcome_cost > 0 else 0.0
|
||||
confirmation_trigger = bool(
|
||||
reduction is not None
|
||||
and reduction
|
||||
>= float(manifest["gates"]["confirmation_trigger_gpu_cost_reduction"])
|
||||
and policies["telemetry"]["reached_acceptable"]
|
||||
)
|
||||
contribution_gate = bool(
|
||||
reduction is not None
|
||||
and reduction >= float(manifest["gates"]["contribution_gpu_cost_reduction"])
|
||||
and policies["telemetry"]["reached_acceptable"]
|
||||
)
|
||||
status = (
|
||||
"TRIGGER_ACTUAL_EARLY_STOP_CONFIRMATION"
|
||||
if confirmation_trigger
|
||||
else "STOP_NO_PROSPECTIVE_GPU_COST_SIGNAL"
|
||||
)
|
||||
normalized_values = [float(row["normalized_slo_goodput"]) for row in rows]
|
||||
costs = list(session_costs.values())
|
||||
invariants = {
|
||||
"controller_complete": state.get("status") == "complete",
|
||||
"four_sessions_complete": len(sessions) == 4
|
||||
and all(item.get("status") == "complete" for item in sessions.values()),
|
||||
"twelve_surface_outcomes": len(rows) == 12,
|
||||
"nonnegative_goodput": all(value >= 0.0 for value in normalized_values),
|
||||
"normalized_goodput_bounded": all(value <= 1.0 + 1e-12 for value in normalized_values),
|
||||
"surface_not_all_identical": len(set(normalized_values)) > 1,
|
||||
"nonnegative_session_costs": all(value >= 0.0 for value in costs),
|
||||
"policy_replay_reaches_oracle_surface": all(
|
||||
policy["reached_acceptable"] for policy in policies.values()
|
||||
),
|
||||
}
|
||||
red_flags = [name for name, passed in invariants.items() if not passed]
|
||||
if red_flags:
|
||||
status = "STOP_SANITY"
|
||||
return {
|
||||
"schema": SCHEMA,
|
||||
"status": status,
|
||||
"claim_boundary": (
|
||||
"Prospective exact-surface replay. Prefix source costs reconstruct the "
|
||||
"measured fixed overhead plus selected replay seconds; actual early-stop "
|
||||
"confirmation is required before claiming GPU-cost reduction."
|
||||
),
|
||||
"manifest": str(manifest_path),
|
||||
"manifest_sha256": sha256_file(manifest_path),
|
||||
"decision": str(decision_path),
|
||||
"decision_sha256": sha256_file(decision_path),
|
||||
"controller_state": str(state_path),
|
||||
"controller_state_sha256": sha256_file(state_path),
|
||||
"surface": surface,
|
||||
"rows": rows,
|
||||
"session_costs_h20_hours": session_costs,
|
||||
"annotation_campaign_h20_hours": float(state["gpu_hours_total"]),
|
||||
"policies": policies,
|
||||
"comparison": {
|
||||
"telemetry_gpu_cost_reduction_fraction": reduction,
|
||||
"confirmation_trigger": confirmation_trigger,
|
||||
"contribution_gate": contribution_gate,
|
||||
"confirmation_trigger_threshold": manifest["gates"][
|
||||
"confirmation_trigger_gpu_cost_reduction"
|
||||
],
|
||||
"contribution_threshold": manifest["gates"][
|
||||
"contribution_gpu_cost_reduction"
|
||||
],
|
||||
"action_changed": policies["outcome_only"]["selected_action"]
|
||||
!= policies["telemetry"]["selected_action"],
|
||||
"measurement_changed": policies["outcome_only"]["measurement_cutoff_s"]
|
||||
!= policies["telemetry"]["measurement_cutoff_s"],
|
||||
},
|
||||
"sanity": {
|
||||
"invariants": invariants,
|
||||
"red_flags": red_flags,
|
||||
"normalized_slo_goodput": numeric(normalized_values),
|
||||
"session_h20_hours": numeric(costs),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--manifest", type=Path, required=True)
|
||||
parser.add_argument("--decision", type=Path, required=True)
|
||||
parser.add_argument("--run-root", type=Path, required=True)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
audit = build_audit(
|
||||
manifest_path=args.manifest,
|
||||
decision_path=args.decision,
|
||||
run_root=args.run_root,
|
||||
)
|
||||
atomic_json(args.output, audit)
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"status": audit["status"],
|
||||
"comparison": audit["comparison"],
|
||||
"sanity": audit["sanity"],
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
324
runs/active-intervention-v0/extract_training.py
Normal file
324
runs/active-intervention-v0/extract_training.py
Normal file
@@ -0,0 +1,324 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Extract paired source/action examples from the accepted action-aware run."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from statistics import fmean
|
||||
from typing import Any, Mapping
|
||||
|
||||
|
||||
PHASES = ("0.25", "0.50", "0.75", "1.00")
|
||||
HERE = Path(__file__).resolve().parent
|
||||
COMMON_STATE = HERE.parent / "telemetry-residual"
|
||||
sys.path.insert(0, str(COMMON_STATE))
|
||||
|
||||
from common_state import summarize_engine # noqa: E402
|
||||
|
||||
|
||||
def sha256_file(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as source:
|
||||
for chunk in iter(lambda: source.read(1 << 20), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def atomic_json(path: Path, payload: Any) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary = path.with_suffix(path.suffix + ".tmp")
|
||||
temporary.write_text(
|
||||
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
||||
)
|
||||
os.replace(temporary, path)
|
||||
|
||||
|
||||
def load_jsonl(path: Path) -> list[dict[str, Any]]:
|
||||
records = []
|
||||
with path.open(encoding="utf-8") as source:
|
||||
for line_number, line in enumerate(source, 1):
|
||||
if not line.strip():
|
||||
continue
|
||||
try:
|
||||
records.append(json.loads(line))
|
||||
except json.JSONDecodeError as error:
|
||||
raise ValueError(f"{path}:{line_number}: invalid JSON") from error
|
||||
if not records:
|
||||
raise ValueError(f"{path}: no request records")
|
||||
return records
|
||||
|
||||
|
||||
def prefix_outcome(
|
||||
requests: list[Mapping[str, Any]], *, cutoff_s: float, offered_total: float
|
||||
) -> dict[str, float]:
|
||||
admitted = [request for request in requests if float(request["arrival_s"]) <= cutoff_s]
|
||||
completed = [
|
||||
request
|
||||
for request in requests
|
||||
if request.get("completed_elapsed_s") is not None
|
||||
and float(request["completed_elapsed_s"]) <= cutoff_s
|
||||
]
|
||||
if not admitted:
|
||||
raise ValueError("prefix has no admitted requests")
|
||||
admitted_ids = {str(request["request_id"]) for request in admitted}
|
||||
if any(str(request["request_id"]) not in admitted_ids for request in completed):
|
||||
raise ValueError("prefix completion precedes admission")
|
||||
passed = sum(bool(request["slo_pass"]) for request in completed)
|
||||
ttft = [float(request["ttft_ms"]) for request in completed]
|
||||
tpot = [float(request["tpot_ms"]) for request in completed]
|
||||
total = len(requests)
|
||||
return {
|
||||
"normalized_slo_goodput": passed / cutoff_s / offered_total,
|
||||
"admitted_fraction": len(admitted) / total,
|
||||
"completed_over_admitted": len(completed) / len(admitted),
|
||||
"completed_pass_rate": passed / max(1, len(completed)),
|
||||
"completed_fail_fraction_of_total": (len(completed) - passed) / total,
|
||||
"outstanding_over_admitted": (len(admitted) - len(completed)) / len(admitted),
|
||||
"ttft_max_over_slo_max": max(ttft, default=0.0) / 6000.0,
|
||||
"ttft_mean_over_slo_max": fmean(ttft) / 6000.0 if ttft else 0.0,
|
||||
"tpot_max_over_slo": max(tpot, default=0.0) / 50.0,
|
||||
"tpot_mean_over_slo": fmean(tpot) / 50.0 if tpot else 0.0,
|
||||
"admitted_input_tokens_mean_over_limit": fmean(
|
||||
float(request["raw_input_tokens"]) for request in admitted
|
||||
)
|
||||
/ 8192.0,
|
||||
}
|
||||
|
||||
|
||||
def telemetry_record(state: Mapping[str, Any]) -> dict[str, float]:
|
||||
common = state["common"]
|
||||
engine = state["engine_only"]
|
||||
executed_steps = int(state["sanity"]["executed_steps"])
|
||||
if executed_steps <= 0:
|
||||
raise ValueError("telemetry phase contains no executed engine steps")
|
||||
return {
|
||||
"scheduler_steps_per_s": float(common["scheduler_steps_per_s"]),
|
||||
"batch_size_mean": float(common["batch_size"]["mean"]),
|
||||
"batch_size_cv": float(common["batch_size"]["cv"]),
|
||||
"batch_tokens_mean": float(common["batch_tokens"]["mean"]),
|
||||
"batch_tokens_cv": float(common["batch_tokens"]["cv"]),
|
||||
"decode_batch_size_mean": float(common["decode_batch_size"]["mean"]),
|
||||
"decode_batch_size_cv": float(common["decode_batch_size"]["cv"]),
|
||||
"prefill_token_fraction": float(common["prefill_token_fraction"]),
|
||||
"queue_waiting_mean": float(common["queue_waiting_mean"]),
|
||||
"queue_running_mean": float(common["queue_running_mean"]),
|
||||
"preemptions_per_step": float(common["preemptions"]) / executed_steps,
|
||||
"kv_usage_mean": float(engine["kv_usage_mean"]),
|
||||
"kv_usage_max": float(engine["kv_usage_max"]),
|
||||
"kv_usage_end_minus_start": float(engine["kv_usage_end_minus_start"]),
|
||||
"graph_none_share": float(engine["graph_none_share"]),
|
||||
"graph_full_share": float(engine["graph_full_share"]),
|
||||
"graph_padding_fraction": float(engine["graph_padding_fraction"]),
|
||||
}
|
||||
|
||||
|
||||
def load_stream(path: Path, *, expected_sha256: str) -> list[dict[str, Any]]:
|
||||
if sha256_file(path) != expected_sha256:
|
||||
raise ValueError(f"engine stream hash mismatch: {path}")
|
||||
decoded = load_jsonl(path)
|
||||
records = [row for row in decoded if "step_index" in row]
|
||||
if not records:
|
||||
raise ValueError(f"engine stream has no Layer-1 records: {path}")
|
||||
return records
|
||||
|
||||
|
||||
def build_dataset(
|
||||
*, audit_path: Path, manifest_path: Path, run_root: Path
|
||||
) -> dict[str, Any]:
|
||||
audit = json.loads(audit_path.read_text(encoding="utf-8"))
|
||||
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
||||
if audit.get("schema") != "action-aware-constraint-pilot-audit-v0":
|
||||
raise ValueError("unexpected action-aware audit schema")
|
||||
if audit["sanity"]["red_flags"]:
|
||||
raise ValueError(f"action-aware audit red flags: {audit['sanity']['red_flags']}")
|
||||
configs = {str(item["id"]): item for item in manifest["configs"]}
|
||||
runs = {
|
||||
(str(run["config_id"]), int(run["repetition"])): run
|
||||
for run in audit["runs"]
|
||||
}
|
||||
source_ids = {str(regime["source"]) for regime in manifest["regimes"].values()}
|
||||
stream_entries = {
|
||||
str(item["config_id"]): item
|
||||
for item in audit["streams"]
|
||||
if str(item["config_id"]) in source_ids
|
||||
}
|
||||
if set(stream_entries) != source_ids:
|
||||
raise ValueError("audit is missing a source config engine stream")
|
||||
streams = {
|
||||
config_id: load_stream(
|
||||
Path(item["stream"]), expected_sha256=str(item["stream_sha256"])
|
||||
)
|
||||
for config_id, item in stream_entries.items()
|
||||
}
|
||||
examples = []
|
||||
request_hashes = []
|
||||
for regime_name, regime in sorted(manifest["regimes"].items()):
|
||||
source_id = str(regime["source"])
|
||||
for repetition in sorted(int(value) for value in manifest["repetitions"]):
|
||||
source_run = runs[(source_id, repetition)]
|
||||
source_config = configs[source_id]
|
||||
request_path = run_root / "sessions" / source_id / f"rep{repetition}" / "requests.jsonl"
|
||||
requests = load_jsonl(request_path)
|
||||
request_hashes.append(sha256_file(request_path))
|
||||
offered_rate_per_gpu = float(
|
||||
manifest["repetitions"][str(repetition)]["selection"][
|
||||
"offered_req_s_per_gpu"
|
||||
]
|
||||
)
|
||||
offered_total = offered_rate_per_gpu * int(manifest["engine"]["tp"])
|
||||
source_goodput = float(source_run["outcome"]["slo_goodput_req_s"])
|
||||
source_normalized = min(1.0, source_goodput / offered_total)
|
||||
decision_id = f"{regime_name}-rep{repetition}"
|
||||
for phase in PHASES:
|
||||
cutoff_s = float(manifest["engine"]["duration_s"]) * float(phase)
|
||||
outcome = prefix_outcome(
|
||||
requests, cutoff_s=cutoff_s, offered_total=offered_total
|
||||
)
|
||||
admitted_count = sum(
|
||||
float(request["arrival_s"]) <= cutoff_s for request in requests
|
||||
)
|
||||
start_ns = int(source_run["state"]["interval"]["start_ns"])
|
||||
phase_state = summarize_engine(
|
||||
streams[source_id],
|
||||
start_ns=start_ns,
|
||||
end_ns=start_ns + round(cutoff_s * 1e9),
|
||||
request_count=admitted_count,
|
||||
)
|
||||
if not all(phase_state["sanity"]["invariants"].values()):
|
||||
raise ValueError(
|
||||
f"engine state invariant failed: {decision_id} phase {phase}"
|
||||
)
|
||||
telemetry = telemetry_record(phase_state)
|
||||
actions = {"noop": source_id, **regime["actions"]}
|
||||
for action_name, target_id in sorted(actions.items()):
|
||||
target_run = runs[(str(target_id), repetition)]
|
||||
target_config = configs[str(target_id)]
|
||||
target_goodput = float(target_run["outcome"]["slo_goodput_req_s"])
|
||||
normalized = target_goodput / offered_total
|
||||
if not 0.0 <= normalized <= 1.0 + 1e-12:
|
||||
raise ValueError("target normalized goodput is outside [0, 1]")
|
||||
examples.append(
|
||||
{
|
||||
"phase": phase,
|
||||
"cutoff_s": cutoff_s,
|
||||
"decision_id": decision_id,
|
||||
"regime": regime_name,
|
||||
"repetition": repetition,
|
||||
"source": {
|
||||
"config_id": source_id,
|
||||
"mns": int(source_config["mns"]),
|
||||
"mbbt": int(source_config["mbbt"]),
|
||||
"offered_rate_per_gpu": offered_rate_per_gpu,
|
||||
"outcome": outcome,
|
||||
"telemetry": telemetry,
|
||||
},
|
||||
"action": {
|
||||
"id": action_name,
|
||||
"target_config_id": str(target_id),
|
||||
"target_mns": int(target_config["mns"]),
|
||||
"target_mbbt": int(target_config["mbbt"]),
|
||||
},
|
||||
"target_slo_goodput_req_s": target_goodput,
|
||||
"target_normalized_goodput": min(1.0, normalized),
|
||||
"source_normalized_goodput": source_normalized,
|
||||
"target_delta_normalized_goodput": min(1.0, normalized)
|
||||
- source_normalized,
|
||||
}
|
||||
)
|
||||
invariants = {
|
||||
"expected_examples": len(examples) == len(PHASES) * 2 * 3 * 3,
|
||||
"four_phases": sorted({example["phase"] for example in examples})
|
||||
== sorted(PHASES),
|
||||
"six_decisions": len({example["decision_id"] for example in examples}) == 6,
|
||||
"three_actions_per_decision_phase": all(
|
||||
sum(
|
||||
item["decision_id"] == decision
|
||||
and item["phase"] == phase
|
||||
for item in examples
|
||||
)
|
||||
== 3
|
||||
for decision in {item["decision_id"] for item in examples}
|
||||
for phase in PHASES
|
||||
),
|
||||
"targets_not_all_identical": len(
|
||||
{example["target_normalized_goodput"] for example in examples}
|
||||
)
|
||||
> 1,
|
||||
"bounded_prefix_ratios": all(
|
||||
0.0 <= float(value) <= 1.0
|
||||
for example in examples
|
||||
for key, value in example["source"]["outcome"].items()
|
||||
if key
|
||||
in {
|
||||
"admitted_fraction",
|
||||
"completed_over_admitted",
|
||||
"completed_pass_rate",
|
||||
"completed_fail_fraction_of_total",
|
||||
"outstanding_over_admitted",
|
||||
}
|
||||
),
|
||||
"direct_telemetry_without_binding_labels": all(
|
||||
not any(token in key for token in ("exclusive", "unresolved", "both"))
|
||||
for example in examples
|
||||
for key in example["source"]["telemetry"]
|
||||
),
|
||||
"treatment_effects_bounded": all(
|
||||
-1.0 <= float(example["target_delta_normalized_goodput"]) <= 1.0
|
||||
for example in examples
|
||||
),
|
||||
}
|
||||
red_flags = [name for name, passed in invariants.items() if not passed]
|
||||
if red_flags:
|
||||
raise RuntimeError(f"training dataset sanity failed: {red_flags}")
|
||||
return {
|
||||
"schema": "active-intervention-training-v0",
|
||||
"status": "VALID",
|
||||
"provenance": {
|
||||
"audit": str(audit_path),
|
||||
"audit_sha256": sha256_file(audit_path),
|
||||
"manifest": str(manifest_path),
|
||||
"manifest_sha256": sha256_file(manifest_path),
|
||||
"run_root": str(run_root),
|
||||
"source_request_sha256": sorted(set(request_hashes)),
|
||||
"source_stream_sha256": sorted(
|
||||
str(item["stream_sha256"]) for item in stream_entries.values()
|
||||
),
|
||||
},
|
||||
"examples": examples,
|
||||
"sanity": {"invariants": invariants, "red_flags": red_flags},
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--audit", type=Path, required=True)
|
||||
parser.add_argument("--manifest", type=Path, required=True)
|
||||
parser.add_argument("--run-root", type=Path, required=True)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
dataset = build_dataset(
|
||||
audit_path=args.audit,
|
||||
manifest_path=args.manifest,
|
||||
run_root=args.run_root,
|
||||
)
|
||||
atomic_json(args.output, dataset)
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"status": dataset["status"],
|
||||
"examples": len(dataset["examples"]),
|
||||
"sanity": dataset["sanity"],
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
287
runs/active-intervention-v0/model.py
Normal file
287
runs/active-intervention-v0/model.py
Normal file
@@ -0,0 +1,287 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Small-data action-response model for the active intervention pilot.
|
||||
|
||||
The model predicts the paired normalized SLO-goodput treatment effect from a
|
||||
source measurement and a full MNS/MBBT action. Telemetry features are direct,
|
||||
continuous engine measurements; there is no diagnosis-to-action rule here.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Iterable, Mapping, Sequence
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
PREFIX_FEATURES = (
|
||||
"normalized_slo_goodput",
|
||||
"admitted_fraction",
|
||||
"completed_over_admitted",
|
||||
"completed_pass_rate",
|
||||
"completed_fail_fraction_of_total",
|
||||
"outstanding_over_admitted",
|
||||
"ttft_max_over_slo_max",
|
||||
"ttft_mean_over_slo_max",
|
||||
"tpot_max_over_slo",
|
||||
"tpot_mean_over_slo",
|
||||
"admitted_input_tokens_mean_over_limit",
|
||||
)
|
||||
|
||||
TELEMETRY_FEATURES = (
|
||||
"scheduler_steps_per_s",
|
||||
"batch_size_mean",
|
||||
"batch_size_cv",
|
||||
"batch_tokens_mean",
|
||||
"batch_tokens_cv",
|
||||
"decode_batch_size_mean",
|
||||
"decode_batch_size_cv",
|
||||
"prefill_token_fraction",
|
||||
"queue_waiting_mean",
|
||||
"queue_running_mean",
|
||||
"preemptions_per_step",
|
||||
"kv_usage_mean",
|
||||
"kv_usage_max",
|
||||
"kv_usage_end_minus_start",
|
||||
"graph_none_share",
|
||||
"graph_full_share",
|
||||
"graph_padding_fraction",
|
||||
)
|
||||
|
||||
|
||||
def _finite(value: Any, name: str) -> float:
|
||||
result = float(value)
|
||||
if not math.isfinite(result):
|
||||
raise ValueError(f"{name} must be finite")
|
||||
return result
|
||||
|
||||
|
||||
def feature_vector(
|
||||
example: Mapping[str, Any], *, include_telemetry: bool
|
||||
) -> tuple[list[str], np.ndarray]:
|
||||
source = example["source"]
|
||||
action = example["action"]
|
||||
source_log_mns = math.log2(_finite(source["mns"], "source MNS"))
|
||||
source_log_mbbt = math.log2(_finite(source["mbbt"], "source MBBT"))
|
||||
target_log_mns = math.log2(_finite(action["target_mns"], "target MNS"))
|
||||
target_log_mbbt = math.log2(_finite(action["target_mbbt"], "target MBBT"))
|
||||
delta_mns = target_log_mns - source_log_mns
|
||||
delta_mbbt = target_log_mbbt - source_log_mbbt
|
||||
names = [
|
||||
"source_log2_mns",
|
||||
"source_log2_mbbt",
|
||||
"target_log2_mns",
|
||||
"target_log2_mbbt",
|
||||
"delta_log2_mns",
|
||||
"delta_log2_mbbt",
|
||||
"delta_product",
|
||||
"offered_rate_per_gpu",
|
||||
]
|
||||
values = [
|
||||
source_log_mns,
|
||||
source_log_mbbt,
|
||||
target_log_mns,
|
||||
target_log_mbbt,
|
||||
delta_mns,
|
||||
delta_mbbt,
|
||||
delta_mns * delta_mbbt,
|
||||
_finite(source["offered_rate_per_gpu"], "offered rate"),
|
||||
]
|
||||
for name in PREFIX_FEATURES:
|
||||
names.append(f"outcome.{name}")
|
||||
values.append(_finite(source["outcome"][name], name))
|
||||
if include_telemetry:
|
||||
for name in TELEMETRY_FEATURES:
|
||||
value = _finite(source["telemetry"][name], name)
|
||||
names.extend(
|
||||
(
|
||||
f"telemetry.{name}",
|
||||
f"telemetry.{name}*delta_mns",
|
||||
f"telemetry.{name}*delta_mbbt",
|
||||
)
|
||||
)
|
||||
values.extend((value, value * delta_mns, value * delta_mbbt))
|
||||
vector = np.asarray(values, dtype=np.float64)
|
||||
if not np.all(np.isfinite(vector)):
|
||||
raise ValueError("feature vector contains a non-finite value")
|
||||
return names, vector
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RidgeModel:
|
||||
feature_names: tuple[str, ...]
|
||||
mean: np.ndarray
|
||||
scale: np.ndarray
|
||||
weights: np.ndarray
|
||||
intercept: float
|
||||
regularization: float
|
||||
|
||||
def predict(self, values: np.ndarray) -> float:
|
||||
if values.shape != self.mean.shape:
|
||||
raise ValueError("ridge prediction feature shape mismatch")
|
||||
normalized = (values - self.mean) / self.scale
|
||||
return float(self.intercept + normalized @ self.weights)
|
||||
|
||||
def to_json(self) -> dict[str, Any]:
|
||||
return {
|
||||
"feature_names": list(self.feature_names),
|
||||
"mean": self.mean.tolist(),
|
||||
"scale": self.scale.tolist(),
|
||||
"weights": self.weights.tolist(),
|
||||
"intercept": self.intercept,
|
||||
"regularization": self.regularization,
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, payload: Mapping[str, Any]) -> "RidgeModel":
|
||||
return cls(
|
||||
feature_names=tuple(str(value) for value in payload["feature_names"]),
|
||||
mean=np.asarray(payload["mean"], dtype=np.float64),
|
||||
scale=np.asarray(payload["scale"], dtype=np.float64),
|
||||
weights=np.asarray(payload["weights"], dtype=np.float64),
|
||||
intercept=float(payload["intercept"]),
|
||||
regularization=float(payload["regularization"]),
|
||||
)
|
||||
|
||||
|
||||
def fit_ridge(
|
||||
examples: Sequence[Mapping[str, Any]],
|
||||
*,
|
||||
include_telemetry: bool,
|
||||
regularization: float,
|
||||
) -> RidgeModel:
|
||||
if not examples:
|
||||
raise ValueError("ridge fit requires examples")
|
||||
if regularization <= 0:
|
||||
raise ValueError("ridge regularization must be positive")
|
||||
encoded = [
|
||||
feature_vector(example, include_telemetry=include_telemetry)
|
||||
for example in examples
|
||||
]
|
||||
names = encoded[0][0]
|
||||
if any(item[0] != names for item in encoded):
|
||||
raise ValueError("feature names changed across examples")
|
||||
x = np.stack([item[1] for item in encoded])
|
||||
y = np.asarray(
|
||||
[
|
||||
_finite(example["target_delta_normalized_goodput"], "target effect")
|
||||
for example in examples
|
||||
],
|
||||
dtype=np.float64,
|
||||
)
|
||||
mean = x.mean(axis=0)
|
||||
scale = x.std(axis=0)
|
||||
scale[scale < 1e-12] = 1.0
|
||||
normalized = (x - mean) / scale
|
||||
intercept = float(y.mean())
|
||||
centered = y - intercept
|
||||
system = normalized.T @ normalized + regularization * np.eye(x.shape[1])
|
||||
weights = np.linalg.solve(system, normalized.T @ centered)
|
||||
return RidgeModel(
|
||||
feature_names=tuple(names),
|
||||
mean=mean,
|
||||
scale=scale,
|
||||
weights=weights,
|
||||
intercept=intercept,
|
||||
regularization=regularization,
|
||||
)
|
||||
|
||||
|
||||
def fit_jackknife_ensemble(
|
||||
examples: Sequence[Mapping[str, Any]],
|
||||
*,
|
||||
include_telemetry: bool,
|
||||
regularization: float,
|
||||
group_key: str = "decision_id",
|
||||
) -> list[RidgeModel]:
|
||||
groups = sorted({str(example[group_key]) for example in examples})
|
||||
if len(groups) < 3:
|
||||
raise ValueError("jackknife ensemble requires at least three groups")
|
||||
models = []
|
||||
for held_out in groups:
|
||||
training = [
|
||||
example for example in examples if str(example[group_key]) != held_out
|
||||
]
|
||||
models.append(
|
||||
fit_ridge(
|
||||
training,
|
||||
include_telemetry=include_telemetry,
|
||||
regularization=regularization,
|
||||
)
|
||||
)
|
||||
return models
|
||||
|
||||
|
||||
def ensemble_predict(
|
||||
models: Sequence[RidgeModel],
|
||||
example: Mapping[str, Any],
|
||||
*,
|
||||
include_telemetry: bool,
|
||||
) -> dict[str, float]:
|
||||
if not models:
|
||||
raise ValueError("ensemble prediction requires models")
|
||||
source = example["source"]
|
||||
action = example["action"]
|
||||
if (
|
||||
int(action["target_mns"]) == int(source["mns"])
|
||||
and int(action["target_mbbt"]) == int(source["mbbt"])
|
||||
):
|
||||
return {"mean": 0.0, "std": 0.0, "min": 0.0, "max": 0.0, "distinct_n": 1}
|
||||
names, values = feature_vector(example, include_telemetry=include_telemetry)
|
||||
if any(model.feature_names != tuple(names) for model in models):
|
||||
raise ValueError("ensemble feature schema mismatch")
|
||||
raw = np.asarray([model.predict(values) for model in models], dtype=np.float64)
|
||||
clipped = np.clip(raw, -1.0, 1.0)
|
||||
return {
|
||||
"mean": float(clipped.mean()),
|
||||
"std": float(clipped.std(ddof=0)),
|
||||
"min": float(clipped.min()),
|
||||
"max": float(clipped.max()),
|
||||
"distinct_n": len(set(float(value) for value in clipped)),
|
||||
}
|
||||
|
||||
|
||||
def select_action(
|
||||
models: Sequence[RidgeModel],
|
||||
candidates: Sequence[Mapping[str, Any]],
|
||||
*,
|
||||
include_telemetry: bool,
|
||||
confidence_z: float = 1.0,
|
||||
minimum_margin: float = 0.02,
|
||||
) -> dict[str, Any]:
|
||||
if len(candidates) < 2:
|
||||
raise ValueError("action selection requires at least two candidates")
|
||||
rows = []
|
||||
for example in candidates:
|
||||
prediction = ensemble_predict(
|
||||
models, example, include_telemetry=include_telemetry
|
||||
)
|
||||
rows.append(
|
||||
{
|
||||
"action_id": str(example["action"]["id"]),
|
||||
"prediction": prediction,
|
||||
"lower": prediction["mean"] - confidence_z * prediction["std"],
|
||||
"upper": prediction["mean"] + confidence_z * prediction["std"],
|
||||
}
|
||||
)
|
||||
rows.sort(key=lambda row: (-row["prediction"]["mean"], row["action_id"]))
|
||||
best, second = rows[:2]
|
||||
margin = float(best["prediction"]["mean"] - second["prediction"]["mean"])
|
||||
confident = bool(
|
||||
margin >= minimum_margin and best["lower"] > second["upper"]
|
||||
)
|
||||
return {
|
||||
"selected_action": best["action_id"],
|
||||
"confident": confident,
|
||||
"predicted_margin": margin,
|
||||
"candidates": rows,
|
||||
}
|
||||
|
||||
|
||||
def models_to_json(models: Iterable[RidgeModel]) -> list[dict[str, Any]]:
|
||||
return [model.to_json() for model in models]
|
||||
|
||||
|
||||
def models_from_json(payload: Iterable[Mapping[str, Any]]) -> list[RidgeModel]:
|
||||
return [RidgeModel.from_json(item) for item in payload]
|
||||
363
runs/active-intervention-v0/prepare_prospective.py
Normal file
363
runs/active-intervention-v0/prepare_prospective.py
Normal file
@@ -0,0 +1,363 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Freeze the unseen-trace 2x2 active intervention development surface."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
AITUNER_ROOT = Path(os.environ.get("AITUNER_ROOT", Path(__file__).resolve().parents[2]))
|
||||
sys.path.insert(0, str(AITUNER_ROOT / "src"))
|
||||
|
||||
from aituner.spec import load_study_spec # noqa: E402
|
||||
from aituner.trace import load_trace_requests, select_requests_for_threshold # noqa: E402
|
||||
|
||||
|
||||
SCHEMA = "active-intervention-prospective-manifest-v0"
|
||||
TP = 4
|
||||
REPETITIONS = (1, 2, 3)
|
||||
DURATION_S = 300.0
|
||||
REPLAY_TIME_SCALE = 0.5
|
||||
OFFERED_RATE_PER_GPU = 2.75
|
||||
TARGET_COUNT = round(OFFERED_RATE_PER_GPU * DURATION_S * TP)
|
||||
WINDOW_ID = "chat_w20260313_1000"
|
||||
ENGINE_VERSION = "0.24.1.dev3+g668cfb7e2"
|
||||
|
||||
|
||||
def atomic_json(path: Path, payload: Any) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary = path.with_suffix(path.suffix + ".tmp")
|
||||
temporary.write_text(
|
||||
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
||||
)
|
||||
os.replace(temporary, path)
|
||||
|
||||
|
||||
def sha256_file(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as source:
|
||||
for chunk in iter(lambda: source.read(1 << 20), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def order_hash(values: list[str]) -> str:
|
||||
return hashlib.sha256("\n".join(values).encode()).hexdigest()
|
||||
|
||||
|
||||
def configs() -> list[dict[str, Any]]:
|
||||
return [
|
||||
{
|
||||
"id": "source_mns32_mbbt4096",
|
||||
"mns": 32,
|
||||
"mbbt": 4096,
|
||||
"repetition_order": [1, 2, 3],
|
||||
},
|
||||
{
|
||||
"id": "mns64_mbbt4096",
|
||||
"mns": 64,
|
||||
"mbbt": 4096,
|
||||
"repetition_order": [2, 3, 1],
|
||||
},
|
||||
{
|
||||
"id": "mns32_mbbt8192",
|
||||
"mns": 32,
|
||||
"mbbt": 8192,
|
||||
"repetition_order": [3, 1, 2],
|
||||
},
|
||||
{
|
||||
"id": "joint_mns64_mbbt8192",
|
||||
"mns": 64,
|
||||
"mbbt": 8192,
|
||||
"repetition_order": [1, 3, 2],
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def partition_trace(source: Path, output_root: Path) -> dict[str, Any]:
|
||||
source_sha = sha256_file(source)
|
||||
output_root.mkdir(parents=True, exist_ok=True)
|
||||
paths = {rep: output_root / f"rep{rep}.jsonl" for rep in REPETITIONS}
|
||||
temporary = {rep: path.with_suffix(".jsonl.tmp") for rep, path in paths.items()}
|
||||
handles = {rep: temporary[rep].open("w", encoding="utf-8") for rep in REPETITIONS}
|
||||
counts = {rep: 0 for rep in REPETITIONS}
|
||||
id_digests = {rep: hashlib.sha256() for rep in REPETITIONS}
|
||||
total = 0
|
||||
try:
|
||||
with source.open(encoding="utf-8") as input_file:
|
||||
for line_number, line in enumerate(input_file, start=1):
|
||||
if not line.strip():
|
||||
continue
|
||||
row = json.loads(line)
|
||||
original_id = str(row.get("request_id") or row.get("id") or line_number)
|
||||
digest = hashlib.sha256(
|
||||
f"{source_sha}:{line_number}:{original_id}".encode()
|
||||
).hexdigest()
|
||||
repetition = int(digest[:16], 16) % len(REPETITIONS) + 1
|
||||
row["request_id"] = f"active-r{repetition}-{digest}"
|
||||
handles[repetition].write(json.dumps(row, ensure_ascii=False) + "\n")
|
||||
counts[repetition] += 1
|
||||
total += 1
|
||||
id_digests[repetition].update(row["request_id"].encode() + b"\n")
|
||||
finally:
|
||||
for handle in handles.values():
|
||||
handle.close()
|
||||
for repetition in REPETITIONS:
|
||||
os.replace(temporary[repetition], paths[repetition])
|
||||
partitions = {
|
||||
str(rep): {
|
||||
"path": str(paths[rep]),
|
||||
"rows": counts[rep],
|
||||
"bytes": paths[rep].stat().st_size,
|
||||
"sha256": sha256_file(paths[rep]),
|
||||
"request_id_order_sha256": id_digests[rep].hexdigest(),
|
||||
}
|
||||
for rep in REPETITIONS
|
||||
}
|
||||
return {
|
||||
"source": str(source),
|
||||
"source_sha256": source_sha,
|
||||
"source_rows": total,
|
||||
"partition_rule": "sha256(source_sha:line_number:original_id) modulo 3",
|
||||
"partitions": partitions,
|
||||
}
|
||||
|
||||
|
||||
def materialize_study(
|
||||
base_study: Path,
|
||||
target: Path,
|
||||
*,
|
||||
repetition: int,
|
||||
trace_path: Path,
|
||||
windows_path: Path,
|
||||
) -> None:
|
||||
payload = json.loads(base_study.read_text(encoding="utf-8"))
|
||||
payload["study_id"] = f"active-intervention-trace13-rep{repetition}"
|
||||
payload["hardware"]["host_candidates"] = ["dash0"]
|
||||
payload["engine"]["engine_version"] = ENGINE_VERSION
|
||||
trace = payload["trace"]
|
||||
trace.update(
|
||||
{
|
||||
"windows_path": str(windows_path),
|
||||
"window_id": WINDOW_ID,
|
||||
"trace_file_override": str(trace_path),
|
||||
"completion_tokens_override": 128,
|
||||
"replay_time_scale": REPLAY_TIME_SCALE,
|
||||
"early_stop_max_lag_s": None,
|
||||
"early_stop_max_elapsed_s": 360.0,
|
||||
"restart_engine_after_early_stop": False,
|
||||
"adaptive_stop": {"enabled": False},
|
||||
}
|
||||
)
|
||||
atomic_json(target, payload)
|
||||
|
||||
|
||||
def attainable_anchor(requests: list[Any], target_count: int) -> tuple[float, list[Any]]:
|
||||
ordered = sorted(float(request.sampling_u) for request in requests)
|
||||
if target_count <= 0 or target_count > len(ordered):
|
||||
raise ValueError(
|
||||
f"target count {target_count} is outside available range 1..{len(ordered)}"
|
||||
)
|
||||
candidates = []
|
||||
for index in sorted({target_count - 1, min(target_count, len(ordered) - 1)}):
|
||||
anchor = ordered[index]
|
||||
selected = select_requests_for_threshold(requests, threshold=anchor)
|
||||
candidates.append((abs(len(selected) - target_count), len(selected), anchor, selected))
|
||||
_error, _count, anchor, selected = min(
|
||||
candidates, key=lambda item: (item[0], item[1], item[2])
|
||||
)
|
||||
return anchor, selected
|
||||
|
||||
|
||||
def selection_record(selected: list[Any]) -> dict[str, Any]:
|
||||
return {
|
||||
"anchor": max(float(request.sampling_u) for request in selected),
|
||||
"selected_count": len(selected),
|
||||
"target_count": TARGET_COUNT,
|
||||
"offered_req_s": len(selected) / DURATION_S,
|
||||
"offered_req_s_per_gpu": len(selected) / DURATION_S / TP,
|
||||
"request_id_order_sha256": order_hash([request.row_id for request in selected]),
|
||||
"arrival_order_sha256": order_hash(
|
||||
[f"{request.arrival_s:.12f}" for request in selected]
|
||||
),
|
||||
"input_length_order_sha256": order_hash(
|
||||
[str(request.prompt_tokens_hint) for request in selected]
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def build(
|
||||
*,
|
||||
base_study: Path,
|
||||
base_action_manifest: Path,
|
||||
source_trace: Path,
|
||||
windows_path: Path,
|
||||
private_root: Path,
|
||||
policy_path: Path,
|
||||
) -> dict[str, Any]:
|
||||
base_manifest = json.loads(base_action_manifest.read_text(encoding="utf-8"))
|
||||
if base_manifest.get("status") != "PASS":
|
||||
raise ValueError("base action-aware manifest did not pass")
|
||||
policy = json.loads(policy_path.read_text(encoding="utf-8"))
|
||||
if policy.get("schema") != "active-intervention-policy-v0":
|
||||
raise ValueError("unexpected frozen policy schema")
|
||||
if policy.get("sanity", {}).get("red_flags"):
|
||||
raise ValueError("frozen policy contains red flags")
|
||||
|
||||
partition = partition_trace(source_trace, private_root / "traces")
|
||||
repetitions = {}
|
||||
selected_sets: list[set[str]] = []
|
||||
for repetition in REPETITIONS:
|
||||
trace_path = Path(partition["partitions"][str(repetition)]["path"])
|
||||
study_path = private_root / "studies" / f"rep{repetition}-tp4.json"
|
||||
materialize_study(
|
||||
base_study,
|
||||
study_path,
|
||||
repetition=repetition,
|
||||
trace_path=trace_path,
|
||||
windows_path=windows_path,
|
||||
)
|
||||
study = load_study_spec(study_path)
|
||||
window, requests = load_trace_requests(study, study_spec_path=study_path)
|
||||
duration_s = float(window.window_end - window.window_start)
|
||||
if not math.isclose(duration_s, DURATION_S, abs_tol=1e-9):
|
||||
raise ValueError(f"rep{repetition}: duration {duration_s} != {DURATION_S}")
|
||||
_anchor, selected = attainable_anchor(requests, TARGET_COUNT)
|
||||
record = selection_record(selected)
|
||||
selected_sets.append({request.row_id for request in selected})
|
||||
repetitions[str(repetition)] = {
|
||||
"study": str(study_path),
|
||||
"study_sha256": sha256_file(study_path),
|
||||
"trace": partition["partitions"][str(repetition)],
|
||||
"available_filtered_requests": len(requests),
|
||||
"selection": record,
|
||||
}
|
||||
|
||||
frozen_configs = configs()
|
||||
config_ids = {str(config["id"]) for config in frozen_configs}
|
||||
invariants = {
|
||||
"three_nonempty_trace_partitions": all(
|
||||
int(item["rows"]) > 0 for item in partition["partitions"].values()
|
||||
),
|
||||
"partition_rows_conserved": sum(
|
||||
int(item["rows"]) for item in partition["partitions"].values()
|
||||
)
|
||||
== int(partition["source_rows"]),
|
||||
"selected_sets_disjoint": all(
|
||||
not selected_sets[left] & selected_sets[right]
|
||||
for left in range(len(selected_sets))
|
||||
for right in range(left + 1, len(selected_sets))
|
||||
),
|
||||
"target_count_attained": all(
|
||||
abs(int(item["selection"]["selected_count"]) - TARGET_COUNT) <= 1
|
||||
for item in repetitions.values()
|
||||
),
|
||||
"four_unique_configs": len(config_ids) == 4,
|
||||
"two_by_two_surface": {
|
||||
(int(config["mns"]), int(config["mbbt"]))
|
||||
for config in frozen_configs
|
||||
}
|
||||
== {(32, 4096), (64, 4096), (32, 8192), (64, 8192)},
|
||||
"repetition_orders_are_permutations": all(
|
||||
sorted(config["repetition_order"]) == list(REPETITIONS)
|
||||
for config in frozen_configs
|
||||
),
|
||||
}
|
||||
red_flags = [name for name, passed in invariants.items() if not passed]
|
||||
return {
|
||||
"schema": SCHEMA,
|
||||
"status": "PASS" if not red_flags else "STOP",
|
||||
"source": {
|
||||
"window_id": WINDOW_ID,
|
||||
"source_trace": str(source_trace),
|
||||
"source_trace_sha256": partition["source_sha256"],
|
||||
"windows_path": str(windows_path),
|
||||
"base_study": str(base_study),
|
||||
"base_study_sha256": sha256_file(base_study),
|
||||
"base_action_manifest": str(base_action_manifest),
|
||||
"base_action_manifest_sha256": sha256_file(base_action_manifest),
|
||||
},
|
||||
"policy": {
|
||||
"path": str(policy_path),
|
||||
"sha256": sha256_file(policy_path),
|
||||
"status": policy["status"],
|
||||
"training": policy["training"],
|
||||
"measurement_policy": policy["measurement_policy"],
|
||||
"launch_reason": (
|
||||
"bounded unseen-trace joint-action test after a negative narrow "
|
||||
"retrospective replay"
|
||||
),
|
||||
},
|
||||
"engine": {
|
||||
"tp": TP,
|
||||
"duration_s": DURATION_S,
|
||||
"client_timeout_s": 450.0,
|
||||
"burnin_max_elapsed_s": 90.0,
|
||||
"disable_slo_early_stop": True,
|
||||
},
|
||||
"burnin": base_manifest["burnin"],
|
||||
"private": {"trace_partition": partition},
|
||||
"repetitions": repetitions,
|
||||
"configs": frozen_configs,
|
||||
"source_config_id": "source_mns32_mbbt4096",
|
||||
"actions": {
|
||||
"noop": "source_mns32_mbbt4096",
|
||||
"mns": "mns64_mbbt4096",
|
||||
"mbbt": "mns32_mbbt8192",
|
||||
"joint": "joint_mns64_mbbt8192",
|
||||
},
|
||||
"checkpoints": {
|
||||
"fractions": [0.25, 0.50, 0.75, 1.0],
|
||||
"seconds": [75.0, 150.0, 225.0, 300.0],
|
||||
},
|
||||
"gates": {
|
||||
"acceptable_regret": 0.02,
|
||||
"source_ceiling_normalized_goodput": 0.98,
|
||||
"confirmation_trigger_gpu_cost_reduction": 0.10,
|
||||
"contribution_gpu_cost_reduction": 0.20,
|
||||
"maximum_task_regret": 0.05,
|
||||
},
|
||||
"budget": {
|
||||
"hard_cap_h20_hours": 6.0,
|
||||
"session_estimate_h20_hours": 1.3,
|
||||
"safety_h20_hours": 0.3,
|
||||
"expected_h20_hours": [4.6, 5.5],
|
||||
"expected_wall_minutes": [75, 100],
|
||||
},
|
||||
"sanity": {"invariants": invariants, "red_flags": red_flags},
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--base-study", type=Path, required=True)
|
||||
parser.add_argument("--base-action-manifest", type=Path, required=True)
|
||||
parser.add_argument("--source-trace", type=Path, required=True)
|
||||
parser.add_argument("--windows-path", type=Path, required=True)
|
||||
parser.add_argument("--private-root", type=Path, required=True)
|
||||
parser.add_argument("--policy", type=Path, required=True)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
payload = build(
|
||||
base_study=args.base_study,
|
||||
base_action_manifest=args.base_action_manifest,
|
||||
source_trace=args.source_trace,
|
||||
windows_path=args.windows_path,
|
||||
private_root=args.private_root,
|
||||
policy_path=args.policy,
|
||||
)
|
||||
atomic_json(args.output, payload)
|
||||
print(json.dumps({"status": payload["status"], "sanity": payload["sanity"]}))
|
||||
if payload["status"] != "PASS":
|
||||
raise SystemExit("prospective manifest preflight failed")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
198
runs/active-intervention-v0/prospective_controller.py
Normal file
198
runs/active-intervention-v0/prospective_controller.py
Normal file
@@ -0,0 +1,198 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run source first, select the next intervention, then annotate the 2x2 surface."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any, Mapping
|
||||
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
ACTION_DIR = HERE.parent / "action-aware-v0"
|
||||
sys.path.insert(0, str(ACTION_DIR))
|
||||
sys.path.insert(0, str(HERE))
|
||||
|
||||
import pilot_controller as action_controller # noqa: E402
|
||||
import prospective_decision # noqa: E402
|
||||
|
||||
|
||||
SCHEMA = "active-intervention-prospective-state-v0"
|
||||
|
||||
|
||||
def validate_inputs(args: argparse.Namespace, manifest: Mapping[str, Any]) -> None:
|
||||
if manifest.get("schema") != "active-intervention-prospective-manifest-v0":
|
||||
raise RuntimeError("unexpected active intervention manifest schema")
|
||||
if manifest.get("status") != "PASS" or manifest["sanity"]["red_flags"]:
|
||||
raise RuntimeError("active intervention manifest did not pass preflight")
|
||||
required = {
|
||||
"manifest": args.manifest,
|
||||
"policy": args.policy,
|
||||
"aituner_root": args.aituner_root,
|
||||
"vllm_source": args.vllm_source,
|
||||
"venv_python": args.venv / "bin/python",
|
||||
"venv_vllm": args.venv / "bin/vllm",
|
||||
"model": args.model,
|
||||
"client": args.client,
|
||||
"burnin_study": Path(manifest["burnin"]["study"]),
|
||||
}
|
||||
for repetition, item in manifest["repetitions"].items():
|
||||
required[f"rep{repetition}_study"] = Path(item["study"])
|
||||
required[f"rep{repetition}_trace"] = Path(item["trace"]["path"])
|
||||
missing = {name: str(path) for name, path in required.items() if not path.exists()}
|
||||
if missing:
|
||||
raise RuntimeError(f"active intervention input paths missing: {missing}")
|
||||
if prospective_decision.sha256_file(args.policy) != manifest["policy"]["sha256"]:
|
||||
raise RuntimeError("active intervention policy hash mismatch")
|
||||
|
||||
|
||||
def dry_run(args: argparse.Namespace, manifest: Mapping[str, Any]) -> dict[str, Any]:
|
||||
plan = action_controller.dry_run_plan(args, manifest)
|
||||
return {
|
||||
"schema": "active-intervention-prospective-dry-run-v0",
|
||||
"status": "PASS",
|
||||
"manifest": str(args.manifest),
|
||||
"policy": str(args.policy),
|
||||
"source_first": manifest["source_config_id"],
|
||||
"post_source_order": "selected by telemetry policy; all remaining cells then annotated",
|
||||
"candidate_actions": manifest["actions"],
|
||||
"projected_h20_hours": plan["projected_h20_hours"],
|
||||
"hard_cap_h20_hours": plan["hard_cap_h20_hours"],
|
||||
"sessions": plan["sessions"],
|
||||
}
|
||||
|
||||
|
||||
def load_or_build_decision(
|
||||
*, args: argparse.Namespace, run_root: Path
|
||||
) -> dict[str, Any]:
|
||||
path = run_root / "active-decision.json"
|
||||
if path.exists():
|
||||
decision = json.loads(path.read_text(encoding="utf-8"))
|
||||
if decision.get("manifest_sha256") != prospective_decision.sha256_file(
|
||||
args.manifest
|
||||
):
|
||||
raise RuntimeError("existing active decision has a different manifest")
|
||||
if decision.get("policy_sha256") != prospective_decision.sha256_file(args.policy):
|
||||
raise RuntimeError("existing active decision has a different policy")
|
||||
return decision
|
||||
decision = prospective_decision.build_decision(
|
||||
manifest_path=args.manifest,
|
||||
policy_path=args.policy,
|
||||
run_root=run_root,
|
||||
)
|
||||
prospective_decision.atomic_json(path, decision)
|
||||
return decision
|
||||
|
||||
|
||||
def parser() -> argparse.ArgumentParser:
|
||||
result = argparse.ArgumentParser()
|
||||
result.add_argument("--manifest", type=Path, required=True)
|
||||
result.add_argument("--policy", type=Path, required=True)
|
||||
result.add_argument("--run-root", type=Path, required=True)
|
||||
result.add_argument("--aituner-root", type=Path, required=True)
|
||||
result.add_argument("--vllm-source", type=Path, required=True)
|
||||
result.add_argument("--venv", type=Path, required=True)
|
||||
result.add_argument("--model", type=Path, required=True)
|
||||
result.add_argument("--client", type=Path, required=True)
|
||||
result.add_argument("--dry-run", action="store_true")
|
||||
return result
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parser().parse_args()
|
||||
manifest = json.loads(args.manifest.read_text(encoding="utf-8"))
|
||||
validate_inputs(args, manifest)
|
||||
action_controller.configure(args, manifest)
|
||||
action_controller.base.MARKER = "active-intervention-prospective-v0"
|
||||
if args.dry_run:
|
||||
print(json.dumps(dry_run(args, manifest), indent=2, sort_keys=True))
|
||||
return
|
||||
|
||||
args.run_root.mkdir(parents=True, exist_ok=True)
|
||||
copied_manifest = args.run_root / "prospective-manifest.json"
|
||||
if not copied_manifest.exists():
|
||||
action_controller.atomic_json(copied_manifest, manifest)
|
||||
state_path = args.run_root / "controller-state.json"
|
||||
state = action_controller.load_state(
|
||||
state_path, float(manifest["budget"]["hard_cap_h20_hours"])
|
||||
)
|
||||
state["schema"] = SCHEMA
|
||||
state["status"] = "running"
|
||||
action_controller.atomic_json(state_path, state)
|
||||
|
||||
configs = {str(item["id"]): dict(item) for item in manifest["configs"]}
|
||||
config_indexes = {
|
||||
str(item["id"]): index for index, item in enumerate(manifest["configs"])
|
||||
}
|
||||
source_id = str(manifest["source_config_id"])
|
||||
action_controller.execute_session(
|
||||
args=args,
|
||||
manifest=manifest,
|
||||
config=configs[source_id],
|
||||
index=config_indexes[source_id],
|
||||
state=state,
|
||||
state_path=state_path,
|
||||
)
|
||||
decision = load_or_build_decision(args=args, run_root=args.run_root)
|
||||
state["active_decision"] = {
|
||||
"path": str(args.run_root / "active-decision.json"),
|
||||
"status": decision["status"],
|
||||
"outcome_only": {
|
||||
key: decision["decisions"]["outcome_only"][key]
|
||||
for key in ("selected_cutoff_s", "decision_kind", "selected_action")
|
||||
},
|
||||
"telemetry": {
|
||||
key: decision["decisions"]["telemetry"][key]
|
||||
for key in ("selected_cutoff_s", "decision_kind", "selected_action")
|
||||
},
|
||||
}
|
||||
action_controller.atomic_json(state_path, state)
|
||||
if decision["status"] != "SELECTED":
|
||||
state["status"] = decision["status"].lower()
|
||||
state["completed_at"] = action_controller.time.time()
|
||||
action_controller.atomic_json(state_path, state)
|
||||
action_controller.wait_all_idle()
|
||||
print(json.dumps({"status": state["status"], "decision": decision["status"]}))
|
||||
return
|
||||
|
||||
action_order = decision["decisions"]["telemetry"]["intervention_order"]
|
||||
execution_order = [source_id]
|
||||
for action_id in action_order:
|
||||
target_id = str(manifest["actions"][action_id])
|
||||
if target_id not in execution_order:
|
||||
execution_order.append(target_id)
|
||||
for config_id in configs:
|
||||
if config_id not in execution_order:
|
||||
execution_order.append(config_id)
|
||||
state["execution_order"] = execution_order
|
||||
action_controller.atomic_json(state_path, state)
|
||||
for config_id in execution_order[1:]:
|
||||
action_controller.execute_session(
|
||||
args=args,
|
||||
manifest=manifest,
|
||||
config=configs[config_id],
|
||||
index=config_indexes[config_id],
|
||||
state=state,
|
||||
state_path=state_path,
|
||||
)
|
||||
state["status"] = "complete"
|
||||
state["completed_at"] = action_controller.time.time()
|
||||
action_controller.atomic_json(state_path, state)
|
||||
action_controller.wait_all_idle()
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"status": state["status"],
|
||||
"completed_sessions": state["completed_sessions"],
|
||||
"gpu_hours_total": state["gpu_hours_total"],
|
||||
"execution_order": execution_order,
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
441
runs/active-intervention-v0/prospective_decision.py
Normal file
441
runs/active-intervention-v0/prospective_decision.py
Normal file
@@ -0,0 +1,441 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Choose measurement horizon and next intervention from a completed source run."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import importlib.util
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import statistics
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
import numpy as np
|
||||
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
COMMON_STATE = HERE.parent / "telemetry-residual"
|
||||
sys.path.insert(0, str(COMMON_STATE))
|
||||
|
||||
from common_state import summarize_engine # noqa: E402
|
||||
|
||||
|
||||
SCHEMA = "active-intervention-prospective-decision-v0"
|
||||
|
||||
|
||||
def load_module(name: str, path: Path):
|
||||
spec = importlib.util.spec_from_file_location(name, path)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
assert spec.loader is not None
|
||||
sys.modules[spec.name] = module
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
MODEL = load_module("active_intervention_prospective_model", HERE / "model.py")
|
||||
EXTRACT = load_module(
|
||||
"active_intervention_prospective_extract", HERE / "extract_training.py"
|
||||
)
|
||||
|
||||
|
||||
def sha256_file(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as source:
|
||||
for chunk in iter(lambda: source.read(1 << 20), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def atomic_json(path: Path, payload: Any) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary = path.with_suffix(path.suffix + ".tmp")
|
||||
temporary.write_text(
|
||||
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
||||
)
|
||||
os.replace(temporary, path)
|
||||
|
||||
|
||||
def numeric(values: Sequence[float]) -> dict[str, Any]:
|
||||
finite = [float(value) for value in values]
|
||||
if not finite or any(not math.isfinite(value) for value in finite):
|
||||
raise ValueError("numeric summary requires finite values")
|
||||
return {
|
||||
"n": len(finite),
|
||||
"min": min(finite),
|
||||
"max": max(finite),
|
||||
"distinct_n": len(set(finite)),
|
||||
}
|
||||
|
||||
|
||||
def load_engine_records(source_root: Path) -> tuple[list[dict[str, Any]], Path]:
|
||||
streams = sorted((source_root / "opprof").glob("*.jsonl"))
|
||||
if len(streams) != 1:
|
||||
raise ValueError(f"expected one source engine stream, found {len(streams)}")
|
||||
records = [
|
||||
row for row in EXTRACT.load_jsonl(streams[0]) if "step_index" in row
|
||||
]
|
||||
if not records:
|
||||
raise ValueError("source engine stream has no Layer-1 records")
|
||||
return records, streams[0]
|
||||
|
||||
|
||||
def candidate_example(
|
||||
*,
|
||||
source_config: Mapping[str, Any],
|
||||
target_config: Mapping[str, Any],
|
||||
action_id: str,
|
||||
offered_rate_per_gpu: float,
|
||||
outcome: Mapping[str, float],
|
||||
telemetry: Mapping[str, float],
|
||||
) -> dict[str, Any]:
|
||||
return {
|
||||
"source": {
|
||||
"mns": int(source_config["mns"]),
|
||||
"mbbt": int(source_config["mbbt"]),
|
||||
"offered_rate_per_gpu": float(offered_rate_per_gpu),
|
||||
"outcome": dict(outcome),
|
||||
"telemetry": dict(telemetry),
|
||||
},
|
||||
"action": {
|
||||
"id": action_id,
|
||||
"target_mns": int(target_config["mns"]),
|
||||
"target_mbbt": int(target_config["mbbt"]),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def aggregate_checkpoint(
|
||||
*,
|
||||
models: Sequence[Any],
|
||||
examples_by_action: Mapping[str, Sequence[Mapping[str, Any]]],
|
||||
include_telemetry: bool,
|
||||
confidence_z: float,
|
||||
minimum_margin: float,
|
||||
) -> dict[str, Any]:
|
||||
rows = []
|
||||
for action_id, examples in sorted(examples_by_action.items()):
|
||||
raw = []
|
||||
for example in examples:
|
||||
source = example["source"]
|
||||
action = example["action"]
|
||||
noop = (
|
||||
int(source["mns"]) == int(action["target_mns"])
|
||||
and int(source["mbbt"]) == int(action["target_mbbt"])
|
||||
)
|
||||
if noop:
|
||||
raw.extend(0.0 for _model in models)
|
||||
continue
|
||||
names, values = MODEL.feature_vector(
|
||||
example, include_telemetry=include_telemetry
|
||||
)
|
||||
if any(model.feature_names != tuple(names) for model in models):
|
||||
raise ValueError("prospective feature schema does not match frozen model")
|
||||
raw.extend(model.predict(values) for model in models)
|
||||
clipped = np.clip(np.asarray(raw, dtype=np.float64), -1.0, 1.0)
|
||||
prediction = {
|
||||
"mean": float(clipped.mean()),
|
||||
"std": float(clipped.std(ddof=0)),
|
||||
"min": float(clipped.min()),
|
||||
"max": float(clipped.max()),
|
||||
"distinct_n": len(set(float(value) for value in clipped)),
|
||||
"sample_n": int(clipped.size),
|
||||
}
|
||||
rows.append(
|
||||
{
|
||||
"action_id": action_id,
|
||||
"prediction": prediction,
|
||||
"lower": prediction["mean"] - confidence_z * prediction["std"],
|
||||
"upper": prediction["mean"] + confidence_z * prediction["std"],
|
||||
}
|
||||
)
|
||||
rows.sort(key=lambda row: (-row["prediction"]["mean"], row["action_id"]))
|
||||
best, second = rows[:2]
|
||||
margin = float(best["prediction"]["mean"] - second["prediction"]["mean"])
|
||||
confident = bool(
|
||||
margin >= minimum_margin and best["lower"] > second["upper"]
|
||||
)
|
||||
return {
|
||||
"selected_action": best["action_id"],
|
||||
"confident": confident,
|
||||
"predicted_margin": margin,
|
||||
"candidates": rows,
|
||||
}
|
||||
|
||||
|
||||
def apply_measurement_and_acquisition(checkpoints: list[dict[str, Any]]) -> dict[str, Any]:
|
||||
selected = checkpoints[-1]
|
||||
stop_reason = "full_measurement_fallback"
|
||||
for previous, current in zip(checkpoints, checkpoints[1:], strict=False):
|
||||
if (
|
||||
previous["confident"]
|
||||
and current["confident"]
|
||||
and previous["selected_action"] == current["selected_action"]
|
||||
):
|
||||
selected = current
|
||||
stop_reason = "two_consecutive_confident_checkpoints"
|
||||
break
|
||||
candidates = selected["candidates"]
|
||||
mean_best = candidates[0]
|
||||
non_noop = [row for row in candidates if row["action_id"] != "noop"]
|
||||
if selected["confident"]:
|
||||
chosen = mean_best
|
||||
decision_kind = "exploit"
|
||||
else:
|
||||
positive_ucb = [row for row in non_noop if float(row["upper"]) > 0.0]
|
||||
if positive_ucb:
|
||||
chosen = max(
|
||||
positive_ucb,
|
||||
key=lambda row: (float(row["upper"]), row["action_id"]),
|
||||
)
|
||||
decision_kind = "diagnostic_ucb"
|
||||
else:
|
||||
chosen = next(row for row in candidates if row["action_id"] == "noop")
|
||||
decision_kind = "abstain_no_positive_ucb"
|
||||
remaining = [row for row in candidates if row["action_id"] != chosen["action_id"]]
|
||||
remaining.sort(key=lambda row: (-float(row["upper"]), row["action_id"]))
|
||||
order = [chosen["action_id"], *(row["action_id"] for row in remaining)]
|
||||
return {
|
||||
"selected_phase": selected["phase"],
|
||||
"selected_cutoff_s": selected["cutoff_s"],
|
||||
"measurement_stop_reason": stop_reason,
|
||||
"decision_kind": decision_kind,
|
||||
"selected_action": chosen["action_id"],
|
||||
"intervention_order": order,
|
||||
"selected_checkpoint": selected,
|
||||
"checkpoints": checkpoints,
|
||||
}
|
||||
|
||||
|
||||
def build_decision(
|
||||
*, manifest_path: Path, policy_path: Path, run_root: Path
|
||||
) -> dict[str, Any]:
|
||||
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
||||
policy = json.loads(policy_path.read_text(encoding="utf-8"))
|
||||
if manifest.get("schema") != "active-intervention-prospective-manifest-v0":
|
||||
raise ValueError("unexpected prospective manifest schema")
|
||||
if policy.get("schema") != "active-intervention-policy-v0":
|
||||
raise ValueError("unexpected frozen policy schema")
|
||||
if sha256_file(policy_path) != manifest["policy"]["sha256"]:
|
||||
raise ValueError("frozen policy hash changed after manifest preparation")
|
||||
configs = {str(item["id"]): item for item in manifest["configs"]}
|
||||
source_id = str(manifest["source_config_id"])
|
||||
source_config = configs[source_id]
|
||||
source_root = run_root / "sessions" / source_id
|
||||
engine_records, stream_path = load_engine_records(source_root)
|
||||
phases = [f"{fraction:.2f}" for fraction in manifest["checkpoints"]["fractions"]]
|
||||
confidence_z = float(policy["measurement_policy"]["confidence_z"])
|
||||
minimum_margin = float(policy["measurement_policy"]["minimum_margin"])
|
||||
|
||||
examples: dict[str, dict[str, dict[str, Mapping[str, Any]]]] = {}
|
||||
source_measurements: dict[str, dict[str, Any]] = {}
|
||||
source_normalized = []
|
||||
telemetry_values = []
|
||||
for repetition in sorted(int(key) for key in manifest["repetitions"]):
|
||||
item = manifest["repetitions"][str(repetition)]
|
||||
result_root = source_root / f"rep{repetition}"
|
||||
result = json.loads((result_root / "result.json").read_text(encoding="utf-8"))
|
||||
if result["selection"]["request_id_order_sha256"] != item["selection"][
|
||||
"request_id_order_sha256"
|
||||
]:
|
||||
raise ValueError(f"source request hash mismatch: rep{repetition}")
|
||||
requests = EXTRACT.load_jsonl(result_root / "requests.jsonl")
|
||||
offered_rate = float(item["selection"]["offered_req_s_per_gpu"])
|
||||
offered_total = offered_rate * int(manifest["engine"]["tp"])
|
||||
source_normalized.append(
|
||||
float(result["slo_pass_count"])
|
||||
/ float(manifest["engine"]["duration_s"])
|
||||
/ offered_total
|
||||
)
|
||||
start_ns = int(result["interval"]["start_mono_ns"])
|
||||
examples[str(repetition)] = {}
|
||||
source_measurements[str(repetition)] = {
|
||||
"result": str(result_root / "result.json"),
|
||||
"result_sha256": sha256_file(result_root / "result.json"),
|
||||
"request_sha256": sha256_file(result_root / "requests.jsonl"),
|
||||
"phases": {},
|
||||
}
|
||||
for phase, cutoff_s in zip(
|
||||
phases, manifest["checkpoints"]["seconds"], strict=True
|
||||
):
|
||||
outcome = EXTRACT.prefix_outcome(
|
||||
requests, cutoff_s=float(cutoff_s), offered_total=offered_total
|
||||
)
|
||||
admitted_count = sum(
|
||||
float(request["arrival_s"]) <= float(cutoff_s)
|
||||
for request in requests
|
||||
)
|
||||
state = summarize_engine(
|
||||
engine_records,
|
||||
start_ns=start_ns,
|
||||
end_ns=start_ns + round(float(cutoff_s) * 1e9),
|
||||
request_count=admitted_count,
|
||||
)
|
||||
if not all(state["sanity"]["invariants"].values()):
|
||||
raise ValueError(
|
||||
f"source engine state invariant failed: rep{repetition} {phase}"
|
||||
)
|
||||
telemetry = EXTRACT.telemetry_record(state)
|
||||
telemetry_values.extend(float(value) for value in telemetry.values())
|
||||
source_measurements[str(repetition)]["phases"][phase] = {
|
||||
"cutoff_s": float(cutoff_s),
|
||||
"outcome": outcome,
|
||||
"telemetry": telemetry,
|
||||
"engine_sanity": state["sanity"],
|
||||
}
|
||||
examples[str(repetition)][phase] = {
|
||||
action_id: candidate_example(
|
||||
source_config=source_config,
|
||||
target_config=configs[str(target_id)],
|
||||
action_id=action_id,
|
||||
offered_rate_per_gpu=offered_rate,
|
||||
outcome=outcome,
|
||||
telemetry=telemetry,
|
||||
)
|
||||
for action_id, target_id in manifest["actions"].items()
|
||||
}
|
||||
|
||||
decisions = {}
|
||||
for mode, include_telemetry in (("outcome_only", False), ("telemetry", True)):
|
||||
checkpoints = []
|
||||
for phase, cutoff_s in zip(
|
||||
phases, manifest["checkpoints"]["seconds"], strict=True
|
||||
):
|
||||
models = MODEL.models_from_json(policy["phases"][phase][mode]["models"])
|
||||
examples_by_action = {
|
||||
action_id: [
|
||||
examples[str(repetition)][phase][action_id]
|
||||
for repetition in sorted(int(key) for key in manifest["repetitions"])
|
||||
]
|
||||
for action_id in manifest["actions"]
|
||||
}
|
||||
checkpoint = aggregate_checkpoint(
|
||||
models=models,
|
||||
examples_by_action=examples_by_action,
|
||||
include_telemetry=include_telemetry,
|
||||
confidence_z=confidence_z,
|
||||
minimum_margin=minimum_margin,
|
||||
)
|
||||
checkpoints.append(
|
||||
{"phase": phase, "cutoff_s": float(cutoff_s), **checkpoint}
|
||||
)
|
||||
decisions[mode] = apply_measurement_and_acquisition(checkpoints)
|
||||
|
||||
ceiling = float(manifest["gates"]["source_ceiling_normalized_goodput"])
|
||||
source_median = float(statistics.median(source_normalized))
|
||||
status = "STOP_SOURCE_CEILING" if source_median >= ceiling else "SELECTED"
|
||||
phase_admission_monotonic = all(
|
||||
all(
|
||||
left <= right + 1e-12
|
||||
for left, right in zip(values, values[1:], strict=False)
|
||||
)
|
||||
for repetition in source_measurements.values()
|
||||
for values in (
|
||||
[
|
||||
float(repetition["phases"][phase]["outcome"]["admitted_fraction"])
|
||||
for phase in phases
|
||||
],
|
||||
)
|
||||
)
|
||||
telemetry_ratio_keys = {
|
||||
"prefill_token_fraction",
|
||||
"kv_usage_mean",
|
||||
"kv_usage_max",
|
||||
"graph_none_share",
|
||||
"graph_full_share",
|
||||
"graph_padding_fraction",
|
||||
}
|
||||
telemetry_records = [
|
||||
measurement["telemetry"]
|
||||
for repetition in source_measurements.values()
|
||||
for measurement in repetition["phases"].values()
|
||||
]
|
||||
invariants = {
|
||||
"three_source_repetitions": len(source_normalized) == 3,
|
||||
"source_goodput_nonnegative": all(value >= 0.0 for value in source_normalized),
|
||||
"source_goodput_bounded": all(
|
||||
value <= 1.0 + 1e-12 for value in source_normalized
|
||||
),
|
||||
"four_actions": set(manifest["actions"]) == {"noop", "mns", "mbbt", "joint"},
|
||||
"four_checkpoints": len(phases) == 4,
|
||||
"finite_telemetry": all(math.isfinite(value) for value in telemetry_values),
|
||||
"nonnegative_telemetry": all(
|
||||
float(value) >= 0.0
|
||||
for record in telemetry_records
|
||||
for key, value in record.items()
|
||||
if key != "kv_usage_end_minus_start"
|
||||
),
|
||||
"telemetry_ratios_bounded": all(
|
||||
0.0 <= float(record[key]) <= 1.0 + 1e-12
|
||||
for record in telemetry_records
|
||||
for key in telemetry_ratio_keys
|
||||
),
|
||||
"telemetry_not_all_identical": len(set(telemetry_values)) > 1,
|
||||
"phase_admission_monotonic": phase_admission_monotonic,
|
||||
"orders_are_permutations": all(
|
||||
set(decisions[mode]["intervention_order"]) == set(manifest["actions"])
|
||||
for mode in decisions
|
||||
),
|
||||
}
|
||||
red_flags = [name for name, passed in invariants.items() if not passed]
|
||||
if red_flags:
|
||||
status = "STOP_SANITY"
|
||||
return {
|
||||
"schema": SCHEMA,
|
||||
"status": status,
|
||||
"manifest": str(manifest_path),
|
||||
"manifest_sha256": sha256_file(manifest_path),
|
||||
"policy": str(policy_path),
|
||||
"policy_sha256": sha256_file(policy_path),
|
||||
"source_stream": str(stream_path),
|
||||
"source_stream_sha256": sha256_file(stream_path),
|
||||
"source_measurements": source_measurements,
|
||||
"source_normalized_goodput": {
|
||||
"values": source_normalized,
|
||||
"median": source_median,
|
||||
**numeric(source_normalized),
|
||||
},
|
||||
"decisions": decisions,
|
||||
"sanity": {
|
||||
"invariants": invariants,
|
||||
"red_flags": red_flags,
|
||||
"telemetry_values": numeric(telemetry_values),
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--manifest", type=Path, required=True)
|
||||
parser.add_argument("--policy", type=Path, required=True)
|
||||
parser.add_argument("--run-root", type=Path, required=True)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
decision = build_decision(
|
||||
manifest_path=args.manifest, policy_path=args.policy, run_root=args.run_root
|
||||
)
|
||||
atomic_json(args.output, decision)
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"status": decision["status"],
|
||||
"source_normalized_goodput": decision["source_normalized_goodput"],
|
||||
"outcome_only": {
|
||||
key: decision["decisions"]["outcome_only"][key]
|
||||
for key in ("selected_cutoff_s", "decision_kind", "selected_action")
|
||||
},
|
||||
"telemetry": {
|
||||
key: decision["decisions"]["telemetry"][key]
|
||||
for key in ("selected_cutoff_s", "decision_kind", "selected_action")
|
||||
},
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
91
runs/active-intervention-v0/test_model.py
Normal file
91
runs/active-intervention-v0/test_model.py
Normal file
@@ -0,0 +1,91 @@
|
||||
#!/usr/bin/env python3
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
|
||||
|
||||
def load_model():
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
"active_intervention_model", HERE / "model.py"
|
||||
)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
assert spec.loader is not None
|
||||
sys.modules[spec.name] = module
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
def example(model, decision: str, action: str, pressure: float, target: float):
|
||||
outcome = {
|
||||
name: 0.5 for name in model.PREFIX_FEATURES
|
||||
}
|
||||
telemetry = {name: 0.0 for name in model.TELEMETRY_FEATURES}
|
||||
telemetry["queue_waiting_mean"] = pressure
|
||||
telemetry["batch_size_mean"] = pressure
|
||||
return {
|
||||
"decision_id": decision,
|
||||
"source": {
|
||||
"mns": 16,
|
||||
"mbbt": 8192,
|
||||
"offered_rate_per_gpu": 2.0,
|
||||
"outcome": outcome,
|
||||
"telemetry": telemetry,
|
||||
},
|
||||
"action": {
|
||||
"id": action,
|
||||
"target_mns": 64 if action == "mns" else 16,
|
||||
"target_mbbt": 8192 if action == "mns" else 16384,
|
||||
},
|
||||
"target_normalized_goodput": target,
|
||||
"target_delta_normalized_goodput": target - 0.5,
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
model = load_model()
|
||||
examples = []
|
||||
for index, pressure in enumerate((0.2, 0.5, 0.8), 1):
|
||||
examples.extend(
|
||||
(
|
||||
example(model, f"d{index}", "mns", pressure, 0.5 + pressure / 2),
|
||||
example(model, f"d{index}", "mbbt", pressure, 0.6 - pressure / 4),
|
||||
)
|
||||
)
|
||||
fitted = model.fit_ridge(
|
||||
examples, include_telemetry=True, regularization=1.0
|
||||
)
|
||||
encoded = fitted.to_json()
|
||||
restored = model.RidgeModel.from_json(encoded)
|
||||
names, values = model.feature_vector(examples[-2], include_telemetry=True)
|
||||
assert tuple(names) == restored.feature_names
|
||||
assert abs(fitted.predict(values) - restored.predict(values)) < 1e-12
|
||||
ensemble = model.fit_jackknife_ensemble(
|
||||
examples, include_telemetry=True, regularization=1.0
|
||||
)
|
||||
decision = model.select_action(
|
||||
ensemble, examples[-2:], include_telemetry=True, minimum_margin=0.0
|
||||
)
|
||||
assert decision["selected_action"] == "mns"
|
||||
assert all(-1.0 <= row["prediction"]["mean"] <= 1.0 for row in decision["candidates"])
|
||||
noop = example(model, "noop", "noop", 0.8, 0.5)
|
||||
noop["action"]["target_mbbt"] = 8192
|
||||
prediction = model.ensemble_predict(
|
||||
ensemble, noop, include_telemetry=True
|
||||
)
|
||||
assert prediction == {
|
||||
"mean": 0.0,
|
||||
"std": 0.0,
|
||||
"min": 0.0,
|
||||
"max": 0.0,
|
||||
"distinct_n": 1,
|
||||
}
|
||||
print("active intervention model: PASS")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
199
runs/active-intervention-v0/test_pipeline.py
Normal file
199
runs/active-intervention-v0/test_pipeline.py
Normal file
@@ -0,0 +1,199 @@
|
||||
#!/usr/bin/env python3
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import json
|
||||
import sys
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
|
||||
|
||||
def load(name: str, path: Path):
|
||||
spec = importlib.util.spec_from_file_location(name, path)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
assert spec.loader is not None
|
||||
sys.modules[spec.name] = module
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
def write_json(path: Path, payload) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(payload) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def write_jsonl(path: Path, rows) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(
|
||||
"".join(json.dumps(row) + "\n" for row in rows), encoding="utf-8"
|
||||
)
|
||||
|
||||
|
||||
def engine_record(index: int, timestamp_ns: int) -> dict:
|
||||
alternate = index % 2
|
||||
return {
|
||||
"step_index": index,
|
||||
"submit_mono_ns": timestamp_ns,
|
||||
"model_executed": True,
|
||||
"scheduled_requests": 1 + alternate,
|
||||
"decode_batch_size": alternate,
|
||||
"prefill_tokens": 8 + alternate,
|
||||
"decode_tokens": alternate,
|
||||
"preemptions": 0,
|
||||
"queues": {"waiting": alternate, "running": 1 + alternate},
|
||||
"kv": {"usage": 0.1 + 0.01 * alternate},
|
||||
"cudagraph": {
|
||||
"runtime_mode": "FULL" if alternate else "NONE",
|
||||
"bucket_tokens": 16,
|
||||
"padding_tokens": alternate,
|
||||
},
|
||||
"dropped_records_before": 0,
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
extractor = load("active_intervention_extract_test", HERE / "extract_training.py")
|
||||
trainer = load("active_intervention_train_test", HERE / "train_policy.py")
|
||||
with tempfile.TemporaryDirectory() as temporary:
|
||||
root = Path(temporary)
|
||||
run_root = root / "runs"
|
||||
configs = [
|
||||
{"id": "a_base", "mns": 16, "mbbt": 8192},
|
||||
{"id": "a_mns", "mns": 64, "mbbt": 8192},
|
||||
{"id": "a_mbbt", "mns": 16, "mbbt": 16384},
|
||||
{"id": "b_base", "mns": 64, "mbbt": 2048},
|
||||
{"id": "b_mns", "mns": 128, "mbbt": 2048},
|
||||
{"id": "b_mbbt", "mns": 64, "mbbt": 8192},
|
||||
]
|
||||
manifest = {
|
||||
"engine": {"duration_s": 300.0, "tp": 4},
|
||||
"configs": configs,
|
||||
"repetitions": {
|
||||
str(rep): {"selection": {"offered_req_s_per_gpu": 0.01}}
|
||||
for rep in (1, 2, 3)
|
||||
},
|
||||
"regimes": {
|
||||
"A": {
|
||||
"source": "a_base",
|
||||
"actions": {"mns": "a_mns", "mbbt": "a_mbbt"},
|
||||
},
|
||||
"B": {
|
||||
"source": "b_base",
|
||||
"actions": {"mns": "b_mns", "mbbt": "b_mbbt"},
|
||||
},
|
||||
},
|
||||
}
|
||||
manifest_path = root / "manifest.json"
|
||||
write_json(manifest_path, manifest)
|
||||
|
||||
streams = []
|
||||
source_starts: dict[tuple[str, int], int] = {}
|
||||
for source_index, source_id in enumerate(("a_base", "b_base")):
|
||||
rows = []
|
||||
index = 0
|
||||
for repetition in (1, 2, 3):
|
||||
start_ns = int((source_index * 2000 + repetition * 400) * 1e9)
|
||||
source_starts[(source_id, repetition)] = start_ns
|
||||
for second in (1, 30, 76, 105, 151, 180, 226, 255):
|
||||
rows.append(engine_record(index, start_ns + int(second * 1e9)))
|
||||
index += 1
|
||||
stream_path = root / f"{source_id}-stream.jsonl"
|
||||
write_jsonl(stream_path, rows)
|
||||
streams.append(
|
||||
{
|
||||
"config_id": source_id,
|
||||
"stream": str(stream_path),
|
||||
"stream_sha256": extractor.sha256_file(stream_path),
|
||||
}
|
||||
)
|
||||
|
||||
request_rows = [
|
||||
{
|
||||
"request_id": f"r{index}",
|
||||
"arrival_s": arrival,
|
||||
"completed_elapsed_s": arrival + 10,
|
||||
"slo_pass": index != 3,
|
||||
"ttft_ms": 1000 + index * 100,
|
||||
"tpot_ms": 20 + index,
|
||||
"raw_input_tokens": 1000 + index * 100,
|
||||
}
|
||||
for index, arrival in enumerate((5.0, 80.0, 155.0, 230.0), 1)
|
||||
]
|
||||
for source_id in ("a_base", "b_base"):
|
||||
for repetition in (1, 2, 3):
|
||||
write_jsonl(
|
||||
run_root
|
||||
/ "sessions"
|
||||
/ source_id
|
||||
/ f"rep{repetition}"
|
||||
/ "requests.jsonl",
|
||||
request_rows,
|
||||
)
|
||||
|
||||
goodput = {
|
||||
"a_base": 0.020,
|
||||
"a_mns": 0.036,
|
||||
"a_mbbt": 0.028,
|
||||
"b_base": 0.032,
|
||||
"b_mns": 0.030,
|
||||
"b_mbbt": 0.038,
|
||||
}
|
||||
runs = []
|
||||
for config in configs:
|
||||
for repetition in (1, 2, 3):
|
||||
item = {
|
||||
"config_id": config["id"],
|
||||
"repetition": repetition,
|
||||
"outcome": {
|
||||
"slo_goodput_req_s": goodput[config["id"]]
|
||||
+ repetition * 0.0001
|
||||
},
|
||||
}
|
||||
if config["id"] in ("a_base", "b_base"):
|
||||
start_ns = source_starts[(config["id"], repetition)]
|
||||
item["state"] = {
|
||||
"interval": {
|
||||
"start_ns": start_ns,
|
||||
"end_ns": start_ns + int(300 * 1e9),
|
||||
}
|
||||
}
|
||||
runs.append(item)
|
||||
audit = {
|
||||
"schema": "action-aware-constraint-pilot-audit-v0",
|
||||
"sanity": {"red_flags": []},
|
||||
"streams": streams,
|
||||
"runs": runs,
|
||||
}
|
||||
audit_path = root / "audit.json"
|
||||
write_json(audit_path, audit)
|
||||
|
||||
dataset = extractor.build_dataset(
|
||||
audit_path=audit_path, manifest_path=manifest_path, run_root=run_root
|
||||
)
|
||||
assert dataset["status"] == "VALID"
|
||||
assert len(dataset["examples"]) == 72
|
||||
assert not dataset["sanity"]["red_flags"]
|
||||
assert all(
|
||||
"exclusive" not in feature
|
||||
for example in dataset["examples"]
|
||||
for feature in example["source"]["telemetry"]
|
||||
)
|
||||
dataset_path = root / "dataset.json"
|
||||
write_json(dataset_path, dataset)
|
||||
policy = trainer.build_policy(dataset_path)
|
||||
assert policy["status"] in {
|
||||
"RETROSPECTIVE_GPU_COST_SIGNAL",
|
||||
"NO_RETROSPECTIVE_GPU_COST_SIGNAL",
|
||||
}
|
||||
assert policy["training"]["acceptable_regret"] == 0.02
|
||||
assert policy["sequential_replay"]["outcome_only"]["decision_n"] == 6
|
||||
assert policy["sequential_replay"]["telemetry"]["decision_n"] == 6
|
||||
assert not policy["sanity"]["red_flags"]
|
||||
print("active intervention pipeline: PASS")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
190
runs/active-intervention-v0/test_prospective.py
Normal file
190
runs/active-intervention-v0/test_prospective.py
Normal file
@@ -0,0 +1,190 @@
|
||||
#!/usr/bin/env python3
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import json
|
||||
import sys
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
|
||||
|
||||
def load(name: str, path: Path):
|
||||
spec = importlib.util.spec_from_file_location(name, path)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
assert spec.loader is not None
|
||||
sys.modules[spec.name] = module
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
def write_json(path: Path, payload) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(payload) + "\n", encoding="utf-8")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
prepare = load("active_intervention_prepare_test", HERE / "prepare_prospective.py")
|
||||
decision_module = load(
|
||||
"active_intervention_decision_test", HERE / "prospective_decision.py"
|
||||
)
|
||||
analyzer = load("active_intervention_audit_test", HERE / "analyze_prospective.py")
|
||||
with tempfile.TemporaryDirectory() as temporary:
|
||||
root = Path(temporary)
|
||||
source = root / "source.jsonl"
|
||||
source.write_text(
|
||||
"".join(
|
||||
json.dumps(
|
||||
{
|
||||
"request_id": f"request-{index}",
|
||||
"timestamp": float(index),
|
||||
"sampling_u": index / 100.0,
|
||||
}
|
||||
)
|
||||
+ "\n"
|
||||
for index in range(60)
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
partition = prepare.partition_trace(source, root / "partitions")
|
||||
assert sum(item["rows"] for item in partition["partitions"].values()) == 60
|
||||
ids = []
|
||||
for item in partition["partitions"].values():
|
||||
assert item["rows"] > 0
|
||||
ids.extend(
|
||||
json.loads(line)["request_id"]
|
||||
for line in Path(item["path"]).read_text(encoding="utf-8").splitlines()
|
||||
)
|
||||
assert len(ids) == len(set(ids)) == 60
|
||||
|
||||
checkpoints = [
|
||||
{
|
||||
"phase": "0.25",
|
||||
"cutoff_s": 75.0,
|
||||
"selected_action": "joint",
|
||||
"confident": True,
|
||||
"candidates": [
|
||||
{"action_id": "joint", "upper": 0.5, "prediction": {"mean": 0.4}},
|
||||
{"action_id": "mns", "upper": 0.2, "prediction": {"mean": 0.1}},
|
||||
{"action_id": "mbbt", "upper": 0.1, "prediction": {"mean": 0.05}},
|
||||
{"action_id": "noop", "upper": 0.0, "prediction": {"mean": 0.0}},
|
||||
],
|
||||
},
|
||||
{
|
||||
"phase": "0.50",
|
||||
"cutoff_s": 150.0,
|
||||
"selected_action": "joint",
|
||||
"confident": True,
|
||||
"candidates": [
|
||||
{"action_id": "joint", "upper": 0.45, "prediction": {"mean": 0.4}},
|
||||
{"action_id": "mns", "upper": 0.2, "prediction": {"mean": 0.1}},
|
||||
{"action_id": "mbbt", "upper": 0.1, "prediction": {"mean": 0.05}},
|
||||
{"action_id": "noop", "upper": 0.0, "prediction": {"mean": 0.0}},
|
||||
],
|
||||
},
|
||||
]
|
||||
selected = decision_module.apply_measurement_and_acquisition(checkpoints)
|
||||
assert selected["selected_cutoff_s"] == 150.0
|
||||
assert selected["selected_action"] == "joint"
|
||||
|
||||
configs = prepare.configs()
|
||||
repetitions = {
|
||||
str(rep): {
|
||||
"selection": {
|
||||
"offered_req_s_per_gpu": 0.25,
|
||||
"request_id_order_sha256": f"hash-{rep}",
|
||||
}
|
||||
}
|
||||
for rep in (1, 2, 3)
|
||||
}
|
||||
manifest = {
|
||||
"schema": "active-intervention-prospective-manifest-v0",
|
||||
"engine": {"duration_s": 300.0, "tp": 4},
|
||||
"repetitions": repetitions,
|
||||
"configs": configs,
|
||||
"source_config_id": "source_mns32_mbbt4096",
|
||||
"actions": {
|
||||
"noop": "source_mns32_mbbt4096",
|
||||
"mns": "mns64_mbbt4096",
|
||||
"mbbt": "mns32_mbbt8192",
|
||||
"joint": "joint_mns64_mbbt8192",
|
||||
},
|
||||
"gates": {
|
||||
"acceptable_regret": 0.02,
|
||||
"confirmation_trigger_gpu_cost_reduction": 0.10,
|
||||
"contribution_gpu_cost_reduction": 0.20,
|
||||
},
|
||||
}
|
||||
manifest_path = root / "manifest.json"
|
||||
write_json(manifest_path, manifest)
|
||||
run_root = root / "run"
|
||||
scores = {
|
||||
"source_mns32_mbbt4096": 0.5,
|
||||
"mns64_mbbt4096": 0.8,
|
||||
"mns32_mbbt8192": 0.7,
|
||||
"joint_mns64_mbbt8192": 1.0,
|
||||
}
|
||||
sessions = {}
|
||||
for config in configs:
|
||||
config_id = config["id"]
|
||||
sessions[config_id] = {"status": "complete", "gpu_hours": 1.2}
|
||||
for repetition in (1, 2, 3):
|
||||
result = {
|
||||
"selection": {
|
||||
"request_id_order_sha256": f"hash-{repetition}"
|
||||
},
|
||||
"slo_pass_count": round(scores[config_id] * 300),
|
||||
"pass_rate": scores[config_id],
|
||||
"interval": {"elapsed_s": 300.0},
|
||||
}
|
||||
write_json(
|
||||
run_root
|
||||
/ "sessions"
|
||||
/ config_id
|
||||
/ f"rep{repetition}"
|
||||
/ "result.json",
|
||||
result,
|
||||
)
|
||||
state = {
|
||||
"status": "complete",
|
||||
"gpu_hours_total": 4.8,
|
||||
"sessions": sessions,
|
||||
}
|
||||
write_json(run_root / "controller-state.json", state)
|
||||
mode_base = {
|
||||
"selected_cutoff_s": 300.0,
|
||||
"selected_action": "mns",
|
||||
"decision_kind": "exploit",
|
||||
"intervention_order": ["mns", "mbbt", "joint", "noop"],
|
||||
}
|
||||
mode_telemetry = {
|
||||
"selected_cutoff_s": 150.0,
|
||||
"selected_action": "joint",
|
||||
"decision_kind": "exploit",
|
||||
"intervention_order": ["joint", "mns", "mbbt", "noop"],
|
||||
}
|
||||
decision = {
|
||||
"schema": "active-intervention-prospective-decision-v0",
|
||||
"manifest_sha256": analyzer.sha256_file(manifest_path),
|
||||
"decisions": {
|
||||
"outcome_only": mode_base,
|
||||
"telemetry": mode_telemetry,
|
||||
},
|
||||
}
|
||||
decision_path = root / "decision.json"
|
||||
write_json(decision_path, decision)
|
||||
audit = analyzer.build_audit(
|
||||
manifest_path=manifest_path,
|
||||
decision_path=decision_path,
|
||||
run_root=run_root,
|
||||
)
|
||||
assert audit["status"] == "TRIGGER_ACTUAL_EARLY_STOP_CONFIRMATION"
|
||||
assert audit["comparison"]["telemetry_gpu_cost_reduction_fraction"] > 0.10
|
||||
assert not audit["sanity"]["red_flags"]
|
||||
print("active intervention prospective pipeline: PASS")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
151
runs/active-intervention-v0/trace13-results.json
Normal file
151
runs/active-intervention-v0/trace13-results.json
Normal file
@@ -0,0 +1,151 @@
|
||||
{
|
||||
"schema": "active-intervention-trace13-result-summary-v0",
|
||||
"status": "STOP_NO_PROSPECTIVE_GPU_COST_SIGNAL",
|
||||
"provenance": {
|
||||
"aituner_commit": "39b767e384fc49da53b99ad06a3e2ca1b6ac37d6",
|
||||
"vllm_commit": "4b253fd8619764b6971a7f2e3a3aa7545f6ace05",
|
||||
"manifest_sha256": "3bd25ae0ca040729a6351635f14447b3c789d4d86fb0fe4d65940735ad225a78",
|
||||
"policy_sha256": "4f096d3a5f8c38771e956dfd576dd6cd5d5691286ab258028dbddbe07b20078c",
|
||||
"decision_sha256": "86b7089151c480e18b5ae6ed65c4e4a3e11159dd0c16d5851312d4d9196d5ca2",
|
||||
"controller_state_sha256": "d10d3cbe5ce20eacc1392f52e79d16ae23b5e15301123d5e85e48d53ae676cda",
|
||||
"audit_sha256": "bd95b45e5d4cb93b5ad2f722b7dc96553d64eb8a5d5aae3a82f29c2d015fe3f6",
|
||||
"remote_root": "/home/admin/cpfs/wjh/active-intervention-prospective-20260715"
|
||||
},
|
||||
"cost": {
|
||||
"annotation_campaign_h20_hours": 5.0379046784506905,
|
||||
"hard_cap_h20_hours": 6.0,
|
||||
"outcome_only_cost_to_acceptable_h20_hours_lower_bound": 2.4284364508172893,
|
||||
"telemetry_cost_to_acceptable_h20_hours_lower_bound": 2.4284364508172893,
|
||||
"telemetry_gpu_cost_reduction_fraction": 0.0,
|
||||
"session_h20_hours": {
|
||||
"source_mns32_mbbt4096": 1.3566088432735868,
|
||||
"mns64_mbbt4096": 1.256969277858734,
|
||||
"mns32_mbbt8192": 1.254111782974667,
|
||||
"joint_mns64_mbbt8192": 1.1702147743437026
|
||||
}
|
||||
},
|
||||
"policy_comparison": {
|
||||
"outcome_only": {
|
||||
"measurement_cutoff_s": 300.0,
|
||||
"selected_action": "joint",
|
||||
"intervention_order": ["joint", "mns", "mbbt", "noop"]
|
||||
},
|
||||
"telemetry": {
|
||||
"measurement_cutoff_s": 300.0,
|
||||
"selected_action": "joint",
|
||||
"intervention_order": ["joint", "mns", "mbbt", "noop"]
|
||||
},
|
||||
"action_changed": false,
|
||||
"measurement_changed": false,
|
||||
"confirmation_trigger": false,
|
||||
"contribution_gate": false
|
||||
},
|
||||
"surface": {
|
||||
"source_mns32_mbbt4096": {
|
||||
"normalized_slo_goodput": [
|
||||
0.40090909090909094,
|
||||
0.3978787878787879,
|
||||
0.42060606060606065
|
||||
],
|
||||
"median": 0.40090909090909094
|
||||
},
|
||||
"mns64_mbbt4096": {
|
||||
"normalized_slo_goodput": [1.0, 0.9996969696969698, 1.0],
|
||||
"median": 1.0
|
||||
},
|
||||
"mns32_mbbt8192": {
|
||||
"normalized_slo_goodput": [
|
||||
0.44393939393939397,
|
||||
0.41515151515151516,
|
||||
0.4260606060606061
|
||||
],
|
||||
"median": 0.42606060606060603
|
||||
},
|
||||
"joint_mns64_mbbt8192": {
|
||||
"normalized_slo_goodput": [1.0, 1.0, 1.0],
|
||||
"median": 1.0
|
||||
}
|
||||
},
|
||||
"selected_checkpoint_prediction": {
|
||||
"actual_median_effect": {
|
||||
"noop": 0.0,
|
||||
"mns": 0.5990909090909091,
|
||||
"mbbt": 0.02515151515151509,
|
||||
"joint": 0.5990909090909091
|
||||
},
|
||||
"outcome_only_predicted_effect": {
|
||||
"noop": 0.0,
|
||||
"mns": 0.2886250281729182,
|
||||
"mbbt": 0.17933598309437812,
|
||||
"joint": 0.3205015384324615
|
||||
},
|
||||
"telemetry_predicted_effect": {
|
||||
"noop": 0.0,
|
||||
"mns": 0.26117798146236215,
|
||||
"mbbt": 0.09686132563074483,
|
||||
"joint": 0.35190199346536294
|
||||
},
|
||||
"actual_joint_minus_mns": 0.0,
|
||||
"outcome_only_joint_minus_mns": 0.0318765102595433,
|
||||
"telemetry_joint_minus_mns": 0.09072401200300079
|
||||
},
|
||||
"engine_mechanism": {
|
||||
"source_mns32_mbbt4096": {
|
||||
"scheduler_records": 41086,
|
||||
"waiting_fraction": 0.9312174463320839,
|
||||
"mns_exclusive_fraction": 0.8536484447256973,
|
||||
"mbbt_exclusive_fraction": 0.01114734946210388,
|
||||
"both_fraction": 0.06642165214428272,
|
||||
"running_utilization_mean": 0.9738878510928297,
|
||||
"token_utilization_mean": 0.15694342925509905,
|
||||
"kv_usage_mean": 0.027507593814715858,
|
||||
"preemptions": 0
|
||||
},
|
||||
"mns64_mbbt4096": {
|
||||
"scheduler_records": 37001,
|
||||
"waiting_fraction": 0.053809356503878275,
|
||||
"mns_exclusive_fraction": 0.0,
|
||||
"mbbt_exclusive_fraction": 0.053809356503878275,
|
||||
"running_utilization_mean": 0.5410364753655307,
|
||||
"token_utilization_mean": 0.17425695631536986,
|
||||
"kv_usage_mean": 0.030549112146415616,
|
||||
"preemptions": 0
|
||||
},
|
||||
"mns32_mbbt8192": {
|
||||
"scheduler_records": 41348,
|
||||
"waiting_fraction": 0.9119425365192996,
|
||||
"mns_exclusive_fraction": 0.9108542130211861,
|
||||
"mbbt_exclusive_fraction": 0.0003627744993711909,
|
||||
"running_utilization_mean": 0.9652567838831383,
|
||||
"token_utilization_mean": 0.0779606414231039,
|
||||
"kv_usage_mean": 0.027355900366122385,
|
||||
"preemptions": 0
|
||||
},
|
||||
"joint_mns64_mbbt8192": {
|
||||
"scheduler_records": 40416,
|
||||
"waiting_fraction": 0.0088826207442597,
|
||||
"mns_exclusive_fraction": 0.0,
|
||||
"mbbt_exclusive_fraction": 0.0088826207442597,
|
||||
"running_utilization_mean": 0.49403392220902614,
|
||||
"token_utilization_mean": 0.07978070546782215,
|
||||
"kv_usage_mean": 0.028003770774946098,
|
||||
"preemptions": 0
|
||||
}
|
||||
},
|
||||
"sanity": {
|
||||
"surface_outcomes": {"n": 12, "min": 0.3978787878787879, "max": 1.0, "distinct_n": 8},
|
||||
"session_h20_hours": {"n": 4, "min": 1.1702147743437026, "max": 1.3566088432735868, "distinct_n": 4},
|
||||
"scheduler_records": {"n": 4, "min": 37001, "max": 41348, "distinct_n": 4},
|
||||
"invariants": {
|
||||
"controller_complete": true,
|
||||
"four_sessions_complete": true,
|
||||
"twelve_surface_outcomes": true,
|
||||
"ratios_bounded": true,
|
||||
"nonnegative_counts_and_costs": true,
|
||||
"surface_not_all_identical": true,
|
||||
"request_hashes_match": true,
|
||||
"no_censored_runs": true
|
||||
},
|
||||
"red_flags": []
|
||||
}
|
||||
}
|
||||
519
runs/active-intervention-v0/train_policy.py
Normal file
519
runs/active-intervention-v0/train_policy.py
Normal file
@@ -0,0 +1,519 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Train and audit outcome-only versus telemetry action-response policies."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import importlib.util
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import sys
|
||||
from collections import defaultdict
|
||||
from pathlib import Path
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
|
||||
|
||||
def _load_model():
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
"active_intervention_model", HERE / "model.py"
|
||||
)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
assert spec.loader is not None
|
||||
sys.modules[spec.name] = module
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
MODEL = _load_model()
|
||||
REGULARIZATION = 10.0
|
||||
MINIMUM_MARGIN = 0.02
|
||||
CONFIDENCE_Z = 1.0
|
||||
ACCEPTABLE_REGRET = 0.02
|
||||
|
||||
|
||||
def sha256_file(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as source:
|
||||
for chunk in iter(lambda: source.read(1 << 20), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def atomic_json(path: Path, payload: Any) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary = path.with_suffix(path.suffix + ".tmp")
|
||||
temporary.write_text(
|
||||
json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8"
|
||||
)
|
||||
os.replace(temporary, path)
|
||||
|
||||
|
||||
def grouped(
|
||||
examples: Sequence[Mapping[str, Any]], key: str
|
||||
) -> dict[str, list[Mapping[str, Any]]]:
|
||||
result: dict[str, list[Mapping[str, Any]]] = defaultdict(list)
|
||||
for example in examples:
|
||||
result[str(example[key])].append(example)
|
||||
return dict(result)
|
||||
|
||||
|
||||
def evaluate_grouped_cv(
|
||||
examples: Sequence[Mapping[str, Any]],
|
||||
*,
|
||||
include_telemetry: bool,
|
||||
holdout_key: str,
|
||||
) -> dict[str, Any]:
|
||||
holdouts = grouped(examples, holdout_key)
|
||||
decision_rows = []
|
||||
for held_out, test_examples in sorted(holdouts.items()):
|
||||
training = [example for example in examples if str(example[holdout_key]) != held_out]
|
||||
if len({str(example["decision_id"]) for example in training}) < 2:
|
||||
continue
|
||||
model = MODEL.fit_ridge(
|
||||
training,
|
||||
include_telemetry=include_telemetry,
|
||||
regularization=REGULARIZATION,
|
||||
)
|
||||
for decision_id, candidates in sorted(grouped(test_examples, "decision_id").items()):
|
||||
predictions = []
|
||||
for candidate in candidates:
|
||||
source = candidate["source"]
|
||||
action = candidate["action"]
|
||||
noop = (
|
||||
int(action["target_mns"]) == int(source["mns"])
|
||||
and int(action["target_mbbt"]) == int(source["mbbt"])
|
||||
)
|
||||
if noop:
|
||||
prediction = 0.0
|
||||
else:
|
||||
names, vector = MODEL.feature_vector(
|
||||
candidate, include_telemetry=include_telemetry
|
||||
)
|
||||
if tuple(names) != model.feature_names:
|
||||
raise ValueError("cross-validation feature schema mismatch")
|
||||
prediction = max(-1.0, min(1.0, model.predict(vector)))
|
||||
predictions.append(
|
||||
{
|
||||
"action_id": str(candidate["action"]["id"]),
|
||||
"prediction": prediction,
|
||||
"real": float(candidate["target_normalized_goodput"]),
|
||||
}
|
||||
)
|
||||
predictions.sort(key=lambda row: (-row["prediction"], row["action_id"]))
|
||||
selected = predictions[0]
|
||||
oracle = max(row["real"] for row in predictions)
|
||||
regret = 1.0 - selected["real"] / oracle if oracle > 0 else 0.0
|
||||
best_actions = {
|
||||
row["action_id"] for row in predictions if math.isclose(row["real"], oracle)
|
||||
}
|
||||
acceptable_actions = {
|
||||
row["action_id"]
|
||||
for row in predictions
|
||||
if oracle <= 0
|
||||
or 1.0 - float(row["real"]) / oracle <= ACCEPTABLE_REGRET + 1e-12
|
||||
}
|
||||
decision_rows.append(
|
||||
{
|
||||
"holdout": held_out,
|
||||
"decision_id": decision_id,
|
||||
"selected_action": selected["action_id"],
|
||||
"best_actions": sorted(best_actions),
|
||||
"acceptable_actions": sorted(acceptable_actions),
|
||||
"correct": regret <= ACCEPTABLE_REGRET + 1e-12,
|
||||
"selected_real": selected["real"],
|
||||
"oracle_real": oracle,
|
||||
"regret": regret,
|
||||
"predictions": predictions,
|
||||
}
|
||||
)
|
||||
if not decision_rows:
|
||||
return {"status": "INSUFFICIENT_GROUPS", "decisions": []}
|
||||
regrets = [float(row["regret"]) for row in decision_rows]
|
||||
return {
|
||||
"status": "VALID",
|
||||
"holdout_key": holdout_key,
|
||||
"acceptable_regret": ACCEPTABLE_REGRET,
|
||||
"decision_n": len(decision_rows),
|
||||
"correct_n": sum(bool(row["correct"]) for row in decision_rows),
|
||||
"accuracy": sum(bool(row["correct"]) for row in decision_rows) / len(decision_rows),
|
||||
"mean_regret": sum(regrets) / len(regrets),
|
||||
"max_regret": max(regrets),
|
||||
"decisions": decision_rows,
|
||||
}
|
||||
|
||||
|
||||
def paired_delta(outcome: Mapping[str, Any], telemetry: Mapping[str, Any]) -> dict[str, Any]:
|
||||
if outcome.get("status") != "VALID" or telemetry.get("status") != "VALID":
|
||||
return {"status": "INSUFFICIENT_GROUPS"}
|
||||
outcome_by_id = {row["decision_id"]: row for row in outcome["decisions"]}
|
||||
telemetry_by_id = {row["decision_id"]: row for row in telemetry["decisions"]}
|
||||
common = sorted(set(outcome_by_id) & set(telemetry_by_id))
|
||||
rows = []
|
||||
for decision_id in common:
|
||||
before = outcome_by_id[decision_id]
|
||||
after = telemetry_by_id[decision_id]
|
||||
rows.append(
|
||||
{
|
||||
"decision_id": decision_id,
|
||||
"outcome_action": before["selected_action"],
|
||||
"telemetry_action": after["selected_action"],
|
||||
"action_changed": before["selected_action"] != after["selected_action"],
|
||||
"regret_delta": float(after["regret"]) - float(before["regret"]),
|
||||
"telemetry_corrected": (not before["correct"]) and bool(after["correct"]),
|
||||
"telemetry_harmed": bool(before["correct"]) and (not after["correct"]),
|
||||
}
|
||||
)
|
||||
return {
|
||||
"status": "VALID",
|
||||
"decision_n": len(rows),
|
||||
"action_changed_n": sum(row["action_changed"] for row in rows),
|
||||
"corrected_n": sum(row["telemetry_corrected"] for row in rows),
|
||||
"harmed_n": sum(row["telemetry_harmed"] for row in rows),
|
||||
"mean_regret_delta": (
|
||||
sum(float(row["regret_delta"]) for row in rows) / len(rows) if rows else 0.0
|
||||
),
|
||||
"rows": rows,
|
||||
}
|
||||
|
||||
|
||||
def evaluate_sequential_measurement_cv(
|
||||
examples: Sequence[Mapping[str, Any]],
|
||||
*,
|
||||
include_telemetry: bool,
|
||||
holdout_key: str,
|
||||
) -> dict[str, Any]:
|
||||
"""Replay a two-consecutive-confident-checkpoint measurement policy."""
|
||||
|
||||
phases = sorted({str(example["phase"]) for example in examples}, key=float)
|
||||
holdouts = grouped(examples, holdout_key)
|
||||
rows = []
|
||||
full_duration_s = max(float(example["cutoff_s"]) for example in examples)
|
||||
for held_out, test_examples in sorted(holdouts.items()):
|
||||
training = [
|
||||
example for example in examples if str(example[holdout_key]) != held_out
|
||||
]
|
||||
if len({str(example["decision_id"]) for example in training}) < 3:
|
||||
continue
|
||||
phase_models = {}
|
||||
for phase in phases:
|
||||
phase_training = [
|
||||
example for example in training if str(example["phase"]) == phase
|
||||
]
|
||||
phase_models[phase] = MODEL.fit_jackknife_ensemble(
|
||||
phase_training,
|
||||
include_telemetry=include_telemetry,
|
||||
regularization=REGULARIZATION,
|
||||
)
|
||||
for decision_id, decision_examples in sorted(
|
||||
grouped(test_examples, "decision_id").items()
|
||||
):
|
||||
checkpoints = []
|
||||
by_phase = grouped(decision_examples, "phase")
|
||||
for phase in phases:
|
||||
candidates = by_phase[phase]
|
||||
decision = MODEL.select_action(
|
||||
phase_models[phase],
|
||||
candidates,
|
||||
include_telemetry=include_telemetry,
|
||||
confidence_z=CONFIDENCE_Z,
|
||||
minimum_margin=MINIMUM_MARGIN,
|
||||
)
|
||||
checkpoints.append(
|
||||
{
|
||||
"phase": phase,
|
||||
"cutoff_s": float(candidates[0]["cutoff_s"]),
|
||||
**decision,
|
||||
}
|
||||
)
|
||||
selected_checkpoint = checkpoints[-1]
|
||||
stop_reason = "full_measurement_fallback"
|
||||
for previous, current in zip(checkpoints, checkpoints[1:], strict=False):
|
||||
if (
|
||||
previous["confident"]
|
||||
and current["confident"]
|
||||
and previous["selected_action"] == current["selected_action"]
|
||||
):
|
||||
selected_checkpoint = current
|
||||
stop_reason = "two_consecutive_confident_checkpoints"
|
||||
break
|
||||
candidates = by_phase[str(selected_checkpoint["phase"])]
|
||||
real_by_action = {
|
||||
str(candidate["action"]["id"]): float(
|
||||
candidate["target_normalized_goodput"]
|
||||
)
|
||||
for candidate in candidates
|
||||
}
|
||||
target_by_action = {
|
||||
str(candidate["action"]["id"]): str(
|
||||
candidate["action"]["target_config_id"]
|
||||
)
|
||||
for candidate in candidates
|
||||
}
|
||||
selected_action = str(selected_checkpoint["selected_action"])
|
||||
oracle = max(real_by_action.values())
|
||||
selected_real = real_by_action[selected_action]
|
||||
regret = 1.0 - selected_real / oracle if oracle > 0 else 0.0
|
||||
source_tp = 4
|
||||
target_s = 0.0 if selected_action == "noop" else full_duration_s
|
||||
replay_gpu_seconds = source_tp * (
|
||||
float(selected_checkpoint["cutoff_s"]) + target_s
|
||||
)
|
||||
rows.append(
|
||||
{
|
||||
"holdout": held_out,
|
||||
"decision_id": decision_id,
|
||||
"selected_phase": str(selected_checkpoint["phase"]),
|
||||
"selected_cutoff_s": float(selected_checkpoint["cutoff_s"]),
|
||||
"stop_reason": stop_reason,
|
||||
"selected_action": selected_action,
|
||||
"selected_target_config_id": target_by_action[selected_action],
|
||||
"selected_real": selected_real,
|
||||
"oracle_real": oracle,
|
||||
"regret": regret,
|
||||
"acceptable": regret <= ACCEPTABLE_REGRET + 1e-12,
|
||||
"replay_gpu_seconds_lower_bound": replay_gpu_seconds,
|
||||
"checkpoints": checkpoints,
|
||||
}
|
||||
)
|
||||
if not rows:
|
||||
return {"status": "INSUFFICIENT_GROUPS", "decisions": []}
|
||||
regrets = [float(row["regret"]) for row in rows]
|
||||
cutoffs = [float(row["selected_cutoff_s"]) for row in rows]
|
||||
costs = [float(row["replay_gpu_seconds_lower_bound"]) for row in rows]
|
||||
return {
|
||||
"status": "VALID",
|
||||
"holdout_key": holdout_key,
|
||||
"measurement_rule": "earliest two consecutive confident checkpoints; otherwise full",
|
||||
"acceptable_regret": ACCEPTABLE_REGRET,
|
||||
"decision_n": len(rows),
|
||||
"acceptable_n": sum(bool(row["acceptable"]) for row in rows),
|
||||
"mean_regret": sum(regrets) / len(regrets),
|
||||
"max_regret": max(regrets),
|
||||
"mean_cutoff_s": sum(cutoffs) / len(cutoffs),
|
||||
"total_replay_gpu_seconds_lower_bound": sum(costs),
|
||||
"total_replay_h20_hours_lower_bound": sum(costs) / 3600.0,
|
||||
"decisions": rows,
|
||||
}
|
||||
|
||||
|
||||
def paired_sequential_delta(
|
||||
outcome: Mapping[str, Any], telemetry: Mapping[str, Any]
|
||||
) -> dict[str, Any]:
|
||||
if outcome.get("status") != "VALID" or telemetry.get("status") != "VALID":
|
||||
return {"status": "INSUFFICIENT_GROUPS"}
|
||||
before_by_id = {row["decision_id"]: row for row in outcome["decisions"]}
|
||||
after_by_id = {row["decision_id"]: row for row in telemetry["decisions"]}
|
||||
rows = []
|
||||
for decision_id in sorted(set(before_by_id) & set(after_by_id)):
|
||||
before = before_by_id[decision_id]
|
||||
after = after_by_id[decision_id]
|
||||
rows.append(
|
||||
{
|
||||
"decision_id": decision_id,
|
||||
"outcome_action": before["selected_action"],
|
||||
"telemetry_action": after["selected_action"],
|
||||
"outcome_cutoff_s": before["selected_cutoff_s"],
|
||||
"telemetry_cutoff_s": after["selected_cutoff_s"],
|
||||
"outcome_regret": before["regret"],
|
||||
"telemetry_regret": after["regret"],
|
||||
"regret_delta": float(after["regret"]) - float(before["regret"]),
|
||||
"gpu_seconds_delta": float(
|
||||
after["replay_gpu_seconds_lower_bound"]
|
||||
)
|
||||
- float(before["replay_gpu_seconds_lower_bound"]),
|
||||
"telemetry_corrected": (not before["acceptable"])
|
||||
and bool(after["acceptable"]),
|
||||
"telemetry_harmed": bool(before["acceptable"])
|
||||
and (not after["acceptable"]),
|
||||
}
|
||||
)
|
||||
outcome_cost = float(outcome["total_replay_gpu_seconds_lower_bound"])
|
||||
telemetry_cost = float(telemetry["total_replay_gpu_seconds_lower_bound"])
|
||||
return {
|
||||
"status": "VALID",
|
||||
"decision_n": len(rows),
|
||||
"corrected_n": sum(row["telemetry_corrected"] for row in rows),
|
||||
"harmed_n": sum(row["telemetry_harmed"] for row in rows),
|
||||
"outcome_replay_gpu_seconds_lower_bound": outcome_cost,
|
||||
"telemetry_replay_gpu_seconds_lower_bound": telemetry_cost,
|
||||
"gpu_cost_reduction_fraction": (
|
||||
1.0 - telemetry_cost / outcome_cost if outcome_cost > 0 else 0.0
|
||||
),
|
||||
"rows": rows,
|
||||
}
|
||||
|
||||
|
||||
def build_policy(dataset_path: Path) -> dict[str, Any]:
|
||||
dataset = json.loads(dataset_path.read_text(encoding="utf-8"))
|
||||
if dataset.get("status") != "VALID" or dataset["sanity"]["red_flags"]:
|
||||
raise ValueError("training dataset is not valid")
|
||||
examples = dataset["examples"]
|
||||
phases = sorted({str(example["phase"]) for example in examples}, key=float)
|
||||
phase_results = {}
|
||||
incremental_candidates = []
|
||||
for phase in phases:
|
||||
selected = [example for example in examples if str(example["phase"]) == phase]
|
||||
outcome_cv = evaluate_grouped_cv(
|
||||
selected, include_telemetry=False, holdout_key="repetition"
|
||||
)
|
||||
telemetry_cv = evaluate_grouped_cv(
|
||||
selected, include_telemetry=True, holdout_key="repetition"
|
||||
)
|
||||
outcome_regime = evaluate_grouped_cv(
|
||||
selected, include_telemetry=False, holdout_key="regime"
|
||||
)
|
||||
telemetry_regime = evaluate_grouped_cv(
|
||||
selected, include_telemetry=True, holdout_key="regime"
|
||||
)
|
||||
delta = paired_delta(outcome_cv, telemetry_cv)
|
||||
outcome_models = MODEL.fit_jackknife_ensemble(
|
||||
selected,
|
||||
include_telemetry=False,
|
||||
regularization=REGULARIZATION,
|
||||
)
|
||||
telemetry_models = MODEL.fit_jackknife_ensemble(
|
||||
selected,
|
||||
include_telemetry=True,
|
||||
regularization=REGULARIZATION,
|
||||
)
|
||||
incremental = bool(
|
||||
delta.get("status") == "VALID"
|
||||
and int(delta["corrected_n"]) >= 1
|
||||
and int(delta["harmed_n"]) == 0
|
||||
and float(delta["mean_regret_delta"]) < -1e-12
|
||||
and float(telemetry_cv["max_regret"]) <= 0.05
|
||||
and telemetry_regime.get("status") == "VALID"
|
||||
and float(telemetry_regime["mean_regret"])
|
||||
<= float(outcome_regime["mean_regret"]) + 1e-12
|
||||
and float(telemetry_regime["max_regret"]) <= 0.05
|
||||
)
|
||||
if incremental:
|
||||
incremental_candidates.append(phase)
|
||||
phase_results[phase] = {
|
||||
"cutoff_s": float(selected[0]["cutoff_s"]),
|
||||
"outcome_only": {
|
||||
"leave_repetition_out": outcome_cv,
|
||||
"leave_regime_out": outcome_regime,
|
||||
"models": MODEL.models_to_json(outcome_models),
|
||||
},
|
||||
"telemetry": {
|
||||
"leave_repetition_out": telemetry_cv,
|
||||
"leave_regime_out": telemetry_regime,
|
||||
"models": MODEL.models_to_json(telemetry_models),
|
||||
},
|
||||
"paired_incremental": delta,
|
||||
"incremental_gate": incremental,
|
||||
}
|
||||
outcome_sequential = evaluate_sequential_measurement_cv(
|
||||
examples, include_telemetry=False, holdout_key="repetition"
|
||||
)
|
||||
telemetry_sequential = evaluate_sequential_measurement_cv(
|
||||
examples, include_telemetry=True, holdout_key="repetition"
|
||||
)
|
||||
sequential_delta = paired_sequential_delta(
|
||||
outcome_sequential, telemetry_sequential
|
||||
)
|
||||
retrospective_cost_gate = bool(
|
||||
sequential_delta.get("status") == "VALID"
|
||||
and int(sequential_delta["harmed_n"]) == 0
|
||||
and int(telemetry_sequential["acceptable_n"])
|
||||
>= int(outcome_sequential["acceptable_n"])
|
||||
and float(telemetry_sequential["max_regret"]) <= 0.05
|
||||
and float(sequential_delta["gpu_cost_reduction_fraction"]) >= 0.10
|
||||
)
|
||||
status = (
|
||||
"RETROSPECTIVE_GPU_COST_SIGNAL"
|
||||
if retrospective_cost_gate
|
||||
else "NO_RETROSPECTIVE_GPU_COST_SIGNAL"
|
||||
)
|
||||
target_values = [float(example["target_normalized_goodput"]) for example in examples]
|
||||
effect_values = [
|
||||
float(example["target_delta_normalized_goodput"]) for example in examples
|
||||
]
|
||||
invariants = {
|
||||
"four_phases": len(phases) == 4,
|
||||
"targets_bounded": all(0.0 <= value <= 1.0 for value in target_values),
|
||||
"targets_not_all_identical": len(set(target_values)) > 1,
|
||||
"effects_bounded": all(-1.0 <= value <= 1.0 for value in effect_values),
|
||||
"effects_not_all_identical": len(set(effect_values)) > 1,
|
||||
"models_present_every_phase": all(
|
||||
phase_results[phase][mode]["models"]
|
||||
for phase in phases
|
||||
for mode in ("outcome_only", "telemetry")
|
||||
),
|
||||
}
|
||||
red_flags = [name for name, passed in invariants.items() if not passed]
|
||||
if red_flags:
|
||||
raise RuntimeError(f"policy sanity failed: {red_flags}")
|
||||
return {
|
||||
"schema": "active-intervention-policy-v0",
|
||||
"status": status,
|
||||
"training": {
|
||||
"dataset": str(dataset_path),
|
||||
"dataset_sha256": sha256_file(dataset_path),
|
||||
"examples": len(examples),
|
||||
"decisions": len({example["decision_id"] for example in examples}),
|
||||
"regularization": REGULARIZATION,
|
||||
"confidence_z": CONFIDENCE_Z,
|
||||
"minimum_margin": MINIMUM_MARGIN,
|
||||
"acceptable_regret": ACCEPTABLE_REGRET,
|
||||
},
|
||||
"measurement_policy": {
|
||||
"rule": "earliest two consecutive confident checkpoints; otherwise full",
|
||||
"checkpoints": [phase_results[phase]["cutoff_s"] for phase in phases],
|
||||
"confidence_z": CONFIDENCE_Z,
|
||||
"minimum_margin": MINIMUM_MARGIN,
|
||||
},
|
||||
"sequential_replay": {
|
||||
"outcome_only": outcome_sequential,
|
||||
"telemetry": telemetry_sequential,
|
||||
"paired_delta": sequential_delta,
|
||||
"retrospective_gpu_cost_gate": retrospective_cost_gate,
|
||||
"minimum_cost_reduction_fraction": 0.10,
|
||||
},
|
||||
"phases": phase_results,
|
||||
"sanity": {
|
||||
"invariants": invariants,
|
||||
"red_flags": red_flags,
|
||||
"target_normalized_goodput": {
|
||||
"n": len(target_values),
|
||||
"min": min(target_values),
|
||||
"max": max(target_values),
|
||||
"distinct_n": len(set(target_values)),
|
||||
},
|
||||
"target_delta_normalized_goodput": {
|
||||
"n": len(effect_values),
|
||||
"min": min(effect_values),
|
||||
"max": max(effect_values),
|
||||
"distinct_n": len(set(effect_values)),
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--dataset", type=Path, required=True)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
policy = build_policy(args.dataset)
|
||||
atomic_json(args.output, policy)
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"status": policy["status"],
|
||||
"measurement_policy": policy["measurement_policy"],
|
||||
"sanity": policy["sanity"],
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -37,6 +37,41 @@ def selection_for(
|
||||
return manifest["cells"][cell]["targets"][level]["selections"][role]
|
||||
|
||||
|
||||
def campaign_gpu_accounting(
|
||||
primary_state_path: Path, prior_state_paths: tuple[Path, ...] = ()
|
||||
) -> dict[str, Any]:
|
||||
attempts = []
|
||||
for role, path in (
|
||||
[("prior_failure", path) for path in prior_state_paths]
|
||||
+ [("primary", primary_state_path)]
|
||||
):
|
||||
state = json.loads(path.read_text(encoding="utf-8"))
|
||||
gpu_hours = float(state["gpu_hours_total"])
|
||||
attempts.append(
|
||||
{
|
||||
"role": role,
|
||||
"path": str(path.resolve()),
|
||||
"sha256": sha256_file(path),
|
||||
"status": state["status"],
|
||||
"h20_hours": gpu_hours,
|
||||
}
|
||||
)
|
||||
total = sum(attempt["h20_hours"] for attempt in attempts)
|
||||
primary = json.loads(primary_state_path.read_text(encoding="utf-8"))
|
||||
hard_cap = float(primary["hard_cap_h20_hours"])
|
||||
return {
|
||||
"attempts": attempts,
|
||||
"aggregate_h20_hours": total,
|
||||
"hard_cap_h20_hours": hard_cap,
|
||||
"invariants": {
|
||||
"costs_nonnegative": all(
|
||||
attempt["h20_hours"] >= 0.0 for attempt in attempts
|
||||
),
|
||||
"aggregate_below_cap": 0.0 <= total < hard_cap,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def build_pilot_examples(
|
||||
manifest: dict[str, Any], run_root: Path, cutoff_s: float
|
||||
) -> tuple[list[PrefixExample], list[dict[str, Any]], list[str]]:
|
||||
@@ -125,11 +160,13 @@ def analyze(
|
||||
manifest_path: Path,
|
||||
model_path: Path,
|
||||
run_root: Path,
|
||||
prior_state_paths: tuple[Path, ...] = (),
|
||||
) -> dict[str, Any]:
|
||||
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
||||
models = json.loads(model_path.read_text(encoding="utf-8"))
|
||||
state_path = run_root / "controller-state.json"
|
||||
state = json.loads(state_path.read_text(encoding="utf-8"))
|
||||
gpu_accounting = campaign_gpu_accounting(state_path, prior_state_paths)
|
||||
cutoff_s = float(models["cutoff_s"])
|
||||
threshold = float(models["accept_probability"])
|
||||
examples, details, red_flags = build_pilot_examples(manifest, run_root, cutoff_s)
|
||||
@@ -171,7 +208,7 @@ def analyze(
|
||||
detail["actual_timestamped_outcomes"] == 0 for detail in details
|
||||
):
|
||||
red_flags.append("no_exact_request_timestamps")
|
||||
if float(state["gpu_hours_total"]) >= float(state["hard_cap_h20_hours"]):
|
||||
if not all(gpu_accounting["invariants"].values()):
|
||||
red_flags.append("hard_cap_exceeded")
|
||||
|
||||
outcome_errors = outcome_policy["false_accept"] + outcome_policy["false_reject"]
|
||||
@@ -232,8 +269,8 @@ def analyze(
|
||||
"opens_expanded_p2": pilot_pass,
|
||||
},
|
||||
"gpu": {
|
||||
"actual_h20_hours": state["gpu_hours_total"],
|
||||
"hard_cap_h20_hours": state["hard_cap_h20_hours"],
|
||||
"primary_attempt_h20_hours": state["gpu_hours_total"],
|
||||
**gpu_accounting,
|
||||
},
|
||||
"sanity": {
|
||||
"red_flags": red_flags,
|
||||
@@ -277,9 +314,15 @@ def main() -> None:
|
||||
parser.add_argument("--manifest", type=Path, required=True)
|
||||
parser.add_argument("--frozen-models", type=Path, required=True)
|
||||
parser.add_argument("--run-root", type=Path, required=True)
|
||||
parser.add_argument("--prior-state", type=Path, action="append", default=[])
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
result = analyze(args.manifest, args.frozen_models, args.run_root)
|
||||
result = analyze(
|
||||
args.manifest,
|
||||
args.frozen_models,
|
||||
args.run_root,
|
||||
tuple(args.prior_state),
|
||||
)
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.output.write_text(json.dumps(result, indent=2, sort_keys=True) + "\n")
|
||||
print(json.dumps({
|
||||
|
||||
429
runs/fidelity-headroom/analyze_pilot_e2e.py
Normal file
429
runs/fidelity-headroom/analyze_pilot_e2e.py
Normal file
@@ -0,0 +1,429 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Replay the P1 simulator shortlist under full and prefix policies."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import math
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from analyze_prefixes import numeric, sha256_file
|
||||
|
||||
|
||||
AITUNER_ROOT = Path(__file__).resolve().parents[2]
|
||||
FROZEN_K = 2
|
||||
CUTOFF_S = 5.0
|
||||
THRESHOLD = 0.95
|
||||
|
||||
|
||||
def git_capture(*arguments: str) -> str:
|
||||
return subprocess.run(
|
||||
["git", "-C", str(AITUNER_ROOT), *arguments],
|
||||
check=True,
|
||||
text=True,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
).stdout
|
||||
|
||||
|
||||
def setup_costs(state: dict[str, Any]) -> dict[str, float]:
|
||||
result = {}
|
||||
for cell, payload in state["cells"].items():
|
||||
tp = int(payload["tp"])
|
||||
annotation_intervals = sum(
|
||||
float(run["elapsed_s"]) * tp / 3600.0
|
||||
for run in payload["runs"]
|
||||
if run["role"] not in {"low1", "high1"}
|
||||
)
|
||||
primary_intervals = sum(
|
||||
float(run["elapsed_s"]) * tp / 3600.0
|
||||
for run in payload["runs"]
|
||||
if run["role"] in {"low1", "high1"}
|
||||
)
|
||||
setup = float(payload["gpu_hours"]) - annotation_intervals - primary_intervals
|
||||
if setup < -1e-12:
|
||||
raise ValueError(f"negative inferred setup cost: {cell}={setup}")
|
||||
result[cell] = max(0.0, setup)
|
||||
return result
|
||||
|
||||
|
||||
def build_candidates(
|
||||
manifest: dict[str, Any],
|
||||
state: dict[str, Any],
|
||||
strong: dict[str, Any],
|
||||
) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
|
||||
baseline_probability = strong["headline"]["sim_plus_outcome"]["probability"]
|
||||
instrument_probability = strong["headline"][
|
||||
"sim_plus_outcome_plus_instrumentation"
|
||||
]["probability"]
|
||||
setup = setup_costs(state)
|
||||
anchors = []
|
||||
for detail, baseline_p, instrument_p in zip(
|
||||
strong["pilot_examples"], baseline_probability, instrument_probability
|
||||
):
|
||||
cell = str(detail["cell"])
|
||||
level = str(detail["level"])
|
||||
role = f"{level}1"
|
||||
selection = manifest["cells"][cell]["targets"][level]["selections"][role]
|
||||
run = next(
|
||||
item for item in state["cells"][cell]["runs"] if item["role"] == role
|
||||
)
|
||||
tp = int(state["cells"][cell]["tp"])
|
||||
full_cost = float(run["elapsed_s"]) * tp / 3600.0
|
||||
prefix_cost = min(CUTOFF_S, float(run["elapsed_s"])) * tp / 3600.0
|
||||
anchors.append(
|
||||
{
|
||||
"cell": cell,
|
||||
"level": level,
|
||||
"role": role,
|
||||
"tp": tp,
|
||||
"real_feasible": bool(detail["adjudicated_feasible"]),
|
||||
"real_goodput_req_s_per_gpu": float(
|
||||
selection["offered_req_s_per_gpu"]
|
||||
),
|
||||
"sim_feasible": bool(detail["sim_slo_feasible"]),
|
||||
"sim_pass_rate": float(detail["sim_slo_pass_rate"]),
|
||||
"sim_throughput_req_s_per_gpu": float(
|
||||
detail["sim_completed_throughput_per_gpu"]
|
||||
),
|
||||
"baseline_probability": float(baseline_p),
|
||||
"instrument_probability": float(instrument_p),
|
||||
"setup_h20_hours": setup[cell],
|
||||
"full_trial_h20_hours": full_cost,
|
||||
"prefix_h20_hours": prefix_cost,
|
||||
}
|
||||
)
|
||||
|
||||
candidates = []
|
||||
for cell in sorted(manifest["cells"]):
|
||||
feasible = [
|
||||
anchor for anchor in anchors if anchor["cell"] == cell and anchor["sim_feasible"]
|
||||
]
|
||||
if not feasible:
|
||||
continue
|
||||
candidates.append(
|
||||
max(feasible, key=lambda anchor: anchor["sim_throughput_req_s_per_gpu"])
|
||||
)
|
||||
candidates.sort(
|
||||
key=lambda anchor: (
|
||||
-anchor["sim_throughput_req_s_per_gpu"],
|
||||
anchor["cell"],
|
||||
)
|
||||
)
|
||||
return anchors, candidates
|
||||
|
||||
|
||||
def expanded_top_k(candidates: list[dict[str, Any]], k: int) -> list[dict[str, Any]]:
|
||||
if not candidates or k <= 0:
|
||||
return []
|
||||
boundary = candidates[min(k, len(candidates)) - 1][
|
||||
"sim_throughput_req_s_per_gpu"
|
||||
]
|
||||
return [
|
||||
candidate
|
||||
for candidate in candidates
|
||||
if candidate["sim_throughput_req_s_per_gpu"] >= boundary - 1e-12
|
||||
]
|
||||
|
||||
|
||||
def selected_result(
|
||||
evaluated: list[dict[str, Any]], feasible_key: str
|
||||
) -> tuple[str | None, float | None]:
|
||||
feasible = [candidate for candidate in evaluated if candidate[feasible_key]]
|
||||
if not feasible:
|
||||
return None, None
|
||||
best = max(feasible, key=lambda candidate: candidate["real_goodput_req_s_per_gpu"])
|
||||
return str(best["cell"]), float(best["real_goodput_req_s_per_gpu"])
|
||||
|
||||
|
||||
def replay(
|
||||
shortlist: list[dict[str, Any]],
|
||||
*,
|
||||
probability_key: str | None,
|
||||
oracle_goodput: float,
|
||||
common_failure_h20_hours: float,
|
||||
) -> dict[str, Any]:
|
||||
evaluated = []
|
||||
online_cost = 0.0
|
||||
early_accept = 0
|
||||
early_reject = 0
|
||||
false_accept = 0
|
||||
false_reject = 0
|
||||
for candidate in shortlist:
|
||||
current = dict(candidate)
|
||||
online_cost += current["setup_h20_hours"]
|
||||
if probability_key is None:
|
||||
predicted_feasible = current["real_feasible"]
|
||||
online_cost += current["full_trial_h20_hours"]
|
||||
action = "full"
|
||||
else:
|
||||
probability = float(current[probability_key])
|
||||
if probability >= THRESHOLD:
|
||||
predicted_feasible = True
|
||||
early_accept += 1
|
||||
online_cost += current["prefix_h20_hours"]
|
||||
action = "early_accept"
|
||||
false_accept += int(not current["real_feasible"])
|
||||
elif probability <= 1.0 - THRESHOLD:
|
||||
predicted_feasible = False
|
||||
early_reject += 1
|
||||
online_cost += current["prefix_h20_hours"]
|
||||
action = "early_reject"
|
||||
false_reject += int(current["real_feasible"])
|
||||
else:
|
||||
predicted_feasible = current["real_feasible"]
|
||||
online_cost += current["full_trial_h20_hours"]
|
||||
action = "continue_full"
|
||||
current["policy_feasible"] = predicted_feasible
|
||||
current["action"] = action
|
||||
evaluated.append(current)
|
||||
selected_cell, selected_goodput = selected_result(evaluated, "policy_feasible")
|
||||
regret = (
|
||||
1.0 - selected_goodput / oracle_goodput
|
||||
if selected_goodput is not None and oracle_goodput > 0
|
||||
else None
|
||||
)
|
||||
return {
|
||||
"selected_cell": selected_cell,
|
||||
"selected_real_goodput_req_s_per_gpu": selected_goodput,
|
||||
"real_regret": regret,
|
||||
"online_h20_hours": online_cost,
|
||||
"conservative_h20_hours_with_prior_failure": (
|
||||
online_cost + common_failure_h20_hours
|
||||
),
|
||||
"early_accept": early_accept,
|
||||
"early_reject": early_reject,
|
||||
"false_accept": false_accept,
|
||||
"false_reject": false_reject,
|
||||
"evaluated": [
|
||||
{
|
||||
"cell": item["cell"],
|
||||
"level": item["level"],
|
||||
"action": item["action"],
|
||||
"real_feasible": item["real_feasible"],
|
||||
"policy_feasible": item["policy_feasible"],
|
||||
}
|
||||
for item in evaluated
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def analyze(
|
||||
manifest_path: Path,
|
||||
state_path: Path,
|
||||
prior_state_path: Path,
|
||||
strong_path: Path,
|
||||
) -> dict[str, Any]:
|
||||
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
||||
state = json.loads(state_path.read_text(encoding="utf-8"))
|
||||
prior = json.loads(prior_state_path.read_text(encoding="utf-8"))
|
||||
strong = json.loads(strong_path.read_text(encoding="utf-8"))
|
||||
anchors, candidates = build_candidates(manifest, state, strong)
|
||||
oracle_anchor = max(
|
||||
(anchor for anchor in anchors if anchor["real_feasible"]),
|
||||
key=lambda anchor: anchor["real_goodput_req_s_per_gpu"],
|
||||
)
|
||||
oracle_goodput = float(oracle_anchor["real_goodput_req_s_per_gpu"])
|
||||
common_failure = float(prior["gpu_hours_total"])
|
||||
by_k = {}
|
||||
for k in (1, 2, 3, 6):
|
||||
shortlist = expanded_top_k(candidates, k)
|
||||
full = replay(
|
||||
shortlist,
|
||||
probability_key=None,
|
||||
oracle_goodput=oracle_goodput,
|
||||
common_failure_h20_hours=common_failure,
|
||||
)
|
||||
baseline = replay(
|
||||
shortlist,
|
||||
probability_key="baseline_probability",
|
||||
oracle_goodput=oracle_goodput,
|
||||
common_failure_h20_hours=common_failure,
|
||||
)
|
||||
instrument = replay(
|
||||
shortlist,
|
||||
probability_key="instrument_probability",
|
||||
oracle_goodput=oracle_goodput,
|
||||
common_failure_h20_hours=common_failure,
|
||||
)
|
||||
for result in (baseline, instrument):
|
||||
result["online_cost_reduction_vs_full"] = (
|
||||
1.0 - result["online_h20_hours"] / full["online_h20_hours"]
|
||||
)
|
||||
result["conservative_cost_reduction_vs_full"] = 1.0 - (
|
||||
result["conservative_h20_hours_with_prior_failure"]
|
||||
/ full["conservative_h20_hours_with_prior_failure"]
|
||||
)
|
||||
by_k[str(k)] = {
|
||||
"actual_shortlist_size": len(shortlist),
|
||||
"shortlist": [candidate["cell"] for candidate in shortlist],
|
||||
"sim_top_k_plus_real_final": full,
|
||||
"sim_plus_outcome": baseline,
|
||||
"sim_plus_outcome_plus_instrumentation": instrument,
|
||||
}
|
||||
|
||||
frozen = by_k[str(FROZEN_K)]
|
||||
full = frozen["sim_top_k_plus_real_final"]
|
||||
baseline = frozen["sim_plus_outcome"]
|
||||
instrument = frozen["sim_plus_outcome_plus_instrumentation"]
|
||||
baseline_safe = baseline["false_accept"] == 0 and baseline["false_reject"] == 0
|
||||
instrument_safe = (
|
||||
instrument["false_accept"] == 0 and instrument["false_reject"] == 0
|
||||
)
|
||||
incremental_reduction = (
|
||||
1.0 - instrument["online_h20_hours"] / baseline["online_h20_hours"]
|
||||
if baseline_safe and instrument_safe and baseline["online_h20_hours"] > 0
|
||||
else None
|
||||
)
|
||||
contribution_gate = {
|
||||
"frozen_k": FROZEN_K,
|
||||
"instrument_safe": instrument_safe,
|
||||
"outcome_baseline_safe": baseline_safe,
|
||||
"instrument_regret_at_most_5pct": (
|
||||
instrument["real_regret"] is not None
|
||||
and instrument["real_regret"] <= 0.05
|
||||
),
|
||||
"instrument_cost_reduction_vs_full_at_least_30pct": (
|
||||
instrument["online_cost_reduction_vs_full"] >= 0.30
|
||||
),
|
||||
"instrument_cost_reduction_vs_outcome_at_least_20pct": (
|
||||
incremental_reduction is not None and incremental_reduction >= 0.20
|
||||
),
|
||||
"incremental_reduction_vs_outcome": incremental_reduction,
|
||||
}
|
||||
contribution_gate["passes"] = all(
|
||||
contribution_gate[key]
|
||||
for key in (
|
||||
"instrument_safe",
|
||||
"outcome_baseline_safe",
|
||||
"instrument_regret_at_most_5pct",
|
||||
"instrument_cost_reduction_vs_full_at_least_30pct",
|
||||
"instrument_cost_reduction_vs_outcome_at_least_20pct",
|
||||
)
|
||||
)
|
||||
|
||||
red_flags = []
|
||||
if state["status"] != "complete" or int(state["completed_cells"]) != 6:
|
||||
red_flags.append("pilot_incomplete")
|
||||
if strong["status"] != "PASS" or strong["sanity"]["red_flags"]:
|
||||
red_flags.append("strong_input_invalid")
|
||||
if len(anchors) != 12 or len(candidates) != 6:
|
||||
red_flags.append("unexpected_surface_size")
|
||||
probabilities = [
|
||||
value
|
||||
for anchor in anchors
|
||||
for value in (anchor["baseline_probability"], anchor["instrument_probability"])
|
||||
]
|
||||
costs = [
|
||||
value
|
||||
for anchor in anchors
|
||||
for value in (
|
||||
anchor["setup_h20_hours"],
|
||||
anchor["full_trial_h20_hours"],
|
||||
anchor["prefix_h20_hours"],
|
||||
)
|
||||
]
|
||||
if not all(0.0 <= value <= 1.0 for value in probabilities):
|
||||
red_flags.append("probability_out_of_range")
|
||||
if not all(value >= 0.0 and math.isfinite(value) for value in costs):
|
||||
red_flags.append("invalid_cost")
|
||||
|
||||
return {
|
||||
"schema": "fidelity-pilot-e2e-v1",
|
||||
"status": "PASS" if not red_flags else "STOP",
|
||||
"scope": "held-out P1 replay; gate diagnostic, not paper-facing evidence",
|
||||
"ranking": [
|
||||
{
|
||||
"rank": rank,
|
||||
"cell": candidate["cell"],
|
||||
"level": candidate["level"],
|
||||
"sim_throughput_req_s_per_gpu": candidate[
|
||||
"sim_throughput_req_s_per_gpu"
|
||||
],
|
||||
"real_feasible": candidate["real_feasible"],
|
||||
"real_goodput_req_s_per_gpu": candidate[
|
||||
"real_goodput_req_s_per_gpu"
|
||||
],
|
||||
}
|
||||
for rank, candidate in enumerate(candidates, start=1)
|
||||
],
|
||||
"real_oracle": {
|
||||
"cell": oracle_anchor["cell"],
|
||||
"level": oracle_anchor["level"],
|
||||
"goodput_req_s_per_gpu": oracle_goodput,
|
||||
},
|
||||
"by_k": by_k,
|
||||
"contribution_gate": contribution_gate,
|
||||
"analysis": {
|
||||
"script": str(Path(__file__).resolve()),
|
||||
"script_sha256": sha256_file(Path(__file__).resolve()),
|
||||
"aituner_git_head": git_capture("rev-parse", "HEAD").strip(),
|
||||
"aituner_git_status_short": git_capture("status", "--short"),
|
||||
},
|
||||
"provenance": {
|
||||
"manifest": str(manifest_path.resolve()),
|
||||
"manifest_sha256": sha256_file(manifest_path),
|
||||
"controller_state": str(state_path.resolve()),
|
||||
"controller_state_sha256": sha256_file(state_path),
|
||||
"prior_state": str(prior_state_path.resolve()),
|
||||
"prior_state_sha256": sha256_file(prior_state_path),
|
||||
"strong_metrics": str(strong_path.resolve()),
|
||||
"strong_metrics_sha256": sha256_file(strong_path),
|
||||
},
|
||||
"sanity": {
|
||||
"red_flags": red_flags,
|
||||
"anchors": numeric([1 for _ in anchors]),
|
||||
"candidates": numeric([1 for _ in candidates]),
|
||||
"probabilities": numeric(probabilities),
|
||||
"costs_h20_hours": numeric(costs),
|
||||
"invariants": {
|
||||
"anchors_12": len(anchors) == 12,
|
||||
"candidates_6": len(candidates) == 6,
|
||||
"probabilities_bounded": all(
|
||||
0.0 <= value <= 1.0 for value in probabilities
|
||||
),
|
||||
"costs_nonnegative": all(value >= 0.0 for value in costs),
|
||||
"per_config_not_all_identical": len(
|
||||
{candidate["sim_throughput_req_s_per_gpu"] for candidate in candidates}
|
||||
)
|
||||
> 1,
|
||||
"tie_expansion_applied": True,
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--manifest", type=Path, required=True)
|
||||
parser.add_argument("--controller-state", type=Path, required=True)
|
||||
parser.add_argument("--prior-state", type=Path, required=True)
|
||||
parser.add_argument("--strong-metrics", type=Path, required=True)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
result = analyze(
|
||||
args.manifest,
|
||||
args.controller_state,
|
||||
args.prior_state,
|
||||
args.strong_metrics,
|
||||
)
|
||||
args.output.write_text(json.dumps(result, indent=2, sort_keys=True) + "\n")
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"status": result["status"],
|
||||
"red_flags": result["sanity"]["red_flags"],
|
||||
"contribution_gate": result["contribution_gate"],
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
if result["status"] != "PASS":
|
||||
raise RuntimeError(result["sanity"]["red_flags"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
298
runs/fidelity-headroom/analyze_strong_baseline.py
Normal file
298
runs/fidelity-headroom/analyze_strong_baseline.py
Normal file
@@ -0,0 +1,298 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Audit telemetry against a simulator-aware outcome calibration baseline.
|
||||
|
||||
This is a retrospective headroom check. It strengthens the earlier
|
||||
outcome-only baseline by giving both nested models the same per-anchor
|
||||
Frontier throughput and SLO predictions. The only additional inputs to the
|
||||
larger model are real engine Layer-1 features.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
|
||||
from analyze_existing import (
|
||||
DEFAULT_REGULARIZATION,
|
||||
REGULARIZATION_SENSITIVITY,
|
||||
_classification_metrics,
|
||||
_fit_logistic,
|
||||
_group_bootstrap_delta,
|
||||
_mcnemar_exact_p,
|
||||
_sigmoid,
|
||||
)
|
||||
from analyze_prefixes import (
|
||||
INSTRUMENTATION_FEATURES,
|
||||
OUTCOME_FEATURES,
|
||||
PrefixExample,
|
||||
build_examples,
|
||||
numeric,
|
||||
policy_metrics,
|
||||
sha256_file,
|
||||
)
|
||||
|
||||
|
||||
SIMULATOR_FEATURES = (
|
||||
"log_sim_completed_throughput_per_gpu",
|
||||
"sim_slo_pass_rate",
|
||||
"sim_slo_feasible",
|
||||
)
|
||||
|
||||
|
||||
def load_simulator_features(raw_root: Path) -> tuple[dict[tuple[str, float], tuple[float, ...]], str]:
|
||||
features: dict[tuple[str, float], tuple[float, ...]] = {}
|
||||
digest = hashlib.sha256()
|
||||
paths = sorted(raw_root.glob("*/trial-0001/run_manifest.json"))
|
||||
for manifest_path in paths:
|
||||
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
||||
run = manifest["run"]
|
||||
if run["mode"] != "frozen-calibrated":
|
||||
continue
|
||||
scorer_path = manifest_path.parent / "scorer_output.json"
|
||||
scorer = json.loads(scorer_path.read_text(encoding="utf-8"))
|
||||
key = (str(run["cell_id"]), float(run["sampling_u"]))
|
||||
if key in features:
|
||||
raise ValueError(f"duplicate frozen simulator run: {key}")
|
||||
throughput = float(scorer["throughput_requests_per_second_per_gpu"])
|
||||
pass_rate = float(scorer["slo"]["pass_rate"])
|
||||
if throughput <= 0 or not 0.0 <= pass_rate <= 1.0:
|
||||
raise ValueError(f"invalid simulator output: {key}")
|
||||
features[key] = (
|
||||
math.log(throughput),
|
||||
pass_rate,
|
||||
float(bool(scorer["slo"]["feasible"])),
|
||||
)
|
||||
for path in (manifest_path, scorer_path):
|
||||
digest.update(str(path.relative_to(raw_root)).encode())
|
||||
digest.update(path.read_bytes())
|
||||
return features, digest.hexdigest()
|
||||
|
||||
|
||||
def simulator_row(
|
||||
example: PrefixExample,
|
||||
features: dict[tuple[str, float], tuple[float, ...]],
|
||||
) -> tuple[float, ...]:
|
||||
matches = [
|
||||
values
|
||||
for (cell, anchor), values in features.items()
|
||||
if cell == example.cell
|
||||
and math.isclose(anchor, example.anchor, rel_tol=0.0, abs_tol=1e-12)
|
||||
]
|
||||
if len(matches) != 1:
|
||||
raise ValueError(
|
||||
f"expected one simulator match for {example.cell}/{example.anchor}: {len(matches)}"
|
||||
)
|
||||
return matches[0]
|
||||
|
||||
|
||||
def grouped_predictions(
|
||||
examples: list[PrefixExample],
|
||||
simulator: dict[tuple[str, float], tuple[float, ...]],
|
||||
*,
|
||||
instrumentation_aware: bool,
|
||||
regularization: float,
|
||||
) -> tuple[np.ndarray, np.ndarray, list[str]]:
|
||||
probabilities: list[float] = []
|
||||
labels: list[int] = []
|
||||
groups: list[str] = []
|
||||
for held_out in sorted({example.cell for example in examples}):
|
||||
train = [example for example in examples if example.cell != held_out]
|
||||
test = [example for example in examples if example.cell == held_out]
|
||||
|
||||
def row(example: PrefixExample) -> np.ndarray:
|
||||
values = example.outcome + simulator_row(example, simulator)
|
||||
if instrumentation_aware:
|
||||
values += example.instrumentation
|
||||
return np.asarray((1.0, *values), dtype=np.float64)
|
||||
|
||||
x_train = np.stack([row(example) for example in train])
|
||||
x_test = np.stack([row(example) for example in test])
|
||||
y_train = np.asarray([example.feasible for example in train], dtype=np.float64)
|
||||
mean = x_train[:, 1:].mean(axis=0)
|
||||
standard_deviation = x_train[:, 1:].std(axis=0)
|
||||
standard_deviation[standard_deviation < 1e-8] = 1.0
|
||||
x_train[:, 1:] = (x_train[:, 1:] - mean) / standard_deviation
|
||||
x_test[:, 1:] = (x_test[:, 1:] - mean) / standard_deviation
|
||||
weights = _fit_logistic(x_train, y_train, regularization)
|
||||
probabilities.extend(_sigmoid(x_test @ weights).tolist())
|
||||
labels.extend(example.feasible for example in test)
|
||||
groups.extend(held_out for _ in test)
|
||||
return (
|
||||
np.asarray(labels, dtype=np.int64),
|
||||
np.asarray(probabilities, dtype=np.float64),
|
||||
groups,
|
||||
)
|
||||
|
||||
|
||||
def analyze(
|
||||
phase6_path: Path,
|
||||
phase6_raw_root: Path,
|
||||
simulator_raw_root: Path,
|
||||
simulator_metrics_path: Path,
|
||||
) -> dict[str, Any]:
|
||||
phase6 = json.loads(phase6_path.read_text(encoding="utf-8"))
|
||||
examples = build_examples(phase6, phase6_raw_root, 5.0)
|
||||
simulator, simulator_raw_sha256 = load_simulator_features(simulator_raw_root)
|
||||
red_flags = []
|
||||
try:
|
||||
matched = [simulator_row(example, simulator) for example in examples]
|
||||
except ValueError as error:
|
||||
matched = []
|
||||
red_flags.append(str(error))
|
||||
|
||||
sensitivity = {}
|
||||
if matched:
|
||||
for regularization in REGULARIZATION_SENSITIVITY:
|
||||
labels, baseline_probability, groups = grouped_predictions(
|
||||
examples,
|
||||
simulator,
|
||||
instrumentation_aware=False,
|
||||
regularization=regularization,
|
||||
)
|
||||
instrument_labels, instrument_probability, instrument_groups = grouped_predictions(
|
||||
examples,
|
||||
simulator,
|
||||
instrumentation_aware=True,
|
||||
regularization=regularization,
|
||||
)
|
||||
if not np.array_equal(labels, instrument_labels) or groups != instrument_groups:
|
||||
raise AssertionError("nested baseline folds differ")
|
||||
baseline_correct = (baseline_probability >= 0.5) == labels
|
||||
instrument_correct = (instrument_probability >= 0.5) == labels
|
||||
paired = {
|
||||
"both_correct": int(np.sum(baseline_correct & instrument_correct)),
|
||||
"sim_outcome_only_correct": int(
|
||||
np.sum(baseline_correct & ~instrument_correct)
|
||||
),
|
||||
"instrumentation_only_correct": int(
|
||||
np.sum(~baseline_correct & instrument_correct)
|
||||
),
|
||||
"both_wrong": int(np.sum(~baseline_correct & ~instrument_correct)),
|
||||
}
|
||||
paired["mcnemar_exact_two_sided_p"] = _mcnemar_exact_p(
|
||||
paired["sim_outcome_only_correct"],
|
||||
paired["instrumentation_only_correct"],
|
||||
)
|
||||
sensitivity[str(regularization)] = {
|
||||
"sim_plus_outcome": {
|
||||
"classification": _classification_metrics(labels, baseline_probability),
|
||||
"policy_0p95": policy_metrics(
|
||||
examples, labels, baseline_probability, 0.95
|
||||
),
|
||||
},
|
||||
"sim_plus_outcome_plus_instrumentation": {
|
||||
"classification": _classification_metrics(labels, instrument_probability),
|
||||
"policy_0p95": policy_metrics(
|
||||
examples, labels, instrument_probability, 0.95
|
||||
),
|
||||
},
|
||||
"paired_correctness": paired,
|
||||
"group_bootstrap": _group_bootstrap_delta(
|
||||
labels,
|
||||
baseline_probability,
|
||||
instrument_probability,
|
||||
groups,
|
||||
),
|
||||
}
|
||||
|
||||
headline = sensitivity.get(str(DEFAULT_REGULARIZATION))
|
||||
simulator_pass_rates = [row[1] for row in matched]
|
||||
labels = [example.feasible for example in examples]
|
||||
if len(examples) != 37:
|
||||
red_flags.append("examples_not_37")
|
||||
if len(simulator) != 92:
|
||||
red_flags.append("frozen_simulator_runs_not_92")
|
||||
if len(set(labels)) != 2:
|
||||
red_flags.append("single_label")
|
||||
if matched and not all(0.0 <= value <= 1.0 for value in simulator_pass_rates):
|
||||
red_flags.append("simulator_pass_rate_out_of_range")
|
||||
|
||||
return {
|
||||
"schema": "fidelity-strong-baseline-v1",
|
||||
"status": "PASS" if not red_flags else "STOP",
|
||||
"scope": "retrospective one-task headroom audit; not contribution evidence",
|
||||
"comparison": (
|
||||
"same 5-second prefix, folds, logistic family, regularization, and frozen "
|
||||
"Frontier outputs; the only nested difference is real Layer-1 engine state"
|
||||
),
|
||||
"features": {
|
||||
"shared_outcome": list(OUTCOME_FEATURES),
|
||||
"shared_simulator": list(SIMULATOR_FEATURES),
|
||||
"instrumentation_only": list(INSTRUMENTATION_FEATURES),
|
||||
},
|
||||
"headline_regularization": DEFAULT_REGULARIZATION,
|
||||
"headline": headline,
|
||||
"regularization_sensitivity": sensitivity,
|
||||
"provenance": {
|
||||
"phase6_metrics": str(phase6_path.resolve()),
|
||||
"phase6_metrics_sha256": sha256_file(phase6_path),
|
||||
"phase6_raw_root": str(phase6_raw_root.resolve()),
|
||||
"simulator_metrics": str(simulator_metrics_path.resolve()),
|
||||
"simulator_metrics_sha256": sha256_file(simulator_metrics_path),
|
||||
"simulator_raw_root": str(simulator_raw_root.resolve()),
|
||||
"frozen_simulator_manifest_scorer_set_sha256": simulator_raw_sha256,
|
||||
},
|
||||
"decision": {
|
||||
"contribution_established": False,
|
||||
"prospective_requirement": (
|
||||
"repeat sim+outcome versus sim+outcome+instrumentation on complete held-out tasks"
|
||||
),
|
||||
},
|
||||
"sanity": {
|
||||
"red_flags": red_flags,
|
||||
"examples": numeric([1 for _ in examples]),
|
||||
"labels": {
|
||||
**numeric(labels),
|
||||
"positive": sum(labels),
|
||||
"negative": len(labels) - sum(labels),
|
||||
},
|
||||
"matched_simulator_pass_rate": numeric(simulator_pass_rates),
|
||||
"frozen_simulator_runs": len(simulator),
|
||||
"invariants": {
|
||||
"all_examples_matched_once": len(matched) == len(examples),
|
||||
"same_nested_folds": True,
|
||||
"simulator_ratios_bounded": all(
|
||||
0.0 <= value <= 1.0 for value in simulator_pass_rates
|
||||
),
|
||||
"labels_not_identical": len(set(labels)) == 2,
|
||||
"per_config_results_not_all_identical": len(set(simulator_pass_rates)) > 1,
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--phase6-metrics", type=Path, required=True)
|
||||
parser.add_argument("--phase6-raw-root", type=Path, required=True)
|
||||
parser.add_argument("--simulator-raw-root", type=Path, required=True)
|
||||
parser.add_argument("--simulator-metrics", type=Path, required=True)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
result = analyze(
|
||||
args.phase6_metrics,
|
||||
args.phase6_raw_root,
|
||||
args.simulator_raw_root,
|
||||
args.simulator_metrics,
|
||||
)
|
||||
args.output.write_text(json.dumps(result, indent=2, sort_keys=True) + "\n")
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"status": result["status"],
|
||||
"output": str(args.output),
|
||||
"red_flags": result["sanity"]["red_flags"],
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
507
runs/fidelity-headroom/analyze_strong_pilot.py
Normal file
507
runs/fidelity-headroom/analyze_strong_pilot.py
Normal file
@@ -0,0 +1,507 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Exploratory P1 audit against the strengthened simulator-aware baseline.
|
||||
|
||||
P1 was already running when the strong baseline was added, so this script is
|
||||
not paper-facing prospective evidence. It trains only on the historical
|
||||
Phase-6 task and evaluates the exact P1 primary probes. Both nested models
|
||||
receive identical Frontier predictions; engine telemetry is the sole feature
|
||||
difference.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import math
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
|
||||
from analyze_existing import (
|
||||
DEFAULT_REGULARIZATION,
|
||||
REGULARIZATION_SENSITIVITY,
|
||||
_classification_metrics,
|
||||
_fit_logistic,
|
||||
_mcnemar_exact_p,
|
||||
_sigmoid,
|
||||
)
|
||||
from analyze_pilot import build_pilot_examples, campaign_gpu_accounting
|
||||
from analyze_prefixes import (
|
||||
INSTRUMENTATION_FEATURES,
|
||||
OUTCOME_FEATURES,
|
||||
PrefixExample,
|
||||
build_examples,
|
||||
numeric,
|
||||
policy_metrics,
|
||||
sha256_file,
|
||||
)
|
||||
from analyze_strong_baseline import (
|
||||
SIMULATOR_FEATURES,
|
||||
load_simulator_features,
|
||||
simulator_row,
|
||||
)
|
||||
|
||||
|
||||
AITUNER_ROOT = Path(__file__).resolve().parents[2]
|
||||
|
||||
|
||||
def git_capture(*arguments: str) -> str:
|
||||
return subprocess.run(
|
||||
["git", "-C", str(AITUNER_ROOT), *arguments],
|
||||
check=True,
|
||||
text=True,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
).stdout
|
||||
|
||||
|
||||
def load_pilot_simulator(
|
||||
path: Path,
|
||||
) -> tuple[dict[tuple[str, str], tuple[float, ...]], list[str]]:
|
||||
payload = json.loads(path.read_text(encoding="utf-8"))
|
||||
red_flags = []
|
||||
if payload.get("status") != "PASS":
|
||||
red_flags.append("pilot_simulator_not_pass")
|
||||
features: dict[tuple[str, str], tuple[float, ...]] = {}
|
||||
for item in payload.get("results", []):
|
||||
key = (str(item["cell"]), str(item["role"]))
|
||||
if key in features:
|
||||
red_flags.append(f"duplicate_pilot_simulator_{key[0]}_{key[1]}")
|
||||
continue
|
||||
scorer = item["scorer"]
|
||||
throughput = float(scorer["throughput_requests_per_second_per_gpu"])
|
||||
pass_rate = float(scorer["slo"]["pass_rate"])
|
||||
if throughput <= 0:
|
||||
red_flags.append(f"nonpositive_pilot_simulator_throughput_{key[0]}_{key[1]}")
|
||||
if not 0.0 <= pass_rate <= 1.0:
|
||||
red_flags.append(f"pilot_simulator_ratio_out_of_range_{key[0]}_{key[1]}")
|
||||
features[key] = (
|
||||
math.log(throughput),
|
||||
pass_rate,
|
||||
float(bool(scorer["slo"]["feasible"])),
|
||||
)
|
||||
if len(features) != 12:
|
||||
red_flags.append("pilot_simulator_entries_not_12")
|
||||
return features, red_flags
|
||||
|
||||
|
||||
def fit_model(
|
||||
examples: list[PrefixExample],
|
||||
simulator: list[tuple[float, ...]],
|
||||
*,
|
||||
instrumentation_aware: bool,
|
||||
regularization: float,
|
||||
) -> dict[str, Any]:
|
||||
rows = []
|
||||
for example, simulator_features in zip(examples, simulator):
|
||||
values = example.outcome + simulator_features
|
||||
if instrumentation_aware:
|
||||
values += example.instrumentation
|
||||
rows.append((1.0, *values))
|
||||
matrix = np.asarray(rows, dtype=np.float64)
|
||||
labels = np.asarray([example.feasible for example in examples], dtype=np.float64)
|
||||
mean = matrix[:, 1:].mean(axis=0)
|
||||
standard_deviation = matrix[:, 1:].std(axis=0)
|
||||
standard_deviation[standard_deviation < 1e-8] = 1.0
|
||||
standardized = matrix.copy()
|
||||
standardized[:, 1:] = (standardized[:, 1:] - mean) / standard_deviation
|
||||
weights = _fit_logistic(standardized, labels, regularization)
|
||||
return {
|
||||
"instrumentation_aware": instrumentation_aware,
|
||||
"regularization": regularization,
|
||||
"feature_mean": mean,
|
||||
"feature_standard_deviation": standard_deviation,
|
||||
"weights": weights,
|
||||
}
|
||||
|
||||
|
||||
def predict_model(
|
||||
model: dict[str, Any],
|
||||
examples: list[PrefixExample],
|
||||
simulator: list[tuple[float, ...]],
|
||||
) -> np.ndarray:
|
||||
rows = []
|
||||
for example, simulator_features in zip(examples, simulator):
|
||||
values = example.outcome + simulator_features
|
||||
if model["instrumentation_aware"]:
|
||||
values += example.instrumentation
|
||||
rows.append((1.0, *values))
|
||||
matrix = np.asarray(rows, dtype=np.float64)
|
||||
matrix[:, 1:] = (
|
||||
matrix[:, 1:] - model["feature_mean"]
|
||||
) / model["feature_standard_deviation"]
|
||||
return _sigmoid(matrix @ model["weights"])
|
||||
|
||||
|
||||
def covariate_shift(
|
||||
training_examples: list[PrefixExample],
|
||||
training_simulator: list[tuple[float, ...]],
|
||||
pilot_examples: list[PrefixExample],
|
||||
pilot_simulator: list[tuple[float, ...]],
|
||||
*,
|
||||
instrumentation_aware: bool,
|
||||
) -> dict[str, Any]:
|
||||
def matrix(
|
||||
examples: list[PrefixExample], simulator: list[tuple[float, ...]]
|
||||
) -> np.ndarray:
|
||||
rows = []
|
||||
for example, simulator_features in zip(examples, simulator):
|
||||
values = example.outcome + simulator_features
|
||||
if instrumentation_aware:
|
||||
values += example.instrumentation
|
||||
rows.append(values)
|
||||
return np.asarray(rows, dtype=np.float64)
|
||||
|
||||
training = matrix(training_examples, training_simulator)
|
||||
pilot = matrix(pilot_examples, pilot_simulator)
|
||||
mean = training.mean(axis=0)
|
||||
standard_deviation = training.std(axis=0)
|
||||
standard_deviation[standard_deviation < 1e-8] = 1.0
|
||||
absolute_z = np.abs((pilot - mean) / standard_deviation)
|
||||
names = [*OUTCOME_FEATURES, *SIMULATOR_FEATURES]
|
||||
if instrumentation_aware:
|
||||
names.extend(INSTRUMENTATION_FEATURES)
|
||||
return {
|
||||
"values": numeric(absolute_z.ravel().tolist()),
|
||||
"count_gt_3": int(np.sum(absolute_z > 3.0)),
|
||||
"count_gt_5": int(np.sum(absolute_z > 5.0)),
|
||||
"total_feature_values": int(absolute_z.size),
|
||||
"per_feature_max_abs_z": {
|
||||
name: float(value) for name, value in zip(names, absolute_z.max(axis=0))
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def comparison(
|
||||
training_examples: list[PrefixExample],
|
||||
training_simulator: list[tuple[float, ...]],
|
||||
pilot_examples: list[PrefixExample],
|
||||
pilot_simulator: list[tuple[float, ...]],
|
||||
regularization: float,
|
||||
) -> dict[str, Any]:
|
||||
labels = np.asarray([example.feasible for example in pilot_examples], dtype=np.int64)
|
||||
baseline_model = fit_model(
|
||||
training_examples,
|
||||
training_simulator,
|
||||
instrumentation_aware=False,
|
||||
regularization=regularization,
|
||||
)
|
||||
instrument_model = fit_model(
|
||||
training_examples,
|
||||
training_simulator,
|
||||
instrumentation_aware=True,
|
||||
regularization=regularization,
|
||||
)
|
||||
baseline_probability = predict_model(
|
||||
baseline_model, pilot_examples, pilot_simulator
|
||||
)
|
||||
instrument_probability = predict_model(
|
||||
instrument_model, pilot_examples, pilot_simulator
|
||||
)
|
||||
baseline_correct = (baseline_probability >= 0.5) == labels
|
||||
instrument_correct = (instrument_probability >= 0.5) == labels
|
||||
paired = {
|
||||
"both_correct": int(np.sum(baseline_correct & instrument_correct)),
|
||||
"sim_outcome_only_correct": int(
|
||||
np.sum(baseline_correct & ~instrument_correct)
|
||||
),
|
||||
"instrumentation_only_correct": int(
|
||||
np.sum(~baseline_correct & instrument_correct)
|
||||
),
|
||||
"both_wrong": int(np.sum(~baseline_correct & ~instrument_correct)),
|
||||
}
|
||||
paired["mcnemar_exact_two_sided_p"] = _mcnemar_exact_p(
|
||||
paired["sim_outcome_only_correct"], paired["instrumentation_only_correct"]
|
||||
)
|
||||
return {
|
||||
"sim_plus_outcome": {
|
||||
"classification": _classification_metrics(labels, baseline_probability),
|
||||
"policy_0p95": policy_metrics(
|
||||
pilot_examples, labels, baseline_probability, 0.95
|
||||
),
|
||||
"probability": baseline_probability.tolist(),
|
||||
},
|
||||
"sim_plus_outcome_plus_instrumentation": {
|
||||
"classification": _classification_metrics(labels, instrument_probability),
|
||||
"policy_0p95": policy_metrics(
|
||||
pilot_examples, labels, instrument_probability, 0.95
|
||||
),
|
||||
"probability": instrument_probability.tolist(),
|
||||
},
|
||||
"paired_correctness": paired,
|
||||
}
|
||||
|
||||
|
||||
def analyze(
|
||||
phase6_path: Path,
|
||||
phase6_raw_root: Path,
|
||||
training_simulator_root: Path,
|
||||
pilot_manifest_path: Path,
|
||||
pilot_run_root: Path,
|
||||
pilot_simulator_path: Path,
|
||||
prior_state_paths: tuple[Path, ...] = (),
|
||||
) -> dict[str, Any]:
|
||||
phase6 = json.loads(phase6_path.read_text(encoding="utf-8"))
|
||||
pilot_manifest = json.loads(pilot_manifest_path.read_text(encoding="utf-8"))
|
||||
pilot_state_path = pilot_run_root / "controller-state.json"
|
||||
pilot_state = json.loads(pilot_state_path.read_text(encoding="utf-8"))
|
||||
gpu_accounting = campaign_gpu_accounting(
|
||||
pilot_state_path, prior_state_paths
|
||||
)
|
||||
training_examples = build_examples(phase6, phase6_raw_root, 5.0)
|
||||
training_simulator_map, training_simulator_sha256 = load_simulator_features(
|
||||
training_simulator_root
|
||||
)
|
||||
training_simulator = [
|
||||
simulator_row(example, training_simulator_map)
|
||||
for example in training_examples
|
||||
]
|
||||
pilot_examples, pilot_details, red_flags = build_pilot_examples(
|
||||
pilot_manifest, pilot_run_root, 5.0
|
||||
)
|
||||
pilot_simulator_map, simulator_red_flags = load_pilot_simulator(
|
||||
pilot_simulator_path
|
||||
)
|
||||
red_flags.extend(simulator_red_flags)
|
||||
pilot_simulator = []
|
||||
for example, detail in zip(pilot_examples, pilot_details):
|
||||
role = f"{detail['level']}1"
|
||||
key = (example.cell, role)
|
||||
if key not in pilot_simulator_map:
|
||||
red_flags.append(f"missing_pilot_simulator_{example.cell}_{role}")
|
||||
pilot_simulator.append((0.0, 0.0, 0.0))
|
||||
else:
|
||||
pilot_simulator.append(pilot_simulator_map[key])
|
||||
|
||||
sensitivity = {}
|
||||
if not red_flags:
|
||||
for regularization in REGULARIZATION_SENSITIVITY:
|
||||
sensitivity[str(regularization)] = comparison(
|
||||
training_examples,
|
||||
training_simulator,
|
||||
pilot_examples,
|
||||
pilot_simulator,
|
||||
regularization,
|
||||
)
|
||||
headline = sensitivity.get(str(DEFAULT_REGULARIZATION))
|
||||
labels = [example.feasible for example in pilot_examples]
|
||||
simulator_pass_rates = [row[1] for row in pilot_simulator]
|
||||
simulator_labels = [int(row[2]) for row in pilot_simulator]
|
||||
if len(training_examples) != 37:
|
||||
red_flags.append("training_examples_not_37")
|
||||
if len(pilot_examples) != 12:
|
||||
red_flags.append("pilot_examples_not_12")
|
||||
if len(set(labels)) != 2:
|
||||
red_flags.append("pilot_single_label")
|
||||
if len(set(simulator_pass_rates)) <= 1:
|
||||
red_flags.append("pilot_simulator_results_identical")
|
||||
if pilot_state.get("status") != "complete" or int(
|
||||
pilot_state.get("completed_cells", 0)
|
||||
) != 6:
|
||||
red_flags.append("pilot_campaign_incomplete")
|
||||
if any(detail["actual_timestamped_outcomes"] == 0 for detail in pilot_details):
|
||||
red_flags.append("pilot_no_exact_request_timestamps")
|
||||
all_cell_validations = all(
|
||||
cell.get("validation") is not None
|
||||
and all(cell["validation"]["invariants"].values())
|
||||
for cell in pilot_state.get("cells", {}).values()
|
||||
)
|
||||
if not all_cell_validations:
|
||||
red_flags.append("pilot_cell_validation_failed")
|
||||
if not all(gpu_accounting["invariants"].values()):
|
||||
red_flags.append("pilot_hard_cap_exceeded")
|
||||
covariate_diagnostics = {
|
||||
"sim_plus_outcome": covariate_shift(
|
||||
training_examples,
|
||||
training_simulator,
|
||||
pilot_examples,
|
||||
pilot_simulator,
|
||||
instrumentation_aware=False,
|
||||
),
|
||||
"sim_plus_outcome_plus_instrumentation": covariate_shift(
|
||||
training_examples,
|
||||
training_simulator,
|
||||
pilot_examples,
|
||||
pilot_simulator,
|
||||
instrumentation_aware=True,
|
||||
),
|
||||
}
|
||||
|
||||
if headline is None:
|
||||
decision = {
|
||||
"strong_incremental_gate": False,
|
||||
"reason": "analysis red flag prevented nested comparison",
|
||||
}
|
||||
else:
|
||||
baseline_policy = headline["sim_plus_outcome"]["policy_0p95"]
|
||||
instrument_policy = headline[
|
||||
"sim_plus_outcome_plus_instrumentation"
|
||||
]["policy_0p95"]
|
||||
baseline_errors = baseline_policy["false_accept"] + baseline_policy["false_reject"]
|
||||
instrument_errors = (
|
||||
instrument_policy["false_accept"] + instrument_policy["false_reject"]
|
||||
)
|
||||
baseline_reduction = baseline_policy["valid_cost_reduction_fraction"]
|
||||
instrument_reduction = instrument_policy["valid_cost_reduction_fraction"]
|
||||
reduction_delta = (
|
||||
instrument_reduction - baseline_reduction
|
||||
if baseline_reduction is not None and instrument_reduction is not None
|
||||
else None
|
||||
)
|
||||
per_lambda_safe_and_better = []
|
||||
for item in sensitivity.values():
|
||||
baseline = item["sim_plus_outcome"]["policy_0p95"]
|
||||
instrument = item["sim_plus_outcome_plus_instrumentation"]["policy_0p95"]
|
||||
base_errors = baseline["false_accept"] + baseline["false_reject"]
|
||||
inst_errors = instrument["false_accept"] + instrument["false_reject"]
|
||||
base_reduction = baseline["valid_cost_reduction_fraction"]
|
||||
inst_reduction = instrument["valid_cost_reduction_fraction"]
|
||||
per_lambda_safe_and_better.append(
|
||||
inst_errors == 0
|
||||
and inst_errors <= base_errors
|
||||
and base_reduction is not None
|
||||
and inst_reduction is not None
|
||||
and inst_reduction > base_reduction
|
||||
)
|
||||
decision = {
|
||||
"strong_incremental_gate": bool(
|
||||
not red_flags
|
||||
and instrument_errors == 0
|
||||
and instrument_errors <= baseline_errors
|
||||
and reduction_delta is not None
|
||||
and reduction_delta >= 0.15
|
||||
),
|
||||
"regularization_robust": all(per_lambda_safe_and_better),
|
||||
"valid_cost_reduction_fraction_delta": reduction_delta,
|
||||
"scope": "exploratory task; may choose P2 design but cannot establish contribution",
|
||||
}
|
||||
|
||||
return {
|
||||
"schema": "fidelity-strong-pilot-v1",
|
||||
"status": "PASS" if not red_flags else "STOP",
|
||||
"scope": (
|
||||
"post-amendment exploratory P1 audit; strong model was not frozen before "
|
||||
"partial P1 outcomes, so this is not prospective contribution evidence"
|
||||
),
|
||||
"features": {
|
||||
"shared_outcome": list(OUTCOME_FEATURES),
|
||||
"shared_simulator": list(SIMULATOR_FEATURES),
|
||||
"instrumentation_only": list(INSTRUMENTATION_FEATURES),
|
||||
},
|
||||
"headline_regularization": DEFAULT_REGULARIZATION,
|
||||
"headline": headline,
|
||||
"regularization_sensitivity": sensitivity,
|
||||
"simulator_only": {
|
||||
"classification": _classification_metrics(
|
||||
np.asarray(labels, dtype=np.int64),
|
||||
np.asarray(simulator_labels, dtype=np.float64),
|
||||
)
|
||||
if labels
|
||||
else None,
|
||||
"predicted_feasible": simulator_labels,
|
||||
},
|
||||
"pilot_examples": [
|
||||
{
|
||||
**detail,
|
||||
"sim_completed_throughput_per_gpu": math.exp(simulator[0]),
|
||||
"sim_slo_pass_rate": simulator[1],
|
||||
"sim_slo_feasible": bool(simulator[2]),
|
||||
}
|
||||
for detail, simulator in zip(pilot_details, pilot_simulator)
|
||||
],
|
||||
"covariate_shift_diagnostic": covariate_diagnostics,
|
||||
"decision": decision,
|
||||
"gpu": {
|
||||
"primary_attempt_h20_hours": pilot_state["gpu_hours_total"],
|
||||
**gpu_accounting,
|
||||
},
|
||||
"analysis": {
|
||||
"script": str(Path(__file__).resolve()),
|
||||
"script_sha256": sha256_file(Path(__file__).resolve()),
|
||||
"aituner_git_head": git_capture("rev-parse", "HEAD").strip(),
|
||||
"aituner_git_status_short": git_capture("status", "--short"),
|
||||
},
|
||||
"provenance": {
|
||||
"phase6_metrics": str(phase6_path.resolve()),
|
||||
"phase6_metrics_sha256": sha256_file(phase6_path),
|
||||
"phase6_raw_root": str(phase6_raw_root.resolve()),
|
||||
"training_simulator_root": str(training_simulator_root.resolve()),
|
||||
"training_simulator_manifest_scorer_set_sha256": training_simulator_sha256,
|
||||
"pilot_manifest": str(pilot_manifest_path.resolve()),
|
||||
"pilot_manifest_sha256": sha256_file(pilot_manifest_path),
|
||||
"pilot_run_root": str(pilot_run_root.resolve()),
|
||||
"pilot_controller_state": str(pilot_state_path.resolve()),
|
||||
"pilot_controller_state_sha256": sha256_file(pilot_state_path),
|
||||
"pilot_simulator": str(pilot_simulator_path.resolve()),
|
||||
"pilot_simulator_sha256": sha256_file(pilot_simulator_path),
|
||||
},
|
||||
"sanity": {
|
||||
"red_flags": red_flags,
|
||||
"training_examples": numeric([1 for _ in training_examples]),
|
||||
"pilot_labels": {
|
||||
**numeric(labels),
|
||||
"positive": sum(labels),
|
||||
"negative": len(labels) - sum(labels),
|
||||
},
|
||||
"pilot_simulator_pass_rate": numeric(simulator_pass_rates),
|
||||
"invariants": {
|
||||
"training_examples_37": len(training_examples) == 37,
|
||||
"pilot_examples_12": len(pilot_examples) == 12,
|
||||
"pilot_cells_6": len({example.cell for example in pilot_examples}) == 6,
|
||||
"pilot_both_labels": len(set(labels)) == 2,
|
||||
"simulator_ratios_bounded": all(
|
||||
0.0 <= value <= 1.0 for value in simulator_pass_rates
|
||||
),
|
||||
"per_config_not_all_identical": len(set(simulator_pass_rates)) > 1,
|
||||
"all_prefixes_exact_monotonic": all(
|
||||
example.completion_time_source in {"exact_monotonic", "none_completed"}
|
||||
for example in pilot_examples
|
||||
),
|
||||
"all_cell_validations": all_cell_validations,
|
||||
"gpu_cost_nonnegative_below_cap": (
|
||||
all(gpu_accounting["invariants"].values())
|
||||
),
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--phase6-metrics", type=Path, required=True)
|
||||
parser.add_argument("--phase6-raw-root", type=Path, required=True)
|
||||
parser.add_argument("--training-simulator-root", type=Path, required=True)
|
||||
parser.add_argument("--pilot-manifest", type=Path, required=True)
|
||||
parser.add_argument("--pilot-run-root", type=Path, required=True)
|
||||
parser.add_argument("--pilot-simulator", type=Path, required=True)
|
||||
parser.add_argument("--prior-state", type=Path, action="append", default=[])
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
args = parser.parse_args()
|
||||
result = analyze(
|
||||
args.phase6_metrics,
|
||||
args.phase6_raw_root,
|
||||
args.training_simulator_root,
|
||||
args.pilot_manifest,
|
||||
args.pilot_run_root,
|
||||
args.pilot_simulator,
|
||||
tuple(args.prior_state),
|
||||
)
|
||||
args.output.write_text(json.dumps(result, indent=2, sort_keys=True) + "\n")
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"status": result["status"],
|
||||
"red_flags": result["sanity"]["red_flags"],
|
||||
"decision": result["decision"],
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
if result["status"] != "PASS":
|
||||
raise RuntimeError(result["sanity"]["red_flags"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
334
runs/fidelity-headroom/prepare_pilot_simulator.py
Normal file
334
runs/fidelity-headroom/prepare_pilot_simulator.py
Normal file
@@ -0,0 +1,334 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Prepare exact Frontier fixtures for the P1 primary low/high probes.
|
||||
|
||||
Prompt-bearing band traces remain under ``--private-root``. The emitted
|
||||
fixtures and public manifest contain token IDs, block IDs, hashes, and
|
||||
aggregate metadata, but no prompt text.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import importlib.util
|
||||
import json
|
||||
import math
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from transformers import AutoTokenizer
|
||||
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
AITUNER_ROOT = HERE.parents[1]
|
||||
sys.path.insert(0, str(HERE))
|
||||
|
||||
import prepare_pilot as pilot # noqa: E402
|
||||
|
||||
|
||||
PRIMARY_ROLES = ("low1", "high1")
|
||||
|
||||
|
||||
def load_module(path: Path):
|
||||
module_root = str(path.parent.resolve())
|
||||
if module_root not in sys.path:
|
||||
sys.path.insert(0, module_root)
|
||||
spec = importlib.util.spec_from_file_location("simfid_s2rb_prepare", path)
|
||||
if spec is None or spec.loader is None:
|
||||
raise ImportError(path)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
sys.modules[spec.name] = module
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
def order_hash(values: list[str]) -> str:
|
||||
return hashlib.sha256("\n".join(values).encode()).hexdigest()
|
||||
|
||||
|
||||
def sha256_file(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as source:
|
||||
for chunk in iter(lambda: source.read(1 << 20), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def git_capture(root: Path, *arguments: str) -> str:
|
||||
return subprocess.run(
|
||||
["git", "-C", str(root), *arguments],
|
||||
check=True,
|
||||
text=True,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
).stdout
|
||||
|
||||
|
||||
def raw_rows(path: Path) -> dict[int, dict[str, Any]]:
|
||||
result = {}
|
||||
with path.open(encoding="utf-8") as source:
|
||||
for index, line in enumerate(source):
|
||||
if line.strip():
|
||||
result[index] = json.loads(line)
|
||||
return result
|
||||
|
||||
|
||||
def selected_hashes(
|
||||
selected: list[Any], rows: dict[int, dict[str, Any]]
|
||||
) -> dict[str, str]:
|
||||
identifiers = []
|
||||
arrivals = []
|
||||
lengths = []
|
||||
for item in selected:
|
||||
row = rows[item.row_index]
|
||||
identifiers.append(str(row.get("request_id") or row.get("id") or item.row_index))
|
||||
arrivals.append(f"{float(item.timestamp) * 0.1:.12f}")
|
||||
lengths.append(str(int(item.input_length)))
|
||||
return {
|
||||
"request_id_order_sha256": order_hash(identifiers),
|
||||
"arrival_order_sha256": order_hash(arrivals),
|
||||
"input_length_order_sha256": order_hash(lengths),
|
||||
}
|
||||
|
||||
|
||||
def kv_blocks(raw_root: Path, cell: str) -> int:
|
||||
stream = next((raw_root / cell / "opprof").glob("*.jsonl"))
|
||||
with stream.open(encoding="utf-8") as source:
|
||||
for line in source:
|
||||
record = json.loads(line)
|
||||
if "step_index" in record:
|
||||
return int(record["kv"]["total_blocks"])
|
||||
raise ValueError(f"no Layer-1 record for {cell}")
|
||||
|
||||
|
||||
def source_window(windows_path: Path, window_id: str) -> tuple[dict[str, Any], Path]:
|
||||
return pilot.resolve_source_trace(windows_path, window_id)
|
||||
|
||||
|
||||
def prepare(args: argparse.Namespace) -> dict[str, Any]:
|
||||
simulator = load_module(args.replayserve_root / "tools/simfid_s2rb_prepare.py")
|
||||
manifest = json.loads(args.pilot_manifest.read_text(encoding="utf-8"))
|
||||
window, trace = source_window(args.source_windows, args.source_window_id)
|
||||
if args.band_root is not None:
|
||||
role_paths = {
|
||||
role: (args.band_root / f"{role}.jsonl").resolve()
|
||||
for role in PRIMARY_ROLES
|
||||
}
|
||||
band_stats = {
|
||||
role: manifest["private"]["band_stats"][role]
|
||||
for role in PRIMARY_ROLES
|
||||
}
|
||||
for role, path in role_paths.items():
|
||||
if sha256_file(path) != band_stats[role]["sha256"]:
|
||||
raise ValueError(f"pre-materialized band hash mismatch: {role}")
|
||||
private_windows = None
|
||||
else:
|
||||
private_windows, all_band_stats = pilot.materialize_bands(
|
||||
trace, window, args.private_root
|
||||
)
|
||||
private_payload = json.loads(private_windows.read_text(encoding="utf-8"))
|
||||
role_paths = {
|
||||
item["fidelity_pilot_role"]: (
|
||||
private_windows.parent / item["trace_file"]
|
||||
).resolve()
|
||||
for item in private_payload["windows"]
|
||||
}
|
||||
band_stats = {
|
||||
role: all_band_stats[role]
|
||||
for role in PRIMARY_ROLES
|
||||
}
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained(
|
||||
args.tokenizer, local_files_only=True, use_fast=True
|
||||
)
|
||||
fixture_root = args.output / "fixtures"
|
||||
config_root = args.output / "configs"
|
||||
fixture_root.mkdir(parents=True, exist_ok=True)
|
||||
config_root.mkdir(parents=True, exist_ok=True)
|
||||
entries = []
|
||||
red_flags = []
|
||||
for role in PRIMARY_ROLES:
|
||||
trace_path = role_paths[role]
|
||||
retained, trace_stats = simulator.scan_trace(trace_path)
|
||||
rows = raw_rows(trace_path)
|
||||
primary_pool = [retained[(index * len(retained)) // 512] for index in range(512)]
|
||||
selections: dict[str, list[Any]] = {}
|
||||
selected_union: set[int] = set()
|
||||
for cell, cell_manifest in sorted(manifest["cells"].items()):
|
||||
level = "low" if role.startswith("low") else "high"
|
||||
expected = cell_manifest["targets"][level]["selections"][role]
|
||||
pool = retained if int(cell_manifest["tp"]) == 4 else primary_pool
|
||||
selected = [item for item in pool if item.sampling_u <= float(expected["anchor"])]
|
||||
selections[cell] = selected
|
||||
selected_union.update(item.row_index for item in selected)
|
||||
hashes = selected_hashes(selected, rows)
|
||||
if len(selected) != int(expected["selected_count"]):
|
||||
red_flags.append(f"selection_count_{cell}_{role}")
|
||||
for key, value in hashes.items():
|
||||
if value != expected[key]:
|
||||
red_flags.append(f"selection_hash_{cell}_{role}_{key}")
|
||||
|
||||
token_gates, selected_records, block_stats = simulator.tokenize_and_hash(
|
||||
trace=trace_path,
|
||||
tokenizer=tokenizer,
|
||||
retained=retained,
|
||||
selected_union=selected_union,
|
||||
)
|
||||
if any(gate["status"] != "pass" for gate in token_gates.values()):
|
||||
red_flags.append(f"token_gate_{role}")
|
||||
for cell, selected in selections.items():
|
||||
cell_manifest = manifest["cells"][cell]
|
||||
level = "low" if role.startswith("low") else "high"
|
||||
expected = cell_manifest["targets"][level]["selections"][role]
|
||||
fixture_id = f"fidelity_p1_{cell}_{role}"
|
||||
cell_record = {
|
||||
"cell_id": cell,
|
||||
"tensor_parallel_size": int(cell_manifest["tp"]),
|
||||
"max_num_seqs": int(cell_manifest["mns"]),
|
||||
"store_role": "companion" if int(cell_manifest["tp"]) == 4 else "primary",
|
||||
"kv_capacity": {
|
||||
"block_size_tokens": 16,
|
||||
"num_blocks": kv_blocks(args.phase6_raw_root, cell),
|
||||
},
|
||||
}
|
||||
probe = {
|
||||
"probe_index": 0 if role == "low1" else 1,
|
||||
"sampling_u": float(expected["anchor"]),
|
||||
}
|
||||
fixture = simulator.create_fixture(
|
||||
fixture_root=fixture_root,
|
||||
fixture_id=fixture_id,
|
||||
cell=cell_record,
|
||||
probe=probe,
|
||||
row_indexes=[item.row_index for item in selected],
|
||||
meta_by_index={item.row_index: item for item in retained},
|
||||
selected_records=selected_records,
|
||||
)
|
||||
config_path = config_root / f"{fixture_id}.json"
|
||||
config = simulator.build_config(
|
||||
path=config_path,
|
||||
cell=cell_record,
|
||||
mode="frozen-calibrated",
|
||||
fixture_ids=[fixture_id],
|
||||
frontier_root=args.frontier_root,
|
||||
cache_dir=args.cache_dir,
|
||||
)
|
||||
entries.append(
|
||||
{
|
||||
"cell": cell,
|
||||
"role": role,
|
||||
"level": level,
|
||||
"anchor": expected["anchor"],
|
||||
"selected_count": len(selected),
|
||||
"fixture_id": fixture_id,
|
||||
"fixture_manifest": str(
|
||||
(fixture_root / fixture_id / "fixture_manifest.json").resolve()
|
||||
),
|
||||
"frontier_csv": fixture["frontier_csv"]["path"],
|
||||
"sidecar": fixture["sidecar_jsonl"]["path"],
|
||||
"config": str(config_path.resolve()),
|
||||
"calibration_scale": config["calibration"]["a_tp"],
|
||||
}
|
||||
)
|
||||
if block_stats["selected_union_records"] != len(selected_union):
|
||||
red_flags.append(f"selected_union_{role}")
|
||||
if trace_stats["retained_inclusive_0_8192"] < 512:
|
||||
red_flags.append(f"retained_too_small_{role}")
|
||||
|
||||
selected_counts = [int(entry["selected_count"]) for entry in entries]
|
||||
calibration = [float(entry["calibration_scale"]) for entry in entries]
|
||||
result = {
|
||||
"schema": "fidelity-p1-frontier-prepared-v1",
|
||||
"status": "PASS" if not red_flags else "STOP",
|
||||
"source": {
|
||||
"pilot_manifest": str(args.pilot_manifest.resolve()),
|
||||
"source_windows": str(args.source_windows.resolve()),
|
||||
"source_window_id": args.source_window_id,
|
||||
"source_trace": str(trace.resolve()),
|
||||
"private_windows": (
|
||||
str(private_windows.resolve()) if private_windows is not None else None
|
||||
),
|
||||
"pre_materialized_band_root": (
|
||||
str(args.band_root.resolve()) if args.band_root is not None else None
|
||||
),
|
||||
"band_stats": band_stats,
|
||||
},
|
||||
"simulator": {
|
||||
"replayserve_root": str(args.replayserve_root.resolve()),
|
||||
"frontier_root": str(args.frontier_root.resolve()),
|
||||
"tokenizer": str(args.tokenizer.resolve()),
|
||||
"mode": "frozen-calibrated",
|
||||
},
|
||||
"generator": {
|
||||
"script": str(Path(__file__).resolve()),
|
||||
"script_sha256": sha256_file(Path(__file__).resolve()),
|
||||
"aituner_git_head": git_capture(AITUNER_ROOT, "rev-parse", "HEAD").strip(),
|
||||
"aituner_git_status_short": git_capture(AITUNER_ROOT, "status", "--short"),
|
||||
},
|
||||
"entries": entries,
|
||||
"sanity": {
|
||||
"red_flags": red_flags,
|
||||
"n": len(entries),
|
||||
"selected_count": {
|
||||
"n": len(selected_counts),
|
||||
"min": min(selected_counts),
|
||||
"max": max(selected_counts),
|
||||
"distinct_n": len(set(selected_counts)),
|
||||
},
|
||||
"calibration_scale": {
|
||||
"n": len(calibration),
|
||||
"min": min(calibration),
|
||||
"max": max(calibration),
|
||||
"distinct_n": len(set(calibration)),
|
||||
},
|
||||
"invariants": {
|
||||
"entries_12": len(entries) == 12,
|
||||
"roles_2": {entry["role"] for entry in entries} == set(PRIMARY_ROLES),
|
||||
"cells_6": len({entry["cell"] for entry in entries}) == 6,
|
||||
"selected_nonnegative": all(value > 0 for value in selected_counts),
|
||||
"per_config_not_identical": len(set(selected_counts)) > 1,
|
||||
},
|
||||
},
|
||||
}
|
||||
args.public_manifest.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.public_manifest.write_text(json.dumps(result, indent=2, sort_keys=True) + "\n")
|
||||
return result
|
||||
|
||||
|
||||
def parser() -> argparse.ArgumentParser:
|
||||
result = argparse.ArgumentParser()
|
||||
result.add_argument("--pilot-manifest", type=Path, required=True)
|
||||
result.add_argument("--source-windows", type=Path, required=True)
|
||||
result.add_argument("--source-window-id", required=True)
|
||||
result.add_argument("--private-root", type=Path, required=True)
|
||||
result.add_argument("--band-root", type=Path)
|
||||
result.add_argument("--output", type=Path, required=True)
|
||||
result.add_argument("--public-manifest", type=Path, required=True)
|
||||
result.add_argument("--phase6-raw-root", type=Path, required=True)
|
||||
result.add_argument("--replayserve-root", type=Path, required=True)
|
||||
result.add_argument("--frontier-root", type=Path, required=True)
|
||||
result.add_argument("--cache-dir", type=Path, required=True)
|
||||
result.add_argument("--tokenizer", type=Path, required=True)
|
||||
return result
|
||||
|
||||
|
||||
def main() -> None:
|
||||
result = prepare(parser().parse_args())
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"status": result["status"],
|
||||
"entries": len(result["entries"]),
|
||||
"red_flags": result["sanity"]["red_flags"],
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
if result["status"] != "PASS":
|
||||
raise RuntimeError(result["sanity"]["red_flags"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
303
runs/fidelity-headroom/run_pilot_simulator.py
Normal file
303
runs/fidelity-headroom/run_pilot_simulator.py
Normal file
@@ -0,0 +1,303 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run and score a frozen Frontier probe manifest, CPU only."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import importlib.util
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
def load_module(name: str, path: Path):
|
||||
module_root = str(path.parent.resolve())
|
||||
if module_root not in sys.path:
|
||||
sys.path.insert(0, module_root)
|
||||
spec = importlib.util.spec_from_file_location(name, path)
|
||||
if spec is None or spec.loader is None:
|
||||
raise ImportError(path)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
sys.modules[spec.name] = module
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
def sha256_file(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as source:
|
||||
for chunk in iter(lambda: source.read(1 << 20), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def atomic_json(path: Path, payload: Any) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
temporary = path.with_suffix(path.suffix + ".tmp")
|
||||
temporary.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n")
|
||||
os.replace(temporary, path)
|
||||
|
||||
|
||||
def git_capture(root: Path, *arguments: str) -> str:
|
||||
return subprocess.run(
|
||||
["git", "-C", str(root), *arguments],
|
||||
check=True,
|
||||
text=True,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
).stdout
|
||||
|
||||
|
||||
def execute(args: argparse.Namespace) -> dict[str, Any]:
|
||||
args.output = args.output.resolve()
|
||||
prepared = json.loads(args.prepared_manifest.read_text(encoding="utf-8"))
|
||||
if prepared["status"] != "PASS":
|
||||
raise RuntimeError("prepared simulator manifest did not pass")
|
||||
expected_runs = int(prepared.get("expected_runs", len(prepared["entries"])))
|
||||
if expected_runs != len(prepared["entries"]):
|
||||
raise RuntimeError(
|
||||
f"prepared manifest expected {expected_runs} runs but contains "
|
||||
f"{len(prepared['entries'])} entries"
|
||||
)
|
||||
driver = load_module(
|
||||
"simfid_execution_driver",
|
||||
args.replayserve_root
|
||||
/ "runs/simfid_s2rb/results/execution_driver.py",
|
||||
)
|
||||
head = git_capture(args.frontier_root, "rev-parse", "HEAD").strip()
|
||||
status_short = git_capture(args.frontier_root, "status", "--short")
|
||||
aituner_root = Path(__file__).resolve().parents[2]
|
||||
aituner_head = git_capture(aituner_root, "rev-parse", "HEAD").strip()
|
||||
aituner_status_short = git_capture(aituner_root, "status", "--short")
|
||||
results = []
|
||||
failures = []
|
||||
gpu_visibility_disabled = True
|
||||
for sequence, entry in enumerate(prepared["entries"]):
|
||||
run_root = args.output / f"{sequence:03d}_{entry['fixture_id']}"
|
||||
scorer_path = run_root / "scorer_output.json"
|
||||
if scorer_path.is_file() and args.resume:
|
||||
scorer = json.loads(scorer_path.read_text(encoding="utf-8"))
|
||||
results.append({**entry, "sequence": sequence, "scorer": scorer, "resumed": True})
|
||||
continue
|
||||
run_root.mkdir(parents=True, exist_ok=True)
|
||||
config_path = Path(entry["config"])
|
||||
config = json.loads(config_path.read_text(encoding="utf-8"))
|
||||
fixture_manifest_path = Path(entry["fixture_manifest"])
|
||||
fixture = json.loads(fixture_manifest_path.read_text(encoding="utf-8"))
|
||||
trace_path = Path(entry["frontier_csv"])
|
||||
sidecar_path = Path(entry["sidecar"])
|
||||
metrics_root = run_root / "frontier_metrics"
|
||||
run_id = f"fidelity_p1_frontier_{sequence:02d}_{entry['cell']}_{entry['role']}"
|
||||
knobs = config["frontier"]["knobs"]
|
||||
command = driver.build_command(
|
||||
trace_path=trace_path,
|
||||
metrics_root=metrics_root,
|
||||
run_id=run_id,
|
||||
knobs=knobs,
|
||||
)
|
||||
driver.audit_command(command, knobs)
|
||||
row = {
|
||||
"hook_path": config["calibration"]["hook_path"],
|
||||
"applied_a_tp": config["calibration"]["a_tp"],
|
||||
"sidecar_path": str(sidecar_path),
|
||||
"request_count": int(fixture["request_count"]),
|
||||
"tensor_parallel_size": int(fixture["tensor_parallel_size"]),
|
||||
}
|
||||
environment = driver.environment_for(row)
|
||||
gpu_visibility_disabled = gpu_visibility_disabled and (
|
||||
environment.get("CUDA_VISIBLE_DEVICES") == ""
|
||||
and environment.get("NVIDIA_VISIBLE_DEVICES") == "void"
|
||||
)
|
||||
run_manifest = {
|
||||
"schema": "fidelity-p1-frontier-run-v1",
|
||||
"sequence": sequence,
|
||||
"cell": entry["cell"],
|
||||
"role": entry["role"],
|
||||
"anchor": entry["anchor"],
|
||||
"request_count": entry["selected_count"],
|
||||
"frontier": {
|
||||
"root": str(args.frontier_root.resolve()),
|
||||
"git_head": head,
|
||||
"git_status_short": status_short,
|
||||
},
|
||||
"runner": {
|
||||
"script": str(Path(__file__).resolve()),
|
||||
"script_sha256": sha256_file(Path(__file__).resolve()),
|
||||
"aituner_git_head": aituner_head,
|
||||
"aituner_git_status_short": aituner_status_short,
|
||||
},
|
||||
"inputs": {
|
||||
"config": str(config_path),
|
||||
"config_sha256": sha256_file(config_path),
|
||||
"fixture_manifest": str(fixture_manifest_path),
|
||||
"fixture_manifest_sha256": sha256_file(fixture_manifest_path),
|
||||
"frontier_csv": str(trace_path),
|
||||
"frontier_csv_sha256": sha256_file(trace_path),
|
||||
"sidecar": str(sidecar_path),
|
||||
"sidecar_sha256": sha256_file(sidecar_path),
|
||||
},
|
||||
"environment": {
|
||||
key: environment[key]
|
||||
for key in (
|
||||
"PYTHONPATH",
|
||||
"FRONTIER_EXECUTION_TIME_SCALE",
|
||||
"CUDA_VISIBLE_DEVICES",
|
||||
"NVIDIA_VISIBLE_DEVICES",
|
||||
"FRONTIER_LOG_LEVEL",
|
||||
)
|
||||
},
|
||||
"command": command,
|
||||
"contains_prompt_text": False,
|
||||
}
|
||||
atomic_json(run_root / "run_manifest.json", run_manifest)
|
||||
start = time.time()
|
||||
with (run_root / "stdout.log").open("w", encoding="utf-8") as stdout, (
|
||||
run_root / "stderr.log"
|
||||
).open("w", encoding="utf-8") as stderr:
|
||||
try:
|
||||
process = subprocess.run(
|
||||
command,
|
||||
cwd=args.frontier_root,
|
||||
env=environment,
|
||||
stdout=stdout,
|
||||
stderr=stderr,
|
||||
timeout=args.timeout_s,
|
||||
)
|
||||
return_code = int(process.returncode)
|
||||
except subprocess.TimeoutExpired:
|
||||
return_code = 124
|
||||
runtime = time.time() - start
|
||||
if return_code != 0:
|
||||
failure = {
|
||||
"sequence": sequence,
|
||||
"cell": entry["cell"],
|
||||
"role": entry["role"],
|
||||
"return_code": return_code,
|
||||
"runtime_s": runtime,
|
||||
}
|
||||
failures.append(failure)
|
||||
atomic_json(run_root / "failure.json", failure)
|
||||
break
|
||||
system_path, request_path = driver.find_metrics(run_root)
|
||||
scorer = driver.score_trial(row, system_path, request_path)
|
||||
scorer["runtime_s"] = runtime
|
||||
atomic_json(scorer_path, scorer)
|
||||
results.append({**entry, "sequence": sequence, "scorer": scorer, "resumed": False})
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"sequence": sequence,
|
||||
"cell": entry["cell"],
|
||||
"role": entry["role"],
|
||||
"runtime_s": runtime,
|
||||
"sim_pass_rate": scorer["slo"]["pass_rate"],
|
||||
"sim_feasible": scorer["slo"]["feasible"],
|
||||
},
|
||||
sort_keys=True,
|
||||
),
|
||||
flush=True,
|
||||
)
|
||||
|
||||
pass_rates = [float(item["scorer"]["slo"]["pass_rate"]) for item in results]
|
||||
throughputs = [
|
||||
float(item["scorer"]["throughput_requests_per_second_per_gpu"])
|
||||
for item in results
|
||||
]
|
||||
runtimes = [float(item["scorer"]["runtime_s"]) for item in results]
|
||||
red_flags = []
|
||||
if failures:
|
||||
red_flags.append("frontier_run_failure")
|
||||
if len(results) != expected_runs:
|
||||
red_flags.append("runs_not_expected")
|
||||
if any(not 0.0 <= value <= 1.0 for value in pass_rates):
|
||||
red_flags.append("pass_rate_out_of_range")
|
||||
if any(value <= 0 for value in throughputs):
|
||||
red_flags.append("nonpositive_throughput")
|
||||
result = {
|
||||
"schema": "fidelity-p1-frontier-result-v1",
|
||||
"status": "PASS" if not red_flags else "STOP",
|
||||
"prepared_manifest": str(args.prepared_manifest.resolve()),
|
||||
"prepared_manifest_sha256": sha256_file(args.prepared_manifest),
|
||||
"frontier": {
|
||||
"root": str(args.frontier_root.resolve()),
|
||||
"git_head": head,
|
||||
"git_status_short": status_short,
|
||||
},
|
||||
"runner": {
|
||||
"script": str(Path(__file__).resolve()),
|
||||
"script_sha256": sha256_file(Path(__file__).resolve()),
|
||||
"aituner_git_head": aituner_head,
|
||||
"aituner_git_status_short": aituner_status_short,
|
||||
},
|
||||
"results": results,
|
||||
"failures": failures,
|
||||
"sanity": {
|
||||
"red_flags": red_flags,
|
||||
"n": len(results),
|
||||
"pass_rate": {
|
||||
"n": len(pass_rates),
|
||||
"min": min(pass_rates) if pass_rates else None,
|
||||
"max": max(pass_rates) if pass_rates else None,
|
||||
"distinct_n": len(set(pass_rates)),
|
||||
},
|
||||
"throughput_per_gpu": {
|
||||
"n": len(throughputs),
|
||||
"min": min(throughputs) if throughputs else None,
|
||||
"max": max(throughputs) if throughputs else None,
|
||||
"distinct_n": len(set(throughputs)),
|
||||
},
|
||||
"runtime_s": {
|
||||
"n": len(runtimes),
|
||||
"min": min(runtimes) if runtimes else None,
|
||||
"max": max(runtimes) if runtimes else None,
|
||||
"distinct_n": len(set(runtimes)),
|
||||
},
|
||||
"invariants": {
|
||||
"runs_expected": len(results) == expected_runs,
|
||||
"expected_runs": expected_runs,
|
||||
"zero_failures": not failures,
|
||||
"ratios_bounded": all(0.0 <= value <= 1.0 for value in pass_rates),
|
||||
"nonnegative_metrics": all(value > 0 for value in throughputs),
|
||||
"per_config_not_identical": len(set(pass_rates)) > 1,
|
||||
"gpu_visibility_disabled": gpu_visibility_disabled,
|
||||
},
|
||||
},
|
||||
}
|
||||
atomic_json(args.output / "metrics.json", result)
|
||||
return result
|
||||
|
||||
|
||||
def parser() -> argparse.ArgumentParser:
|
||||
result = argparse.ArgumentParser()
|
||||
result.add_argument("--prepared-manifest", type=Path, required=True)
|
||||
result.add_argument("--output", type=Path, required=True)
|
||||
result.add_argument("--replayserve-root", type=Path, required=True)
|
||||
result.add_argument("--frontier-root", type=Path, required=True)
|
||||
result.add_argument("--timeout-s", type=float, default=900.0)
|
||||
result.add_argument("--resume", action="store_true")
|
||||
return result
|
||||
|
||||
|
||||
def main() -> None:
|
||||
result = execute(parser().parse_args())
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"status": result["status"],
|
||||
"runs": len(result["results"]),
|
||||
"red_flags": result["sanity"]["red_flags"],
|
||||
},
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
if result["status"] != "PASS":
|
||||
raise RuntimeError(result["sanity"]["red_flags"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
477
runs/fidelity-headroom/strong-baseline-metrics.json
Normal file
477
runs/fidelity-headroom/strong-baseline-metrics.json
Normal file
@@ -0,0 +1,477 @@
|
||||
{
|
||||
"comparison": "same 5-second prefix, folds, logistic family, regularization, and frozen Frontier outputs; the only nested difference is real Layer-1 engine state",
|
||||
"decision": {
|
||||
"contribution_established": false,
|
||||
"prospective_requirement": "repeat sim+outcome versus sim+outcome+instrumentation on complete held-out tasks"
|
||||
},
|
||||
"features": {
|
||||
"instrumentation_only": [
|
||||
"model_steps_per_second",
|
||||
"waiting_mean",
|
||||
"waiting_max",
|
||||
"waiting_nonzero_share",
|
||||
"running_mean",
|
||||
"running_max",
|
||||
"decode_batch_mean",
|
||||
"decode_batch_max",
|
||||
"decode_batch_cv",
|
||||
"kv_usage_mean",
|
||||
"kv_usage_max",
|
||||
"kv_usage_end_minus_start",
|
||||
"graph_none_share",
|
||||
"graph_full_share",
|
||||
"padding_fraction",
|
||||
"prefill_token_fraction",
|
||||
"preemptions"
|
||||
],
|
||||
"shared_outcome": [
|
||||
"log_offered_rate_per_gpu",
|
||||
"log2_tp",
|
||||
"log2_max_num_seqs",
|
||||
"admitted_fraction",
|
||||
"completed_over_admitted",
|
||||
"completed_pass_rate",
|
||||
"completed_fail_fraction_of_total",
|
||||
"outstanding_over_admitted",
|
||||
"ttft_max_over_slo_max",
|
||||
"ttft_mean_over_slo_max",
|
||||
"tpot_max_over_slo",
|
||||
"tpot_mean_over_slo",
|
||||
"admitted_input_tokens_mean_over_limit"
|
||||
],
|
||||
"shared_simulator": [
|
||||
"log_sim_completed_throughput_per_gpu",
|
||||
"sim_slo_pass_rate",
|
||||
"sim_slo_feasible"
|
||||
]
|
||||
},
|
||||
"headline": {
|
||||
"group_bootstrap": {
|
||||
"accuracy_delta_instrumentation_minus_outcome": {
|
||||
"ci95": [
|
||||
0.0,
|
||||
0.18181818181818188
|
||||
],
|
||||
"point": 0.08108108108108103
|
||||
},
|
||||
"brier_delta_instrumentation_minus_outcome": {
|
||||
"ci95": [
|
||||
-0.04292727744470806,
|
||||
0.019924730979981074
|
||||
],
|
||||
"point": -0.010145365131402809
|
||||
},
|
||||
"replicates": 10000,
|
||||
"seed": 20260714,
|
||||
"semantics": "group bootstrap over cells; diagnostic confidence interval"
|
||||
},
|
||||
"paired_correctness": {
|
||||
"both_correct": 30,
|
||||
"both_wrong": 4,
|
||||
"instrumentation_only_correct": 3,
|
||||
"mcnemar_exact_two_sided_p": 0.25,
|
||||
"sim_outcome_only_correct": 0
|
||||
},
|
||||
"sim_plus_outcome": {
|
||||
"classification": {
|
||||
"accuracy": 0.8108108108108109,
|
||||
"balanced_accuracy": 0.7242063492063493,
|
||||
"brier": 0.1058226346682949,
|
||||
"confusion": {
|
||||
"false_negative": 3,
|
||||
"false_positive": 4,
|
||||
"true_negative": 5,
|
||||
"true_positive": 25
|
||||
},
|
||||
"log_loss": 0.3011048455679668
|
||||
},
|
||||
"policy_0p95": {
|
||||
"abstain_continue_full": 17,
|
||||
"correctly_saved_h20_hours": 0.5429431818208333,
|
||||
"decision_coverage": 0.5405405405405406,
|
||||
"early_accept": 16,
|
||||
"early_reject": 4,
|
||||
"false_accept": 0,
|
||||
"false_accept_examples": [],
|
||||
"false_reject": 0,
|
||||
"false_reject_examples": [],
|
||||
"full_trial_h20_hours": 1.0669595034675,
|
||||
"invalidly_saved_h20_hours": 0.0,
|
||||
"remaining_h20_hours_at_cutoff": 0.957237281245278,
|
||||
"saved_h20_hours_if_decisions_used": 0.5429431818208333,
|
||||
"threshold": 0.95,
|
||||
"valid_cost_reduction_fraction": 0.5088695307144538,
|
||||
"valid_zero_error_policy": true
|
||||
}
|
||||
},
|
||||
"sim_plus_outcome_plus_instrumentation": {
|
||||
"classification": {
|
||||
"accuracy": 0.8918918918918919,
|
||||
"balanced_accuracy": 0.8154761904761905,
|
||||
"brier": 0.0956772695368921,
|
||||
"confusion": {
|
||||
"false_negative": 1,
|
||||
"false_positive": 3,
|
||||
"true_negative": 6,
|
||||
"true_positive": 27
|
||||
},
|
||||
"log_loss": 0.288823031828762
|
||||
},
|
||||
"policy_0p95": {
|
||||
"abstain_continue_full": 12,
|
||||
"correctly_saved_h20_hours": 0.7360063646722222,
|
||||
"decision_coverage": 0.6756756756756757,
|
||||
"early_accept": 20,
|
||||
"early_reject": 5,
|
||||
"false_accept": 0,
|
||||
"false_accept_examples": [],
|
||||
"false_reject": 0,
|
||||
"false_reject_examples": [],
|
||||
"full_trial_h20_hours": 1.0669595034675,
|
||||
"invalidly_saved_h20_hours": 0.0,
|
||||
"remaining_h20_hours_at_cutoff": 0.957237281245278,
|
||||
"saved_h20_hours_if_decisions_used": 0.7360063646722222,
|
||||
"threshold": 0.95,
|
||||
"valid_cost_reduction_fraction": 0.6898165884274738,
|
||||
"valid_zero_error_policy": true
|
||||
}
|
||||
}
|
||||
},
|
||||
"headline_regularization": 1.0,
|
||||
"provenance": {
|
||||
"frozen_simulator_manifest_scorer_set_sha256": "833842d96ecaa0b059ef99852621752f7989e63d100118b6025425fb119b7a55",
|
||||
"phase6_metrics": "/home/gahow/phd/aituner/runs/opprof-phase6/phase6/metrics.json",
|
||||
"phase6_metrics_sha256": "290ba7fcb8727291166de7e4d47afdc84e230052495c81dd087db0ace9f93a16",
|
||||
"phase6_raw_root": "/home/gahow/phd/aituner/runs/opprof-phase6/phase6/solo-authoritative/cells",
|
||||
"simulator_metrics": "/home/gahow/phd/replayserve/runs/simfid_s2rb/results/metrics.json",
|
||||
"simulator_metrics_sha256": "55edb37d5692e979ab6f6dc6c65913a9db0aa0a836c350e4c05d9c38eee78206",
|
||||
"simulator_raw_root": "/home/gahow/phd/replayserve/runs/simfid_s2rb/results/raw"
|
||||
},
|
||||
"regularization_sensitivity": {
|
||||
"0.1": {
|
||||
"group_bootstrap": {
|
||||
"accuracy_delta_instrumentation_minus_outcome": {
|
||||
"ci95": [
|
||||
-0.17500000000000004,
|
||||
0.0
|
||||
],
|
||||
"point": -0.08108108108108103
|
||||
},
|
||||
"brier_delta_instrumentation_minus_outcome": {
|
||||
"ci95": [
|
||||
-0.026383192545085435,
|
||||
0.0607951286646285
|
||||
],
|
||||
"point": 0.019228316404518567
|
||||
},
|
||||
"replicates": 10000,
|
||||
"seed": 20260714,
|
||||
"semantics": "group bootstrap over cells; diagnostic confidence interval"
|
||||
},
|
||||
"paired_correctness": {
|
||||
"both_correct": 30,
|
||||
"both_wrong": 4,
|
||||
"instrumentation_only_correct": 0,
|
||||
"mcnemar_exact_two_sided_p": 0.25,
|
||||
"sim_outcome_only_correct": 3
|
||||
},
|
||||
"sim_plus_outcome": {
|
||||
"classification": {
|
||||
"accuracy": 0.8918918918918919,
|
||||
"balanced_accuracy": 0.8154761904761905,
|
||||
"brier": 0.10990776306815446,
|
||||
"confusion": {
|
||||
"false_negative": 1,
|
||||
"false_positive": 3,
|
||||
"true_negative": 6,
|
||||
"true_positive": 27
|
||||
},
|
||||
"log_loss": 0.328357763455984
|
||||
},
|
||||
"policy_0p95": {
|
||||
"abstain_continue_full": 12,
|
||||
"correctly_saved_h20_hours": 0.7402314096841667,
|
||||
"decision_coverage": 0.6756756756756757,
|
||||
"early_accept": 20,
|
||||
"early_reject": 5,
|
||||
"false_accept": 0,
|
||||
"false_accept_examples": [],
|
||||
"false_reject": 0,
|
||||
"false_reject_examples": [],
|
||||
"full_trial_h20_hours": 1.0669595034675,
|
||||
"invalidly_saved_h20_hours": 0.0,
|
||||
"remaining_h20_hours_at_cutoff": 0.957237281245278,
|
||||
"saved_h20_hours_if_decisions_used": 0.7402314096841667,
|
||||
"threshold": 0.95,
|
||||
"valid_cost_reduction_fraction": 0.6937764809990414,
|
||||
"valid_zero_error_policy": true
|
||||
}
|
||||
},
|
||||
"sim_plus_outcome_plus_instrumentation": {
|
||||
"classification": {
|
||||
"accuracy": 0.8108108108108109,
|
||||
"balanced_accuracy": 0.7619047619047619,
|
||||
"brier": 0.12913607947267303,
|
||||
"confusion": {
|
||||
"false_negative": 4,
|
||||
"false_positive": 3,
|
||||
"true_negative": 6,
|
||||
"true_positive": 24
|
||||
},
|
||||
"log_loss": 0.4373556318820343
|
||||
},
|
||||
"policy_0p95": {
|
||||
"abstain_continue_full": 9,
|
||||
"correctly_saved_h20_hours": 0.7469523484622221,
|
||||
"decision_coverage": 0.7567567567567568,
|
||||
"early_accept": 22,
|
||||
"early_reject": 6,
|
||||
"false_accept": 2,
|
||||
"false_accept_examples": [
|
||||
{
|
||||
"anchor": 0.49609375,
|
||||
"cell": "tp2_mns8",
|
||||
"label_feasible": false,
|
||||
"probability_feasible": 0.9869795738005246,
|
||||
"remaining_h20_hours": 0.010117910306111111
|
||||
},
|
||||
{
|
||||
"anchor": 0.033717411016,
|
||||
"cell": "tp4_mns16",
|
||||
"label_feasible": false,
|
||||
"probability_feasible": 0.9855364057197005,
|
||||
"remaining_h20_hours": 0.023106262014444445
|
||||
}
|
||||
],
|
||||
"false_reject": 0,
|
||||
"false_reject_examples": [],
|
||||
"full_trial_h20_hours": 1.0669595034675,
|
||||
"invalidly_saved_h20_hours": 0.03322417232055556,
|
||||
"remaining_h20_hours_at_cutoff": 0.957237281245278,
|
||||
"saved_h20_hours_if_decisions_used": 0.7801765207827777,
|
||||
"threshold": 0.95,
|
||||
"valid_cost_reduction_fraction": null,
|
||||
"valid_zero_error_policy": false
|
||||
}
|
||||
}
|
||||
},
|
||||
"1.0": {
|
||||
"group_bootstrap": {
|
||||
"accuracy_delta_instrumentation_minus_outcome": {
|
||||
"ci95": [
|
||||
0.0,
|
||||
0.18181818181818188
|
||||
],
|
||||
"point": 0.08108108108108103
|
||||
},
|
||||
"brier_delta_instrumentation_minus_outcome": {
|
||||
"ci95": [
|
||||
-0.04292727744470806,
|
||||
0.019924730979981074
|
||||
],
|
||||
"point": -0.010145365131402809
|
||||
},
|
||||
"replicates": 10000,
|
||||
"seed": 20260714,
|
||||
"semantics": "group bootstrap over cells; diagnostic confidence interval"
|
||||
},
|
||||
"paired_correctness": {
|
||||
"both_correct": 30,
|
||||
"both_wrong": 4,
|
||||
"instrumentation_only_correct": 3,
|
||||
"mcnemar_exact_two_sided_p": 0.25,
|
||||
"sim_outcome_only_correct": 0
|
||||
},
|
||||
"sim_plus_outcome": {
|
||||
"classification": {
|
||||
"accuracy": 0.8108108108108109,
|
||||
"balanced_accuracy": 0.7242063492063493,
|
||||
"brier": 0.1058226346682949,
|
||||
"confusion": {
|
||||
"false_negative": 3,
|
||||
"false_positive": 4,
|
||||
"true_negative": 5,
|
||||
"true_positive": 25
|
||||
},
|
||||
"log_loss": 0.3011048455679668
|
||||
},
|
||||
"policy_0p95": {
|
||||
"abstain_continue_full": 17,
|
||||
"correctly_saved_h20_hours": 0.5429431818208333,
|
||||
"decision_coverage": 0.5405405405405406,
|
||||
"early_accept": 16,
|
||||
"early_reject": 4,
|
||||
"false_accept": 0,
|
||||
"false_accept_examples": [],
|
||||
"false_reject": 0,
|
||||
"false_reject_examples": [],
|
||||
"full_trial_h20_hours": 1.0669595034675,
|
||||
"invalidly_saved_h20_hours": 0.0,
|
||||
"remaining_h20_hours_at_cutoff": 0.957237281245278,
|
||||
"saved_h20_hours_if_decisions_used": 0.5429431818208333,
|
||||
"threshold": 0.95,
|
||||
"valid_cost_reduction_fraction": 0.5088695307144538,
|
||||
"valid_zero_error_policy": true
|
||||
}
|
||||
},
|
||||
"sim_plus_outcome_plus_instrumentation": {
|
||||
"classification": {
|
||||
"accuracy": 0.8918918918918919,
|
||||
"balanced_accuracy": 0.8154761904761905,
|
||||
"brier": 0.0956772695368921,
|
||||
"confusion": {
|
||||
"false_negative": 1,
|
||||
"false_positive": 3,
|
||||
"true_negative": 6,
|
||||
"true_positive": 27
|
||||
},
|
||||
"log_loss": 0.288823031828762
|
||||
},
|
||||
"policy_0p95": {
|
||||
"abstain_continue_full": 12,
|
||||
"correctly_saved_h20_hours": 0.7360063646722222,
|
||||
"decision_coverage": 0.6756756756756757,
|
||||
"early_accept": 20,
|
||||
"early_reject": 5,
|
||||
"false_accept": 0,
|
||||
"false_accept_examples": [],
|
||||
"false_reject": 0,
|
||||
"false_reject_examples": [],
|
||||
"full_trial_h20_hours": 1.0669595034675,
|
||||
"invalidly_saved_h20_hours": 0.0,
|
||||
"remaining_h20_hours_at_cutoff": 0.957237281245278,
|
||||
"saved_h20_hours_if_decisions_used": 0.7360063646722222,
|
||||
"threshold": 0.95,
|
||||
"valid_cost_reduction_fraction": 0.6898165884274738,
|
||||
"valid_zero_error_policy": true
|
||||
}
|
||||
}
|
||||
},
|
||||
"10.0": {
|
||||
"group_bootstrap": {
|
||||
"accuracy_delta_instrumentation_minus_outcome": {
|
||||
"ci95": [
|
||||
-0.13333333333333341,
|
||||
0.05555555555555558
|
||||
],
|
||||
"point": -0.027027027027027084
|
||||
},
|
||||
"brier_delta_instrumentation_minus_outcome": {
|
||||
"ci95": [
|
||||
-0.03091105649870874,
|
||||
0.01684192005239855
|
||||
],
|
||||
"point": -0.007318433328714388
|
||||
},
|
||||
"replicates": 10000,
|
||||
"seed": 20260714,
|
||||
"semantics": "group bootstrap over cells; diagnostic confidence interval"
|
||||
},
|
||||
"paired_correctness": {
|
||||
"both_correct": 30,
|
||||
"both_wrong": 4,
|
||||
"instrumentation_only_correct": 1,
|
||||
"mcnemar_exact_two_sided_p": 1.0,
|
||||
"sim_outcome_only_correct": 2
|
||||
},
|
||||
"sim_plus_outcome": {
|
||||
"classification": {
|
||||
"accuracy": 0.8648648648648649,
|
||||
"balanced_accuracy": 0.7222222222222222,
|
||||
"brier": 0.10613344425735322,
|
||||
"confusion": {
|
||||
"false_negative": 0,
|
||||
"false_positive": 5,
|
||||
"true_negative": 4,
|
||||
"true_positive": 28
|
||||
},
|
||||
"log_loss": 0.3404203142465075
|
||||
},
|
||||
"policy_0p95": {
|
||||
"abstain_continue_full": 32,
|
||||
"correctly_saved_h20_hours": 0.21727432337249997,
|
||||
"decision_coverage": 0.13513513513513514,
|
||||
"early_accept": 5,
|
||||
"early_reject": 0,
|
||||
"false_accept": 0,
|
||||
"false_accept_examples": [],
|
||||
"false_reject": 0,
|
||||
"false_reject_examples": [],
|
||||
"full_trial_h20_hours": 1.0669595034675,
|
||||
"invalidly_saved_h20_hours": 0.0,
|
||||
"remaining_h20_hours_at_cutoff": 0.957237281245278,
|
||||
"saved_h20_hours_if_decisions_used": 0.21727432337249997,
|
||||
"threshold": 0.95,
|
||||
"valid_cost_reduction_fraction": 0.20363877229302757,
|
||||
"valid_zero_error_policy": true
|
||||
}
|
||||
},
|
||||
"sim_plus_outcome_plus_instrumentation": {
|
||||
"classification": {
|
||||
"accuracy": 0.8378378378378378,
|
||||
"balanced_accuracy": 0.7420634920634921,
|
||||
"brier": 0.09881501092863883,
|
||||
"confusion": {
|
||||
"false_negative": 2,
|
||||
"false_positive": 4,
|
||||
"true_negative": 5,
|
||||
"true_positive": 26
|
||||
},
|
||||
"log_loss": 0.312914193285738
|
||||
},
|
||||
"policy_0p95": {
|
||||
"abstain_continue_full": 30,
|
||||
"correctly_saved_h20_hours": 0.2384080185036111,
|
||||
"decision_coverage": 0.1891891891891892,
|
||||
"early_accept": 6,
|
||||
"early_reject": 1,
|
||||
"false_accept": 0,
|
||||
"false_accept_examples": [],
|
||||
"false_reject": 0,
|
||||
"false_reject_examples": [],
|
||||
"full_trial_h20_hours": 1.0669595034675,
|
||||
"invalidly_saved_h20_hours": 0.0,
|
||||
"remaining_h20_hours_at_cutoff": 0.957237281245278,
|
||||
"saved_h20_hours_if_decisions_used": 0.2384080185036111,
|
||||
"threshold": 0.95,
|
||||
"valid_cost_reduction_fraction": 0.22344617366339725,
|
||||
"valid_zero_error_policy": true
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"sanity": {
|
||||
"examples": {
|
||||
"distinct_n": 1,
|
||||
"max": 1.0,
|
||||
"min": 1.0,
|
||||
"n": 37
|
||||
},
|
||||
"frozen_simulator_runs": 92,
|
||||
"invariants": {
|
||||
"all_examples_matched_once": true,
|
||||
"labels_not_identical": true,
|
||||
"per_config_results_not_all_identical": true,
|
||||
"same_nested_folds": true,
|
||||
"simulator_ratios_bounded": true
|
||||
},
|
||||
"labels": {
|
||||
"distinct_n": 2,
|
||||
"max": 1.0,
|
||||
"min": 0.0,
|
||||
"n": 37,
|
||||
"negative": 9,
|
||||
"positive": 28
|
||||
},
|
||||
"matched_simulator_pass_rate": {
|
||||
"distinct_n": 12,
|
||||
"max": 1.0,
|
||||
"min": 0.06884057971014493,
|
||||
"n": 37
|
||||
},
|
||||
"red_flags": []
|
||||
},
|
||||
"schema": "fidelity-strong-baseline-v1",
|
||||
"scope": "retrospective one-task headroom audit; not contribution evidence",
|
||||
"status": "PASS"
|
||||
}
|
||||
50
runs/fidelity-headroom/test_pilot_e2e.py
Normal file
50
runs/fidelity-headroom/test_pilot_e2e.py
Normal file
@@ -0,0 +1,50 @@
|
||||
#!/usr/bin/env python3
|
||||
from __future__ import annotations
|
||||
|
||||
from analyze_pilot_e2e import expanded_top_k, replay
|
||||
|
||||
|
||||
def candidate(
|
||||
cell: str,
|
||||
sim_score: float,
|
||||
real_feasible: bool,
|
||||
probability: float,
|
||||
) -> dict[str, object]:
|
||||
return {
|
||||
"cell": cell,
|
||||
"level": "high",
|
||||
"sim_throughput_req_s_per_gpu": sim_score,
|
||||
"real_goodput_req_s_per_gpu": sim_score,
|
||||
"real_feasible": real_feasible,
|
||||
"setup_h20_hours": 0.1,
|
||||
"full_trial_h20_hours": 0.05,
|
||||
"prefix_h20_hours": 0.01,
|
||||
"instrument_probability": probability,
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
candidates = [
|
||||
candidate("a", 3.0, True, 0.5),
|
||||
candidate("b", 2.0, False, 0.01),
|
||||
candidate("c", 2.0, True, 0.99),
|
||||
]
|
||||
shortlist = expanded_top_k(candidates, 2)
|
||||
assert [item["cell"] for item in shortlist] == ["a", "b", "c"]
|
||||
result = replay(
|
||||
shortlist,
|
||||
probability_key="instrument_probability",
|
||||
oracle_goodput=3.0,
|
||||
common_failure_h20_hours=0.02,
|
||||
)
|
||||
assert result["selected_cell"] == "a"
|
||||
assert result["false_accept"] == 0
|
||||
assert result["false_reject"] == 0
|
||||
assert result["early_accept"] == 1
|
||||
assert result["early_reject"] == 1
|
||||
assert result["online_h20_hours"] > 0
|
||||
print("fidelity pilot e2e: PASS")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -14,6 +14,7 @@ sys.path.insert(0, str(HERE))
|
||||
|
||||
import pilot_controller as controller # noqa: E402
|
||||
import prepare_pilot as prepare # noqa: E402
|
||||
from analyze_pilot import campaign_gpu_accounting # noqa: E402
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -69,6 +70,32 @@ def main() -> None:
|
||||
assert row["fidelity_pilot_band"] == role
|
||||
assert abs(float(row["sampling_u"]) - 0.5) < 1e-12
|
||||
|
||||
prior = root / "prior-state.json"
|
||||
primary = root / "primary-state.json"
|
||||
prior.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"status": "failed",
|
||||
"gpu_hours_total": 0.02,
|
||||
"hard_cap_h20_hours": 3.5,
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
primary.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"status": "complete",
|
||||
"gpu_hours_total": 1.5,
|
||||
"hard_cap_h20_hours": 3.5,
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
accounting = campaign_gpu_accounting(primary, (prior,))
|
||||
assert math.isclose(accounting["aggregate_h20_hours"], 1.52)
|
||||
assert all(accounting["invariants"].values())
|
||||
|
||||
assert len(controller.ORDER) == 6
|
||||
assert set(controller.ORDER) == set(prepare.CELLS)
|
||||
assert math.isclose(
|
||||
|
||||
37
runs/fidelity-headroom/test_strong_baseline.py
Normal file
37
runs/fidelity-headroom/test_strong_baseline.py
Normal file
@@ -0,0 +1,37 @@
|
||||
#!/usr/bin/env python3
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from analyze_strong_baseline import analyze
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
REPLAYSERVE = ROOT.parent / "replayserve"
|
||||
|
||||
|
||||
def main() -> None:
|
||||
result = analyze(
|
||||
ROOT / "runs/opprof-phase6/phase6/metrics.json",
|
||||
ROOT / "runs/opprof-phase6/phase6/solo-authoritative/cells",
|
||||
REPLAYSERVE / "runs/simfid_s2rb/results/raw",
|
||||
REPLAYSERVE / "runs/simfid_s2rb/results/metrics.json",
|
||||
)
|
||||
assert result["status"] == "PASS", json.dumps(result["sanity"], indent=2)
|
||||
assert result["sanity"]["frozen_simulator_runs"] == 92
|
||||
assert result["sanity"]["labels"]["n"] == 37
|
||||
headline = result["headline"]
|
||||
assert headline["sim_plus_outcome"]["policy_0p95"]["false_accept"] == 0
|
||||
assert headline["sim_plus_outcome"]["policy_0p95"]["false_reject"] == 0
|
||||
assert (
|
||||
headline["sim_plus_outcome_plus_instrumentation"]["policy_0p95"][
|
||||
"false_accept"
|
||||
]
|
||||
== 0
|
||||
)
|
||||
print("fidelity strong baseline: PASS")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
100
runs/fidelity-headroom/test_strong_pilot.py
Normal file
100
runs/fidelity-headroom/test_strong_pilot.py
Normal file
@@ -0,0 +1,100 @@
|
||||
#!/usr/bin/env python3
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
from analyze_prefixes import PrefixExample
|
||||
from prepare_pilot_simulator import load_module as load_prepare_module
|
||||
from run_pilot_simulator import load_module as load_run_module
|
||||
from analyze_strong_pilot import (
|
||||
covariate_shift,
|
||||
fit_model,
|
||||
load_pilot_simulator,
|
||||
predict_model,
|
||||
)
|
||||
|
||||
|
||||
def example(index: int) -> PrefixExample:
|
||||
label = int(index >= 4)
|
||||
return PrefixExample(
|
||||
cell=f"cell-{index // 2}",
|
||||
anchor=float(index),
|
||||
cutoff_s=5.0,
|
||||
tp=1,
|
||||
full_elapsed_s=10.0,
|
||||
feasible=label,
|
||||
primary_feasible=label,
|
||||
outcome=tuple(float(index + offset) for offset in range(13)),
|
||||
instrumentation=tuple(float(index * offset + 1) for offset in range(17)),
|
||||
completion_time_source="exact_monotonic",
|
||||
)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
examples = [example(index) for index in range(8)]
|
||||
simulator = [(float(index), index / 10.0, float(index >= 4)) for index in range(8)]
|
||||
for instrumentation_aware in (False, True):
|
||||
model = fit_model(
|
||||
examples,
|
||||
simulator,
|
||||
instrumentation_aware=instrumentation_aware,
|
||||
regularization=1.0,
|
||||
)
|
||||
probability = predict_model(model, examples, simulator)
|
||||
assert probability.shape == (8,)
|
||||
assert np.all((probability >= 0.0) & (probability <= 1.0))
|
||||
shift = covariate_shift(
|
||||
examples,
|
||||
simulator,
|
||||
examples,
|
||||
simulator,
|
||||
instrumentation_aware=instrumentation_aware,
|
||||
)
|
||||
assert shift["values"]["min"] >= 0.0
|
||||
assert shift["count_gt_3"] == 0
|
||||
|
||||
payload = {
|
||||
"status": "PASS",
|
||||
"results": [
|
||||
{
|
||||
"cell": f"cell-{index // 2}",
|
||||
"role": "low1" if index % 2 == 0 else "high1",
|
||||
"scorer": {
|
||||
"throughput_requests_per_second_per_gpu": 1.0 + index,
|
||||
"slo": {
|
||||
"pass_rate": index / 12.0,
|
||||
"feasible": index % 2 == 0,
|
||||
},
|
||||
},
|
||||
}
|
||||
for index in range(12)
|
||||
],
|
||||
}
|
||||
with tempfile.TemporaryDirectory() as temporary:
|
||||
path = Path(temporary) / "metrics.json"
|
||||
path.write_text(json.dumps(payload), encoding="utf-8")
|
||||
features, red_flags = load_pilot_simulator(path)
|
||||
assert len(features) == 12
|
||||
assert red_flags == []
|
||||
|
||||
with tempfile.TemporaryDirectory() as temporary:
|
||||
root = Path(temporary)
|
||||
(root / "prepare_dependency.py").write_text("VALUE = 17\n", encoding="utf-8")
|
||||
(root / "prepare_target.py").write_text(
|
||||
"from prepare_dependency import VALUE\n", encoding="utf-8"
|
||||
)
|
||||
assert load_prepare_module(root / "prepare_target.py").VALUE == 17
|
||||
(root / "run_dependency.py").write_text("VALUE = 23\n", encoding="utf-8")
|
||||
(root / "run_target.py").write_text(
|
||||
"from run_dependency import VALUE\n", encoding="utf-8"
|
||||
)
|
||||
assert load_run_module("run_target", root / "run_target.py").VALUE == 23
|
||||
print("fidelity strong pilot: PASS")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
4
runs/frontier-attn-structured-v0/.gitignore
vendored
Normal file
4
runs/frontier-attn-structured-v0/.gitignore
vendored
Normal file
@@ -0,0 +1,4 @@
|
||||
__pycache__/
|
||||
cache/
|
||||
replay/
|
||||
figure-prototype.svg
|
||||
@@ -0,0 +1,249 @@
|
||||
From 1f8900a4ac64e45754b03d0aa7c1dddab65785cf Mon Sep 17 00:00:00 2001
|
||||
From: Gahow Wang <gahow.wang@gmail.com>
|
||||
Date: Thu, 23 Jul 2026 15:27:40 +0800
|
||||
Subject: [PATCH] Experiment with structured attention prefill predictor
|
||||
|
||||
---
|
||||
.../shared_prediction_model_manager.py | 16 +++-
|
||||
.../sklearn_execution_time_predictor.py | 16 +++-
|
||||
.../structured_attention_prefill.py | 79 +++++++++++++++++++
|
||||
.../unit/test_structured_attention_prefill.py | 59 ++++++++++++++
|
||||
4 files changed, 165 insertions(+), 5 deletions(-)
|
||||
create mode 100644 frontier/execution_time_predictor/structured_attention_prefill.py
|
||||
create mode 100644 tests/unit/test_structured_attention_prefill.py
|
||||
|
||||
diff --git a/frontier/execution_time_predictor/shared_prediction_model_manager.py b/frontier/execution_time_predictor/shared_prediction_model_manager.py
|
||||
index 8a65a49..4a21165 100644
|
||||
--- a/frontier/execution_time_predictor/shared_prediction_model_manager.py
|
||||
+++ b/frontier/execution_time_predictor/shared_prediction_model_manager.py
|
||||
@@ -19,6 +19,9 @@ from frontier.execution_time_predictor.attention_tp_policy import (
|
||||
from frontier.execution_time_predictor.attention_dataset_contract import (
|
||||
enforce_mixed_attention_input_contract,
|
||||
)
|
||||
+from frontier.execution_time_predictor.structured_attention_prefill import (
|
||||
+ StructuredAttentionPrefillRegressor,
|
||||
+)
|
||||
from frontier.logger import init_logger
|
||||
from frontier.moe_gating_runtime import (
|
||||
DEFAULT_MOE_GATING_RUNTIME_CONTEXT,
|
||||
@@ -1254,7 +1257,10 @@ class ExecutionTimePredictionModelManager:
|
||||
raise ValueError(
|
||||
"Missing required column 'prefill_chunk_size' in attention profiling data."
|
||||
)
|
||||
- standard_prefill_df = prefill_df[prefill_df["prefill_chunk_size"] > 0].copy()
|
||||
+ standard_prefill_df = prefill_df[
|
||||
+ (prefill_df["prefill_chunk_size"] > 0)
|
||||
+ & (prefill_df["batch_size"] == 1)
|
||||
+ ].copy()
|
||||
|
||||
prefill_model_signature = f"attn_prefill_{attention_signature}"
|
||||
if prefill_model_signature not in trained_model_signatures:
|
||||
@@ -1742,7 +1748,13 @@ class ExecutionTimePredictionModelManager:
|
||||
# initialization to generate missing cache files.
|
||||
# ============================================================
|
||||
|
||||
- estimator, grid_search_params = self._create_estimator_and_params(execution_time_predictor_config)
|
||||
+ if model_name == "attn_prefill":
|
||||
+ estimator = StructuredAttentionPrefillRegressor()
|
||||
+ grid_search_params = {}
|
||||
+ else:
|
||||
+ estimator, grid_search_params = self._create_estimator_and_params(
|
||||
+ execution_time_predictor_config
|
||||
+ )
|
||||
|
||||
cv = min(execution_time_predictor_config.k_fold_cv_splits, len(df)) if len(df) >= 2 else 2
|
||||
|
||||
diff --git a/frontier/execution_time_predictor/sklearn_execution_time_predictor.py b/frontier/execution_time_predictor/sklearn_execution_time_predictor.py
|
||||
index 27b62bf..b7f5350 100644
|
||||
--- a/frontier/execution_time_predictor/sklearn_execution_time_predictor.py
|
||||
+++ b/frontier/execution_time_predictor/sklearn_execution_time_predictor.py
|
||||
@@ -45,6 +45,9 @@ from frontier.execution_time_predictor.attention_tp_policy import (
|
||||
from frontier.execution_time_predictor.attention_dataset_contract import (
|
||||
enforce_mixed_attention_input_contract,
|
||||
)
|
||||
+from frontier.execution_time_predictor.structured_attention_prefill import (
|
||||
+ StructuredAttentionPrefillRegressor,
|
||||
+)
|
||||
from frontier.logger import init_logger
|
||||
from frontier.moe_gating_runtime import get_moe_gating_base_model_name
|
||||
from frontier.profiling.cpu_overhead.schema import (
|
||||
@@ -2573,8 +2576,12 @@ class SklearnExecutionTimePredictor(BaseExecutionTimePredictor):
|
||||
if cached_model:
|
||||
return cached_model
|
||||
|
||||
- model = self._get_estimator()
|
||||
- grid_search_params = self._get_grid_search_params()
|
||||
+ if model_name == "attn_prefill":
|
||||
+ model = StructuredAttentionPrefillRegressor()
|
||||
+ grid_search_params = {}
|
||||
+ else:
|
||||
+ model = self._get_estimator()
|
||||
+ grid_search_params = self._get_grid_search_params()
|
||||
|
||||
if len(df) < self._config.k_fold_cv_splits:
|
||||
cv = 2
|
||||
@@ -2869,7 +2876,10 @@ class SklearnExecutionTimePredictor(BaseExecutionTimePredictor):
|
||||
raise ValueError(
|
||||
"Missing required column 'prefill_chunk_size' in attention profiling data."
|
||||
)
|
||||
- standard_prefill_df = prefill_df[prefill_df["prefill_chunk_size"] > 0].copy()
|
||||
+ standard_prefill_df = prefill_df[
|
||||
+ (prefill_df["prefill_chunk_size"] > 0)
|
||||
+ & (prefill_df["batch_size"] == 1)
|
||||
+ ].copy()
|
||||
if len(standard_prefill_df) == 0:
|
||||
raise ValueError(
|
||||
"No standard prefill rows (prefill_chunk_size > 0) found in eager attention profiling data."
|
||||
diff --git a/frontier/execution_time_predictor/structured_attention_prefill.py b/frontier/execution_time_predictor/structured_attention_prefill.py
|
||||
new file mode 100644
|
||||
index 0000000..1829047
|
||||
--- /dev/null
|
||||
+++ b/frontier/execution_time_predictor/structured_attention_prefill.py
|
||||
@@ -0,0 +1,79 @@
|
||||
+"""Structured latency model for single-request chunked prefill attention."""
|
||||
+
|
||||
+from typing import Any
|
||||
+
|
||||
+import numpy as np
|
||||
+from sklearn.base import BaseEstimator, RegressorMixin
|
||||
+from sklearn.isotonic import IsotonicRegression
|
||||
+from sklearn.linear_model import LinearRegression
|
||||
+
|
||||
+
|
||||
+class StructuredAttentionPrefillRegressor(RegressorMixin, BaseEstimator):
|
||||
+ """Model attention as a monotone base curve plus continuous KV growth.
|
||||
+
|
||||
+ Input columns retain the existing Frontier contract:
|
||||
+ ``[kv_cache_size, prefill_chunk_size_squared]``.
|
||||
+ """
|
||||
+
|
||||
+ def fit(self, X: Any, y: Any) -> "StructuredAttentionPrefillRegressor":
|
||||
+ values = self._as_feature_array(X)
|
||||
+ target = np.asarray(y, dtype=float)
|
||||
+ kv_cache_size = values[:, 0]
|
||||
+ prefill_chunk_size = np.sqrt(np.maximum(values[:, 1], 0.0))
|
||||
+
|
||||
+ base_mask = np.isclose(kv_cache_size, 0.0)
|
||||
+ growth_mask = kv_cache_size > 0.0
|
||||
+ if not np.any(base_mask) or not np.any(growth_mask):
|
||||
+ raise ValueError(
|
||||
+ "structured attn_prefill training requires both KV=0 base rows "
|
||||
+ "and KV>0 growth rows"
|
||||
+ )
|
||||
+
|
||||
+ base_q = prefill_chunk_size[base_mask]
|
||||
+ base_y = target[base_mask]
|
||||
+ unique_q = np.unique(base_q)
|
||||
+ grouped_y = np.asarray(
|
||||
+ [np.mean(base_y[np.isclose(base_q, q)]) for q in unique_q],
|
||||
+ dtype=float,
|
||||
+ )
|
||||
+ self._base_model = IsotonicRegression(
|
||||
+ increasing=True,
|
||||
+ out_of_bounds="clip",
|
||||
+ ).fit(unique_q, grouped_y)
|
||||
+
|
||||
+ growth_q = prefill_chunk_size[growth_mask]
|
||||
+ growth_kv = kv_cache_size[growth_mask]
|
||||
+ growth_base = self._base_model.predict(growth_q)
|
||||
+ growth_features = np.column_stack(
|
||||
+ (growth_kv, growth_q * growth_kv)
|
||||
+ )
|
||||
+ self._growth_model = LinearRegression(
|
||||
+ fit_intercept=False,
|
||||
+ positive=True,
|
||||
+ ).fit(growth_features, target[growth_mask] - growth_base)
|
||||
+
|
||||
+ self.n_features_in_ = 2
|
||||
+ self._frontier_base_q_min = float(unique_q.min())
|
||||
+ self._frontier_base_q_max = float(unique_q.max())
|
||||
+ self._frontier_growth_kv_max = float(growth_kv.max())
|
||||
+ return self
|
||||
+
|
||||
+ def predict(self, X: Any) -> np.ndarray:
|
||||
+ values = self._as_feature_array(X)
|
||||
+ kv_cache_size = values[:, 0]
|
||||
+ prefill_chunk_size = np.sqrt(np.maximum(values[:, 1], 0.0))
|
||||
+ base = self._base_model.predict(prefill_chunk_size)
|
||||
+ growth_features = np.column_stack(
|
||||
+ (kv_cache_size, prefill_chunk_size * kv_cache_size)
|
||||
+ )
|
||||
+ return np.maximum(base + self._growth_model.predict(growth_features), 0.0)
|
||||
+
|
||||
+ @staticmethod
|
||||
+ def _as_feature_array(X: Any) -> np.ndarray:
|
||||
+ values = np.asarray(X, dtype=float)
|
||||
+ if values.ndim != 2 or values.shape[1] != 2:
|
||||
+ raise ValueError(
|
||||
+ "structured attn_prefill expects exactly two features: "
|
||||
+ "kv_cache_size and prefill_chunk_size_squared"
|
||||
+ )
|
||||
+ return values
|
||||
diff --git a/tests/unit/test_structured_attention_prefill.py b/tests/unit/test_structured_attention_prefill.py
|
||||
new file mode 100644
|
||||
index 0000000..12c4247
|
||||
--- /dev/null
|
||||
+++ b/tests/unit/test_structured_attention_prefill.py
|
||||
@@ -0,0 +1,59 @@
|
||||
+import pickle
|
||||
+import unittest
|
||||
+
|
||||
+import numpy as np
|
||||
+
|
||||
+from frontier.execution_time_predictor.structured_attention_prefill import (
|
||||
+ StructuredAttentionPrefillRegressor,
|
||||
+)
|
||||
+
|
||||
+
|
||||
+class StructuredAttentionPrefillRegressorTest(unittest.TestCase):
|
||||
+ def setUp(self) -> None:
|
||||
+ q = np.asarray([64, 128, 256, 512, 1024, 2048, 4096, 8192], dtype=float)
|
||||
+ base = 0.05 + 1e-4 * q + 4e-8 * q**2
|
||||
+ context_q = np.asarray([2048, 4096, 8192] * 3, dtype=float)
|
||||
+ context_kv = np.repeat([8192, 16384, 24576], 3).astype(float)
|
||||
+ context_y = (
|
||||
+ np.interp(context_q, q, base)
|
||||
+ + 1.5e-5 * context_kv
|
||||
+ + 3e-8 * context_q * context_kv
|
||||
+ )
|
||||
+ self.X = np.column_stack(
|
||||
+ (
|
||||
+ np.concatenate((np.zeros_like(q), context_kv)),
|
||||
+ np.concatenate((q**2, context_q**2)),
|
||||
+ )
|
||||
+ )
|
||||
+ self.y = np.concatenate((base, context_y))
|
||||
+
|
||||
+ def test_recovers_structured_curve(self) -> None:
|
||||
+ model = StructuredAttentionPrefillRegressor().fit(self.X, self.y)
|
||||
+ np.testing.assert_allclose(model.predict(self.X), self.y, rtol=1e-6)
|
||||
+
|
||||
+ def test_prediction_is_nonnegative_and_monotone(self) -> None:
|
||||
+ model = StructuredAttentionPrefillRegressor().fit(self.X, self.y)
|
||||
+ q = np.arange(1, 8193, dtype=float)
|
||||
+ for kv in (0, 8192, 32768, 40912):
|
||||
+ X = np.column_stack((np.full_like(q, kv), q**2))
|
||||
+ prediction = model.predict(X)
|
||||
+ self.assertTrue(np.all(prediction >= 0))
|
||||
+ self.assertTrue(np.all(np.diff(prediction) >= -1e-12))
|
||||
+
|
||||
+ kv = np.arange(0, 40913, 64, dtype=float)
|
||||
+ for q_value in (64, 2048, 8192):
|
||||
+ X = np.column_stack((kv, np.full_like(kv, q_value**2)))
|
||||
+ self.assertTrue(np.all(np.diff(model.predict(X)) >= -1e-12))
|
||||
+
|
||||
+ def test_pickle_round_trip(self) -> None:
|
||||
+ model = StructuredAttentionPrefillRegressor().fit(self.X, self.y)
|
||||
+ restored = pickle.loads(pickle.dumps(model))
|
||||
+ np.testing.assert_allclose(restored.predict(self.X), self.y, rtol=1e-6)
|
||||
+
|
||||
+ def test_requires_base_and_growth_rows(self) -> None:
|
||||
+ with self.assertRaisesRegex(ValueError, "KV=0 base rows"):
|
||||
+ StructuredAttentionPrefillRegressor().fit(self.X[:8], self.y[:8])
|
||||
+
|
||||
+
|
||||
+if __name__ == "__main__":
|
||||
+ unittest.main()
|
||||
--
|
||||
2.43.0
|
||||
|
||||
256
runs/frontier-attn-structured-v0/analyze_predictor_ablation.py
Normal file
256
runs/frontier-attn-structured-v0/analyze_predictor_ablation.py
Normal file
@@ -0,0 +1,256 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Offline predictor ablation for EXP-ATTN-STRUCTURED.
|
||||
|
||||
This is deliberately profile-only: it decides whether the structured model is
|
||||
good enough to justify the expensive 7-cell trace replay.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from sklearn.ensemble import RandomForestRegressor
|
||||
|
||||
ROOT = Path(__file__).resolve().parent
|
||||
REPO = ROOT.parents[1]
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument(
|
||||
"--profile",
|
||||
type=Path,
|
||||
default=REPO
|
||||
/ "runs/frontier-prefill-kvgrowth-fix-v0/profiles/"
|
||||
"profile-v5-kvgrowth/attention.csv",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--frontier-checkout",
|
||||
type=Path,
|
||||
default=Path("/tmp/frontier-attn-structured-v0"),
|
||||
)
|
||||
parser.add_argument("--output-root", type=Path, default=ROOT / "results")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def normalize_bool(series: pd.Series) -> pd.Series:
|
||||
return series.astype(str).str.strip().str.lower().isin(
|
||||
{"1", "true", "t", "yes", "y"}
|
||||
)
|
||||
|
||||
|
||||
def load_profile(path: Path) -> pd.DataFrame:
|
||||
df = pd.read_csv(path).drop_duplicates()
|
||||
for column in ("is_prefill", "is_true_mixed_batch"):
|
||||
df[column] = normalize_bool(df[column])
|
||||
df = df[
|
||||
(df["n_embd"] == 2048)
|
||||
& (df["n_q_head"] == 32)
|
||||
& (df["n_kv_head"] == 4)
|
||||
& (df["block_size"] == 16)
|
||||
& df["is_prefill"]
|
||||
& ~df["is_true_mixed_batch"]
|
||||
& (df["prefill_chunk_size"] > 0)
|
||||
].copy()
|
||||
df["prefill_chunk_size_squared"] = df["prefill_chunk_size"] ** 2
|
||||
return df
|
||||
|
||||
|
||||
def mape(actual: np.ndarray, predicted: np.ndarray) -> float:
|
||||
return float(np.mean(np.abs((predicted - actual) / actual)) * 100)
|
||||
|
||||
|
||||
def make_rf() -> RandomForestRegressor:
|
||||
# Exact best parameters selected by the current profile-v5 GridSearchCV.
|
||||
return RandomForestRegressor(
|
||||
random_state=0,
|
||||
n_estimators=250,
|
||||
max_depth=8,
|
||||
min_samples_split=2,
|
||||
)
|
||||
|
||||
|
||||
def features(df: pd.DataFrame) -> pd.DataFrame:
|
||||
return df[["kv_cache_size", "prefill_chunk_size_squared"]]
|
||||
|
||||
|
||||
def score_model(
|
||||
name: str,
|
||||
estimator: Any,
|
||||
train: pd.DataFrame,
|
||||
single: pd.DataFrame,
|
||||
grid: pd.DataFrame,
|
||||
) -> dict[str, Any]:
|
||||
target = "time_stats.attn_prefill.median"
|
||||
estimator.fit(features(train), train[target])
|
||||
grid_prediction = estimator.predict(features(grid))
|
||||
single_prediction = estimator.predict(features(single))
|
||||
|
||||
heldout_actual: list[float] = []
|
||||
heldout_prediction: list[float] = []
|
||||
for context in sorted(grid["kv_cache_size"].unique()):
|
||||
test = grid[grid["kv_cache_size"] == context]
|
||||
fold_train = train.drop(index=test.index, errors="ignore")
|
||||
fold_model = (
|
||||
make_rf()
|
||||
if name.startswith("rf")
|
||||
else estimator.__class__()
|
||||
)
|
||||
fold_model.fit(features(fold_train), fold_train[target])
|
||||
heldout_actual.extend(test[target].astype(float))
|
||||
heldout_prediction.extend(fold_model.predict(features(test)))
|
||||
|
||||
q = np.arange(1, 8193, dtype=float)
|
||||
q_deltas: list[float] = []
|
||||
prediction_min: list[float] = []
|
||||
for context in (0, 8192, 16384, 24576, 32768, 40912):
|
||||
X = pd.DataFrame(
|
||||
{
|
||||
"kv_cache_size": np.full_like(q, context),
|
||||
"prefill_chunk_size_squared": q**2,
|
||||
}
|
||||
)
|
||||
prediction = estimator.predict(X)
|
||||
prediction_min.append(float(prediction.min()))
|
||||
q_deltas.append(float(np.diff(prediction).min()))
|
||||
|
||||
kv = np.arange(0, 40913, 64, dtype=float)
|
||||
kv_deltas: list[float] = []
|
||||
for query in (64, 512, 2048, 4096, 8192):
|
||||
X = pd.DataFrame(
|
||||
{
|
||||
"kv_cache_size": kv,
|
||||
"prefill_chunk_size_squared": np.full_like(kv, query**2),
|
||||
}
|
||||
)
|
||||
kv_deltas.append(float(np.diff(estimator.predict(X)).min()))
|
||||
|
||||
heldout_actual_array = np.asarray(heldout_actual)
|
||||
heldout_prediction_array = np.asarray(heldout_prediction)
|
||||
return {
|
||||
"candidate": name,
|
||||
"training_rows": len(train),
|
||||
"grid_fit_mape_pct": mape(
|
||||
grid[target].to_numpy(), np.asarray(grid_prediction)
|
||||
),
|
||||
"single_fit_mape_pct": mape(
|
||||
single[target].to_numpy(), np.asarray(single_prediction)
|
||||
),
|
||||
"heldout_context_mape_pct": mape(
|
||||
heldout_actual_array, heldout_prediction_array
|
||||
),
|
||||
"heldout_context_max_abs_error_pct": float(
|
||||
np.max(
|
||||
np.abs(
|
||||
(heldout_prediction_array - heldout_actual_array)
|
||||
/ heldout_actual_array
|
||||
)
|
||||
)
|
||||
* 100
|
||||
),
|
||||
"prediction_min_ms": min(prediction_min),
|
||||
"q_min_delta_ms": min(q_deltas),
|
||||
"kv_min_delta_ms": min(kv_deltas),
|
||||
"monotone_and_nonnegative": (
|
||||
min(prediction_min) >= 0
|
||||
and min(q_deltas) >= -1e-12
|
||||
and min(kv_deltas) >= -1e-12
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parse_args()
|
||||
sys.path.insert(0, str(args.frontier_checkout))
|
||||
from frontier.execution_time_predictor.structured_attention_prefill import (
|
||||
StructuredAttentionPrefillRegressor,
|
||||
)
|
||||
|
||||
df = load_profile(args.profile)
|
||||
records: list[dict[str, Any]] = []
|
||||
data_audit: dict[str, Any] = {}
|
||||
for tp in (1, 2, 4):
|
||||
tp_df = df[df["num_tensor_parallel_workers"] == tp].copy()
|
||||
single = tp_df[tp_df["batch_size"] == 1].copy()
|
||||
grid = single[
|
||||
single["prefill_chunk_size"].isin((2048, 4096, 8192))
|
||||
& (single["kv_cache_size"] > 0)
|
||||
].copy()
|
||||
duplicate_groups = (
|
||||
tp_df.groupby(
|
||||
["kv_cache_size", "prefill_chunk_size_squared"]
|
||||
)
|
||||
.size()
|
||||
.gt(1)
|
||||
.sum()
|
||||
)
|
||||
data_audit[f"tp{tp}"] = {
|
||||
"standard_rows": len(tp_df),
|
||||
"single_request_rows": len(single),
|
||||
"target_grid_rows": len(grid),
|
||||
"duplicate_feature_groups": int(duplicate_groups),
|
||||
}
|
||||
|
||||
candidates = (
|
||||
("rf_all", make_rf(), tp_df),
|
||||
("rf_single", make_rf(), single),
|
||||
(
|
||||
"structured_single",
|
||||
StructuredAttentionPrefillRegressor(),
|
||||
single,
|
||||
),
|
||||
)
|
||||
for name, model, train in candidates:
|
||||
result = score_model(name, model, train, single, grid)
|
||||
result["tp"] = tp
|
||||
records.append(result)
|
||||
|
||||
structured = [r for r in records if r["candidate"] == "structured_single"]
|
||||
checks = {
|
||||
"heldout_context_mape_le_5pct": all(
|
||||
r["heldout_context_mape_pct"] <= 5 for r in structured
|
||||
),
|
||||
"monotone_and_nonnegative": all(
|
||||
r["monotone_and_nonnegative"] for r in structured
|
||||
),
|
||||
}
|
||||
checks["profile_gate"] = all(checks.values())
|
||||
payload = {
|
||||
"schema": "frontier-attn-structured-ablation-v1",
|
||||
"profile": str(args.profile.resolve()),
|
||||
"frontier_checkout": str(args.frontier_checkout.resolve()),
|
||||
"data_audit": data_audit,
|
||||
"results": records,
|
||||
"checks": checks,
|
||||
}
|
||||
|
||||
args.output_root.mkdir(parents=True, exist_ok=True)
|
||||
(args.output_root / "predictor-ablation.json").write_text(
|
||||
json.dumps(payload, indent=2)
|
||||
)
|
||||
with (args.output_root / "predictor-ablation.csv").open(
|
||||
"w", newline=""
|
||||
) as stream:
|
||||
writer = csv.DictWriter(stream, fieldnames=list(records[0]))
|
||||
writer.writeheader()
|
||||
writer.writerows(records)
|
||||
print(json.dumps(checks, indent=2))
|
||||
for row in records:
|
||||
print(
|
||||
f"TP{row['tp']} {row['candidate']:18s} "
|
||||
f"grid={row['grid_fit_mape_pct']:.2f}% "
|
||||
f"heldout={row['heldout_context_mape_pct']:.2f}% "
|
||||
f"max={row['heldout_context_max_abs_error_pct']:.2f}% "
|
||||
f"monotone={row['monotone_and_nonnegative']}"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
304
runs/frontier-attn-structured-v0/analyze_trace_verdict.py
Normal file
304
runs/frontier-attn-structured-v0/analyze_trace_verdict.py
Normal file
@@ -0,0 +1,304 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Trial-aware verdict for the seven structured-attention trace replays."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import csv
|
||||
import json
|
||||
import math
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
ROOT = Path(__file__).resolve().parent
|
||||
REPO = ROOT.parents[1]
|
||||
S3_REAL = REPO / "runs/frontier-s3-real-v0"
|
||||
V5 = REPO / "runs/frontier-prefill-kvgrowth-fix-v0"
|
||||
|
||||
CELLS = {
|
||||
"tp1_rho0p00125": {
|
||||
"real": "frontier-tp1-real-r0p00125-t*",
|
||||
"old": V5 / "sim-replay-tp1/v5/tp1_rho0p00125",
|
||||
},
|
||||
"tp1_rho0p0025": {
|
||||
"real": "frontier-tp1-real-r0p0025-t*",
|
||||
"old": V5 / "sim-replay-tp1/v5/tp1_rho0p0025",
|
||||
},
|
||||
"tp2_rho0p0025": {
|
||||
"real": "frontier-s3-real-full-r0p0025-tp2-t*",
|
||||
"old": V5 / "sim-replay/tp2_rho0p0025",
|
||||
},
|
||||
"tp2_rho0p005": {
|
||||
"real": "frontier-s3-real-full-r0p005-tp2-t*",
|
||||
"old": V5 / "sim-replay/tp2_rho0p005",
|
||||
},
|
||||
"tp4_rho0p0025": {
|
||||
"real": "frontier-s3-real-full-r0p0025-tp4-t*",
|
||||
"old": V5 / "sim-replay/tp4_rho0p0025",
|
||||
},
|
||||
"tp4_rho0p005": {
|
||||
"real": "frontier-s3-real-full-r0p005-tp4-t*",
|
||||
"old": V5 / "sim-replay/tp4_rho0p005",
|
||||
},
|
||||
"tp4_rho0p01": {
|
||||
"real": "frontier-s3-real-full-r0p01-tp4-t*",
|
||||
"old": V5 / "sim-replay/tp4_rho0p01",
|
||||
},
|
||||
}
|
||||
METRICS = {
|
||||
"ttft": ("ttft_ms", "ttft"),
|
||||
"tpot": ("tpot_ms", "tpot"),
|
||||
"e2e": ("e2e_ms", "request_e2e_time"),
|
||||
}
|
||||
QUANTILES = {"mean": None, "p50": 0.5, "p90": 0.9, "p99": 0.99}
|
||||
|
||||
|
||||
def percentile(values: list[float], quantile: float) -> float:
|
||||
ordered = sorted(values)
|
||||
position = (len(ordered) - 1) * quantile
|
||||
lower, upper = math.floor(position), math.ceil(position)
|
||||
if lower == upper:
|
||||
return ordered[lower]
|
||||
return (
|
||||
ordered[lower] * (upper - position)
|
||||
+ ordered[upper] * (position - lower)
|
||||
)
|
||||
|
||||
|
||||
def summarize(values: list[float]) -> dict[str, float]:
|
||||
return {
|
||||
name: (
|
||||
sum(values) / len(values)
|
||||
if quantile is None
|
||||
else percentile(values, quantile)
|
||||
)
|
||||
for name, quantile in QUANTILES.items()
|
||||
}
|
||||
|
||||
|
||||
def load_real_trials(pattern: str) -> list[list[dict[str, Any]]]:
|
||||
trials = []
|
||||
for run_root in sorted((S3_REAL / "fleet-artifacts").glob(pattern)):
|
||||
results = list(
|
||||
run_root.glob(
|
||||
"artifacts/outputs/full-real/*/*/trial-*/results/result.json"
|
||||
)
|
||||
)
|
||||
if len(results) != 1:
|
||||
raise ValueError(f"expected one result in {run_root}, got {results}")
|
||||
trials.append(json.loads(results[0].read_text())["requests"])
|
||||
if len(trials) != 2:
|
||||
raise ValueError(f"expected two real trials for {pattern}, got {len(trials)}")
|
||||
return trials
|
||||
|
||||
|
||||
def load_sim(root: Path) -> list[dict[str, str]]:
|
||||
matches = list((root / "metrics").rglob("request_metrics.csv"))
|
||||
if len(matches) != 1:
|
||||
raise ValueError(f"expected one request_metrics.csv below {root}: {matches}")
|
||||
rows = list(csv.DictReader(matches[0].open()))
|
||||
rows.sort(key=lambda row: int(float(row["Request Id"])))
|
||||
return rows
|
||||
|
||||
|
||||
def distribution_bias(
|
||||
real_rows: list[dict[str, Any]],
|
||||
sim_rows: list[dict[str, str]],
|
||||
) -> dict[str, dict[str, float]]:
|
||||
output: dict[str, dict[str, float]] = {}
|
||||
for metric, (real_key, sim_key) in METRICS.items():
|
||||
pairs = [
|
||||
(float(real[real_key]), float(sim[sim_key]))
|
||||
for real, sim in zip(real_rows, sim_rows)
|
||||
if real.get("success")
|
||||
]
|
||||
real_summary = summarize([pair[0] for pair in pairs])
|
||||
sim_summary = summarize([pair[1] for pair in pairs])
|
||||
output[metric] = {
|
||||
name: (sim_summary[name] - real_summary[name]) / real_summary[name]
|
||||
for name in QUANTILES
|
||||
}
|
||||
return output
|
||||
|
||||
|
||||
def paired_relative_error(
|
||||
real_rows: list[dict[str, Any]],
|
||||
sim_rows: list[dict[str, str]],
|
||||
) -> dict[str, dict[str, float]]:
|
||||
output: dict[str, dict[str, float]] = {}
|
||||
for metric, (real_key, sim_key) in METRICS.items():
|
||||
errors = [
|
||||
(float(sim[sim_key]) - float(real[real_key])) / float(real[real_key])
|
||||
for real, sim in zip(real_rows, sim_rows)
|
||||
if real.get("success") and float(real[real_key]) != 0
|
||||
]
|
||||
output[metric] = summarize(errors)
|
||||
return output
|
||||
|
||||
|
||||
def aggregate_trial_bias(
|
||||
trial_biases: list[dict[str, dict[str, float]]],
|
||||
) -> dict[str, dict[str, dict[str, float]]]:
|
||||
return {
|
||||
metric: {
|
||||
quantile: {
|
||||
"mean": sum(values) / len(values),
|
||||
"min": min(values),
|
||||
"max": max(values),
|
||||
}
|
||||
for quantile in QUANTILES
|
||||
for values in [
|
||||
[trial[metric][quantile] for trial in trial_biases]
|
||||
]
|
||||
}
|
||||
for metric in METRICS
|
||||
}
|
||||
|
||||
|
||||
def legacy_pooled_bias(
|
||||
real_trials: list[list[dict[str, Any]]],
|
||||
sim_rows: list[dict[str, str]],
|
||||
) -> dict[str, dict[str, float]]:
|
||||
output: dict[str, dict[str, float]] = {}
|
||||
for metric, (real_key, sim_key) in METRICS.items():
|
||||
real_values = [
|
||||
float(row[real_key])
|
||||
for trial in real_trials
|
||||
for row in trial[: len(sim_rows)]
|
||||
if row.get("success")
|
||||
]
|
||||
sim_values = [float(row[sim_key]) for row in sim_rows]
|
||||
real_summary = summarize(real_values)
|
||||
sim_summary = summarize(sim_values)
|
||||
output[metric] = {
|
||||
name: (sim_summary[name] - real_summary[name]) / real_summary[name]
|
||||
for name in QUANTILES
|
||||
}
|
||||
return output
|
||||
|
||||
|
||||
def waiting_p99(sim_rows: list[dict[str, str]]) -> float:
|
||||
return percentile(
|
||||
[float(row["request_waiting_time_total"]) for row in sim_rows], 0.99
|
||||
)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
results: dict[str, Any] = {}
|
||||
flat_rows: list[dict[str, Any]] = []
|
||||
for label, paths in CELLS.items():
|
||||
real_trials = load_real_trials(paths["real"])
|
||||
old_sim = load_sim(paths["old"])
|
||||
new_sim = load_sim(ROOT / "replay" / label)
|
||||
old_trial_bias = [
|
||||
distribution_bias(trial, old_sim) for trial in real_trials
|
||||
]
|
||||
new_trial_bias = [
|
||||
distribution_bias(trial, new_sim) for trial in real_trials
|
||||
]
|
||||
old_legacy = legacy_pooled_bias(real_trials, old_sim)
|
||||
new_legacy = legacy_pooled_bias(real_trials, new_sim)
|
||||
wait_p99 = waiting_p99(new_sim)
|
||||
results[label] = {
|
||||
"old": {
|
||||
"trialwise_distribution_bias": old_trial_bias,
|
||||
"trialwise_distribution_bias_summary": aggregate_trial_bias(
|
||||
old_trial_bias
|
||||
),
|
||||
"legacy_pooled_distribution_bias": old_legacy,
|
||||
},
|
||||
"new": {
|
||||
"trialwise_distribution_bias": new_trial_bias,
|
||||
"trialwise_distribution_bias_summary": aggregate_trial_bias(
|
||||
new_trial_bias
|
||||
),
|
||||
"paired_relative_error": [
|
||||
paired_relative_error(trial, new_sim)
|
||||
for trial in real_trials
|
||||
],
|
||||
"legacy_pooled_distribution_bias": new_legacy,
|
||||
"waiting_p99_ms": wait_p99,
|
||||
"validity": (
|
||||
"PASS_SUBCRITICAL"
|
||||
if wait_p99 < 1000
|
||||
else "GATE_FAIL_DIAGNOSTIC"
|
||||
),
|
||||
},
|
||||
}
|
||||
for metric in METRICS:
|
||||
for quantile in QUANTILES:
|
||||
flat_rows.append(
|
||||
{
|
||||
"cell": label,
|
||||
"metric": metric,
|
||||
"quantile": quantile,
|
||||
"old_bias": old_legacy[metric][quantile],
|
||||
"new_bias": new_legacy[metric][quantile],
|
||||
"abs_bias_delta_pp": 100
|
||||
* (
|
||||
abs(new_legacy[metric][quantile])
|
||||
- abs(old_legacy[metric][quantile])
|
||||
),
|
||||
"validity": results[label]["new"]["validity"],
|
||||
}
|
||||
)
|
||||
|
||||
tp1_checks = []
|
||||
for cell in ("tp1_rho0p00125", "tp1_rho0p0025"):
|
||||
for quantile in ("mean", "p99"):
|
||||
old = results[cell]["old"]["legacy_pooled_distribution_bias"]["ttft"][
|
||||
quantile
|
||||
]
|
||||
new = results[cell]["new"]["legacy_pooled_distribution_bias"]["ttft"][
|
||||
quantile
|
||||
]
|
||||
tp1_checks.append(abs(old) - abs(new) >= 0.05)
|
||||
|
||||
regressions = [
|
||||
row
|
||||
for row in flat_rows
|
||||
if row["cell"].startswith(("tp2", "tp4"))
|
||||
and row["metric"] in ("ttft", "e2e")
|
||||
and row["abs_bias_delta_pp"] > 5
|
||||
]
|
||||
checks = {
|
||||
"tp1_ttft_mean_p99_improve_ge_5pp": all(tp1_checks),
|
||||
"tp2_tp4_ttft_e2e_no_abs_regression_gt_5pp": not regressions,
|
||||
"regressions": regressions,
|
||||
}
|
||||
checks["trace_gate"] = (
|
||||
checks["tp1_ttft_mean_p99_improve_ge_5pp"]
|
||||
and checks["tp2_tp4_ttft_e2e_no_abs_regression_gt_5pp"]
|
||||
)
|
||||
|
||||
payload = {
|
||||
"schema": "frontier-attn-structured-trial-aware-verdict-v1",
|
||||
"metric_note": (
|
||||
"Primary values are per-real-trial distribution biases with request "
|
||||
"alignment by index. legacy_pooled reproduces the old milestone "
|
||||
"quantile convention only for direct comparison."
|
||||
),
|
||||
"cells": results,
|
||||
"checks": checks,
|
||||
}
|
||||
output = ROOT / "results"
|
||||
output.mkdir(parents=True, exist_ok=True)
|
||||
(output / "trace-verdict.json").write_text(json.dumps(payload, indent=2))
|
||||
with (output / "trace-verdict.csv").open("w", newline="") as stream:
|
||||
writer = csv.DictWriter(stream, fieldnames=list(flat_rows[0]))
|
||||
writer.writeheader()
|
||||
writer.writerows(flat_rows)
|
||||
|
||||
print(json.dumps(checks, indent=2))
|
||||
for label, result in results.items():
|
||||
old = result["old"]["legacy_pooled_distribution_bias"]["ttft"]
|
||||
new = result["new"]["legacy_pooled_distribution_bias"]["ttft"]
|
||||
print(
|
||||
f"{label}: TTFT mean {old['mean']:+.1%}->{new['mean']:+.1%}, "
|
||||
f"p99 {old['p99']:+.1%}->{new['p99']:+.1%}, "
|
||||
f"waiting_p99={result['new']['waiting_p99_ms']:.0f}ms "
|
||||
f"{result['new']['validity']}"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
82
runs/frontier-attn-structured-v0/experiment-card.md
Normal file
82
runs/frontier-attn-structured-v0/experiment-card.md
Normal file
@@ -0,0 +1,82 @@
|
||||
# 实验 EXP-ATTN-STRUCTURED:结构化 predictor 能否关闭大 KV 端的 RF 欠拟合
|
||||
|
||||
> **状态:** 已完成(profile gate PASS;global merge gate FAIL)
|
||||
>
|
||||
> Parent campaign:[`../frontier-simulator-gap-campaign-v0/README.md`](../frontier-simulator-gap-campaign-v0/README.md)
|
||||
|
||||
## Claim 与决策
|
||||
|
||||
- **Parent claim:** profile-v5 已补齐 chunked-prefill KV-context 测量,但当前 RF 仍在 TP1/2/4 的新网格上产生约 12%--14% self-fit MAPE,并在 TP1 真实 trace 中留下 −13% 到 −22% TTFT 偏差。
|
||||
- **目的:** 检查该 residual 是否来自可工程修复的 predictor representation,而不是 profile 数据或 serving path。
|
||||
- **Competing hypotheses:**
|
||||
- H1:standard prefill 模型错误混入 pure multi-request rows,且 RF 对连续 attention scaling 作阶梯平滑;使用单请求数据和结构化 `base(q)+KV×(a+bq)` 可关闭残余。
|
||||
- H2:残余主要来自未建模的 serving-path 组件;替换 predictor 不会改善 7-cell trace fidelity。
|
||||
- **事前预测:**
|
||||
- H1:held-out context MAPE ≤5%,TP1 TTFT mean/p99 绝对偏差至少改善 5 pp。
|
||||
- H2:profile gate 失败,或 profile gate 通过但 trace TTFT 几乎不动。
|
||||
- **判定规则:**
|
||||
- profile gate:TP1/2/4 held-out context MAPE 均 ≤5%;q/KV 单调且预测非负。
|
||||
- trace gate:两个 TP1 cell 的 TTFT mean/p99 |bias| 各改善 ≥5 pp;TP2/TP4 任一 TTFT/E2E quantile 不恶化 >5 pp。
|
||||
- profile gate 失败即停止;trace gate 失败则回退 patch,不进入 EXP-2。
|
||||
|
||||
## Setup
|
||||
|
||||
- **自变量:**
|
||||
- A:现有 RF,standard prefill 全部非 true-mixed rows。
|
||||
- B:现有 RF,但仅 `batch_size=1`。
|
||||
- C:仅 `batch_size=1` 的 structured predictor:
|
||||
- `base(q)`:KV=0 profile 的单调分段线性插值;
|
||||
- growth:非负 least-squares `KV×(a+bq)`。
|
||||
- **控制变量:** attention/linear/MoE/collective profile、trace、prefix cache、scheduler、graph mode、KV blocks、MNS、全部 argv。
|
||||
- **System context:** Qwen3-30B-A3B BF16;H20;Frontier `deadc4a3`;TP1/2/4;MNS16;chunk 8192;prefix caching。
|
||||
- **Workload 或 trace:** 现有 7-cell 60-min production chat trace matrix;real 侧每 cell 两个 trial。
|
||||
- **Baselines:** `docs/assets/frontier-fidelity/full-matrix.csv` 的 sim-v5。
|
||||
- **Metrics:**
|
||||
- profile:grid fit MAPE、leave-one-context MAPE/max error、q/KV monotonicity;
|
||||
- trace:request-ID paired bias;每个 real trial 单独计算后报告 mean 与 trial interval;
|
||||
- queue validity:waiting p99,TP1 超过 1 s 的 cell 标为 diagnostic。
|
||||
|
||||
## 预期产物与 review
|
||||
|
||||
- **预期数据:** `results/predictor-ablation.{json,csv}`、`replay/<cell>/`、`results/paired-verdict.json`。
|
||||
- **Figure prototype:** `figure-prototype.png`;左图为 q8k 随 KV 增长的 actual/RF/structured,右图为 7-cell TTFT bias 的事前期望。
|
||||
- **人工 review:** 已按 campaign 顺序批准执行。
|
||||
- **Review 意见:** 只改 standard single-request predictor;不得改 mixed predictor 或任何 profile row。
|
||||
|
||||
## 复现信息
|
||||
|
||||
- **Code:** Frontier base `deadc4a321f0baaa534c6ebd17f974123733cdc2`;实验 patch 将保存为 `frontier-structured-attn.patch` 并记录 SHA256。
|
||||
- **Environment:** 本地 CPU replay;Python dependency roots 复用 `runs/frontier-collective-joint-v0/counterfactual/joint-r2/manifest.json`。
|
||||
- **产物路径:** 本目录。
|
||||
- **已知 deviation:** milestone 文档将 7-cell 口径称为“逐 request paired”,但旧脚本实际 pool 两个 real trial 后比较 quantile;本实验会修正分析口径,不改旧结果文件。
|
||||
|
||||
## 预分析事实
|
||||
|
||||
- 现有训练代码使用 `["kv_cache_size", "prefill_chunk_size_squared"]` 与 RF grid search。
|
||||
- runtime cache 注释明确 standard model 是 per-request;多请求 prefill 在模型存在时走 `attn_prefill_mixed`。
|
||||
- profile-v5 的 standard 训练集每 TP 有 29 行,其中单请求 23 行;有 4 组相同 `(KV,q²)` feature 对应多个 pure-batch 标签。
|
||||
- 初步 structured candidate 的 leave-one-context MAPE:TP1 0.84%、TP2 1.61%、TP4 3.01%;max error 分别 2.04%、3.47%、5.49%。这些是实现前的临时计算,须由版本化脚本复现后才进入结果。
|
||||
|
||||
## 结果
|
||||
|
||||
- **观察事实:**
|
||||
- structured held-out-context MAPE 为 TP1/2/4=`0.84%/1.60%/3.01%`;
|
||||
当前 RF 为 `44.40%/44.20%/43.54%`。单调/非负 gate 通过。
|
||||
- TP1 两点 TTFT mean bias `−13.5/−17.7% → −6.3/−9.5%`,p99
|
||||
`−16.8/−22.1% → −8.6/−14.4%`。
|
||||
- TP2 两点 TTFT mean bias `−11.2/−14.1% → −4.5/−7.1%`,p99
|
||||
`−17.3/−19.4% → −7.7/−9.0%`。
|
||||
- TP4 三点 TTFT mean bias `+2.5/+2.9/−0.1% → +7.8/+8.4/+6.0%`;
|
||||
三点均使绝对误差恶化 `5.3--5.8 pp`,触发预设回归 gate。
|
||||
- validity 重新审计:TP1 两点 waiting p99=`1.34/1.89 s`;TP2
|
||||
ρ=.005=`1.17 s`,均标为 `GATE_FAIL_DIAGNOSTIC`。其余四点通过。
|
||||
- **异常:** TP4 ρ=.005 的 TPOT p99 从 `+31.2%` 变为 `+36.6%`,
|
||||
表明该 tail 对 prefill/mixed-decode 相位敏感,不是本 patch 能关闭的稳定
|
||||
decode predictor 偏差。
|
||||
- **含义:** H1 的 representation 机制得到支持,但“全局替换 RF 可直接提升
|
||||
7-cell fidelity”被反驳。TP4 原先接近零的 mean TTFT 含有 predictor
|
||||
欠拟合与其它正向 residual 的误差抵消;单独修正 attention 会揭开后者。
|
||||
- **Claim update:** structured predictor 是明确的工程候选,但必须与 TP4
|
||||
residual 联合收敛后才可 merge;当前 patch 只保留为 ablation。
|
||||
- **下一步:** EXP-2 先重算 structured 分支的 TP2 chunk-level residual;
|
||||
仅 residual ≥10% 才运行 GPU serving-path 三臂 profile。
|
||||
BIN
runs/frontier-attn-structured-v0/figure-prototype.png
Normal file
BIN
runs/frontier-attn-structured-v0/figure-prototype.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 119 KiB |
525
runs/frontier-attn-structured-v0/frontier-reference.json
Normal file
525
runs/frontier-attn-structured-v0/frontier-reference.json
Normal file
@@ -0,0 +1,525 @@
|
||||
{
|
||||
"cc_cache": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/cc-cache",
|
||||
"cells": {
|
||||
"tp1_mns16": {
|
||||
"argv": [
|
||||
"/usr/bin/python3",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/run_frontier_with_curves.py",
|
||||
"--simulation_mode",
|
||||
"online",
|
||||
"--sys_arch",
|
||||
"co-location",
|
||||
"--cc_backend_config_type",
|
||||
"vidur",
|
||||
"--cluster_config_num_replicas",
|
||||
"1",
|
||||
"--cluster_scheduler_config_type",
|
||||
"sticky_round_robin",
|
||||
"--replica_config_model_name",
|
||||
"qwen3-a3b-30b-moe",
|
||||
"--replica_config_device",
|
||||
"h20",
|
||||
"--replica_config_network_device",
|
||||
"h20_dgx",
|
||||
"--replica_config_attn_tensor_parallel_size",
|
||||
"1",
|
||||
"--replica_config_attn_data_parallel_size",
|
||||
"1",
|
||||
"--replica_config_moe_tensor_parallel_size",
|
||||
"1",
|
||||
"--replica_config_moe_expert_parallel_size",
|
||||
"1",
|
||||
"--replica_config_num_pipeline_stages",
|
||||
"1",
|
||||
"--replica_scheduler_config_type",
|
||||
"vllm_v1",
|
||||
"--decode_cuda_graph_mode",
|
||||
"piecewise",
|
||||
"--vllm_v1_scheduler_config_batch_size_cap",
|
||||
"16",
|
||||
"--vllm_v1_scheduler_config_max_tokens_in_batch",
|
||||
"8192",
|
||||
"--vllm_v1_scheduler_config_long_prefill_token_threshold",
|
||||
"0",
|
||||
"--vllm_v1_scheduler_config_block_size",
|
||||
"16",
|
||||
"--vllm_v1_scheduler_config_num_blocks_mode",
|
||||
"explicit",
|
||||
"--vllm_v1_scheduler_config_gpu_memory_utilization",
|
||||
"0.92",
|
||||
"--vllm_v1_scheduler_config_non_kv_cache_overhead_bytes",
|
||||
"0",
|
||||
"--request_generator_config_type",
|
||||
"trace_replay",
|
||||
"--trace_request_generator_config_trace_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp1-frontier.csv",
|
||||
"--trace_request_generator_config_max_tokens",
|
||||
"40960",
|
||||
"--metrics_config_output_dir",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/sim/tp1_mns16/metrics",
|
||||
"--metrics_config_run_id",
|
||||
"joint_tp1_mns16",
|
||||
"--metrics_config_write_metrics",
|
||||
"--metrics_config_store_request_metrics",
|
||||
"--metrics_config_store_batch_metrics",
|
||||
"--metrics_config_store_token_completion_metrics",
|
||||
"--metrics_config_store_utilization_metrics",
|
||||
"--no-metrics_config_store_plots",
|
||||
"--no-metrics_config_enable_chrome_trace",
|
||||
"--no-metrics_config_write_json_trace",
|
||||
"--metrics_config_store_frontier_stage_batch_ledger",
|
||||
"--no-random_forrest_execution_time_predictor_config_enable_dummy_mode",
|
||||
"--random_forrest_execution_time_predictor_config_linear_op_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/linear_op.csv",
|
||||
"--random_forrest_execution_time_predictor_config_atten_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/attention.csv",
|
||||
"--random_forrest_execution_time_predictor_config_moe_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/moe.csv",
|
||||
"--random_forrest_execution_time_predictor_config_linear_op_kernel_only_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/linear_op.csv",
|
||||
"--random_forrest_execution_time_predictor_config_atten_kernel_only_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/attention.csv",
|
||||
"--random_forrest_execution_time_predictor_config_moe_kernel_only_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/moe.csv",
|
||||
"--random_forrest_execution_time_predictor_config_prediction_max_prefill_chunk_size",
|
||||
"8192",
|
||||
"--random_forrest_execution_time_predictor_config_prediction_max_batch_size",
|
||||
"32",
|
||||
"--random_forrest_execution_time_predictor_config_prediction_max_tokens_per_request",
|
||||
"40960",
|
||||
"--random_forrest_execution_time_predictor_config_no_cache",
|
||||
"--random_forrest_execution_time_predictor_config_skip_cpu_overhead_modeling",
|
||||
"--vllm_v1_scheduler_config_num_blocks",
|
||||
"20128",
|
||||
"--vllm_v1_scheduler_config_enable_chunked_prefill",
|
||||
"--random_forrest_execution_time_predictor_config_num_training_job_threads",
|
||||
"4",
|
||||
"--cudagraph_capture_sizes",
|
||||
"1",
|
||||
"2",
|
||||
"4",
|
||||
"8",
|
||||
"16",
|
||||
"24",
|
||||
"32",
|
||||
"--vidur_cc_backend_config_all_reduce_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/measured-allreduce.csv",
|
||||
"--vidur_cc_backend_config_cache_dir",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/cc-cache",
|
||||
"--vidur_cc_backend_config_k_fold_cv_splits",
|
||||
"6",
|
||||
"--vidur_cc_backend_config_num_training_job_threads",
|
||||
"1",
|
||||
"--metrics_config_cache_dir",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/model-cache"
|
||||
],
|
||||
"log": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/logs/tp1_mns16.log",
|
||||
"source_command": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp1_mns16/tp1/command.json",
|
||||
"source_command_sha256": "a9815797b1601bf6f6cdf0269e84acb376a84945609e338868dc8347aab650e6",
|
||||
"usage": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/usage/tp1_mns16.json"
|
||||
},
|
||||
"tp2_mns16": {
|
||||
"argv": [
|
||||
"/usr/bin/python3",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/run_frontier_with_curves.py",
|
||||
"--simulation_mode",
|
||||
"online",
|
||||
"--sys_arch",
|
||||
"co-location",
|
||||
"--cc_backend_config_type",
|
||||
"vidur",
|
||||
"--cluster_config_num_replicas",
|
||||
"1",
|
||||
"--cluster_scheduler_config_type",
|
||||
"sticky_round_robin",
|
||||
"--replica_config_model_name",
|
||||
"qwen3-a3b-30b-moe",
|
||||
"--replica_config_device",
|
||||
"h20",
|
||||
"--replica_config_network_device",
|
||||
"h20_dgx",
|
||||
"--replica_config_attn_tensor_parallel_size",
|
||||
"2",
|
||||
"--replica_config_attn_data_parallel_size",
|
||||
"1",
|
||||
"--replica_config_moe_tensor_parallel_size",
|
||||
"2",
|
||||
"--replica_config_moe_expert_parallel_size",
|
||||
"1",
|
||||
"--replica_config_num_pipeline_stages",
|
||||
"1",
|
||||
"--replica_scheduler_config_type",
|
||||
"vllm_v1",
|
||||
"--decode_cuda_graph_mode",
|
||||
"piecewise",
|
||||
"--vllm_v1_scheduler_config_batch_size_cap",
|
||||
"16",
|
||||
"--vllm_v1_scheduler_config_max_tokens_in_batch",
|
||||
"8192",
|
||||
"--vllm_v1_scheduler_config_long_prefill_token_threshold",
|
||||
"0",
|
||||
"--vllm_v1_scheduler_config_block_size",
|
||||
"16",
|
||||
"--vllm_v1_scheduler_config_num_blocks_mode",
|
||||
"explicit",
|
||||
"--vllm_v1_scheduler_config_gpu_memory_utilization",
|
||||
"0.92",
|
||||
"--vllm_v1_scheduler_config_non_kv_cache_overhead_bytes",
|
||||
"0",
|
||||
"--request_generator_config_type",
|
||||
"trace_replay",
|
||||
"--trace_request_generator_config_trace_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp2-frontier.csv",
|
||||
"--trace_request_generator_config_max_tokens",
|
||||
"40960",
|
||||
"--metrics_config_output_dir",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/sim/tp2_mns16/metrics",
|
||||
"--metrics_config_run_id",
|
||||
"joint_tp2_mns16",
|
||||
"--metrics_config_write_metrics",
|
||||
"--metrics_config_store_request_metrics",
|
||||
"--metrics_config_store_batch_metrics",
|
||||
"--metrics_config_store_token_completion_metrics",
|
||||
"--metrics_config_store_utilization_metrics",
|
||||
"--no-metrics_config_store_plots",
|
||||
"--no-metrics_config_enable_chrome_trace",
|
||||
"--no-metrics_config_write_json_trace",
|
||||
"--metrics_config_store_frontier_stage_batch_ledger",
|
||||
"--no-random_forrest_execution_time_predictor_config_enable_dummy_mode",
|
||||
"--random_forrest_execution_time_predictor_config_linear_op_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/linear_op.csv",
|
||||
"--random_forrest_execution_time_predictor_config_atten_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/attention.csv",
|
||||
"--random_forrest_execution_time_predictor_config_moe_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/moe.csv",
|
||||
"--random_forrest_execution_time_predictor_config_linear_op_kernel_only_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/linear_op.csv",
|
||||
"--random_forrest_execution_time_predictor_config_atten_kernel_only_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/attention.csv",
|
||||
"--random_forrest_execution_time_predictor_config_moe_kernel_only_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/moe.csv",
|
||||
"--random_forrest_execution_time_predictor_config_prediction_max_prefill_chunk_size",
|
||||
"8192",
|
||||
"--random_forrest_execution_time_predictor_config_prediction_max_batch_size",
|
||||
"32",
|
||||
"--random_forrest_execution_time_predictor_config_prediction_max_tokens_per_request",
|
||||
"40960",
|
||||
"--random_forrest_execution_time_predictor_config_no_cache",
|
||||
"--random_forrest_execution_time_predictor_config_skip_cpu_overhead_modeling",
|
||||
"--vllm_v1_scheduler_config_num_blocks",
|
||||
"76620",
|
||||
"--vllm_v1_scheduler_config_enable_chunked_prefill",
|
||||
"--random_forrest_execution_time_predictor_config_num_training_job_threads",
|
||||
"4",
|
||||
"--cudagraph_capture_sizes",
|
||||
"1",
|
||||
"2",
|
||||
"4",
|
||||
"8",
|
||||
"16",
|
||||
"24",
|
||||
"32",
|
||||
"--vidur_cc_backend_config_all_reduce_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/measured-allreduce.csv",
|
||||
"--vidur_cc_backend_config_cache_dir",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/cc-cache",
|
||||
"--vidur_cc_backend_config_k_fold_cv_splits",
|
||||
"6",
|
||||
"--vidur_cc_backend_config_num_training_job_threads",
|
||||
"1",
|
||||
"--metrics_config_cache_dir",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/model-cache"
|
||||
],
|
||||
"log": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/logs/tp2_mns16.log",
|
||||
"source_command": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp2_mns16/tp2/command.json",
|
||||
"source_command_sha256": "61788a8810be301c9dbc006624aa19b6a932bc44d341b836861087833cffc3df",
|
||||
"usage": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/usage/tp2_mns16.json"
|
||||
},
|
||||
"tp4_mns16": {
|
||||
"argv": [
|
||||
"/usr/bin/python3",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/run_frontier_with_curves.py",
|
||||
"--simulation_mode",
|
||||
"online",
|
||||
"--sys_arch",
|
||||
"co-location",
|
||||
"--cc_backend_config_type",
|
||||
"vidur",
|
||||
"--cluster_config_num_replicas",
|
||||
"1",
|
||||
"--cluster_scheduler_config_type",
|
||||
"sticky_round_robin",
|
||||
"--replica_config_model_name",
|
||||
"qwen3-a3b-30b-moe",
|
||||
"--replica_config_device",
|
||||
"h20",
|
||||
"--replica_config_network_device",
|
||||
"h20_dgx",
|
||||
"--replica_config_attn_tensor_parallel_size",
|
||||
"4",
|
||||
"--replica_config_attn_data_parallel_size",
|
||||
"1",
|
||||
"--replica_config_moe_tensor_parallel_size",
|
||||
"4",
|
||||
"--replica_config_moe_expert_parallel_size",
|
||||
"1",
|
||||
"--replica_config_num_pipeline_stages",
|
||||
"1",
|
||||
"--replica_scheduler_config_type",
|
||||
"vllm_v1",
|
||||
"--decode_cuda_graph_mode",
|
||||
"piecewise",
|
||||
"--vllm_v1_scheduler_config_batch_size_cap",
|
||||
"16",
|
||||
"--vllm_v1_scheduler_config_max_tokens_in_batch",
|
||||
"8192",
|
||||
"--vllm_v1_scheduler_config_long_prefill_token_threshold",
|
||||
"0",
|
||||
"--vllm_v1_scheduler_config_block_size",
|
||||
"16",
|
||||
"--vllm_v1_scheduler_config_num_blocks_mode",
|
||||
"explicit",
|
||||
"--vllm_v1_scheduler_config_gpu_memory_utilization",
|
||||
"0.92",
|
||||
"--vllm_v1_scheduler_config_non_kv_cache_overhead_bytes",
|
||||
"0",
|
||||
"--request_generator_config_type",
|
||||
"trace_replay",
|
||||
"--trace_request_generator_config_trace_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp4-frontier.csv",
|
||||
"--trace_request_generator_config_max_tokens",
|
||||
"40960",
|
||||
"--metrics_config_output_dir",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/sim/tp4_mns16/metrics",
|
||||
"--metrics_config_run_id",
|
||||
"joint_tp4_mns16",
|
||||
"--metrics_config_write_metrics",
|
||||
"--metrics_config_store_request_metrics",
|
||||
"--metrics_config_store_batch_metrics",
|
||||
"--metrics_config_store_token_completion_metrics",
|
||||
"--metrics_config_store_utilization_metrics",
|
||||
"--no-metrics_config_store_plots",
|
||||
"--no-metrics_config_enable_chrome_trace",
|
||||
"--no-metrics_config_write_json_trace",
|
||||
"--metrics_config_store_frontier_stage_batch_ledger",
|
||||
"--no-random_forrest_execution_time_predictor_config_enable_dummy_mode",
|
||||
"--random_forrest_execution_time_predictor_config_linear_op_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/linear_op.csv",
|
||||
"--random_forrest_execution_time_predictor_config_atten_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/attention.csv",
|
||||
"--random_forrest_execution_time_predictor_config_moe_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/moe.csv",
|
||||
"--random_forrest_execution_time_predictor_config_linear_op_kernel_only_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/linear_op.csv",
|
||||
"--random_forrest_execution_time_predictor_config_atten_kernel_only_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/attention.csv",
|
||||
"--random_forrest_execution_time_predictor_config_moe_kernel_only_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/moe.csv",
|
||||
"--random_forrest_execution_time_predictor_config_prediction_max_prefill_chunk_size",
|
||||
"8192",
|
||||
"--random_forrest_execution_time_predictor_config_prediction_max_batch_size",
|
||||
"32",
|
||||
"--random_forrest_execution_time_predictor_config_prediction_max_tokens_per_request",
|
||||
"40960",
|
||||
"--random_forrest_execution_time_predictor_config_no_cache",
|
||||
"--random_forrest_execution_time_predictor_config_skip_cpu_overhead_modeling",
|
||||
"--vllm_v1_scheduler_config_num_blocks",
|
||||
"191882",
|
||||
"--vllm_v1_scheduler_config_enable_chunked_prefill",
|
||||
"--random_forrest_execution_time_predictor_config_num_training_job_threads",
|
||||
"4",
|
||||
"--cudagraph_capture_sizes",
|
||||
"1",
|
||||
"2",
|
||||
"4",
|
||||
"8",
|
||||
"16",
|
||||
"24",
|
||||
"32",
|
||||
"--vidur_cc_backend_config_all_reduce_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/measured-allreduce.csv",
|
||||
"--vidur_cc_backend_config_cache_dir",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/cc-cache",
|
||||
"--vidur_cc_backend_config_k_fold_cv_splits",
|
||||
"6",
|
||||
"--vidur_cc_backend_config_num_training_job_threads",
|
||||
"1",
|
||||
"--metrics_config_cache_dir",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/model-cache"
|
||||
],
|
||||
"log": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/logs/tp4_mns16.log",
|
||||
"source_command": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp4_mns16/tp4/command.json",
|
||||
"source_command_sha256": "9bbcf10446336ba5885193f391dd628cd18ff64d91a51ebcd463ffd24be95532",
|
||||
"usage": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/usage/tp4_mns16.json"
|
||||
},
|
||||
"tp4_mns32": {
|
||||
"argv": [
|
||||
"/usr/bin/python3",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/run_frontier_with_curves.py",
|
||||
"--simulation_mode",
|
||||
"online",
|
||||
"--sys_arch",
|
||||
"co-location",
|
||||
"--cc_backend_config_type",
|
||||
"vidur",
|
||||
"--cluster_config_num_replicas",
|
||||
"1",
|
||||
"--cluster_scheduler_config_type",
|
||||
"sticky_round_robin",
|
||||
"--replica_config_model_name",
|
||||
"qwen3-a3b-30b-moe",
|
||||
"--replica_config_device",
|
||||
"h20",
|
||||
"--replica_config_network_device",
|
||||
"h20_dgx",
|
||||
"--replica_config_attn_tensor_parallel_size",
|
||||
"4",
|
||||
"--replica_config_attn_data_parallel_size",
|
||||
"1",
|
||||
"--replica_config_moe_tensor_parallel_size",
|
||||
"4",
|
||||
"--replica_config_moe_expert_parallel_size",
|
||||
"1",
|
||||
"--replica_config_num_pipeline_stages",
|
||||
"1",
|
||||
"--replica_scheduler_config_type",
|
||||
"vllm_v1",
|
||||
"--decode_cuda_graph_mode",
|
||||
"piecewise",
|
||||
"--vllm_v1_scheduler_config_batch_size_cap",
|
||||
"32",
|
||||
"--vllm_v1_scheduler_config_max_tokens_in_batch",
|
||||
"8192",
|
||||
"--vllm_v1_scheduler_config_long_prefill_token_threshold",
|
||||
"0",
|
||||
"--vllm_v1_scheduler_config_block_size",
|
||||
"16",
|
||||
"--vllm_v1_scheduler_config_num_blocks_mode",
|
||||
"explicit",
|
||||
"--vllm_v1_scheduler_config_gpu_memory_utilization",
|
||||
"0.92",
|
||||
"--vllm_v1_scheduler_config_non_kv_cache_overhead_bytes",
|
||||
"0",
|
||||
"--request_generator_config_type",
|
||||
"trace_replay",
|
||||
"--trace_request_generator_config_trace_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp4-frontier.csv",
|
||||
"--trace_request_generator_config_max_tokens",
|
||||
"40960",
|
||||
"--metrics_config_output_dir",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/sim/tp4_mns32/metrics",
|
||||
"--metrics_config_run_id",
|
||||
"joint_tp4_mns32",
|
||||
"--metrics_config_write_metrics",
|
||||
"--metrics_config_store_request_metrics",
|
||||
"--metrics_config_store_batch_metrics",
|
||||
"--metrics_config_store_token_completion_metrics",
|
||||
"--metrics_config_store_utilization_metrics",
|
||||
"--no-metrics_config_store_plots",
|
||||
"--no-metrics_config_enable_chrome_trace",
|
||||
"--no-metrics_config_write_json_trace",
|
||||
"--metrics_config_store_frontier_stage_batch_ledger",
|
||||
"--no-random_forrest_execution_time_predictor_config_enable_dummy_mode",
|
||||
"--random_forrest_execution_time_predictor_config_linear_op_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/linear_op.csv",
|
||||
"--random_forrest_execution_time_predictor_config_atten_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/attention.csv",
|
||||
"--random_forrest_execution_time_predictor_config_moe_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/profile-v4-trace-final/moe.csv",
|
||||
"--random_forrest_execution_time_predictor_config_linear_op_kernel_only_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/linear_op.csv",
|
||||
"--random_forrest_execution_time_predictor_config_atten_kernel_only_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/attention.csv",
|
||||
"--random_forrest_execution_time_predictor_config_moe_kernel_only_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/frozen-kernel-only/moe.csv",
|
||||
"--random_forrest_execution_time_predictor_config_prediction_max_prefill_chunk_size",
|
||||
"8192",
|
||||
"--random_forrest_execution_time_predictor_config_prediction_max_batch_size",
|
||||
"64",
|
||||
"--random_forrest_execution_time_predictor_config_prediction_max_tokens_per_request",
|
||||
"40960",
|
||||
"--random_forrest_execution_time_predictor_config_no_cache",
|
||||
"--random_forrest_execution_time_predictor_config_skip_cpu_overhead_modeling",
|
||||
"--vllm_v1_scheduler_config_num_blocks",
|
||||
"191786",
|
||||
"--vllm_v1_scheduler_config_enable_chunked_prefill",
|
||||
"--random_forrest_execution_time_predictor_config_num_training_job_threads",
|
||||
"4",
|
||||
"--cudagraph_capture_sizes",
|
||||
"1",
|
||||
"2",
|
||||
"4",
|
||||
"8",
|
||||
"16",
|
||||
"24",
|
||||
"32",
|
||||
"40",
|
||||
"48",
|
||||
"56",
|
||||
"64",
|
||||
"--vidur_cc_backend_config_all_reduce_input_file",
|
||||
"/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-profiles/measured-allreduce.csv",
|
||||
"--vidur_cc_backend_config_cache_dir",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/cc-cache",
|
||||
"--vidur_cc_backend_config_k_fold_cv_splits",
|
||||
"6",
|
||||
"--vidur_cc_backend_config_num_training_job_threads",
|
||||
"1",
|
||||
"--metrics_config_cache_dir",
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/model-cache"
|
||||
],
|
||||
"log": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/logs/tp4_mns32.log",
|
||||
"source_command": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp4_mns32/tp4/command.json",
|
||||
"source_command_sha256": "fbc7dee55590b415ed1cde8072de835ed155a0c20ba0eb305c3cb22aa8065a51",
|
||||
"usage": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/usage/tp4_mns32.json"
|
||||
}
|
||||
},
|
||||
"collective_curve": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/results/collective-curve.json",
|
||||
"collective_curve_sha256": "f9543649d4ea78f08240bf1284ab74083aa5cf5671ed47e386047f1453300b36",
|
||||
"collective_curve_variant": "drop_mean",
|
||||
"frontier_checkout": "/tmp/frontier-attn-structured-v0",
|
||||
"frontier_commit": "1f8900a4ac64e45754b03d0aa7c1dddab65785cf",
|
||||
"mode": "joint",
|
||||
"model_cache": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/model-cache",
|
||||
"moe_curve": "/home/gahow/phd/aituner/runs/frontier-fused-moe-profile-v0/results/fused-moe-curve.json",
|
||||
"moe_curve_sha256": "b94d65d9d581adefcc6c14ed4920cce6a1136f74f1014737dc3e4249bc8250d2",
|
||||
"python": "/usr/bin/python3",
|
||||
"python_dependency_roots": [
|
||||
"/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/python-deps",
|
||||
"/home/gahow/.cache/uv/archive-v0/-_kzErLcPO5nASZFX8b9k",
|
||||
"/home/gahow/.cache/uv/archive-v0/FbaBs_QJ9QKEbQ9V_4aIR",
|
||||
"/home/gahow/.cache/uv/archive-v0/fuHsGXD0Lv_UjFC8yI4-7",
|
||||
"/home/gahow/.cache/uv/archive-v0/jFGdqQLpB1eopfm9VxT3j",
|
||||
"/home/gahow/.cache/uv/archive-v0/YWW6ExSJuPVvv4-qYQTin",
|
||||
"/home/gahow/.cache/uv/archive-v0/3_qxZ5Ll-EpVAGZfbksfe"
|
||||
],
|
||||
"traces": {
|
||||
"1": {
|
||||
"first_arrival_s": 0.0,
|
||||
"last_arrival_s": 595.348837209302,
|
||||
"requests": 129,
|
||||
"source_request_metrics": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp1_mns16/tp1/metrics/qwen3_a3b_30b_moe/online_serving/qwen30_trace_tp1_mns16_tp1/request_metrics.csv",
|
||||
"source_sha256": "0b82e09644a5884fcd10d894b68495daefdabb32b770146c2f9ece37b8469f4f",
|
||||
"trace": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp1-frontier.csv",
|
||||
"trace_sha256": "59dd8996ff879ef94330004104dfdf515b791bce4036576eccc93290e9206dad"
|
||||
},
|
||||
"2": {
|
||||
"first_arrival_s": 0.0,
|
||||
"last_arrival_s": 297.674418604651,
|
||||
"requests": 129,
|
||||
"source_request_metrics": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp2_mns16/tp2/metrics/qwen3_a3b_30b_moe/online_serving/qwen30_trace_tp2_mns16_tp2/request_metrics.csv",
|
||||
"source_sha256": "33983081bb20dd5e2053e9e3d13def8732e958150c9b47a8609ba345123f2316",
|
||||
"trace": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp2-frontier.csv",
|
||||
"trace_sha256": "64fc077b38274a76a8279884ac4115836cd1157c95119c64fabac50d81124f69"
|
||||
},
|
||||
"4": {
|
||||
"first_arrival_s": 0.0,
|
||||
"last_arrival_s": 148.837209302326,
|
||||
"requests": 129,
|
||||
"source_request_metrics": "/home/gahow/phd/aituner/runs/frontier-split-rootcause-v0/frozen-inputs/q30-lo-fixed-pd-cells/sim/fixed-pd/runs/tp4_mns16/tp4/metrics/qwen3_a3b_30b_moe/online_serving/qwen30_trace_tp4_mns16_tp4/request_metrics.csv",
|
||||
"source_sha256": "b36cd383c07b546d2c1f2fac754d5dbb92efd6880316b4233a7aef9fa1115a36",
|
||||
"trace": "/home/gahow/phd/aituner/runs/frontier-collective-joint-v0/counterfactual/joint-r2/inputs/tp4-frontier.csv",
|
||||
"trace_sha256": "adb3d6f3932a44c86c3d9e7cf1e57739594e8c48d19a26b5aa54b37dce0e0c19"
|
||||
}
|
||||
}
|
||||
}
|
||||
72
runs/frontier-attn-structured-v0/plot_figure_prototype.py
Normal file
72
runs/frontier-attn-structured-v0/plot_figure_prototype.py
Normal file
@@ -0,0 +1,72 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Schematic figure frozen before EXP-ATTN-STRUCTURED execution."""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import matplotlib
|
||||
|
||||
matplotlib.use("Agg")
|
||||
import matplotlib.pyplot as plt
|
||||
import numpy as np
|
||||
|
||||
ROOT = Path(__file__).resolve().parent
|
||||
SURFACE = "#fcfcfb"
|
||||
INK = "#111111"
|
||||
MUTED = "#77736c"
|
||||
GRID = "#dedbd2"
|
||||
RF = "#d95f02"
|
||||
STRUCTURED = "#1b75bc"
|
||||
|
||||
fig, axes = plt.subplots(1, 2, figsize=(10.8, 4.2), dpi=160)
|
||||
fig.patch.set_facecolor(SURFACE)
|
||||
|
||||
for ax in axes:
|
||||
ax.set_facecolor(SURFACE)
|
||||
ax.grid(axis="y", color=GRID, linewidth=0.8)
|
||||
ax.set_axisbelow(True)
|
||||
ax.spines[["top", "right"]].set_visible(False)
|
||||
ax.tick_params(colors=MUTED, labelsize=8)
|
||||
|
||||
kv = np.array([8, 16, 24], dtype=float)
|
||||
actual = np.array([11.97, 19.92, 27.84])
|
||||
rf = np.array([9.46, 16.40, 24.52])
|
||||
structured_expected = np.array([12.0, 19.9, 27.9])
|
||||
axes[0].plot(kv, actual, "o-", color=INK, label="profile actual")
|
||||
axes[0].plot(kv, rf, "s--", color=RF, label="current RF")
|
||||
axes[0].plot(
|
||||
kv,
|
||||
structured_expected,
|
||||
"^:",
|
||||
color=STRUCTURED,
|
||||
label="structured (expected)",
|
||||
)
|
||||
axes[0].set_xlabel("KV context (ktok)")
|
||||
axes[0].set_ylabel("TP1 q8k attention time (ms)")
|
||||
axes[0].set_title("(a) Continuous KV growth", loc="left", fontsize=10)
|
||||
axes[0].legend(frameon=False, fontsize=8)
|
||||
|
||||
labels = ["TP1\n.00125", "TP1\n.0025", "TP2\n.0025", "TP2\n.005",
|
||||
"TP4\n.0025", "TP4\n.005", "TP4\n.01"]
|
||||
x = np.arange(len(labels))
|
||||
v5_mean = np.array([-13.5, -17.7, -11.2, -14.1, 2.5, 2.9, -0.2])
|
||||
expected = np.array([-5, -8, -9, -11, 3, 3, 0])
|
||||
axes[1].axhspan(-15, 15, color=GRID, alpha=0.5)
|
||||
axes[1].axhline(0, color=MUTED, linewidth=0.8)
|
||||
axes[1].plot(x, v5_mean, "o-", color=RF, label="sim-v5 measured")
|
||||
axes[1].plot(x, expected, "s--", color=STRUCTURED, label="H1 expected")
|
||||
axes[1].set_xticks(x, labels)
|
||||
axes[1].set_ylabel("TTFT mean bias (%)")
|
||||
axes[1].set_title("(b) 7-cell trace gate", loc="left", fontsize=10)
|
||||
axes[1].legend(frameon=False, fontsize=8)
|
||||
|
||||
fig.suptitle(
|
||||
"MOCK / schematic — EXP-ATTN-STRUCTURED (not measured results)",
|
||||
x=0.01,
|
||||
ha="left",
|
||||
color=RF,
|
||||
fontsize=9,
|
||||
)
|
||||
fig.tight_layout(rect=(0, 0, 1, 0.95))
|
||||
fig.savefig(ROOT / "figure-prototype.png", facecolor=SURFACE)
|
||||
fig.savefig(ROOT / "figure-prototype.svg", facecolor=SURFACE)
|
||||
print(ROOT / "figure-prototype.png")
|
||||
@@ -0,0 +1,10 @@
|
||||
candidate,training_rows,grid_fit_mape_pct,single_fit_mape_pct,heldout_context_mape_pct,heldout_context_max_abs_error_pct,prediction_min_ms,q_min_delta_ms,kv_min_delta_ms,monotone_and_nonnegative,tp
|
||||
rf_all,29,15.074275515235467,19.842994757611535,44.40139318281943,82.78279487156401,0.06029164119272453,0.0,-4.2841601371801374e-05,False,1
|
||||
rf_single,23,11.834992526698676,22.893412972496023,34.45356428541224,62.49334437588834,0.059327708247725125,-0.00022153525203457564,-1.4336001873005433e-05,False,1
|
||||
structured_single,23,0.8432511364168856,2.227110646811972,0.841926169535806,2.0408978739639134,0.05679146709541477,0.0,0.0016745062683911627,True,1
|
||||
rf_all,29,13.812689689573157,17.879870012606048,44.20143320278334,82.0656368501208,0.06000113548192927,-0.004361070463210395,-0.009949388915300408,False,2
|
||||
rf_single,23,10.738962684891058,20.818820413545826,34.84360402100271,66.67253880294443,0.059971319361210015,-0.003811210796127021,-0.009949388915300408,False,2
|
||||
structured_single,23,1.518412258922364,6.226292654776418,1.6047958673086566,3.464671475193195,0.05767893331746252,0.0,0.0013929374121726124,True,2
|
||||
rf_all,29,14.117015331381916,15.49366794487052,43.53650868837558,81.33508178007524,0.05866772018640992,-0.0009967416035880083,-5.5955198407176e-05,False,4
|
||||
rf_single,23,12.07517666989496,17.672692652721008,34.14562332866605,62.31478818862995,0.058213693721655094,-0.0012244979345549661,-8.259841203689389e-05,False,4
|
||||
structured_single,23,3.088740104976349,5.969109476280061,3.0103379904473164,5.490598706238697,0.05747733327249683,0.0,0.0012048051417407057,True,4
|
||||
|
149
runs/frontier-attn-structured-v0/results/predictor-ablation.json
Normal file
149
runs/frontier-attn-structured-v0/results/predictor-ablation.json
Normal file
@@ -0,0 +1,149 @@
|
||||
{
|
||||
"schema": "frontier-attn-structured-ablation-v1",
|
||||
"profile": "/home/gahow/phd/aituner/runs/frontier-prefill-kvgrowth-fix-v0/profiles/profile-v5-kvgrowth/attention.csv",
|
||||
"frontier_checkout": "/tmp/frontier-attn-structured-v0",
|
||||
"data_audit": {
|
||||
"tp1": {
|
||||
"standard_rows": 29,
|
||||
"single_request_rows": 23,
|
||||
"target_grid_rows": 10,
|
||||
"duplicate_feature_groups": 4
|
||||
},
|
||||
"tp2": {
|
||||
"standard_rows": 29,
|
||||
"single_request_rows": 23,
|
||||
"target_grid_rows": 10,
|
||||
"duplicate_feature_groups": 4
|
||||
},
|
||||
"tp4": {
|
||||
"standard_rows": 29,
|
||||
"single_request_rows": 23,
|
||||
"target_grid_rows": 10,
|
||||
"duplicate_feature_groups": 4
|
||||
}
|
||||
},
|
||||
"results": [
|
||||
{
|
||||
"candidate": "rf_all",
|
||||
"training_rows": 29,
|
||||
"grid_fit_mape_pct": 15.074275515235467,
|
||||
"single_fit_mape_pct": 19.842994757611535,
|
||||
"heldout_context_mape_pct": 44.40139318281943,
|
||||
"heldout_context_max_abs_error_pct": 82.78279487156401,
|
||||
"prediction_min_ms": 0.06029164119272453,
|
||||
"q_min_delta_ms": 0.0,
|
||||
"kv_min_delta_ms": -4.2841601371801374e-05,
|
||||
"monotone_and_nonnegative": false,
|
||||
"tp": 1
|
||||
},
|
||||
{
|
||||
"candidate": "rf_single",
|
||||
"training_rows": 23,
|
||||
"grid_fit_mape_pct": 11.834992526698676,
|
||||
"single_fit_mape_pct": 22.893412972496023,
|
||||
"heldout_context_mape_pct": 34.45356428541224,
|
||||
"heldout_context_max_abs_error_pct": 62.49334437588834,
|
||||
"prediction_min_ms": 0.059327708247725125,
|
||||
"q_min_delta_ms": -0.00022153525203457564,
|
||||
"kv_min_delta_ms": -1.4336001873005433e-05,
|
||||
"monotone_and_nonnegative": false,
|
||||
"tp": 1
|
||||
},
|
||||
{
|
||||
"candidate": "structured_single",
|
||||
"training_rows": 23,
|
||||
"grid_fit_mape_pct": 0.8432511364168856,
|
||||
"single_fit_mape_pct": 2.227110646811972,
|
||||
"heldout_context_mape_pct": 0.841926169535806,
|
||||
"heldout_context_max_abs_error_pct": 2.0408978739639134,
|
||||
"prediction_min_ms": 0.05679146709541477,
|
||||
"q_min_delta_ms": 0.0,
|
||||
"kv_min_delta_ms": 0.0016745062683911627,
|
||||
"monotone_and_nonnegative": true,
|
||||
"tp": 1
|
||||
},
|
||||
{
|
||||
"candidate": "rf_all",
|
||||
"training_rows": 29,
|
||||
"grid_fit_mape_pct": 13.812689689573157,
|
||||
"single_fit_mape_pct": 17.879870012606048,
|
||||
"heldout_context_mape_pct": 44.20143320278334,
|
||||
"heldout_context_max_abs_error_pct": 82.0656368501208,
|
||||
"prediction_min_ms": 0.06000113548192927,
|
||||
"q_min_delta_ms": -0.004361070463210395,
|
||||
"kv_min_delta_ms": -0.009949388915300408,
|
||||
"monotone_and_nonnegative": false,
|
||||
"tp": 2
|
||||
},
|
||||
{
|
||||
"candidate": "rf_single",
|
||||
"training_rows": 23,
|
||||
"grid_fit_mape_pct": 10.738962684891058,
|
||||
"single_fit_mape_pct": 20.818820413545826,
|
||||
"heldout_context_mape_pct": 34.84360402100271,
|
||||
"heldout_context_max_abs_error_pct": 66.67253880294443,
|
||||
"prediction_min_ms": 0.059971319361210015,
|
||||
"q_min_delta_ms": -0.003811210796127021,
|
||||
"kv_min_delta_ms": -0.009949388915300408,
|
||||
"monotone_and_nonnegative": false,
|
||||
"tp": 2
|
||||
},
|
||||
{
|
||||
"candidate": "structured_single",
|
||||
"training_rows": 23,
|
||||
"grid_fit_mape_pct": 1.518412258922364,
|
||||
"single_fit_mape_pct": 6.226292654776418,
|
||||
"heldout_context_mape_pct": 1.6047958673086566,
|
||||
"heldout_context_max_abs_error_pct": 3.464671475193195,
|
||||
"prediction_min_ms": 0.05767893331746252,
|
||||
"q_min_delta_ms": 0.0,
|
||||
"kv_min_delta_ms": 0.0013929374121726124,
|
||||
"monotone_and_nonnegative": true,
|
||||
"tp": 2
|
||||
},
|
||||
{
|
||||
"candidate": "rf_all",
|
||||
"training_rows": 29,
|
||||
"grid_fit_mape_pct": 14.117015331381916,
|
||||
"single_fit_mape_pct": 15.49366794487052,
|
||||
"heldout_context_mape_pct": 43.53650868837558,
|
||||
"heldout_context_max_abs_error_pct": 81.33508178007524,
|
||||
"prediction_min_ms": 0.05866772018640992,
|
||||
"q_min_delta_ms": -0.0009967416035880083,
|
||||
"kv_min_delta_ms": -5.5955198407176e-05,
|
||||
"monotone_and_nonnegative": false,
|
||||
"tp": 4
|
||||
},
|
||||
{
|
||||
"candidate": "rf_single",
|
||||
"training_rows": 23,
|
||||
"grid_fit_mape_pct": 12.07517666989496,
|
||||
"single_fit_mape_pct": 17.672692652721008,
|
||||
"heldout_context_mape_pct": 34.14562332866605,
|
||||
"heldout_context_max_abs_error_pct": 62.31478818862995,
|
||||
"prediction_min_ms": 0.058213693721655094,
|
||||
"q_min_delta_ms": -0.0012244979345549661,
|
||||
"kv_min_delta_ms": -8.259841203689389e-05,
|
||||
"monotone_and_nonnegative": false,
|
||||
"tp": 4
|
||||
},
|
||||
{
|
||||
"candidate": "structured_single",
|
||||
"training_rows": 23,
|
||||
"grid_fit_mape_pct": 3.088740104976349,
|
||||
"single_fit_mape_pct": 5.969109476280061,
|
||||
"heldout_context_mape_pct": 3.0103379904473164,
|
||||
"heldout_context_max_abs_error_pct": 5.490598706238697,
|
||||
"prediction_min_ms": 0.05747733327249683,
|
||||
"q_min_delta_ms": 0.0,
|
||||
"kv_min_delta_ms": 0.0012048051417407057,
|
||||
"monotone_and_nonnegative": true,
|
||||
"tp": 4
|
||||
}
|
||||
],
|
||||
"checks": {
|
||||
"heldout_context_mape_le_5pct": true,
|
||||
"monotone_and_nonnegative": true,
|
||||
"profile_gate": true
|
||||
}
|
||||
}
|
||||
85
runs/frontier-attn-structured-v0/results/trace-verdict.csv
Normal file
85
runs/frontier-attn-structured-v0/results/trace-verdict.csv
Normal file
@@ -0,0 +1,85 @@
|
||||
cell,metric,quantile,old_bias,new_bias,abs_bias_delta_pp,validity
|
||||
tp1_rho0p00125,ttft,mean,-0.13461915993830945,-0.06328385143753473,-7.133530850077471,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p00125,ttft,p50,-0.1888032883165517,-0.1117698463137391,-7.7033442002812595,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p00125,ttft,p90,-0.23223319000679374,-0.0884727580604037,-14.376043194639005,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p00125,ttft,p99,-0.16824856840168442,-0.08585570369326061,-8.239286470842382,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p00125,tpot,mean,0.1305653876178269,0.14303898259628217,1.2473594978455265,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p00125,tpot,p50,0.16837991778063455,0.1688780144991309,0.04980967184963492,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p00125,tpot,p90,0.012855564422932954,0.02293841255516459,1.0082848132231637,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p00125,tpot,p99,-0.08151080694091946,-0.06480973978913974,-1.6701067151779714,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p00125,e2e,mean,0.06539157436639341,0.0838094906426525,1.8417916276259092,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p00125,e2e,p50,0.10867934171755954,0.11142188099086506,0.2742539273305519,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p00125,e2e,p90,0.10946179160934073,0.11872607178176105,0.9264280172420314,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p00125,e2e,p99,-0.0657146472433028,-0.04585643943465822,-1.9858207808644577,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p0025,ttft,mean,-0.17677087458166194,-0.09487906467830536,-8.189180990335657,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p0025,ttft,p50,0.008311436147272566,0.015508635458377175,0.7197199311104608,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p0025,ttft,p90,-0.2608553178054708,-0.129865645852306,-13.098967195316478,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p0025,ttft,p99,-0.22116442296103195,-0.14410151018893788,-7.706291277209407,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p0025,tpot,mean,0.014380733682430142,0.052302860177306544,3.7922126494876403,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p0025,tpot,p50,0.13351145963877706,0.14244154194999165,0.8930082311214588,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p0025,tpot,p90,-0.05456950130438105,0.018069551387063856,-3.649994991731719,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p0025,tpot,p99,-0.23635406197836167,-0.1790206784227079,-5.733338355565376,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p0025,e2e,mean,-0.020164189983441452,0.01563919326555167,-0.4524996717889782,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p0025,e2e,p50,0.07632815851795742,0.10505951594320918,2.8731357425251765,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p0025,e2e,p90,0.049809009042946335,0.08165607475145953,3.1847065708513194,GATE_FAIL_DIAGNOSTIC
|
||||
tp1_rho0p0025,e2e,p99,-0.18303183751478602,-0.15333409130015813,-2.96977462146279,GATE_FAIL_DIAGNOSTIC
|
||||
tp2_rho0p0025,ttft,mean,-0.11161154024124531,-0.045433491316312524,-6.617804892493279,PASS_SUBCRITICAL
|
||||
tp2_rho0p0025,ttft,p50,-0.18420934047220774,-0.0875990583320057,-9.661028214020204,PASS_SUBCRITICAL
|
||||
tp2_rho0p0025,ttft,p90,-0.21836264490202395,-0.1142200855839675,-10.414255931805645,PASS_SUBCRITICAL
|
||||
tp2_rho0p0025,ttft,p99,-0.17283197594971011,-0.07674087497700505,-9.609110097270507,PASS_SUBCRITICAL
|
||||
tp2_rho0p0025,tpot,mean,0.13711499081487563,0.15073709978689778,1.362210897202215,PASS_SUBCRITICAL
|
||||
tp2_rho0p0025,tpot,p50,0.17555321305308488,0.18527285925405948,0.9719646200974597,PASS_SUBCRITICAL
|
||||
tp2_rho0p0025,tpot,p90,0.0591579975137338,0.07863996768588354,1.9481970172149734,PASS_SUBCRITICAL
|
||||
tp2_rho0p0025,tpot,p99,0.1300483675091633,0.1661995397125918,3.6151172203428503,PASS_SUBCRITICAL
|
||||
tp2_rho0p0025,e2e,mean,0.10925353865257875,0.12696146154030977,1.7707922887731016,PASS_SUBCRITICAL
|
||||
tp2_rho0p0025,e2e,p50,0.1339165600755408,0.14796251184179712,1.404595176625631,PASS_SUBCRITICAL
|
||||
tp2_rho0p0025,e2e,p90,0.10666828724664539,0.12767235986604622,2.1004072619400835,PASS_SUBCRITICAL
|
||||
tp2_rho0p0025,e2e,p99,-0.07437942975605877,-0.03965040123988098,-3.472902851617779,PASS_SUBCRITICAL
|
||||
tp2_rho0p005,ttft,mean,-0.14129871969878843,-0.07082500585057615,-7.047371384821228,GATE_FAIL_DIAGNOSTIC
|
||||
tp2_rho0p005,ttft,p50,-0.2042045530944649,-0.1284088888361358,-7.579566425832909,GATE_FAIL_DIAGNOSTIC
|
||||
tp2_rho0p005,ttft,p90,-0.19003454588767263,-0.111136573344055,-7.889797254361763,GATE_FAIL_DIAGNOSTIC
|
||||
tp2_rho0p005,ttft,p99,-0.19373351009539902,-0.09012556161973535,-10.360794847566366,GATE_FAIL_DIAGNOSTIC
|
||||
tp2_rho0p005,tpot,mean,0.03117337438562004,0.05457336599027876,2.3399991604658723,GATE_FAIL_DIAGNOSTIC
|
||||
tp2_rho0p005,tpot,p50,0.07668249597302182,0.0872512146277895,1.0568718654767675,GATE_FAIL_DIAGNOSTIC
|
||||
tp2_rho0p005,tpot,p90,-0.03724827658429621,0.01132352325010614,-2.5924753334190074,GATE_FAIL_DIAGNOSTIC
|
||||
tp2_rho0p005,tpot,p99,-0.10050436157089844,-0.05948329733667248,-4.102106423422596,GATE_FAIL_DIAGNOSTIC
|
||||
tp2_rho0p005,e2e,mean,0.023867807624916495,0.050164236759908075,2.629642913499158,GATE_FAIL_DIAGNOSTIC
|
||||
tp2_rho0p005,e2e,p50,0.07536668131278851,0.0912493994395978,1.588271812680929,GATE_FAIL_DIAGNOSTIC
|
||||
tp2_rho0p005,e2e,p90,-0.03100470462321266,-0.0004979191794830456,-3.0506785443729614,GATE_FAIL_DIAGNOSTIC
|
||||
tp2_rho0p005,e2e,p99,-0.04909581632382212,-0.0014369075488634014,-4.765890877495872,GATE_FAIL_DIAGNOSTIC
|
||||
tp4_rho0p0025,ttft,mean,0.025022574738277282,0.07766646565061346,5.2643890912336175,PASS_SUBCRITICAL
|
||||
tp4_rho0p0025,ttft,p50,-0.040042171846277425,0.029153379043297147,-1.0888792802980278,PASS_SUBCRITICAL
|
||||
tp4_rho0p0025,ttft,p90,-0.056230097634382616,0.025222989748299444,-3.100710788608317,PASS_SUBCRITICAL
|
||||
tp4_rho0p0025,ttft,p99,-0.07535478089127973,0.0205549324689376,-5.479984842234213,PASS_SUBCRITICAL
|
||||
tp4_rho0p0025,tpot,mean,0.2170232618103144,0.2295170016795841,1.2493739869269715,PASS_SUBCRITICAL
|
||||
tp4_rho0p0025,tpot,p50,0.22406751004936917,0.22406940610958842,0.00018960602192474862,PASS_SUBCRITICAL
|
||||
tp4_rho0p0025,tpot,p90,0.1725668492041552,0.1795965483385546,0.7029699134399409,PASS_SUBCRITICAL
|
||||
tp4_rho0p0025,tpot,p99,0.1603764334794579,0.254199450927048,9.382301744759008,PASS_SUBCRITICAL
|
||||
tp4_rho0p0025,e2e,mean,0.18328079455378776,0.19190660476104554,0.8625810207257778,PASS_SUBCRITICAL
|
||||
tp4_rho0p0025,e2e,p50,0.2057109615696404,0.21253242300694286,0.6821461437302473,PASS_SUBCRITICAL
|
||||
tp4_rho0p0025,e2e,p90,0.1869703879211648,0.19279855916666536,0.5828171245500557,PASS_SUBCRITICAL
|
||||
tp4_rho0p0025,e2e,p99,0.14838304065885655,0.15160588346499027,0.3222842806133719,PASS_SUBCRITICAL
|
||||
tp4_rho0p005,ttft,mean,0.028648879997638963,0.08356973519379125,5.492085519615229,PASS_SUBCRITICAL
|
||||
tp4_rho0p005,ttft,p50,0.028237979190582876,0.09021324737193111,6.197526818134823,PASS_SUBCRITICAL
|
||||
tp4_rho0p005,ttft,p90,-0.055961238578361966,-0.0006012940489499138,-5.535994452941205,PASS_SUBCRITICAL
|
||||
tp4_rho0p005,ttft,p99,-0.045491187615110146,0.05253708684194205,0.7045899226831902,PASS_SUBCRITICAL
|
||||
tp4_rho0p005,tpot,mean,0.1707193936623511,0.1813506819061229,1.0631288243771824,PASS_SUBCRITICAL
|
||||
tp4_rho0p005,tpot,p50,0.16851374859025317,0.1743324539248605,0.5818705334607321,PASS_SUBCRITICAL
|
||||
tp4_rho0p005,tpot,p90,0.10492621353626864,0.11816822907155744,1.3242015535288796,PASS_SUBCRITICAL
|
||||
tp4_rho0p005,tpot,p99,0.31231888769471766,0.3662267201704473,5.390783247572961,PASS_SUBCRITICAL
|
||||
tp4_rho0p005,e2e,mean,0.151102823467785,0.1630793421767109,1.197651870892591,PASS_SUBCRITICAL
|
||||
tp4_rho0p005,e2e,p50,0.1594213531394918,0.17360527090575292,1.418391776626113,PASS_SUBCRITICAL
|
||||
tp4_rho0p005,e2e,p90,0.1266352410718406,0.13652104764058856,0.9885806568747962,PASS_SUBCRITICAL
|
||||
tp4_rho0p005,e2e,p99,0.15576537699445703,0.17677450343779522,2.100912644333819,PASS_SUBCRITICAL
|
||||
tp4_rho0p01,ttft,mean,-0.0014652143973501086,0.059650620031540064,5.8185405634189955,PASS_SUBCRITICAL
|
||||
tp4_rho0p01,ttft,p50,0.22936601881498542,0.24159106387124998,1.2225045056264565,PASS_SUBCRITICAL
|
||||
tp4_rho0p01,ttft,p90,-0.0646093465409875,-0.011548419893895705,-5.30609266470918,PASS_SUBCRITICAL
|
||||
tp4_rho0p01,ttft,p99,-0.09621678853313553,-0.015475807392170575,-8.074098114096495,PASS_SUBCRITICAL
|
||||
tp4_rho0p01,tpot,mean,0.06675609059150077,0.10596101263201793,3.9204922040517163,PASS_SUBCRITICAL
|
||||
tp4_rho0p01,tpot,p50,0.08477302587551214,0.09703754188844527,1.2264516012933129,PASS_SUBCRITICAL
|
||||
tp4_rho0p01,tpot,p90,0.0379183273767328,0.08369800201750718,4.577967464077439,PASS_SUBCRITICAL
|
||||
tp4_rho0p01,tpot,p99,-0.010663954751357074,0.06667813160571406,5.601417685435699,PASS_SUBCRITICAL
|
||||
tp4_rho0p01,e2e,mean,0.07212904317306096,0.09777288053702092,2.564383736395996,PASS_SUBCRITICAL
|
||||
tp4_rho0p01,e2e,p50,0.12069215463307655,0.1388377217196832,1.8145567086606653,PASS_SUBCRITICAL
|
||||
tp4_rho0p01,e2e,p90,0.03317293927357903,0.058060760567474216,2.4887821293895183,PASS_SUBCRITICAL
|
||||
tp4_rho0p01,e2e,p99,0.0338378271163308,0.06380410696893425,2.996627985260345,PASS_SUBCRITICAL
|
||||
|
2219
runs/frontier-attn-structured-v0/results/trace-verdict.json
Normal file
2219
runs/frontier-attn-structured-v0/results/trace-verdict.json
Normal file
File diff suppressed because it is too large
Load Diff
126
runs/frontier-attn-structured-v0/run_replay.py
Normal file
126
runs/frontier-attn-structured-v0/run_replay.py
Normal file
@@ -0,0 +1,126 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Replay one real-trace cell with the structured-attention experiment commit."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import importlib.util
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent
|
||||
REPO = ROOT.parents[1]
|
||||
S3_REAL = REPO / "runs/frontier-s3-real-v0"
|
||||
BASE_REFERENCE = (
|
||||
REPO
|
||||
/ "runs/frontier-collective-joint-v0/counterfactual/joint-r2/manifest.json"
|
||||
)
|
||||
BASE_COMMIT = "deadc4a321f0baaa534c6ebd17f974123733cdc2"
|
||||
EXPERIMENT_COMMIT = "1f8900a4ac64e45754b03d0aa7c1dddab65785cf"
|
||||
PATCH = ROOT / "0001-Experiment-with-structured-attention-prefill-predict.patch"
|
||||
|
||||
|
||||
def load_s3_module():
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
"s3_prefix_replay", S3_REAL / "run_frontier_prefix_replay.py"
|
||||
)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
sys.path.insert(0, str(S3_REAL))
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
def git(checkout: Path, *args: str) -> str:
|
||||
return subprocess.check_output(
|
||||
["git", "-C", str(checkout), *args], text=True
|
||||
).strip()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--trace", type=Path, required=True)
|
||||
parser.add_argument("--output-root", type=Path, required=True)
|
||||
parser.add_argument(
|
||||
"--config",
|
||||
choices=("tp4_mns16", "tp2_mns16", "tp1_mns16"),
|
||||
required=True,
|
||||
)
|
||||
parser.add_argument("--label", required=True)
|
||||
parser.add_argument("--max-tokens", type=int, required=True)
|
||||
parser.add_argument("--duration-s", type=float)
|
||||
parser.add_argument("--cache-root", type=Path, required=True)
|
||||
parser.add_argument(
|
||||
"--frontier-checkout",
|
||||
type=Path,
|
||||
default=Path("/tmp/frontier-attn-structured-v0"),
|
||||
)
|
||||
parser.add_argument(
|
||||
"--attention-profile",
|
||||
type=Path,
|
||||
default=REPO
|
||||
/ "runs/frontier-prefill-kvgrowth-fix-v0/profiles/"
|
||||
"profile-v5-kvgrowth/attention.csv",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
frontier = args.frontier_checkout.resolve()
|
||||
profile = args.attention_profile.resolve()
|
||||
if git(frontier, "rev-parse", "HEAD") != EXPERIMENT_COMMIT:
|
||||
raise SystemExit(f"unexpected experiment checkout HEAD: {frontier}")
|
||||
if git(frontier, "rev-parse", "HEAD^") != BASE_COMMIT:
|
||||
raise SystemExit("experiment commit is not directly based on frozen Frontier")
|
||||
if git(frontier, "status", "--porcelain"):
|
||||
raise SystemExit("experiment Frontier checkout must be clean")
|
||||
if not profile.is_file():
|
||||
raise SystemExit(f"attention profile missing: {profile}")
|
||||
|
||||
reference = json.loads(BASE_REFERENCE.read_text())
|
||||
reference["frontier_checkout"] = str(frontier)
|
||||
reference["frontier_commit"] = EXPERIMENT_COMMIT
|
||||
generated_reference = ROOT / "frontier-reference.json"
|
||||
generated_reference.write_text(json.dumps(reference, indent=2))
|
||||
|
||||
module = load_s3_module()
|
||||
module.REFERENCE = generated_reference
|
||||
module.EXPECTED_FRONTIER_COMMIT = EXPERIMENT_COMMIT
|
||||
original_replace = module.replace_flag
|
||||
|
||||
def replace_and_override(argv: list[str], flag: str, value: str) -> None:
|
||||
original_replace(argv, flag, value)
|
||||
if flag.endswith("trace_file"):
|
||||
atten_flag = (
|
||||
"--random_forrest_execution_time_predictor_config_atten_input_file"
|
||||
)
|
||||
original_replace(argv, atten_flag, str(profile))
|
||||
no_cache = (
|
||||
"--random_forrest_execution_time_predictor_config_no_cache"
|
||||
)
|
||||
if no_cache in argv:
|
||||
argv.remove(no_cache)
|
||||
|
||||
module.replace_flag = replace_and_override
|
||||
module.parse_args = lambda: args
|
||||
module.main()
|
||||
|
||||
manifest_path = args.output_root / "manifest.json"
|
||||
manifest = json.loads(manifest_path.read_text())
|
||||
manifest.update(
|
||||
{
|
||||
"schema": "frontier-attn-structured-replay-v1",
|
||||
"frontier_base_commit": BASE_COMMIT,
|
||||
"frontier_experiment_commit": EXPERIMENT_COMMIT,
|
||||
"frontier_patch": str(PATCH.resolve()),
|
||||
"frontier_patch_sha256": module.sha256(PATCH),
|
||||
"attention_profile_override": str(profile),
|
||||
"attention_profile_sha256": module.sha256(profile),
|
||||
"model_cache_enabled": True,
|
||||
}
|
||||
)
|
||||
manifest_path.write_text(json.dumps(manifest, indent=2))
|
||||
print(f"structured replay done: {args.output_root}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
153
runs/frontier-code-trace-v0/README.md
Normal file
153
runs/frontier-code-trace-v0/README.md
Normal file
@@ -0,0 +1,153 @@
|
||||
# Frontier code-trace campaign handoff
|
||||
|
||||
Phase A code prefill+decode 已进入 61min real matrix。data/profile、
|
||||
max-length、Frontier rho calibration 和 TP2/TP4 paired canary 均已完成;
|
||||
第一批 TP4 三个 load 与 TP2 low-rho diagnostic 正在 dash1–dash4 并行
|
||||
运行。code prefill-only 的独立 sim calibration 也已完成。
|
||||
|
||||
完整设计与 gate 见 [`experiment-card.md`](experiment-card.md)。
|
||||
|
||||
## 当前资产与下一步
|
||||
|
||||
- development window:0513 `[3480,7140)`,61min;
|
||||
- held-out window:0529 `[2640,6240)`,只在 development 判据冻结后使用;
|
||||
- profile:`profiles/profile-v6-code-longctx/`,覆盖 TP1/2/4 和 131072
|
||||
KV context;
|
||||
- full paired inputs(CPFS):
|
||||
`runs/frontier-code-trace-v0/inputs/full-r0p{0002,0004,0008,0016}-v1/`;
|
||||
- compact provenance:`results/calibration-summary.json`、
|
||||
`results/prefill-only-calibration-summary.json`、
|
||||
`results/canary-analysis-tp{2,4}-v*.json` 和
|
||||
`results/paired-input-manifests/`;
|
||||
- 当前 A4 wave 1:TP4 `rho={0.0002,0.0008,0.0016}` trial 1,以及
|
||||
TP2 `rho=0.0002` trial 1 diagnostic;
|
||||
- TP4 canary 的 TTFT/E2E、prefix hit 与 decode batch 通过;TP2
|
||||
TTFT p90 低估 32.1%,因此 TP2 其余 cell 暂不扩展;
|
||||
- prefill-only:TP2 已冻结 `rho={0.0004,0.0008,0.0016}`;
|
||||
TP4 到 `0.0032` 仍亚临界,需追加更高 rho 后冻结 near-knee。
|
||||
|
||||
source trace 的远端位置是:
|
||||
|
||||
```text
|
||||
/home/admin/cpfs/wjh/ali-trace/trace-glm5.1-formatted/
|
||||
```
|
||||
|
||||
以下命令保留为从 source 重新构建时的复现入口。
|
||||
|
||||
## 1. 审计所有 1h+ code source
|
||||
|
||||
在持有 trace 的机器、repo 根目录执行:
|
||||
|
||||
```bash
|
||||
python3 runs/frontier-code-trace-v0/audit_code_trace.py \
|
||||
--trace-root ~/ali-trace/trace-glm5.1-formatted \
|
||||
--output runs/frontier-code-trace-v0/inputs/code-audit.json
|
||||
```
|
||||
|
||||
如果目录里混有非 request JSONL,先只读列举文件,再用多个 `--source` 显式指定。审计输出必须满足:
|
||||
|
||||
```text
|
||||
data_gate = PASS
|
||||
selected.hash_contract.exact_source_block_size != null
|
||||
max_model_len_recommendation != null
|
||||
selected.selected_window_stats.max_model_len_coverage[推荐值].coverage = 1.0
|
||||
```
|
||||
|
||||
全量审计已确认 source block size=512;development source window 若 100%
|
||||
覆盖需要 262144,但正式 server cap 以 session-sampled paired cell 的实际
|
||||
`ISL+OSL max` 向上对齐,不能把 full-window 262144 无条件套到低 rho cell。
|
||||
审计会单独记录并排除 `input_length<=0` 或 `output_length<=0` 的 source
|
||||
行;这些行只有在 raw trace 同样显示 zero usage/empty response 时才按
|
||||
“未发生模型执行”处理,不能无记录过滤。
|
||||
|
||||
## 2. 物化稳定窗口
|
||||
|
||||
```bash
|
||||
python3 runs/frontier-code-trace-v0/prepare_code_window.py \
|
||||
--audit runs/frontier-code-trace-v0/inputs/code-audit.json \
|
||||
--output-root runs/frontier-code-trace-v0/inputs/code-window
|
||||
```
|
||||
|
||||
输出是 60–75min `code-raw-window.jsonl` 和 manifest。source 文件不修改。
|
||||
|
||||
## 3. 生成 P+D paired trace
|
||||
|
||||
若没有 prompt sidecar,先生成 shape/prefix-faithful synthetic prompts:
|
||||
|
||||
```bash
|
||||
python3 runs/frontier-s3-real-v0/remap_hash_blocks.py \
|
||||
--input runs/frontier-code-trace-v0/inputs/code-window/code-raw-window.jsonl \
|
||||
--output-root runs/frontier-code-trace-v0/inputs/code-pd-rho-max \
|
||||
--source-block-size 512 \
|
||||
--workload-mode prefill_decode \
|
||||
--rho 1.0 \
|
||||
--max-total-tokens 131072 \
|
||||
--validate-parents
|
||||
```
|
||||
|
||||
命令中的 `512` 和 `131072` 必须替换为 audit manifest 值。若存在对齐 prompt sidecar,加 `--prompt` 与 `--tokenizer`,并要求 synthetic fallback 为 0。
|
||||
|
||||
正式 rho 不能直接用 1.0;先从最大 remap cache 按 session-coherent `sampling_u` 过滤,分别标定 low/mid/near-knee。
|
||||
|
||||
## 4. 生成 prefill-only paired trace
|
||||
|
||||
对 chat/code 使用同一个转换接口:
|
||||
|
||||
```bash
|
||||
python3 runs/frontier-s3-real-v0/remap_hash_blocks.py \
|
||||
--input INPUT_WINDOW.jsonl \
|
||||
--output-root OUTPUT_ROOT \
|
||||
--source-block-size SOURCE_BLOCK_SIZE \
|
||||
--workload-mode prefill_only \
|
||||
--rho RHO \
|
||||
--max-total-tokens MAX_MODEL_LEN \
|
||||
--validate-parents
|
||||
```
|
||||
|
||||
该模式会同时把 Frontier `num_decode_tokens`、real request `min/max_tokens` 和 remapped row 的 `output_length` 固定为 1。
|
||||
|
||||
## 5. max-model-len 真机 gate
|
||||
|
||||
现有 real runner 新增了三个显式环境变量,chat 默认行为不变:
|
||||
|
||||
```bash
|
||||
MAX_MODEL_LEN=ACTUAL_CELL_MAX_ROUNDED_UP \
|
||||
TRACE_INPUT_ROOT=/absolute/path/to/materialized/code-cell \
|
||||
ALLOW_SYNTHETIC_PROMPTS=true \
|
||||
OUTPUT_ROOT=/absolute/path/to/new/output \
|
||||
bash runs/frontier-s3-real-v0/run_full_real.sh RHO_LABEL tp4_mns16 1 PORT
|
||||
```
|
||||
|
||||
- `MAX_MODEL_LEN` 必须覆盖 manifest 中该 paired cell 的实际最大请求;
|
||||
- `TRACE_INPUT_ROOT` 内必须有 `real_requests.jsonl` 和 `manifest.json`;
|
||||
- synthetic prompt 默认拒绝,只有在 experiment card 明确降级 claim 后才设为 `true`;
|
||||
- runner 会在启动前扫描 paired requests,若任何 `ISL+OSL` 超 cap 立即失败。
|
||||
|
||||
长上下文 server 必须同时设置
|
||||
`VLLM_ALLOW_LONG_MAX_MODEL_LEN=1` 和
|
||||
`--hf-overrides '{"max_position_embeddings":MAX_MODEL_LEN}'`;runner 已在
|
||||
`ALLOW_LONG_CONTEXT_SERVER=true` 时自动处理。长上下文默认使用 host-local
|
||||
vLLM compile cache,并按 topology 复用 FlashInfer workspace;启动 compile
|
||||
不进入 workload latency。
|
||||
|
||||
## 6. decode-only
|
||||
|
||||
当前 materializer 故意不提供 `decode_only` 选项。已安装 vLLM 0.20.0
|
||||
包含 `DecodeBenchConnector`,但它在首次 admission 后同步填 dummy KV;
|
||||
fill time 必须与 KV-ready arrival 分离。Frontier `Request` 支持
|
||||
`num_processed_tokens`,当前 trace generator 尚未从 CSV 注入该值。
|
||||
只有 real 首步无 prefill、sim ledger 首步为 decode 的 C0 gate 通过后,
|
||||
才创建 strict decode-only jobs。
|
||||
|
||||
## 本地验证
|
||||
|
||||
```bash
|
||||
python3 -m unittest -v \
|
||||
runs/frontier-code-trace-v0/test_code_trace_preflight.py \
|
||||
runs/frontier-s3-real-v0/test_remap_hash_blocks.py \
|
||||
runs/frontier-s3-real-v0/test_select_chat_window.py
|
||||
python3 -m py_compile \
|
||||
runs/frontier-code-trace-v0/*.py \
|
||||
runs/frontier-s3-real-v0/*.py
|
||||
bash -n runs/frontier-s3-real-v0/run_full_real.sh
|
||||
```
|
||||
424
runs/frontier-code-trace-v0/analyze_canary.py
Normal file
424
runs/frontier-code-trace-v0/analyze_canary.py
Normal file
@@ -0,0 +1,424 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Analyze one paired 10-minute code-trace real/sim canary topology."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
import re
|
||||
import statistics
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
METRICS = ("ttft", "tpot", "e2e")
|
||||
TPOT_MIN_OUTPUT_TOKENS = (2, 8, 32)
|
||||
SLO_TARGET_PASS_RATE = 0.95
|
||||
PROM_COUNTERS = ("vllm:prefix_cache_queries_total", "vllm:prefix_cache_hits_total")
|
||||
csv.field_size_limit(16 * 1024 * 1024)
|
||||
ITERATION_RE = re.compile(
|
||||
r"Iteration.*?:\s+"
|
||||
r"(?P<context_requests>\d+) context requests, "
|
||||
r"(?P<context_tokens>\d+) context tokens, "
|
||||
r"(?P<generation_requests>\d+) generation requests, "
|
||||
r"(?P<generation_tokens>\d+) generation tokens"
|
||||
)
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--input-root", type=Path, required=True)
|
||||
parser.add_argument("--sim-root", type=Path, required=True)
|
||||
parser.add_argument("--real-root", type=Path, action="append", required=True)
|
||||
parser.add_argument("--topology", required=True)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def sha256(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as stream:
|
||||
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def percentile(values: list[float], q: float) -> float:
|
||||
ordered = sorted(values)
|
||||
position = (len(ordered) - 1) * q
|
||||
lower = math.floor(position)
|
||||
upper = math.ceil(position)
|
||||
if lower == upper:
|
||||
return ordered[lower]
|
||||
return ordered[lower] * (upper - position) + ordered[upper] * (position - lower)
|
||||
|
||||
|
||||
def distribution(values: list[float]) -> dict[str, float | int]:
|
||||
if not values:
|
||||
raise ValueError("empty distribution")
|
||||
return {
|
||||
"count": len(values),
|
||||
"mean": statistics.fmean(values),
|
||||
"p50": percentile(values, 0.5),
|
||||
"p90": percentile(values, 0.9),
|
||||
"p95": percentile(values, 0.95),
|
||||
"p99": percentile(values, 0.99),
|
||||
"max": max(values),
|
||||
}
|
||||
|
||||
|
||||
def read_csv(path: Path) -> list[dict[str, str]]:
|
||||
with path.open(newline="") as stream:
|
||||
return list(csv.DictReader(stream))
|
||||
|
||||
|
||||
def find_one(root: Path, name: str) -> Path:
|
||||
matches = list(root.glob(f"**/{name}"))
|
||||
if len(matches) != 1:
|
||||
raise ValueError(f"expected one {name} below {root}, found {matches}")
|
||||
return matches[0]
|
||||
|
||||
|
||||
def prom_counter(path: Path, name: str) -> float:
|
||||
values = []
|
||||
with path.open() as stream:
|
||||
for line in stream:
|
||||
if line.startswith(name + "{") or line.startswith(name + " "):
|
||||
values.append(float(line.rsplit(maxsplit=1)[1]))
|
||||
if not values:
|
||||
raise ValueError(f"{path}: missing Prometheus counter {name}")
|
||||
return sum(values)
|
||||
|
||||
|
||||
def prefix_cache_delta(root: Path) -> dict[str, float]:
|
||||
before = root / "metrics/before.prom"
|
||||
after = root / "metrics/after.prom"
|
||||
deltas = {
|
||||
name: prom_counter(after, name) - prom_counter(before, name)
|
||||
for name in PROM_COUNTERS
|
||||
}
|
||||
queries = deltas[PROM_COUNTERS[0]]
|
||||
hits = deltas[PROM_COUNTERS[1]]
|
||||
if queries <= 0 or hits < 0 or hits > queries:
|
||||
raise ValueError(f"{root}: invalid prefix counter deltas {deltas}")
|
||||
return {
|
||||
"query_tokens": queries,
|
||||
"hit_tokens": hits,
|
||||
"hit_ratio": hits / queries,
|
||||
}
|
||||
|
||||
|
||||
def real_decode_batch(root: Path) -> dict[str, Any]:
|
||||
counts: Counter[int] = Counter()
|
||||
mixed_steps = 0
|
||||
for path in sorted(root.rglob("server.log")):
|
||||
with path.open(errors="replace") as stream:
|
||||
for line in stream:
|
||||
match = ITERATION_RE.search(line)
|
||||
if match is None:
|
||||
continue
|
||||
context_requests = int(match.group("context_requests"))
|
||||
generation_requests = int(match.group("generation_requests"))
|
||||
generation_tokens = int(match.group("generation_tokens"))
|
||||
if context_requests:
|
||||
mixed_steps += 1
|
||||
continue
|
||||
if generation_requests and generation_tokens == generation_requests:
|
||||
counts[generation_requests] += 1
|
||||
|
||||
if not counts:
|
||||
return {
|
||||
"steps": 0,
|
||||
"mixed_steps_excluded": mixed_steps,
|
||||
"max": None,
|
||||
"share_gt_1": None,
|
||||
"histogram": {},
|
||||
}
|
||||
steps = sum(counts.values())
|
||||
return {
|
||||
"steps": steps,
|
||||
"mixed_steps_excluded": mixed_steps,
|
||||
"max": max(counts),
|
||||
"share_gt_1": sum(value for key, value in counts.items() if key > 1) / steps,
|
||||
"histogram": {str(key): value for key, value in sorted(counts.items())},
|
||||
}
|
||||
|
||||
|
||||
def load_sim(root: Path, trace: list[dict[str, str]], trace_sha: str) -> dict[str, Any]:
|
||||
manifest = json.loads((root / "manifest.json").read_text())
|
||||
if manifest["trace_sha256"] != trace_sha:
|
||||
raise ValueError(f"{root}: sim/input trace SHA mismatch")
|
||||
rows = read_csv(find_one(root / "metrics", "request_metrics.csv"))
|
||||
if len(rows) != len(trace):
|
||||
raise ValueError(f"{root}: sim/input request count mismatch")
|
||||
values = {
|
||||
"ttft": [float(row["ttft"]) for row in rows],
|
||||
"tpot": [float(row["tpot"]) for row in rows if row["tpot"].strip()],
|
||||
"e2e": [float(row["request_e2e_time"]) for row in rows],
|
||||
"waiting": [float(row["request_waiting_time_total"]) for row in rows],
|
||||
}
|
||||
completions = [
|
||||
float(trace_row["arrived_at"]) + float(metric_row["request_e2e_time"]) / 1000
|
||||
for trace_row, metric_row in zip(trace, rows)
|
||||
]
|
||||
tail_index = max(range(len(completions)), key=completions.__getitem__)
|
||||
last_arrival = max(float(row["arrived_at"]) for row in trace)
|
||||
summary = json.loads((root / "summary.json").read_text())
|
||||
slo_pass = []
|
||||
for trace_row, metric_row in zip(trace, rows):
|
||||
input_tokens = int(trace_row["num_prefill_tokens"])
|
||||
ttft_threshold_ms = 1000 + 1000 * input_tokens / 8000
|
||||
tpot = (
|
||||
float(metric_row["tpot"])
|
||||
if metric_row["tpot"].strip()
|
||||
else None
|
||||
)
|
||||
slo_pass.append(
|
||||
float(metric_row["ttft"]) <= ttft_threshold_ms
|
||||
and (tpot is None or tpot <= 150)
|
||||
)
|
||||
return {
|
||||
"values": values,
|
||||
"tpot_by_min_output_tokens": {
|
||||
str(threshold): [
|
||||
float(metric_row["tpot"])
|
||||
for trace_row, metric_row in zip(trace, rows)
|
||||
if int(trace_row["num_decode_tokens"]) >= threshold
|
||||
and metric_row["tpot"].strip()
|
||||
]
|
||||
for threshold in TPOT_MIN_OUTPUT_TOKENS
|
||||
},
|
||||
"slo": {
|
||||
"passed": sum(slo_pass),
|
||||
"pass_rate": sum(slo_pass) / len(slo_pass),
|
||||
"feasible": sum(slo_pass) / len(slo_pass) >= SLO_TARGET_PASS_RATE,
|
||||
},
|
||||
"summary": summary,
|
||||
"drain": {
|
||||
"last_arrival_s": last_arrival,
|
||||
"last_completion_s": completions[tail_index],
|
||||
"tail_after_last_arrival_s": completions[tail_index] - last_arrival,
|
||||
"tail_driver": {
|
||||
"request_index": tail_index,
|
||||
"arrival_s": float(trace[tail_index]["arrived_at"]),
|
||||
"arrival_before_cutoff_s": last_arrival
|
||||
- float(trace[tail_index]["arrived_at"]),
|
||||
"input_tokens": int(trace[tail_index]["num_prefill_tokens"]),
|
||||
"output_tokens": int(trace[tail_index]["num_decode_tokens"]),
|
||||
"waiting_ms": values["waiting"][tail_index],
|
||||
"e2e_ms": values["e2e"][tail_index],
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def load_real(
|
||||
root: Path,
|
||||
input_manifest: dict[str, Any],
|
||||
trace: list[dict[str, str]],
|
||||
) -> dict[str, Any]:
|
||||
result_path = root / "results/result.json"
|
||||
result = json.loads(result_path.read_text())
|
||||
if result["contract"]["row_vector_sha256"] != input_manifest["paired_row_vector_sha256"]:
|
||||
raise ValueError(f"{root}: real/input row digest mismatch")
|
||||
requests = result["requests"]
|
||||
if len(requests) != len(trace) or not all(row["success"] for row in requests):
|
||||
raise ValueError(f"{root}: incomplete or failed real request vector")
|
||||
for index, (request, trace_row) in enumerate(zip(requests, trace)):
|
||||
observed = (int(request["input_tokens"]), int(request["requested_output_tokens"]))
|
||||
expected = (
|
||||
int(trace_row["num_prefill_tokens"]),
|
||||
int(trace_row["num_decode_tokens"]),
|
||||
)
|
||||
if observed != expected:
|
||||
raise ValueError(f"{root}: request {index} shape {observed} != {expected}")
|
||||
values = {
|
||||
metric: [
|
||||
float(request[f"{metric}_ms"])
|
||||
for request in requests
|
||||
if request.get(f"{metric}_ms") is not None
|
||||
]
|
||||
for metric in METRICS
|
||||
}
|
||||
completions = [
|
||||
float(request["admitted_s"]) + float(request["e2e_ms"]) / 1000
|
||||
for request in requests
|
||||
]
|
||||
tail_index = max(range(len(completions)), key=completions.__getitem__)
|
||||
last_arrival = max(float(request["scheduled_s"]) for request in requests)
|
||||
return {
|
||||
"root": str(root),
|
||||
"result_sha256": sha256(result_path),
|
||||
"values": values,
|
||||
"tpot_by_min_output_tokens": {
|
||||
str(threshold): [
|
||||
float(request["tpot_ms"])
|
||||
for request in requests
|
||||
if int(request["requested_output_tokens"]) >= threshold
|
||||
and request.get("tpot_ms") is not None
|
||||
]
|
||||
for threshold in TPOT_MIN_OUTPUT_TOKENS
|
||||
},
|
||||
"slo": {
|
||||
"passed": sum(bool(request["slo_pass"]) for request in requests),
|
||||
"pass_rate": sum(bool(request["slo_pass"]) for request in requests)
|
||||
/ len(requests),
|
||||
"feasible": sum(bool(request["slo_pass"]) for request in requests)
|
||||
/ len(requests)
|
||||
>= SLO_TARGET_PASS_RATE,
|
||||
},
|
||||
"summary": result["summary"],
|
||||
"prefix_cache": prefix_cache_delta(root),
|
||||
"decode_batch": real_decode_batch(root),
|
||||
"drain": {
|
||||
"last_arrival_s": last_arrival,
|
||||
"last_completion_s": completions[tail_index],
|
||||
"tail_after_last_arrival_s": completions[tail_index] - last_arrival,
|
||||
"tail_driver": {
|
||||
"request_index": tail_index,
|
||||
"arrival_s": float(requests[tail_index]["scheduled_s"]),
|
||||
"arrival_before_cutoff_s": last_arrival
|
||||
- float(requests[tail_index]["scheduled_s"]),
|
||||
"input_tokens": int(requests[tail_index]["input_tokens"]),
|
||||
"output_tokens": int(requests[tail_index]["requested_output_tokens"]),
|
||||
"admission_lag_ms": float(requests[tail_index]["admission_lag_ms"]),
|
||||
"e2e_ms": float(requests[tail_index]["e2e_ms"]),
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parse_args()
|
||||
input_manifest = json.loads((args.input_root / "manifest.json").read_text())
|
||||
trace_path = args.input_root / "frontier.csv"
|
||||
trace = read_csv(trace_path)
|
||||
if len(trace) != input_manifest["requests"]:
|
||||
raise ValueError("input manifest/trace request count mismatch")
|
||||
trace_sha = sha256(trace_path)
|
||||
sim = load_sim(args.sim_root, trace, trace_sha)
|
||||
reals = [load_real(root, input_manifest, trace) for root in args.real_root]
|
||||
pooled = {
|
||||
metric: [value for real in reals for value in real["values"][metric]]
|
||||
for metric in METRICS
|
||||
}
|
||||
latency = {}
|
||||
for metric in METRICS:
|
||||
real_dist = distribution(pooled[metric])
|
||||
real_per_trial = [
|
||||
distribution(real["values"][metric]) for real in reals
|
||||
]
|
||||
real_reference = {
|
||||
statistic: statistics.fmean(
|
||||
float(trial[statistic]) for trial in real_per_trial
|
||||
)
|
||||
for statistic in ("mean", "p50", "p90", "p95", "p99")
|
||||
}
|
||||
sim_dist = distribution(sim["values"][metric])
|
||||
latency[metric] = {
|
||||
"real": real_dist,
|
||||
"real_trial_statistic_mean": real_reference,
|
||||
"sim": sim_dist,
|
||||
"relative_bias_percent": {
|
||||
statistic: 100
|
||||
* (float(sim_dist[statistic]) - real_reference[statistic])
|
||||
/ real_reference[statistic]
|
||||
for statistic in ("mean", "p50", "p90", "p95", "p99")
|
||||
},
|
||||
"real_per_trial": real_per_trial,
|
||||
}
|
||||
tpot_sensitivity = {}
|
||||
for threshold in TPOT_MIN_OUTPUT_TOKENS:
|
||||
key = str(threshold)
|
||||
real_per_trial = [
|
||||
distribution(real["tpot_by_min_output_tokens"][key])
|
||||
for real in reals
|
||||
]
|
||||
real_values = [
|
||||
value
|
||||
for real in reals
|
||||
for value in real["tpot_by_min_output_tokens"][key]
|
||||
]
|
||||
sim_values = sim["tpot_by_min_output_tokens"][key]
|
||||
real_dist = distribution(real_values)
|
||||
real_reference = {
|
||||
statistic: statistics.fmean(
|
||||
float(trial[statistic]) for trial in real_per_trial
|
||||
)
|
||||
for statistic in ("mean", "p50", "p90", "p95", "p99")
|
||||
}
|
||||
sim_dist = distribution(sim_values)
|
||||
tpot_sensitivity[key] = {
|
||||
"real": real_dist,
|
||||
"real_trial_statistic_mean": real_reference,
|
||||
"real_per_trial": real_per_trial,
|
||||
"sim": sim_dist,
|
||||
"relative_bias_percent": {
|
||||
statistic: 100
|
||||
* (float(sim_dist[statistic]) - real_reference[statistic])
|
||||
/ real_reference[statistic]
|
||||
for statistic in ("mean", "p50", "p90", "p95", "p99")
|
||||
},
|
||||
}
|
||||
payload = {
|
||||
"schema": "frontier-code-trace-canary-analysis-v1",
|
||||
"topology": args.topology,
|
||||
"requests_per_trial": len(trace),
|
||||
"trials": len(reals),
|
||||
"input": {
|
||||
"manifest": str(args.input_root / "manifest.json"),
|
||||
"paired_row_vector_sha256": input_manifest["paired_row_vector_sha256"],
|
||||
"frontier_csv_sha256": trace_sha,
|
||||
},
|
||||
"latency_ms": latency,
|
||||
"tpot_by_min_output_tokens": tpot_sensitivity,
|
||||
"slo": {
|
||||
"definition": {
|
||||
"ttft_ms": "1000 + 1000 * input_tokens / 8000",
|
||||
"tpot_ms": 150,
|
||||
"target_pass_rate": SLO_TARGET_PASS_RATE,
|
||||
},
|
||||
"real_per_trial": [real["slo"] for real in reals],
|
||||
"sim": sim["slo"],
|
||||
"feasibility_flip": any(
|
||||
real["slo"]["feasible"] != sim["slo"]["feasible"]
|
||||
for real in reals
|
||||
),
|
||||
},
|
||||
"prefix_cache": {
|
||||
"real_per_trial": [real["prefix_cache"] for real in reals],
|
||||
"real_hit_ratio_mean": statistics.fmean(
|
||||
real["prefix_cache"]["hit_ratio"] for real in reals
|
||||
),
|
||||
"sim": sim["summary"]["prefix_cache"],
|
||||
},
|
||||
"drain": {
|
||||
"interpretation": (
|
||||
"Report the max-completion request explicitly; a response that "
|
||||
"arrived well before the cutoff can create a long drain tail "
|
||||
"without implying queue accumulation."
|
||||
),
|
||||
"real_per_trial": [real["drain"] for real in reals],
|
||||
"sim": sim["drain"],
|
||||
},
|
||||
"sim_decode_batch": sim["summary"]["decode_batch"],
|
||||
"real_decode_batch_per_trial": [real["decode_batch"] for real in reals],
|
||||
"real_artifacts": [
|
||||
{
|
||||
"root": real["root"],
|
||||
"result_sha256": real["result_sha256"],
|
||||
}
|
||||
for real in reals
|
||||
],
|
||||
}
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.output.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n")
|
||||
print(json.dumps({"output": str(args.output), "topology": args.topology}))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
382
runs/frontier-code-trace-v0/audit_code_trace.py
Normal file
382
runs/frontier-code-trace-v0/audit_code_trace.py
Normal file
@@ -0,0 +1,382 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Audit long code traces before choosing a replay window and max model length."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import math
|
||||
import statistics
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable, Sequence
|
||||
|
||||
|
||||
BLOCK_SIZE_CANDIDATES = (16, 32, 64, 128, 256, 512, 1024)
|
||||
MAX_MODEL_LEN_CANDIDATES = (40960, 65536, 98304, 131072, 262144)
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--trace-root", type=Path)
|
||||
parser.add_argument("--source", type=Path, action="append")
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
parser.add_argument("--min-minutes", type=int, default=60)
|
||||
parser.add_argument("--max-minutes", type=int, default=75)
|
||||
parser.add_argument("--bin-seconds", type=int, default=60)
|
||||
parser.add_argument("--max-acceptable-gap-s", type=float, default=5.0)
|
||||
parser.add_argument("--model-position-limit", type=int, default=262144)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def percentile(values: Sequence[int | float], fraction: float) -> float | None:
|
||||
if not values:
|
||||
return None
|
||||
ordered = sorted(float(value) for value in values)
|
||||
position = (len(ordered) - 1) * fraction
|
||||
lower = math.floor(position)
|
||||
upper = math.ceil(position)
|
||||
if lower == upper:
|
||||
return ordered[lower]
|
||||
return ordered[lower] * (upper - position) + ordered[upper] * (position - lower)
|
||||
|
||||
|
||||
def distribution(values: Sequence[int | float]) -> dict[str, int | float | None]:
|
||||
return {
|
||||
"count": len(values),
|
||||
"min": min(values) if values else None,
|
||||
"p50": percentile(values, 0.50),
|
||||
"p90": percentile(values, 0.90),
|
||||
"p95": percentile(values, 0.95),
|
||||
"p99": percentile(values, 0.99),
|
||||
"max": max(values) if values else None,
|
||||
"mean": statistics.fmean(values) if values else None,
|
||||
}
|
||||
|
||||
|
||||
def parse_hash_ids(value: Any) -> list[Any]:
|
||||
if isinstance(value, list):
|
||||
return value
|
||||
if isinstance(value, str):
|
||||
stripped = value.strip()
|
||||
if not stripped:
|
||||
return []
|
||||
if stripped.startswith("["):
|
||||
decoded = json.loads(stripped)
|
||||
if not isinstance(decoded, list):
|
||||
raise ValueError("hash_ids JSON must decode to a list")
|
||||
return decoded
|
||||
delimiter = "|" if "|" in stripped else ","
|
||||
return [part for part in stripped.split(delimiter) if part.strip()]
|
||||
if value is None:
|
||||
return []
|
||||
return [value]
|
||||
|
||||
|
||||
def iter_jsonl(path: Path) -> Iterable[tuple[int, dict[str, Any]]]:
|
||||
with path.open() as stream:
|
||||
for line_number, line in enumerate(stream, 1):
|
||||
if not line.strip():
|
||||
continue
|
||||
row = json.loads(line)
|
||||
if not isinstance(row, dict):
|
||||
raise ValueError(f"{path}:{line_number}: row must be an object")
|
||||
yield line_number, row
|
||||
|
||||
|
||||
def choose_window(
|
||||
*,
|
||||
counts: Sequence[int],
|
||||
max_gaps: Sequence[float],
|
||||
first_timestamp: float,
|
||||
min_minutes: int,
|
||||
max_minutes: int,
|
||||
bin_seconds: int,
|
||||
max_acceptable_gap_s: float,
|
||||
) -> dict[str, Any] | None:
|
||||
candidates = []
|
||||
for minutes in range(max_minutes, min_minutes - 1, -1):
|
||||
bins = math.ceil(minutes * 60 / bin_seconds)
|
||||
for start_bin in range(0, len(counts) - bins + 1):
|
||||
selected = counts[start_bin : start_bin + bins]
|
||||
mean = statistics.fmean(selected)
|
||||
cv = statistics.pstdev(selected) / mean if mean else math.inf
|
||||
max_gap = max(max_gaps[start_bin : start_bin + bins], default=0.0)
|
||||
candidates.append(
|
||||
{
|
||||
"_score": (
|
||||
max_gap > max_acceptable_gap_s,
|
||||
cv,
|
||||
max_gap,
|
||||
-minutes,
|
||||
start_bin,
|
||||
),
|
||||
"start_bin": start_bin,
|
||||
"minutes": minutes,
|
||||
"count_mean_per_bin": mean,
|
||||
"count_cv": cv,
|
||||
"count_min_per_bin": min(selected),
|
||||
"count_max_per_bin": max(selected),
|
||||
"max_gap_s": max_gap,
|
||||
}
|
||||
)
|
||||
if not candidates:
|
||||
return None
|
||||
chosen = min(candidates, key=lambda item: item["_score"])
|
||||
chosen.pop("_score")
|
||||
chosen["start_timestamp"] = first_timestamp + chosen["start_bin"] * bin_seconds
|
||||
chosen["end_timestamp"] = chosen["start_timestamp"] + chosen["minutes"] * 60
|
||||
return chosen
|
||||
|
||||
|
||||
def scan_source(path: Path, args: argparse.Namespace) -> dict[str, Any]:
|
||||
rows = 0
|
||||
source_rows = 0
|
||||
invalid_zero_token_rows = 0
|
||||
invalid_zero_token_examples: list[dict[str, Any]] = []
|
||||
first_timestamp = None
|
||||
last_timestamp = None
|
||||
previous_timestamp = None
|
||||
counts: Counter[int] = Counter()
|
||||
max_gaps: dict[int, float] = {}
|
||||
input_lengths: list[int] = []
|
||||
output_lengths: list[int] = []
|
||||
total_lengths: list[int] = []
|
||||
hash_rows = 0
|
||||
hash_matches = Counter()
|
||||
prompt_rows = 0
|
||||
sampling_rows = 0
|
||||
schema_keys: Counter[str] = Counter()
|
||||
for line_number, row in iter_jsonl(path):
|
||||
source_rows += 1
|
||||
missing = [
|
||||
key
|
||||
for key in ("timestamp", "input_length", "output_length")
|
||||
if key not in row
|
||||
]
|
||||
if missing:
|
||||
raise ValueError(f"{path}:{line_number}: missing required fields {missing}")
|
||||
timestamp = float(row["timestamp"])
|
||||
input_tokens = int(row["input_length"])
|
||||
output_tokens = int(row["output_length"])
|
||||
schema_keys.update(row.keys())
|
||||
if input_tokens <= 0 or output_tokens <= 0:
|
||||
invalid_zero_token_rows += 1
|
||||
if len(invalid_zero_token_examples) < 20:
|
||||
invalid_zero_token_examples.append(
|
||||
{
|
||||
"line_number": line_number,
|
||||
"chat_id": row.get("chat_id"),
|
||||
"timestamp": timestamp,
|
||||
"input_length": input_tokens,
|
||||
"output_length": output_tokens,
|
||||
}
|
||||
)
|
||||
continue
|
||||
if first_timestamp is None:
|
||||
first_timestamp = timestamp
|
||||
if previous_timestamp is not None and timestamp < previous_timestamp:
|
||||
raise ValueError(
|
||||
f"{path}:{line_number}: timestamp {timestamp} < {previous_timestamp}"
|
||||
)
|
||||
bin_index = math.floor((timestamp - first_timestamp) / args.bin_seconds)
|
||||
counts[bin_index] += 1
|
||||
if previous_timestamp is not None:
|
||||
previous_bin = math.floor(
|
||||
(previous_timestamp - first_timestamp) / args.bin_seconds
|
||||
)
|
||||
max_gaps[previous_bin] = max(
|
||||
max_gaps.get(previous_bin, 0.0),
|
||||
timestamp - previous_timestamp,
|
||||
)
|
||||
input_lengths.append(input_tokens)
|
||||
output_lengths.append(output_tokens)
|
||||
total_lengths.append(input_tokens + output_tokens)
|
||||
hashes = parse_hash_ids(row.get("hash_ids"))
|
||||
if hashes:
|
||||
hash_rows += 1
|
||||
for block_size in BLOCK_SIZE_CANDIDATES:
|
||||
if len(hashes) == math.ceil(input_tokens / block_size):
|
||||
hash_matches[block_size] += 1
|
||||
prompt_rows += int(
|
||||
isinstance(row.get("prompt"), (str, list)) and bool(row.get("prompt"))
|
||||
)
|
||||
sampling_rows += int("sampling_u" in row)
|
||||
rows += 1
|
||||
previous_timestamp = timestamp
|
||||
last_timestamp = timestamp
|
||||
if not rows or first_timestamp is None or last_timestamp is None:
|
||||
raise ValueError(f"{path}: empty trace")
|
||||
total_bins = math.floor((last_timestamp - first_timestamp) / args.bin_seconds) + 1
|
||||
chosen = choose_window(
|
||||
counts=[counts[index] for index in range(total_bins)],
|
||||
max_gaps=[max_gaps.get(index, 0.0) for index in range(total_bins)],
|
||||
first_timestamp=first_timestamp,
|
||||
min_minutes=args.min_minutes,
|
||||
max_minutes=args.max_minutes,
|
||||
bin_seconds=args.bin_seconds,
|
||||
max_acceptable_gap_s=args.max_acceptable_gap_s,
|
||||
)
|
||||
return {
|
||||
"source": str(path.resolve()),
|
||||
"rows": rows,
|
||||
"source_rows": source_rows,
|
||||
"invalid_zero_token_rows": invalid_zero_token_rows,
|
||||
"invalid_zero_token_fraction": invalid_zero_token_rows / source_rows,
|
||||
"invalid_zero_token_examples": invalid_zero_token_examples,
|
||||
"first_timestamp": first_timestamp,
|
||||
"last_timestamp": last_timestamp,
|
||||
"span_s": last_timestamp - first_timestamp,
|
||||
"request_rate_per_s": rows / max(last_timestamp - first_timestamp, 1.0),
|
||||
"input_length": distribution(input_lengths),
|
||||
"output_length": distribution(output_lengths),
|
||||
"total_length": distribution(total_lengths),
|
||||
"over_max_model_len": {
|
||||
str(limit): {
|
||||
"requests": sum(value > limit for value in total_lengths),
|
||||
"fraction": sum(value > limit for value in total_lengths) / rows,
|
||||
}
|
||||
for limit in MAX_MODEL_LEN_CANDIDATES
|
||||
},
|
||||
"hash_contract": {
|
||||
"rows_with_hash_ids": hash_rows,
|
||||
"candidate_exact_match_rows": {
|
||||
str(size): hash_matches[size] for size in BLOCK_SIZE_CANDIDATES
|
||||
},
|
||||
"exact_source_block_size": next(
|
||||
(
|
||||
size
|
||||
for size in BLOCK_SIZE_CANDIDATES
|
||||
if hash_rows and hash_matches[size] == hash_rows
|
||||
),
|
||||
None,
|
||||
),
|
||||
},
|
||||
"prompt_rows": prompt_rows,
|
||||
"sampling_u_rows": sampling_rows,
|
||||
"schema_field_counts": dict(sorted(schema_keys.items())),
|
||||
"stable_window": chosen,
|
||||
}
|
||||
|
||||
|
||||
def scan_window(source: Path, window: dict[str, Any]) -> dict[str, Any]:
|
||||
start = float(window["start_timestamp"])
|
||||
end = float(window["end_timestamp"])
|
||||
inputs: list[int] = []
|
||||
outputs: list[int] = []
|
||||
totals: list[int] = []
|
||||
for _, row in iter_jsonl(source):
|
||||
timestamp = float(row["timestamp"])
|
||||
if timestamp < start:
|
||||
continue
|
||||
if timestamp >= end:
|
||||
break
|
||||
input_tokens = int(row["input_length"])
|
||||
output_tokens = int(row["output_length"])
|
||||
if input_tokens <= 0 or output_tokens <= 0:
|
||||
continue
|
||||
inputs.append(input_tokens)
|
||||
outputs.append(output_tokens)
|
||||
totals.append(input_tokens + output_tokens)
|
||||
return {
|
||||
"requests": len(totals),
|
||||
"input_length": distribution(inputs),
|
||||
"output_length": distribution(outputs),
|
||||
"total_length": distribution(totals),
|
||||
"max_model_len_coverage": {
|
||||
str(limit): {
|
||||
"covered_requests": sum(value <= limit for value in totals),
|
||||
"excluded_requests": sum(value > limit for value in totals),
|
||||
"coverage": sum(value <= limit for value in totals) / len(totals),
|
||||
}
|
||||
for limit in MAX_MODEL_LEN_CANDIDATES
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def resolve_sources(args: argparse.Namespace) -> list[Path]:
|
||||
if args.source:
|
||||
return [path.resolve() for path in args.source]
|
||||
if args.trace_root is None:
|
||||
raise ValueError("provide --trace-root or one or more --source")
|
||||
sources = sorted(
|
||||
path.resolve()
|
||||
for path in args.trace_root.glob("*.jsonl")
|
||||
if "prompt" not in path.stem.lower()
|
||||
)
|
||||
if not sources:
|
||||
raise FileNotFoundError(f"no non-prompt JSONL files under {args.trace_root}")
|
||||
return sources
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parse_args()
|
||||
if not 0 < args.min_minutes <= args.max_minutes:
|
||||
raise ValueError("require 0 < min_minutes <= max_minutes")
|
||||
sources = resolve_sources(args)
|
||||
files = [scan_source(path, args) for path in sources]
|
||||
eligible = [item for item in files if item["stable_window"] is not None]
|
||||
if not eligible:
|
||||
chosen = None
|
||||
data_gate = "BLOCKED_NO_1H_WINDOW"
|
||||
else:
|
||||
chosen = min(
|
||||
eligible,
|
||||
key=lambda item: (
|
||||
item["stable_window"]["max_gap_s"] > args.max_acceptable_gap_s,
|
||||
item["stable_window"]["count_cv"],
|
||||
-item["stable_window"]["minutes"],
|
||||
item["source"],
|
||||
),
|
||||
)
|
||||
chosen["selected_window_stats"] = scan_window(
|
||||
Path(chosen["source"]), chosen["stable_window"]
|
||||
)
|
||||
exact_block_size = chosen["hash_contract"]["exact_source_block_size"]
|
||||
max_total = chosen["selected_window_stats"]["total_length"]["max"]
|
||||
data_gate = (
|
||||
"PASS"
|
||||
if exact_block_size is not None
|
||||
and max_total is not None
|
||||
and max_total <= args.model_position_limit
|
||||
else "BLOCKED_HASH_OR_POSITION_CONTRACT"
|
||||
)
|
||||
recommendation = None
|
||||
if chosen is not None:
|
||||
maximum = chosen["selected_window_stats"]["total_length"]["max"]
|
||||
recommendation = next(
|
||||
(
|
||||
limit
|
||||
for limit in MAX_MODEL_LEN_CANDIDATES
|
||||
if maximum <= limit <= args.model_position_limit
|
||||
),
|
||||
None,
|
||||
)
|
||||
payload = {
|
||||
"schema": "frontier-code-trace-audit-v1",
|
||||
"trace_root": str(args.trace_root.resolve()) if args.trace_root else None,
|
||||
"sources": [str(path) for path in sources],
|
||||
"window_policy": {
|
||||
"min_minutes": args.min_minutes,
|
||||
"max_minutes": args.max_minutes,
|
||||
"bin_seconds": args.bin_seconds,
|
||||
"max_acceptable_gap_s": args.max_acceptable_gap_s,
|
||||
"selection": "lowest density CV after rejecting anomalous-gap windows",
|
||||
},
|
||||
"model_position_limit": args.model_position_limit,
|
||||
"files": files,
|
||||
"selected": chosen,
|
||||
"max_model_len_recommendation": recommendation,
|
||||
"data_gate": data_gate,
|
||||
"runtime_gate": (
|
||||
"PENDING: vLLM startup must prove enough KV blocks and nonzero "
|
||||
"max concurrency at the recommended max_model_len for each TP"
|
||||
),
|
||||
}
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.output.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n")
|
||||
print(json.dumps({"data_gate": data_gate, "output": str(args.output)}))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
79
runs/frontier-code-trace-v0/check_longctx_profile_repeats.py
Normal file
79
runs/frontier-code-trace-v0/check_longctx_profile_repeats.py
Normal file
@@ -0,0 +1,79 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Check fresh-process repeat stability for the code long-context grid."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--first", type=Path, nargs="+", required=True)
|
||||
parser.add_argument("--second", type=Path, nargs="+", required=True)
|
||||
parser.add_argument("--output", type=Path, required=True)
|
||||
parser.add_argument("--max-relative-difference", type=float, default=0.05)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def load(paths: list[Path]) -> dict[tuple[int, str], dict]:
|
||||
rows: dict[tuple[int, str], dict] = {}
|
||||
for path in paths:
|
||||
payload = json.loads(path.read_text())
|
||||
for row in payload["rows"]:
|
||||
if row.get("error"):
|
||||
raise ValueError(
|
||||
f"{path}: failed profile row {row['config']['batch_spec']}"
|
||||
)
|
||||
key = (
|
||||
int(row["tensor_parallel_size"]),
|
||||
str(row["config"]["batch_spec"]),
|
||||
)
|
||||
if key in rows:
|
||||
raise ValueError(f"duplicate row {key}")
|
||||
rows[key] = row
|
||||
return rows
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parse_args()
|
||||
first = load(args.first)
|
||||
second = load(args.second)
|
||||
if first.keys() != second.keys():
|
||||
raise ValueError(
|
||||
f"repeat key mismatch: first_only={sorted(first.keys()-second.keys())}, "
|
||||
f"second_only={sorted(second.keys()-first.keys())}"
|
||||
)
|
||||
comparisons = []
|
||||
for key in sorted(first):
|
||||
left = float(first[key]["mean_time"])
|
||||
right = float(second[key]["mean_time"])
|
||||
relative = abs(left - right) / ((left + right) / 2)
|
||||
comparisons.append(
|
||||
{
|
||||
"tp": key[0],
|
||||
"batch_spec": key[1],
|
||||
"first_mean_s": left,
|
||||
"second_mean_s": right,
|
||||
"relative_difference": relative,
|
||||
"pass": relative <= args.max_relative_difference,
|
||||
}
|
||||
)
|
||||
maximum = max(item["relative_difference"] for item in comparisons)
|
||||
payload = {
|
||||
"schema": "frontier-code-longctx-repeat-check-v1",
|
||||
"threshold": args.max_relative_difference,
|
||||
"maximum_relative_difference": maximum,
|
||||
"status": "PASS" if maximum <= args.max_relative_difference else "FAIL",
|
||||
"comparisons": comparisons,
|
||||
}
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.output.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n")
|
||||
print(json.dumps({"status": payload["status"], "max": maximum}))
|
||||
if payload["status"] != "PASS":
|
||||
raise SystemExit(1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
346
runs/frontier-code-trace-v0/experiment-card.md
Normal file
346
runs/frontier-code-trace-v0/experiment-card.md
Normal file
@@ -0,0 +1,346 @@
|
||||
# EXP-CODE-TRACE:从 chat 1h trace 扩展到 code 与 phase-separated replay
|
||||
|
||||
> **状态:RUNNING(Phase A code P+D)。** A0 数据/profile、A1
|
||||
> max-length smoke、A2 paired canary 与 A3 sim calibration 已完成;
|
||||
> A4 第一批 61min real jobs 正在 `dash1`--`dash4` 运行。禁止使用 `dash0`。
|
||||
|
||||
## 目标与成功定义
|
||||
|
||||
当前 1h+ 证据只覆盖 Qwen3-30B-A3B 的生产 chat trace、prefill+decode(P+D)和亚临界负载。本 campaign 分两步扩展:
|
||||
|
||||
1. **主任务:** 使用 `~/ali-trace/trace-glm5.1-formatted/` 中的 1h+ code trace,先完成 P+D real-vs-Frontier 回放;
|
||||
2. **后续 phase matrix:** 对 chat/code 都补 prefill-only 和严格 decode-only。
|
||||
|
||||
本轮不是只看“能否跑完”。每个正式 cell 必须满足:同一 request vector、同一 arrival、同一 token shape、同一 prefix/initial-KV 合约、real 零失败、无持续 backlog,并同时报告 TTFT/TPOT/E2E、queue/batch、KV/prefix state 与 5min 分窗漂移。
|
||||
|
||||
## 当前决策快照
|
||||
|
||||
| 项目 | 当前结论 | 下一 gate |
|
||||
|---|---|---|
|
||||
| code P+D 数据 | 61min development window、long-context profile-v6、四个 paired full 输入已冻结 | 第一批 1h real 运行中 |
|
||||
| TP4 负载 | low/mid/near-knee=`rho 0.0002/0.0008/0.0016`;paired canary 的 TTFT/E2E、KV、batch 通过 | 三个 load 的 trial 1 运行中 |
|
||||
| TP2 负载 | `rho 0.0004` canary 暴露 TTFT p90 `-32.1%` bad case | 只跑 `rho 0.0002` 1h diagnostic,暂不铺满 |
|
||||
| code prefill-only | TP2 已冻结 `rho 0.0004/0.0008/0.0016`,`0.0032` 过载;TP4 到 `0.0032` 仍亚临界 | TP4 追加更高 rho 边界 |
|
||||
| strict decode-only | vLLM 0.20.0 有 `DecodeBenchConnector`;Frontier trace generator 尚不能注入 initial computed tokens | C0 contract canary,未进入正式结果 |
|
||||
|
||||
## 三种 workload mode 的冻结定义
|
||||
|
||||
| Mode | 保留 | 改写 | 主指标 | 明确不声称 |
|
||||
|---|---|---|---|---|
|
||||
| P+D | 原 ISL/OSL、arrival、session/prefix | 仅做 source block→16-token runtime block 映射 | TTFT、TPOT、E2E、hit ratio、batch/queue | 不代表 PD 分离 |
|
||||
| prefill-only | 原 ISL、arrival、session/prefix | OSL 固定为 1,real `min_tokens=max_tokens=1`,sim decode tokens=1 | TTFT、prefill service/tokens/s、prefix hit、queue | TPOT 不定义;1-token decode 只用于完成请求 |
|
||||
| strict decode-only | 原 OSL、context length、arrival burst | arrival 定义为 **KV-ready time**;请求进入 decode 时已有 ISL 长度的 initial KV | TPOT、decode tokens/s、batch/queue、preemption | 不包含 prefill 与 KV transfer latency,不把短 prompt proxy 称为 decode-only |
|
||||
|
||||
strict decode-only 必须同时具备:
|
||||
|
||||
- real:vLLM `DecodeBenchConnector`(或等价、经验证的 initial-KV 注入);
|
||||
- sim:Frontier request 在 admission 时已拥有相同长度/块布局的 computed KV;
|
||||
- 两侧都不在 decode critical path 重做 prefill;
|
||||
- arrival 以 KV-ready time 对齐。若只保留原 trace 的相对到达形状,结论限定为 decode engine compute/scheduling fidelity。
|
||||
|
||||
在该合约完成前,只允许跑并标注为 **decode-dominant proxy**,不能进入 strict decode-only 结果表。
|
||||
|
||||
## 为什么 code P+D 不能直接复用 chat 配置
|
||||
|
||||
已知历史探查显示 code trace ISL p90 约 81.9k,约 32.6% 请求超过旧 `40960` 上限;真实数值必须由本 campaign 重新审计。至少有四个独立适配面:
|
||||
|
||||
1. **Serving cap:** `max_model_len` 必须覆盖 `ISL+OSL`,不能只看 ISL,也不能静默丢掉超长请求;
|
||||
2. **KV capacity:** Qwen3-30B 模型 position limit 为 262144,但 TP1/2/4 在 H20 上是否有足够 KV blocks 是 runtime gate,不由 config.json 自动保证;
|
||||
3. **Prefix block:** code source hash 预计为 512-token block,chat harness 原先固定 64→16;
|
||||
4. **Profile support:** 当前修复后的 attention profile 只覆盖到约 32k KV context。即使 vLLM 能跑 128k,Frontier 对 32k–128k 仍会出 profile 支撑域;在补 long-context 网格前只能做诊断 replay,不能做 fidelity claim。
|
||||
|
||||
## Hypotheses
|
||||
|
||||
- **H-code-generalizes:** 在补齐 long-context profile 支撑域后,code P+D 的 TTFT/TPOT/E2E 分布统计偏差仍处于当前 chat 量级,且 1h 残差不发散。
|
||||
- **H-longctx-gap:** code 的主要新增 gap 来自 32k 以上 KV-context 外推;补到 trace p99/max 对应的网格后,TTFT bias 随 ISL 的二次项显著收敛。
|
||||
- **H-phase-specific:** prefill-only 主要暴露 long-context/profile gap;strict decode-only 主要暴露 batch-conditioned whole-layer service 与 scheduler fixed-point gap。二者不能用 P+D 的误差抵消来互相证明准确。
|
||||
|
||||
## Preflight gates(按顺序,任一失败即停止后续真机矩阵)
|
||||
|
||||
### G0:数据位置与 provenance
|
||||
|
||||
- 只读列举 `trace-glm5.1-formatted/*.jsonl`,记录文件大小与 SHA256;
|
||||
- 确认至少两个独立日期段:一个作为 development,一个 held-out;
|
||||
- 本机当前没有该目录;仓库历史记录的远端位置为
|
||||
`/home/admin/cpfs/wjh/ali-trace/trace-glm5.1-formatted/`。恢复机器后先确认 `~/ali-trace/...` 是否为同一路径/软链,不能假设。
|
||||
|
||||
### G1:1h window、schema 与 block contract
|
||||
|
||||
运行 `audit_code_trace.py`,要求:
|
||||
|
||||
- timestamp 单调,存在 60–75min 连续稳定窗口;
|
||||
- `timestamp/input_length/output_length` 全行存在;
|
||||
- `input_length<=0` 或 `output_length<=0` 的行不进入 replay,但必须计数并保留样例。已抽查的两条 0→0 行在 raw trace 中同时满足 `usage.total_tokens=0`、`response_message={}`,属于未发生模型执行的 source request,不是 full-cache decode;
|
||||
- `hash_ids` 数量与某个 source block size 在全行严格满足
|
||||
`ceil(ISL/source_block_size)`;预计值 512,但以审计结果为准;
|
||||
- 记录 ISL/OSL/ISL+OSL 的 p50/p90/p95/p99/max、gap、request rate、prompt/sampling 字段覆盖。
|
||||
|
||||
选择窗口后用 `prepare_code_window.py` 物化只读派生文件,并按 session root 生成确定性的 `sampling_u`。另一日期段不参与 rho 与 profile 选择。
|
||||
|
||||
### G2:`max_model_len` data gate
|
||||
|
||||
source-window audit 先用 `40960/65536/98304/131072/262144` 给出完整
|
||||
窗口上界;真实 server 则使用能 **100% 覆盖该 rho 实际 paired requests
|
||||
的 `ISL+OSL`** 的最小 16-token 对齐值。规则:
|
||||
|
||||
- sampling 只按 session-coherent `sampling_u`,不得按 token length 过滤;
|
||||
- full source window 的 cap 用于记录 workload envelope,不强迫低 rho cell
|
||||
为未被抽中的 outlier 预留 KV capacity;
|
||||
- 若某 paired cell max≤131072,使用 131072 或更小的对齐值;超过
|
||||
131072 时按该 cell 实际 max 向上对齐,而不是直接跳到 262144;
|
||||
- Frontier 的 trace max tokens、predictor max tokens/request、vLLM `--max-model-len` 三处使用同一个 manifest 值。
|
||||
|
||||
### G3:prompt 与 prefix fidelity
|
||||
|
||||
优先级:
|
||||
|
||||
1. 有对齐 prompt text sidecar:用 Qwen tokenizer 重分词,要求 token length 与 trace ISL 全行一致;
|
||||
2. trace 内已有 prompt text/token IDs:同样做长度与 hash relation 检查;
|
||||
3. 两者都没有:允许用 source hash 确定性展开为 synthetic Qwen token IDs,但结果降级为 **length/arrival/prefix-shape faithful**,不声称 prompt-content 或 MoE routing faithful。
|
||||
|
||||
不论走哪条路径,source→16 映射冲突、runtime identity collision、parent prefix violation 都必须为 0。P+D/prefill-only 两侧 prefix caching 同开;先用 5–10min TP4/MNS16 做 hit-ratio audit。
|
||||
|
||||
### G4:long-context profile support
|
||||
|
||||
现有 profile-v5 的 KV context 上界约 32k,对 code 不足。根据 development window 的 uncached-ISL 分布生成 profile-v6-code-longctx:
|
||||
|
||||
- full chunk:`q8k`,context 至少覆盖 40k/56k/72k/88k/104k/120k/128k;
|
||||
- tail chunk:从真实 `ISL mod 8192` 的 p50/p90 选择 2–4k/4–6k 代表点;
|
||||
- TP1/2/4 分开采集,复测 `q1ks8k/q8ks32k` anchor;
|
||||
- 每点至少两次 fresh-process repeat;CV≤5%,anchor drift≤10%;
|
||||
- profile max context 必须 ≥ development window p99;正式 max claim 要求 ≥ max。若只覆盖 p99,max 以上请求单独列为 out-of-support,不进入总体准确度数字。
|
||||
|
||||
这是 code P+D 正式 fidelity 的硬 gate。可以先用旧 profile 跑 diagnostic sim 来估 load,但不得与真机组成最终 gap。
|
||||
|
||||
### G5:vLLM max-length/KV runtime gate
|
||||
|
||||
对每个候选 topology(先 TP4,再 TP2,TP1 后置):
|
||||
|
||||
1. fresh server,以 manifest cap 启动;
|
||||
2. 记录 vLLM 版本、model config、GPU KV blocks、maximum concurrency、启动日志;
|
||||
3. 发 3 个单请求:ISL p50、p99、max(OSL=1),usage 必须逐 token 对齐;
|
||||
4. 发 5min sampled P+D canary,零 OOM/timeout/preemption storm;
|
||||
5. 只有 maximum concurrency>1 且 canary drain tail≤窗口时长 10% 才进入 rho calibration。
|
||||
|
||||
`max_model_len` 变大不等于每个请求都预占最大 KV,但会改变启动合法性与可表达的单请求上界;实际 KV 压力仍由并发 token state 决定。
|
||||
|
||||
对本模型,`VLLM_ALLOW_LONG_MAX_MODEL_LEN=1` 只放宽 scheduler/config
|
||||
校验,不会扩展模型内部 RoPE cache;server 还必须显式传
|
||||
`--hf-overrides '{"max_position_embeddings":147456}'`。runner 对
|
||||
`MAX_MODEL_LEN>40960` 自动同时设置这两层。长上下文 job 默认使用
|
||||
host-local vLLM compile cache;FlashInfer workspace 按 topology 复用,
|
||||
避免每个 rho/trial 重编译同一组 fused-MoE kernels。两者只影响启动,
|
||||
不进入 replay latency。
|
||||
|
||||
### G6:每种 mode 独立标定 rho
|
||||
|
||||
不能复用 P+D rho:
|
||||
|
||||
- P+D 同时按 raw/prefix-adjusted prefill tokens/s 与 decode tokens/s 看 knee;
|
||||
- prefill-only 因 OSL=1,重新按 prefill work 标定;
|
||||
- strict decode-only 因无 prefill,按 decode tokens/s 和 batch fixed point 标定。
|
||||
|
||||
每种 workload×mode 选择 `low/mid/near-knee` 三点;正式点必须亚临界:全请求完成、无持续 backlog、drain tail≤10%、waiting p99 不单调随时间增长。跨 knee 点若运行,只作为 overload boundary,不支持“不发散”结论。
|
||||
|
||||
`drain tail` 必须同时列出最后完成请求的 arrival、ISL、OSL、waiting
|
||||
和 E2E。早于 cutoff 到达但 OSL 很长的请求可以在最后 arrival 后继续
|
||||
decode;这属于 intrinsic response tail,不等价于 arrival cutoff 时仍有
|
||||
持续增长的 queue backlog。亚临界判断以 queue/waiting trajectory 和
|
||||
tail driver 分解共同决定,不能只用一个 drain 秒数。
|
||||
|
||||
### G7:strict decode-only capability gate
|
||||
|
||||
先在 10min synthetic trace 上验证:
|
||||
|
||||
- real connector 确认没有执行 prefill kernel;
|
||||
- Frontier ledger 第一个阶段就是 decode,computed tokens=ISL;
|
||||
- 相同 context length 下两侧 KV block count 一致;
|
||||
- connector preload/transfer 时间独立记账,不混入 TPOT;
|
||||
- decode batch telemetry 能覆盖 b1 到目标 batch。
|
||||
|
||||
若 vLLM 0.20 community stack 没有等价 connector,严格 case 保持 BLOCKED;可另跑 decode-dominant proxy,但单独命名和汇报。
|
||||
|
||||
## 正式实验矩阵与推进顺序
|
||||
|
||||
### Phase A:code P+D(第一优先级)
|
||||
|
||||
1. **A0 CPU/data:** G0–G4;
|
||||
2. **A1 max-len smoke:** TP4→TP2;TP1 只在 KV gate 通过后加入;
|
||||
3. **A2 paired 10min canary:** TP4/MNS16,low rho,real+sim;
|
||||
4. **A3 calibration:** 各 rho 只先跑 sim,冻结 low/mid/near-knee;
|
||||
5. **A4 full:** TP4/MNS16、TP2/MNS16 × 3 rho × 2 trial × 60–75min;
|
||||
6. **A5 held-out:** 只在 development window 判据冻结后,对第二日期段跑 TP4 的 mid/near-knee。
|
||||
|
||||
若某 topology 的 near-knee 过载,像现有 chat TP2/ρ0.01 一样排除,不为凑齐矩阵强跑。
|
||||
|
||||
当前状态:A0–A3 完成。A4 第一批为 TP4
|
||||
`rho={0.0002,0.0008,0.0016}` trial 1,以及 TP2 `rho=0.0002`
|
||||
trial 1 diagnostic;其余 TP2 cell 等该 diagnostic 验证 canary bad case
|
||||
后再决定是否扩展。
|
||||
|
||||
### Phase B:chat/code prefill-only
|
||||
|
||||
- 复用各自已物化 window,只把 OSL 改为 1;
|
||||
- primary:TP4/MNS16、TP2/MNS16 × 3 独立 rho × 2 trial;
|
||||
- 报 TTFT/CDF/quantiles、prefill tokens/s、prefix hit、waiting、chunk/context 分带 residual;
|
||||
- TPOT 记为 N/A,E2E 仅作为“一 token completion”辅助值;
|
||||
- code 必须继续使用 profile-v6 long-context;chat 使用已验证 profile-v5。
|
||||
|
||||
### Phase C:chat/code strict decode-only
|
||||
|
||||
先做 batch-sensitive screening,再决定是否铺满:
|
||||
|
||||
- **C0 capability canary:** 两 workload × TP4 × MNS{16,128},10min;
|
||||
- **C1 core full:** TP{2,4} × MNS{16,128} × rho{low,near-knee} × 2 trial;
|
||||
- **C2 conditional expansion:** 只有当 C1 的 batch 分布从 b≤8 跨到 b>8,或 accuracy gap 随 MNS 改变>5pp,才补 MNS{32,64} 与 mid rho。
|
||||
|
||||
decode profile/serving anchors 至少覆盖实际 batch p99。当前 whole-layer grid 只对少数 b≤8 有证据,且 b6 有长尾;在 MNS128 case 前必须补 b{1,2,4,8,16,32,64,128} 或实际访问 bucket,不能把 b8 常数外推到 b128。
|
||||
|
||||
已安装 vLLM 0.20.0 的 `DecodeBenchConnector` 会在首次 schedule 时把
|
||||
`request.num_tokens-num_computed_tokens-1` 个 token 标为 external,
|
||||
同步向已分配的每层 KV blocks 写 dummy non-zero values,再从最后一个
|
||||
prompt token 开始 forward。因此它适合测大 context 下的 decode
|
||||
compute/scheduling,但 connector fill 发生在 client admission 之后:
|
||||
fill time 必须单独记账并从 KV-ready arrival/TPOT 口径中排除。
|
||||
dummy KV 也不提供真实 prompt-content 或 MoE-routing fidelity。
|
||||
|
||||
Frontier commit `deadc4a3` 的 `Request` 已支持构造
|
||||
`num_processed_tokens`,但 `TraceReplayRequestGenerator` 不读取该列;
|
||||
因此 sim 侧仍需一个显式、可测试的 `initial_computed_tokens` trace
|
||||
contract。C0 必须同时证明 real 首个 model step 是 decode、Frontier
|
||||
首个 ledger stage 是 decode,之后才能解除 strict decode-only 的 BLOCKED。
|
||||
|
||||
## 指标与判据
|
||||
|
||||
共同口径:
|
||||
|
||||
- 分布统计偏差:`(sim statistic-real statistic)/real statistic`,不是 per-request MAPE;
|
||||
- mean/p50/p90/p99 与 empirical CDF;
|
||||
- 5min 分窗,前 15min warmup 不进漂移 slope;
|
||||
- batch histogram、time-weighted running/waiting、drain tail、preemption;
|
||||
- 两 trial pooled 结果和 trial-to-trial noise floor 分开报告。
|
||||
|
||||
判据分两层:
|
||||
|
||||
1. **准确度:** primary latency mean/p90/p99 的 |bias|≤15% 为强通过,15–30% 为有界但需标注 correction,>30% 立 bad case;任何 topology 排序或 SLO feasibility 翻转都单独判 failure,不能被平均值掩盖。
|
||||
2. **长时稳定:** `|residual Theil–Sen slope|×12 / real noise floor < 1` 为 H-BOUNDED;只适用于亚临界 cell。
|
||||
|
||||
mode-specific:
|
||||
|
||||
- P+D:TTFT/TPOT/E2E 全部 primary;
|
||||
- prefill-only:TTFT primary,TPOT N/A;
|
||||
- strict decode-only:TPOT primary,TTFT 仅表示 admission/connector overhead,不进入 compute-fidelity gate。
|
||||
|
||||
## 成本与调度
|
||||
|
||||
- Phase A core:12 个 60–75min jobs(2 topology×3 load×2 trial),约 15 host-hours;按 TP 加权约 45 H20-GPU-hours,加 2–4 个 smoke/canary;
|
||||
- Phase B 两 workload:24 个 full jobs,按相同 75min 上界约 90 H20-GPU-hours;
|
||||
- Phase C 不一次铺满。C0 4 个 10min canary;C1 32 个 full jobs;C2 按触发条件追加。
|
||||
|
||||
每个 job fresh server;只在 `dash1`–`dash4` 全 8 卡 idle/healthy 时启动。即使 TP2/TP4 job 只用部分 GPU,也不在同一 host 并跑,避免 fresh-server 空窗竞态。每一批使用新的 jobs TOML,现有 dispatcher 非幂等。
|
||||
|
||||
## 预期产物
|
||||
|
||||
- `inputs/code-audit.json`、`inputs/code-window/window-manifest.json`;
|
||||
- P+D/prefill-only 的 paired `frontier.csv`、`real_requests.jsonl` 与 manifest;
|
||||
- profile-v6-code-longctx raw/merged profile 与 variance report;
|
||||
- 每 cell real/sim request metrics、server telemetry、stage ledger;
|
||||
- `results/code-pd-fidelity.md`;
|
||||
- 最终 `chat/code × P+D/prefill-only/decode-only` compatibility table。
|
||||
|
||||
## 已知边界
|
||||
|
||||
- code trace 来自 GLM5.1 业务,serving model 是 Qwen3-30B;若无原 prompt text,测试只能保持 shape/prefix 结构,不能证明内容相关 routing fidelity;
|
||||
- `max_model_len=128k/256k` 解决的是接入上界,不自动解决 32k 以上 profile 外推;
|
||||
- strict decode-only 只测 decode engine;完整 PD 分离还需要单独建模 prefill、KV transfer、backpressure 与 KV-ready arrival。
|
||||
|
||||
## 执行记录(2026-07-23)
|
||||
|
||||
- fleet probe:dash1–dash4 均为 8×H20;32 张卡 memory.used=0、
|
||||
utilization=0、无 compute process、uncorrected ECC=0;
|
||||
- 两个 formatted trace 都严格满足 512-token source hash contract;
|
||||
- 0513:2,108,130 个有效请求、6090 个 zero-usage source 行;稳定
|
||||
development window=`[3480,7140)`,61min、1,078,928 请求;
|
||||
- 0529:1,977,423 个有效请求、6031 个 zero-usage source 行;冻结为
|
||||
held-out,稳定候选 window=`[2640,6240)`;
|
||||
- development window:ISL p50/p90/p99/max =
|
||||
20,051/88,224/125,803/202,371;OSL p50/p90/p99/max =
|
||||
78/758/6449/131,072;`ISL+OSL max=202,745`;
|
||||
- full-window 131072 coverage=99.399%,262144 coverage=100%。但
|
||||
session sampling 的候选 `rho<=0.0032` 实际 max total=137,016,因此
|
||||
primary server cap 将按最终 cell max 对齐,不为未抽中的 202k outlier
|
||||
直接预留 262k;
|
||||
- source 无 Qwen-aligned prompt/token IDs。raw canonical prompt 使用 GLM
|
||||
token contract,不能同时保持 Qwen token content 与 trace ISL;本 campaign
|
||||
采用 synthetic Qwen tokens 保持 length/hash/prefix shape,并降级内容 claim。
|
||||
- selected rho=0.0032 中有 1355 个可检查 parent links:tail rewrite
|
||||
p50/p90/p95/p99/max=1/1/4/57/169 个 source blocks,说明 coder
|
||||
`parent_chat_id` 不等价于 append-only prompt。source hash 序列作为 prefix
|
||||
truth;synthetic content block 生成后再计算 parent-sensitive runtime
|
||||
identities,避免“相同内容块出现在不同前缀后”造成 Frontier false hit。
|
||||
- 远端 Qwen3-30B `config.json` 的原生 position limit 是 40960
|
||||
(`rope_theta=1e6`,无 rope_scaling)。147456 profile smoke 在显式
|
||||
`VLLM_ALLOW_LONG_MAX_MODEL_LEN=1` 下成功;该 override 只支持
|
||||
performance/shape fidelity,不形成生成质量或模型长上下文正确性 claim,
|
||||
并作为 provenance 中的显式实验变量。
|
||||
- profile-v6-code-longctx 覆盖 TP1/2/4、KV context 到 131072,
|
||||
33 个 long-context rows;两次 fresh-process repeat 的最大相对差
|
||||
4.648%,旧 anchor drift 最大 1.7%。attention profile SHA256 =
|
||||
`fbcf7e1f95789a6f6d771e24d1fc60958b7daf04eb0db260d27869a19d71d550`。
|
||||
- TP4 `max_model_len=147456` smoke 已在 dash4 通过。server 日志同时确认
|
||||
`max_model_len=147456` 与 `hf_overrides.max_position_embeddings=147456`;
|
||||
20,051+78、119,702+68、136,774+242 三个 shape 均成功。对应
|
||||
TTFT=853.29/12,699.04/3,820.55ms,TPOT=16.65/7.21/8.13ms。
|
||||
最长请求的非单调 TTFT 来自 cold compile/cache state,因此这里只作为
|
||||
runtime support gate,不作为 profile accuracy 数据。
|
||||
- calibration 全部使用同一 profile-v6 SHA。TP4 的 `rho=0.0016`
|
||||
decode batch max=16、drain=21.07s,仍通过 10% 亚临界 gate;TP2 的
|
||||
`rho=0.0016` waiting p50=332.97s、drain=976.86s,明确过载并排除。
|
||||
完整 compact table 在 `results/calibration-summary.json`。
|
||||
- TP2/TP4 的 `rho={0.0002,0.0004,0.0008,0.0016}` 61min full paired
|
||||
inputs 已在 CPFS 物化;每个 paired row digest 和 Frontier CSV SHA
|
||||
均与 calibration input 逐项一致。manifest 副本在
|
||||
`results/paired-input-manifests/`。最大一个目录约 901MiB,不把大型
|
||||
token arrays 提交进 Git。
|
||||
- code prefill-only 的最大 calibration cache 已物化:
|
||||
3477 requests、总 prefill 115,828,371 tokens、OSL 全为 1,
|
||||
paired digest=`40865068e02414612ba1cd4595894e20e85e01fd34d73f8660185552d531ecea`。
|
||||
- 多 host 并发 server startup 暴露出 shared CPFS AOT cache 和每-job
|
||||
FlashInfer JIT 的 apparatus cost。它发生在 readiness 前,不进入 TTFT;
|
||||
runner commit `d5bb974` 改为长上下文默认使用 host-local vLLM cache,
|
||||
并按 topology 复用 FlashInfer workspace。
|
||||
- 第一轮 paired real canary 的 3 次旧 client 运行都只在同一个
|
||||
`106709+197` 请求失败,根因是 `return_token_ids` 把 100k+ prompt
|
||||
vector 放进单条 SSE event,超过 aiohttp 默认 512KiB line limit。
|
||||
commit `e1f2557` 把 exact client read buffer 提到 8MiB;700KiB
|
||||
单-event runtime 对照和随后 TP4×2、TP2×1 的 53/53 replay 均通过。
|
||||
- TP4 canary 的 real-vs-sim prefix hit ratio =
|
||||
`0.239908/0.239973`;pure-decode batch max 都为 4,
|
||||
`share(b>1)=15.87%/15.69%`(real 两 trial)vs `16.35%`(sim)。
|
||||
TTFT mean/p50/p90/p95/p99 bias =
|
||||
`-8.3/-4.1/-8.8/-13.1/-6.3%`,E2E =
|
||||
`+2.2/+6.1/+11.0/+1.4/-1.3%`。长 drain 的同一
|
||||
`61976+21361` 请求 real=92.61/92.34s、sim=91.37s,不是 backlog。
|
||||
- TP4 若把 OSL=4 请求纳入 TPOT,mean/p99 bias 会被单个
|
||||
`~213ms/token` 样本放大到 `-29.9%/-71.5%`;OSL≥8 后
|
||||
mean/p50/p90/p95/p99 bias =
|
||||
`+8.7/+8.2/+0.6/+13.4/+3.7%`。因此 raw TPOT 仍保留,但正式报告必须
|
||||
同时给 OSL threshold sensitivity,不能把短输出的三段 inter-token
|
||||
interval 当作稳定 decode service。
|
||||
- TP4 canary 两 trial 的 real SLO pass rate 都是 `50/53=94.34%`,
|
||||
sim 为 `52/53=98.11%`,在 95% feasibility threshold 上发生翻转;
|
||||
这由两个临界 TTFT 请求和上述 OSL=4 请求共同造成,作为明确 bad case
|
||||
进入 1h 检验,不能被总体 latency gap 掩盖。
|
||||
- TP2 canary 的 cache/batch/drain 仍对齐,但 TTFT p90 bias=`-32.1%`,
|
||||
E2E p90/p95=`-18.4%/-23.8%`。因此先只启动 low-rho 1h diagnostic,
|
||||
不直接铺满 TP2 六个正式 jobs。
|
||||
- code prefill-only 已完成 10-cell Frontier calibration。TP2
|
||||
`rho=0.0032` drain=1384.59s、waiting p50=751.01s,明确过载;
|
||||
`0.0004/0.0008/0.0016` 冻结为 low/mid/near-knee。TP4 到
|
||||
`rho=0.0032` 仍只有 9.09s drain,暂称 highest-tested,追加更高 rho
|
||||
后才冻结 near-knee。compact table 在
|
||||
`results/prefill-only-calibration-summary.json`。
|
||||
- 2026-07-23 18:37 UTC 启动 A4 wave 1:dash1=`TP4/rho0.0002/t1`、
|
||||
dash3=`TP4/rho0.0008/t1`、dash4=`TP4/rho0.0016/t1`、
|
||||
dash2=`TP2/rho0.0002/t1 diagnostic`;四台启动前再次确认 8×H20
|
||||
memory/utilization=0、无 compute process、uncorrected ECC=0。
|
||||
116
runs/frontier-code-trace-v0/prepare_code_window.py
Normal file
116
runs/frontier-code-trace-v0/prepare_code_window.py
Normal file
@@ -0,0 +1,116 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Materialize the stable code window selected by audit_code_trace.py."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--audit", type=Path, required=True)
|
||||
parser.add_argument("--output-root", type=Path, required=True)
|
||||
parser.add_argument("--sample-seed", type=int, default=20260723)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def session_uniform(seed: int, window_id: str, session_root: Any) -> float:
|
||||
payload = json.dumps(
|
||||
{"seed": seed, "window_id": window_id, "session_root": session_root},
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
).encode()
|
||||
return int.from_bytes(hashlib.blake2b(payload, digest_size=8).digest(), "big") / (
|
||||
1 << 64
|
||||
)
|
||||
|
||||
|
||||
def sha256(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as stream:
|
||||
for chunk in iter(lambda: stream.read(1 << 20), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parse_args()
|
||||
audit = json.loads(args.audit.read_text())
|
||||
if audit["data_gate"] != "PASS":
|
||||
raise ValueError(f"trace data gate is not PASS: {audit['data_gate']}")
|
||||
selected = audit["selected"]
|
||||
source = Path(selected["source"])
|
||||
window = selected["stable_window"]
|
||||
start = float(window["start_timestamp"])
|
||||
end = float(window["end_timestamp"])
|
||||
if args.output_root.exists():
|
||||
raise ValueError(f"refusing to overwrite {args.output_root}")
|
||||
args.output_root.mkdir(parents=True)
|
||||
destination = args.output_root / "code-raw-window.jsonl"
|
||||
root_of: dict[Any, Any] = {}
|
||||
request_count = 0
|
||||
with source.open() as input_stream, destination.open("w") as output_stream:
|
||||
for source_index, line in enumerate(input_stream):
|
||||
if not line.strip():
|
||||
continue
|
||||
row = json.loads(line)
|
||||
timestamp = float(row["timestamp"])
|
||||
if timestamp < start:
|
||||
continue
|
||||
if timestamp >= end:
|
||||
break
|
||||
if int(row["input_length"]) <= 0 or int(row["output_length"]) <= 0:
|
||||
continue
|
||||
chat = row.get("chat_id", source_index)
|
||||
parent = row.get("parent_chat_id")
|
||||
has_parent = parent not in (None, "", -1, "-1")
|
||||
session_root = root_of.get(parent, parent) if has_parent else chat
|
||||
root_of[chat] = session_root
|
||||
materialized = {
|
||||
**row,
|
||||
"source_index": source_index,
|
||||
"session_root": session_root,
|
||||
"sampling_u": session_uniform(
|
||||
args.sample_seed,
|
||||
f"code-{start:.6f}-{end:.6f}",
|
||||
session_root,
|
||||
),
|
||||
}
|
||||
output_stream.write(
|
||||
json.dumps(materialized, ensure_ascii=False, separators=(",", ":"))
|
||||
+ "\n"
|
||||
)
|
||||
request_count += 1
|
||||
expected = int(selected["selected_window_stats"]["requests"])
|
||||
if request_count != expected:
|
||||
raise ValueError(f"window request mismatch: materialized={request_count}, audit={expected}")
|
||||
manifest = {
|
||||
"schema": "frontier-code-window-v1",
|
||||
"audit": str(args.audit.resolve()),
|
||||
"audit_sha256": sha256(args.audit),
|
||||
"source": str(source.resolve()),
|
||||
"source_block_size": selected["hash_contract"]["exact_source_block_size"],
|
||||
"target_block_size": 16,
|
||||
"start_timestamp": start,
|
||||
"end_timestamp": end,
|
||||
"duration_s": end - start,
|
||||
"requests": request_count,
|
||||
"sample_seed": args.sample_seed,
|
||||
"sampling_rule": "session-coherent deterministic sampling_u",
|
||||
"max_model_len": audit["max_model_len_recommendation"],
|
||||
"window_stats": selected["selected_window_stats"],
|
||||
"raw_window": str(destination.resolve()),
|
||||
"raw_window_sha256": sha256(destination),
|
||||
}
|
||||
(args.output_root / "window-manifest.json").write_text(
|
||||
json.dumps(manifest, indent=2, sort_keys=True) + "\n"
|
||||
)
|
||||
print(json.dumps({"requests": request_count, "output_root": str(args.output_root)}))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,277 @@
|
||||
time_stats.attn_input_reshape.min,time_stats.attn_input_reshape.max,time_stats.attn_input_reshape.mean,time_stats.attn_input_reshape.median,time_stats.attn_input_reshape.std,time_stats.attn_kv_cache_save.min,time_stats.attn_kv_cache_save.max,time_stats.attn_kv_cache_save.mean,time_stats.attn_kv_cache_save.median,time_stats.attn_kv_cache_save.std,time_stats.attn_prefill.min,time_stats.attn_prefill.max,time_stats.attn_prefill.mean,time_stats.attn_prefill.median,time_stats.attn_prefill.std,time_stats.attn_decode.min,time_stats.attn_decode.max,time_stats.attn_decode.mean,time_stats.attn_decode.median,time_stats.attn_decode.std,time_stats.attn_output_reshape.min,time_stats.attn_output_reshape.max,time_stats.attn_output_reshape.mean,time_stats.attn_output_reshape.median,time_stats.attn_output_reshape.std,n_embd,n_q_head,n_kv_head,block_size,num_tensor_parallel_workers,max_model_len,batch_size,prefill_chunk_size,kv_cache_size,is_prefill,attention_backend,is_mixed_batch,mode,seq_lens,total_tokens,max_seq_len,min_seq_len,avg_seq_len,equal_seq_len,seq_len_variance,seq_len_std,seq_len_cv,is_chunked_prefill_sample,chunk_start_token,chunk_end_token,total_prefill_tokens,profiling_precision,model_arch,quant_signature,measurement_type,is_true_mixed_batch,prefill_seq_lens,prefill_kv_cache_sizes,decode_kv_cache_sizes,num_prefill_seqs,num_decode_seqs,decode_batch_size,total_batch_size,total_decode_tokens,decode_avg_kv_cache_size,batch_composition_ratio,batch_spec,projection_policy
|
||||
0.0,0.0,0.0,0.0,0.0,0.01414399966597557,0.028863999992609024,0.019705599918961526,0.01771199982613325,0.005157200849836681,0.047968000173568726,0.07046400010585785,0.05810240097343922,0.05810240097343922,0.007477463486041561,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,64,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[64],64,64,64,64.0,True,0.0,0.0,0.0,False,0.0,64.0,64,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q64,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.04947200044989586,0.020412799902260303,0.01635199971497059,0.010107497379722417,0.046560000628232956,0.08323200047016144,0.05587520003318787,0.05587520003318787,0.011126758739503428,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,128,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[128],128,128,128,128.0,True,0.0,0.0,0.0,False,0.0,128.0,128,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01484800036996603,0.022207999601960182,0.017033600155264138,0.015312000177800655,0.002819235991970241,0.05104000121355057,0.07692799717187881,0.056396800279617305,0.056396800279617305,0.007481982178637539,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,256,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[256],256,256,256,256.0,True,0.0,0.0,0.0,False,0.0,256.0,256,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q256,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015072000212967396,0.022272000089287758,0.01706880023702979,0.01616000011563301,0.002460889579319197,0.06931199878454208,0.0838719978928566,0.07432000041007995,0.07432000041007995,0.004777766433175866,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,512,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,False,0.0,512.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.018592000007629395,0.028543999418616295,0.02095999978482723,0.019183999858796597,0.003198175496053494,0.12179200351238251,0.15408000349998474,0.1307712011039257,0.1307712011039257,0.00858807797538298,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,1024,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[1024],1024,1024,1024,1024.0,True,0.0,0.0,0.0,False,0.0,1024.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.027775999158620834,0.03385600075125694,0.030131200328469276,0.029680000618100166,0.0021152558103575215,0.32678401470184326,0.3450239896774292,0.33396480381488797,0.33396480381488797,0.0045872424917606375,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,2048,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,2048.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04438399896025658,0.05084799975156784,0.046540799736976626,0.04531199857592583,0.002277905811237223,1.0959680080413818,1.1151360273361206,1.0999775886535645,1.0999775886535645,0.005694403246120485,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.078015998005867,0.08691199868917465,0.08114239946007729,0.08019199967384338,0.00292795706334475,4.070400238037109,4.113152027130127,4.087088012695312,4.087088012695312,0.013660567012509554,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.016063999384641647,0.05167999863624573,0.022115200012922286,0.017583999782800674,0.010340094822340818,0.05196800082921982,0.09011200070381165,0.06328320093452933,0.06328320093452933,0.012557341255467452,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,64,448.0,True,FLASH_ATTN,False,vllm020_batch_spec,[64],64,64,64,64.0,True,0.0,0.0,0.0,True,448.0,512.0,64,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q64s512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01583999954164028,0.026623999699950218,0.018927999772131443,0.017376000061631203,0.003514650316260619,0.06681600213050842,0.07993599772453308,0.0725280001759529,0.0725280001759529,0.004343502558613716,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,128,896.0,True,FLASH_ATTN,False,vllm020_batch_spec,[128],128,128,128,128.0,True,0.0,0.0,0.0,True,896.0,1024.0,128,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q128s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01648000068962574,0.030880000442266464,0.01945280022919178,0.017967999912798405,0.004096211183007485,0.1311360001564026,0.1546880006790161,0.13908160030841826,0.13908160030841826,0.007511906874366178,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,256,1792.0,True,FLASH_ATTN,False,vllm020_batch_spec,[256],256,256,256,256.0,True,0.0,0.0,0.0,True,1792.0,2048.0,256,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q256s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.017855999991297722,0.03558399900794029,0.020851199887692927,0.018559999763965607,0.005235911594130716,0.32950401306152344,0.350271999835968,0.33912960588932034,0.33912960588932034,0.006027400986663648,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,512,3584.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,True,3584.0,4096.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.019328000023961067,0.040608000010252,0.022790400311350822,0.020704000256955624,0.006113051965778337,1.1415679454803467,1.1518720388412476,1.144483208656311,1.144483208656311,0.0032332311374389127,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,1024,7168.0,True,FLASH_ATTN,False,vllm020_batch_spec,[1024],1024,1024,1024,1024.0,True,0.0,0.0,0.0,True,7168.0,8192.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1ks8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015807999297976494,0.030688000842928886,0.019596799835562707,0.01774400006979704,0.004343384771033462,0.0,0.0,0.0,0.0,0.0,0.049056001007556915,0.07580800354480743,0.05948160067200661,0.05948160067200661,0.009031541471446955,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01603199914097786,0.02486399933695793,0.01923839971423149,0.018079999834299088,0.0032282528537266424,0.0,0.0,0.0,0.0,0.0,0.05142400041222572,0.07353600114583969,0.059328000620007516,0.059328000620007516,0.0073307807735143084,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.016543999314308167,0.03977600112557411,0.021379199624061585,0.018511999398469925,0.006593576176246171,0.0,0.0,0.0,0.0,0.0,0.0488319993019104,0.06435199826955795,0.05479039996862411,0.05479039996862411,0.005672522998491864,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01635199971497059,0.02844800055027008,0.019267200119793416,0.017952000722289085,0.0035068687666949577,0.0,0.0,0.0,0.0,0.0,0.049855999648571014,0.07798399776220322,0.05986879989504815,0.05986879989504815,0.01043914754878828,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.016383999958634377,0.026079999282956123,0.01923519968986511,0.017791999503970146,0.0032161974331284568,0.0,0.0,0.0,0.0,0.0,0.058111999183893204,0.1045759990811348,0.06708480007946492,0.06708480007946492,0.013479022462646494,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014720000326633453,0.04057599976658821,0.019100800156593323,0.015455999877303839,0.007512281243011577,0.0,0.0,0.0,0.0,0.0,0.05363199859857559,0.07782399654388428,0.06090559959411622,0.06090559959411622,0.007544620176348091,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014431999996304512,0.02191999927163124,0.016684799920767546,0.01563199982047081,0.0024293118621811216,0.0,0.0,0.0,0.0,0.0,0.0629120022058487,0.07891199737787247,0.06891520097851753,0.06891520097851753,0.005472695665695425,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014560000039637089,0.038943998515605927,0.018313600029796363,0.01561600062996149,0.007127270260115769,0.0,0.0,0.0,0.0,0.0,0.08675199747085571,0.10662399977445602,0.09391999915242194,0.09391999915242194,0.006988099589086635,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014527999795973301,0.054687999188899994,0.021439999900758268,0.01539199985563755,0.012052764849597775,0.0,0.0,0.0,0.0,0.0,0.13488000631332397,0.1528639942407608,0.1431359991431236,0.1431359991431236,0.005436271464033599,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.041919998824596405,0.01899839974939823,0.015343999955803156,0.007989843526623287,0.0,0.0,0.0,0.0,0.0,0.06176000088453293,0.08374399691820145,0.06747519969940186,0.06747519969940186,0.0066067747128778,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.016256000846624374,0.11353600025177002,0.042761600017547606,0.028960000723600388,0.029104301538020762,0.0,0.0,0.0,0.0,0.0,0.09734400361776352,0.14422400295734406,0.11392960175871848,0.11392960175871848,0.013198594600417867,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014688000082969666,0.034143999218940735,0.018918400071561335,0.01643200032413006,0.005500943993080684,0.0,0.0,0.0,0.0,0.0,0.12918399274349213,0.15087999403476715,0.13807999789714814,0.13807999789714814,0.007658330538677587,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.016063999384641647,0.03641600161790848,0.0198208000510931,0.01780799962580204,0.0057264128169845765,0.0,0.0,0.0,0.0,0.0,0.22099199891090393,0.23904000222682953,0.2293503984808922,0.2293503984808922,0.004861342907006028,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014751999638974667,0.035840000957250595,0.018908800091594458,0.015792000107467175,0.0064374817924757475,0.0,0.0,0.0,0.0,0.0,0.10134399682283401,0.12201599776744843,0.10896319895982742,0.10896319895982742,0.006336330809165179,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.013856000266969204,0.03417599946260452,0.017846399918198586,0.014800000004470348,0.006495007539635255,0.0,0.0,0.0,0.0,0.0,0.13126400113105774,0.15561600029468536,0.1389280006289482,0.1389280006289482,0.008381472811075022,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014431999996304512,0.03519999980926514,0.019168000388890504,0.01600000075995922,0.005995522477654695,0.0,0.0,0.0,0.0,0.0,0.21728000044822693,0.2343679964542389,0.2231455981731415,0.2231455981731415,0.004720730646739123,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014047999866306782,0.03670400008559227,0.018441599886864425,0.015023999847471714,0.006596535162793127,0.0,0.0,0.0,0.0,0.0,0.39321601390838623,0.4524799883365631,0.4058080047369003,0.4058080047369003,0.01578349755088941,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014336000196635723,0.022143999114632607,0.016912000067532063,0.015168000012636185,0.0028156089295136347,0.0,0.0,0.0,0.0,0.0,0.15587200224399567,0.3079040050506592,0.17838079929351805,0.17838079929351805,0.04355865575265927,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014271999709308147,0.02112000063061714,0.015516800060868263,0.01488000014796853,0.001940870731593904,0.0,0.0,0.0,0.0,0.0,0.21587200462818146,0.23561599850654602,0.22250880002975468,0.22250880002975468,0.006181951170646666,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015552000142633915,0.039264000952243805,0.023609600123018028,0.02131200022995472,0.007236979625711548,0.0,0.0,0.0,0.0,0.0,0.408735990524292,0.470335990190506,0.4336863994598388,0.4336863994598388,0.01844662383160074,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.025407999753952026,0.016227199975401164,0.014960000291466713,0.0031617375441736185,0.0,0.0,0.0,0.0,0.0,0.7412800192832947,0.7627840042114258,0.7464000046253203,0.7464000046253203,0.006112167448837547,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01462399959564209,0.02812799997627735,0.020652799773961304,0.021359999664127827,0.004308706957613102,0.028383498565450627,0.039859687970646644,0.032492258074592426,0.032492258074592426,0.00453597266208597,0.029312501176103633,0.04116431058787463,0.03355574193327539,0.03355574193327539,0.004684436757701648,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,72,,,,,,,,False,,,64,BF16,generic,none,CUDA_EVENT,True,[64],[0],"[512, 512, 512, 512, 512, 512, 512, 512]",1,8,8,9,8,512.0,0.1111111111111111,q64_8q1s512,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.015552000142633915,0.034143999218940735,0.023171199765056372,0.024255999363958836,0.005908614918096971,0.03333159243114438,0.038935341782478095,0.03580428402241854,0.03580428402241854,0.002082270297044095,0.03633240903369937,0.04244065945634484,0.03902771507087562,0.03902771507087562,0.0022697354261490147,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,136,,,,,,,,False,,,128,BF16,generic,none,CUDA_EVENT,True,[128],[0],"[1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024]",1,8,8,9,8,1024.0,0.1111111111111111,q128_8q1s1k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.02611199952661991,0.019635199941694735,0.018112000077962875,0.0038298050749257795,0.04189529417991216,0.057484239920526384,0.04744885718421094,0.04744885718421094,0.004779830748455743,0.051672703037266184,0.07089975418195164,0.05852234134479412,0.05852234134479412,0.005895334539786378,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,144,,,,,,,,False,,,128,BF16,generic,none,CUDA_EVENT,True,[128],[0],"[1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024]",1,16,16,17,16,1024.0,0.058823529411764705,q128_16q1s1k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014720000326633453,0.029311999678611755,0.018662399891763926,0.01673599984496832,0.0044017112162725355,0.04322973959325901,0.05049827064705393,0.04535414343408448,0.04535414343408448,0.0022944156000240697,0.08733025617719538,0.10201372836398578,0.09162185574241775,0.09162185574241775,0.004635047631846023,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,272,,,,,,,,False,,,256,BF16,generic,none,CUDA_EVENT,True,[256],[0],"[2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048]",1,16,16,17,16,2048.0,0.058823529411764705,q256_16q1s2k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014592000283300877,0.026367999613285065,0.016336000058799982,0.01515199989080429,0.003415353455946402,0.06031842775160765,0.06618323188375198,0.06274415549817247,0.06274415549817247,0.001985647289644255,0.14768157653992678,0.16204076249052324,0.15362064543185072,0.15362064543185072,0.004861590945216247,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,288,,,,,,,,False,,,256,BF16,generic,none,CUDA_EVENT,True,[256],[0],"[2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048]",1,32,32,33,32,2048.0,0.030303030303030304,q256_32q1s2k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014751999638974667,0.02454400062561035,0.017167999967932702,0.016191999427974224,0.0028685016454498436,0.09128700688359712,0.09689150775996329,0.09360396051475776,0.09360396051475776,0.0013477542744916764,0.2740889887523892,0.2909164873209846,0.2810456356995366,0.2810456356995366,0.004046628526808561,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,544,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,32,32,33,32,4096.0,0.030303030303030304,q512_32q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.01881599985063076,0.03097599931061268,0.024598400108516216,0.024848000146448612,0.0037539437391565975,0.1035249255866932,0.10603132147437412,0.10399797220840973,0.10399797220840973,0.0007025109429056814,0.5652750707893444,0.5789606938874117,0.5678580377517648,0.5678580377517648,0.003835906384194897,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,576,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,64,64,65,64,4096.0,0.015384615384615385,q512_64q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.018624000251293182,0.043487999588251114,0.024460799992084503,0.019600000232458115,0.008922081629781394,0.196169204945307,0.2076140047945799,0.2008444429250931,0.2008444429250931,0.00394350953292805,1.1196708009293268,1.1849940417370974,1.1463555573610091,1.1463555573610091,0.022508285530529783,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1088,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192]",1,64,64,65,64,8192.0,0.015384615384615385,q1k_64q1s8k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.027135999873280525,0.03667199984192848,0.02945920005440712,0.028256000019609928,0.0029446678408169553,0.37757279619664613,0.3898113624476372,0.38075520430942184,0.38075520430942184,0.0036600203831042254,0.2522831942132049,0.2604606493092597,0.25440958703617433,0.25440958703617433,0.0024455194930253126,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,32,32,33,32,4096.0,0.030303030303030304,q2k_32q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.0435199998319149,0.049695998430252075,0.04502719938755036,0.04395199939608574,0.002035785660169308,1.1048984388245497,1.1189053886476108,1.1112371236754128,1.1112371236754128,0.0050220696724624985,0.1395495077239122,0.14131859607071193,0.14035008841031085,0.14035008841031085,0.0006342911944856293,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,4112,,,,,,,,False,,,4096,BF16,generic,none,CUDA_EVENT,True,[4096],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,16,16,17,16,4096.0,0.058823529411764705,q4k_16q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.018783999606966972,0.024288000538945198,0.020627199858427047,0.019567999988794327,0.0019457052717059358,0.09455999732017517,0.12185599654912949,0.10207359939813614,0.10207359939813614,0.00753346544014261,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,2,1024,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512]",1024,512,512,512.0,True,0.0,0.0,0.0,False,0.0,1024.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,2q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.027295999228954315,0.034752000123262405,0.02943360023200512,0.028528000228106976,0.0023159160681123767,0.14416000247001648,0.15904000401496887,0.14979200065135959,0.14979200065135959,0.00480624675866005,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,4,2048,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512]",2048,512,512,512.0,True,0.0,0.0,0.0,False,0.0,2048.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,4q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.043455999344587326,0.05084799975156784,0.045500800386071204,0.04411200061440468,0.002490556062831049,0.24454399943351746,0.25865599513053894,0.2504959970712662,0.2504959970712662,0.003778494140301353,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512, 512, 512, 512, 512]",4096,512,512,512.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.07673600316047668,0.08374399691820145,0.07970559895038605,0.07873599976301193,0.0022907976135004057,0.445248007774353,0.4758400022983551,0.45409599840641024,0.45409599840641024,0.009133230340383306,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512]",8192,512,512,512.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.043296001851558685,0.051072001457214355,0.044972800090909,0.04399999976158142,0.002451216824441962,0.6147199869155884,0.6290879845619202,0.6204927921295166,0.6204927921295166,0.004344846739788608,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,2,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[2048, 2048]",4096,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,2q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.07737600058317184,0.09123200178146362,0.08094720020890236,0.07980800047516823,0.004065264798368611,1.1744320392608643,1.1887680292129517,1.178323209285736,1.178323209285736,0.004395876469319665,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,4,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[2048, 2048, 2048, 2048]",8192,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,4q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014688000082969666,0.03254399821162224,0.01810879958793521,0.01611199975013733,0.005058478889747962,0.0,0.0,0.0,0.0,0.0,0.06339199841022491,0.0841279998421669,0.07019519805908202,0.07019519805908202,0.006070126182471763,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01484800036996603,0.030271999537944794,0.018124799989163876,0.015711999498307705,0.004696400814632773,0.0,0.0,0.0,0.0,0.0,0.2739199995994568,0.2922239899635315,0.28281279802322384,0.28281279802322384,0.00633037978599564,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014495999552309513,0.022431999444961548,0.016844799742102623,0.015471999999135733,0.0027624803153841917,0.0,0.0,0.0,0.0,0.0,0.3909119963645935,0.4160960018634796,0.39996159672737125,0.39996159672737125,0.007037174636804898,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014240000396966934,0.035071998834609985,0.018764800019562246,0.015583999920636415,0.006305594700523817,0.0,0.0,0.0,0.0,0.0,0.7383679747581482,0.7597119808197021,0.7430047929286957,0.7430047929286957,0.006192645960911466,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.028991999104619026,0.019401599932461978,0.01665600063279271,0.005434953793780546,0.0,0.0,0.0,0.0,0.0,1.427008032798767,1.4517120122909546,1.4361984014511109,1.4361984014511109,0.008802755821028187,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014271999709308147,0.02844800055027008,0.01820160010829568,0.014944000169634819,0.004942002036909087,0.0,0.0,0.0,0.0,0.0,0.08361600339412689,0.11027199774980545,0.09058240056037903,0.09058240056037903,0.00872938604423622,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014112000353634357,0.023104000836610794,0.01648960020393133,0.015632000286132097,0.0027628828340349578,0.0,0.0,0.0,0.0,0.0,0.5063040256500244,0.5497599840164185,0.5168287932872773,0.5168287932872773,0.013557229606313293,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01616000011563301,0.029055999591946602,0.018780799955129622,0.017215999774634838,0.003794434947431941,0.0,0.0,0.0,0.0,0.0,0.7439360022544861,0.7719680070877075,0.7502080142498017,0.7502080142498017,0.00839205598262567,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015968000516295433,0.03046399913728237,0.018822400271892546,0.01726400014013052,0.004212536831927695,0.0,0.0,0.0,0.0,0.0,1.4256000518798828,1.449504017829895,1.4325888037681578,1.4325888037681578,0.0065448028108711165,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014271999709308147,0.027712000533938408,0.01633920017629862,0.014944000169634819,0.0038488220071983326,0.0,0.0,0.0,0.0,0.0,2.798719882965088,3.0184640884399414,2.8293471813201903,2.8293471813201903,0.06375662767704417,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014720000326633453,0.021407999098300934,0.01637439979240298,0.014992000069469213,0.0024463594774459265,0.0,0.0,0.0,0.0,0.0,0.09347199648618698,0.10473600029945374,0.09820479974150656,0.09820479974150656,0.004061668684999473,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014303999952971935,0.04790399968624115,0.020179199893027543,0.016447999514639378,0.009635010283506967,0.0,0.0,0.0,0.0,0.0,0.6221439838409424,0.6444799900054932,0.6277984082698822,0.6277984082698822,0.007234140179397069,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,8,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014271999709308147,0.03276799991726875,0.01921279989182949,0.015232000034302473,0.0065506148511265076,0.0,0.0,0.0,0.0,0.0,0.9111359715461731,0.9307839870452881,0.9151648044586183,0.9151648044586183,0.005528712062927265,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,16,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014175999909639359,0.02707199938595295,0.016512000095099212,0.014688000082969666,0.0038989080209321414,0.0,0.0,0.0,0.0,0.0,1.7645119428634644,1.7849279642105103,1.7684095859527589,1.7684095859527589,0.005876963340184193,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,32,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014240000396966934,0.029184000566601753,0.016825600154697896,0.014752000104635954,0.00457910514148248,0.0,0.0,0.0,0.0,0.0,3.4781761169433594,3.5388801097869873,3.490892815589905,3.490892815589905,0.01818910600692522,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,64,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014976000413298607,0.03139200061559677,0.01898880014196038,0.016848000697791576,0.005083715524393418,0.13752702814163098,0.14117629917511842,0.13848196486571557,0.13848196486571557,0.0009880564964713527,0.523336967410432,0.5372236808930884,0.5269708253945994,0.5269708253945994,0.00375988994658546,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,520,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,8,8,9,8,16384.0,0.1111111111111111,q512_8q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.01833599992096424,0.02707199938595295,0.020995199866592883,0.01976000051945448,0.0028392656352050185,0.17817885890237703,0.18622915300900206,0.18109068484526028,0.18109068484526028,0.0027040006080457624,0.5449571488834201,0.5695788427872015,0.5538629212834322,0.5538629212834322,0.008270141985514729,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1040,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,16,16,17,16,16384.0,0.058823529411764705,q1k_16q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.02703999914228916,0.03308799862861633,0.028883199393749236,0.02759999968111515,0.002285758578932753,0.4688266550410815,0.48301665772804186,0.4727750380198454,0.4727750380198454,0.004396396389258256,1.0430453981052807,1.0746153117238624,1.051829759343436,1.051829759343436,0.009781101336187197,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,32,32,33,32,16384.0,0.030303030303030304,q2k_32q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.04368000105023384,0.059328000992536545,0.04809600040316582,0.04617599956691265,0.005158792915002645,1.3509461459747514,1.3759066409666496,1.3589365122072,1.3589365122072,0.008008877716183384,0.9213738861449998,0.9383974724214122,0.9268234851606594,0.9268234851606594,0.005462224239661061,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,4112,,,,,,,,False,,,4096,BF16,generic,none,CUDA_EVENT,True,[4096],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,16,16,17,16,32768.0,0.058823529411764705,q4k_16q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.026944000273942947,0.040511999279260635,0.030291200056672095,0.02817599941045046,0.004386386489305873,0.5071830964059985,0.5142792136001758,0.5096392092471257,0.5096392092471257,0.0022059272685869546,2.175632932188972,2.2060727208328075,2.186168772243963,2.186168772243963,0.009462633998570079,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,32,32,33,32,32768.0,0.030303030303030304,q2k_32q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.018303999677300453,0.03017600066959858,0.02039040010422468,0.018943999893963337,0.0034612051983613614,0.21341429693945616,0.21552776300295579,0.21393496609766816,0.21393496609766816,0.0005736773262009222,4.617401495144284,4.663128147488988,4.628666619290515,4.628666619290515,0.012412001359412127,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1088,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,64,64,65,64,32768.0,0.015384615384615385,q1k_64q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014720000326633453,0.02364799939095974,0.01809599995613098,0.016352000646293163,0.0035481127058959038,0.04822399839758873,0.08566399663686752,0.05961279980838299,0.05961279980838299,0.011445665413968877,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,64,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[64],64,64,64,64.0,True,0.0,0.0,0.0,False,0.0,64.0,64,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q64,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014175999909639359,0.020864000543951988,0.01668160008266568,0.015664000064134598,0.0025769063833097584,0.049695998430252075,0.08057600259780884,0.05882879942655563,0.05882879942655563,0.009515126108519331,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,128,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[128],128,128,128,128.0,True,0.0,0.0,0.0,False,0.0,128.0,128,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.028704000636935234,0.017430400010198355,0.015056000091135502,0.004349294613335555,0.049855999648571014,0.07366400212049484,0.05459520071744919,0.05459520071744919,0.0069610920757925574,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,256,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[256],256,256,256,256.0,True,0.0,0.0,0.0,False,0.0,256.0,256,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q256,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014336000196635723,0.03161599859595299,0.017788799852132796,0.015711999963968992,0.005005385723318659,0.06102399900555611,0.08179199695587158,0.06650560013949873,0.06650560013949873,0.006947995595801105,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,512,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,False,0.0,512.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.04416000097990036,0.019200000166893005,0.015488000120967627,0.008561241323364038,0.08054400235414505,0.09196799993515015,0.08607039973139763,0.08607039973139763,0.004145329035483754,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,1024,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[1024],1024,1024,1024,1024.0,True,0.0,0.0,0.0,False,0.0,1024.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.017855999991297722,0.023391999304294586,0.019667199812829494,0.018463999964296818,0.0022577669687832585,0.18729600310325623,0.20585599541664124,0.19546559900045393,0.19546559900045393,0.006339068824303663,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,2048,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,2048.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.026240000501275063,0.03446400165557861,0.028297600522637367,0.027088000439107418,0.0025148441522922374,0.5754240155220032,0.5889919996261597,0.5800191938877105,0.5800191938877105,0.003858869273829596,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04182400181889534,0.047200001776218414,0.043036799877882004,0.0423360001295805,0.0017073405772076728,2.063199996948242,2.0787200927734375,2.067151999473572,2.067151999473572,0.004959651271127835,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.05648000165820122,0.02270399993285537,0.017935999669134617,0.012377915150727689,0.049536000937223434,0.07196799665689468,0.056015999615192415,0.056015999615192415,0.0070476637552742884,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,64,448.0,True,FLASH_ATTN,False,vllm020_batch_spec,[64],64,64,64,64.0,True,0.0,0.0,0.0,True,448.0,512.0,64,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q64s512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01408000010997057,0.021344000473618507,0.01611520005390048,0.014928000047802925,0.0025499884993961702,0.07072000205516815,0.2642880082130432,0.1588256008923054,0.1588256008923054,0.053220347086102376,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,128,896.0,True,FLASH_ATTN,False,vllm020_batch_spec,[128],128,128,128,128.0,True,0.0,0.0,0.0,True,896.0,1024.0,128,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q128s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01360000018030405,0.03014400042593479,0.016975999902933837,0.015056000091135502,0.004715159697364111,0.08393599838018417,0.11036799848079681,0.09160000011324881,0.09160000011324881,0.007911669434472792,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,256,1792.0,True,FLASH_ATTN,False,vllm020_batch_spec,[256],256,256,256,256.0,True,0.0,0.0,0.0,True,1792.0,2048.0,256,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q256s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014495999552309513,0.020767999812960625,0.016233599931001663,0.015392000321298838,0.002202615907995427,0.1998399943113327,0.22070400416851044,0.20855360180139543,0.20855360180139543,0.00689307200230667,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,512,3584.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,True,3584.0,4096.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01484800036996603,0.033440001308918,0.018684800155460833,0.016048000194132328,0.00553991005639174,0.6043199896812439,0.635807991027832,0.6126143991947173,0.6126143991947173,0.008745933953408096,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,1024,7168.0,True,FLASH_ATTN,False,vllm020_batch_spec,[1024],1024,1024,1024,1024.0,True,0.0,0.0,0.0,True,7168.0,8192.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1ks8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.013824000023305416,0.03359999880194664,0.017811199743300678,0.014944000169634819,0.005926140483565472,0.0,0.0,0.0,0.0,0.0,0.045471999794244766,0.07539200037717819,0.054758400097489356,0.054758400097489356,0.010253548506101549,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014047999866306782,0.02409599907696247,0.01641279999166727,0.01473599998280406,0.003347158638713987,0.0,0.0,0.0,0.0,0.0,0.0461760014295578,0.07529599964618683,0.05460800044238568,0.05460800044238568,0.009748937798340135,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014271999709308147,0.028416000306606293,0.018611199874430894,0.01598400017246604,0.005209875093863647,0.0,0.0,0.0,0.0,0.0,0.048128001391887665,0.07897599786520004,0.061353600397706036,0.061353600397706036,0.010488153655157845,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014112000353634357,0.025728000327944756,0.016092800162732603,0.01512000011280179,0.003300888123052605,0.0,0.0,0.0,0.0,0.0,0.04864000156521797,0.07648000121116638,0.05810560062527656,0.05810560062527656,0.009473544218404408,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.03551999852061272,0.018588799890130757,0.016064000315964222,0.006071057437992447,0.0,0.0,0.0,0.0,0.0,0.04822399839758873,0.07862400263547897,0.05660480037331582,0.05660480037331582,0.009565394213730401,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014303999952971935,0.028831999748945236,0.01699519995599985,0.015024000313133001,0.004285000744868188,0.0,0.0,0.0,0.0,0.0,0.04854400083422661,0.06719999760389328,0.05778240002691746,0.05778240002691746,0.006852125554679805,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.0144640002399683,0.024927999824285507,0.0161183999851346,0.01521599991247058,0.0029774808524673907,0.0,0.0,0.0,0.0,0.0,0.05379199981689453,0.08966399729251862,0.06270079985260964,0.06270079985260964,0.009983591277092696,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014431999996304512,0.03001599945127964,0.017648000083863736,0.01547200046479702,0.004585126309484848,0.0,0.0,0.0,0.0,0.0,0.061664000153541565,0.07692799717187881,0.06704320013523103,0.06704320013523103,0.005500191798490629,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014175999909639359,0.026367999613285065,0.016947199776768684,0.015039999969303608,0.0037951566103550205,0.0,0.0,0.0,0.0,0.0,0.08799999952316284,0.111455999314785,0.0964031994342804,0.0964031994342804,0.007397541615558088,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01369599997997284,0.021247999742627144,0.015359999984502793,0.014640000183135271,0.0020942770950814317,0.0,0.0,0.0,0.0,0.0,0.051711998879909515,0.07065600156784058,0.058054400235414506,0.058054400235414506,0.006633034815910025,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014208000153303146,0.0307839997112751,0.017148799914866685,0.015039999969303608,0.004882374686312284,0.0,0.0,0.0,0.0,0.0,0.061919998377561576,0.07843200117349625,0.06628479920327664,0.06628479920327664,0.004962852689801192,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.013887999579310417,0.02969600073993206,0.017500799987465142,0.015232000034302473,0.00467273319705314,0.0,0.0,0.0,0.0,0.0,0.08819200098514557,0.11097600311040878,0.09493440166115762,0.09493440166115762,0.007509235042985577,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.022112000733613968,0.01751680001616478,0.016047999262809753,0.0030306621792915785,0.0,0.0,0.0,0.0,0.0,0.13065600395202637,0.15110400319099426,0.13857279866933822,0.13857279866933822,0.00750137841249771,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.013919999822974205,0.04012800008058548,0.01892479993402958,0.01550400024279952,0.007459545267816961,0.0,0.0,0.0,0.0,0.0,0.06278400123119354,0.08259200304746628,0.0704512007534504,0.0704512007534504,0.005979055984382744,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014336000196635723,0.02236800082027912,0.016332800220698118,0.014928000047802925,0.002784044363186731,0.0,0.0,0.0,0.0,0.0,0.1003199964761734,0.1279360055923462,0.10921279862523078,0.10921279862523078,0.00862769617046716,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.013919999822974205,0.0225600004196167,0.016128000058233737,0.014800000004470348,0.002958953332547708,0.0,0.0,0.0,0.0,0.0,0.13116799294948578,0.14812800288200378,0.13783999979496003,0.13783999979496003,0.005696384361148053,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014208000153303146,0.029983999207615852,0.017468800116330386,0.01508800033479929,0.0047379550962483065,0.0,0.0,0.0,0.0,0.0,0.217631995677948,0.2447360008955002,0.22715839892625808,0.22715839892625808,0.008831828221847138,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014047999866306782,0.02304000034928322,0.016512000095099212,0.014512000139802694,0.0035026774828624254,0.0,0.0,0.0,0.0,0.0,0.11020799726247787,0.12307199835777283,0.11600959971547126,0.11600959971547126,0.004667637266950902,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014112000353634357,0.042080000042915344,0.017510399967432023,0.014607999939471483,0.00823117883530167,0.0,0.0,0.0,0.0,0.0,0.15702399611473083,0.17587199807167053,0.16399359852075576,0.16399359852075576,0.006588299074676393,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.013919999822974205,0.0208320003002882,0.015299199987202883,0.01462399959564209,0.001916768086505386,0.0,0.0,0.0,0.0,0.0,0.21779200434684753,0.2415360063314438,0.2260768011212349,0.2260768011212349,0.007251352080685236,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014015999622642994,0.021088000386953354,0.015619200188666582,0.01473599998280406,0.002013227452302141,0.0,0.0,0.0,0.0,0.0,0.3959999978542328,0.4152640104293823,0.40332479774951924,0.40332479774951924,0.006942401431914052,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014175999909639359,0.03379200026392937,0.020595200080424547,0.019600000232458115,0.006548754248881344,0.026192623739694512,0.040006882507168426,0.029155180178492036,0.029155180178492036,0.00408828748438241,0.024591375524545756,0.037561119537986146,0.027372820355088746,0.027372820355088746,0.0038383559348575944,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,72,,,,,,,,False,,,64,BF16,generic,none,CUDA_EVENT,True,[64],[0],"[512, 512, 512, 512, 512, 512, 512, 512]",1,8,8,9,8,512.0,0.1111111111111111,q64_8q1s512,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.013856000266969204,0.020896000787615776,0.015318400040268899,0.014431999996304512,0.0019640312960926966,0.029349018208693862,0.06236262941356679,0.03714313592014963,0.03714313592014963,0.009618313791872569,0.028826981462526914,0.06125337308649041,0.036482463672774496,0.036482463672774496,0.009447230957011863,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,136,,,,,,,,False,,,128,BF16,generic,none,CUDA_EVENT,True,[128],[0],"[1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024]",1,8,8,9,8,1024.0,0.1111111111111111,q128_8q1s1k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.013856000266969204,0.032127998769283295,0.017286399938166143,0.014479999896138906,0.005536495446636193,0.03273085874558354,0.04394578491937082,0.03563837490653034,0.03563837490653034,0.003429601730256567,0.03488514202593899,0.04683821345079978,0.037984025407085474,0.037984025407085474,0.0036553316361902684,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,144,,,,,,,,False,,,128,BF16,generic,none,CUDA_EVENT,True,[128],[0],"[1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024]",1,16,16,17,16,1024.0,0.058823529411764705,q128_16q1s1k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.013856000266969204,0.021503999829292297,0.015078400075435639,0.01425600005313754,0.002238435975478849,0.042100813549974185,0.051784144690147756,0.045378692890289625,0.045378692890289625,0.003184109940650555,0.05111518843748309,0.06287185663927425,0.05509490773570219,0.05509490773570219,0.0038658725544299132,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,272,,,,,,,,False,,,256,BF16,generic,none,CUDA_EVENT,True,[256],[0],"[2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048]",1,16,16,17,16,2048.0,0.058823529411764705,q256_16q1s2k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.02687999978661537,0.017667199857532977,0.01566399959847331,0.003864811846387173,0.048475323773821306,0.05599956978723733,0.05123499252968345,0.05123499252968345,0.002427379820423941,0.08429268135885387,0.09737642835214408,0.0890914090616175,0.0890914090616175,0.0042209177332077005,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,288,,,,,,,,False,,,256,BF16,generic,none,CUDA_EVENT,True,[256],[0],"[2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048]",1,32,32,33,32,2048.0,0.030303030303030304,q256_32q1s2k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.01462399959564209,0.03359999880194664,0.019804799742996693,0.016032000072300434,0.006750519908145474,0.07121508474579985,0.07292308367136396,0.07194410845270183,0.07194410845270183,0.0005500065915943844,0.14760091249712767,0.15114092373010238,0.14911189243564577,0.14911189243564577,0.0011399477384397005,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,544,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,32,32,33,32,4096.0,0.030303030303030304,q512_32q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.021856000646948814,0.016835200227797033,0.015263999812304974,0.00283854448975065,0.08146338272142935,0.08560865714384096,0.0834932625520734,0.0834932625520734,0.0012259635905347874,0.2782486219401307,0.29240733788179385,0.2851819365989658,0.2851819365989658,0.004187435731481665,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,576,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,64,64,65,64,4096.0,0.015384615384615385,q512_64q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.04495999962091446,0.02143679987639189,0.015856000129133463,0.009709987872683342,0.1194459208702178,0.12302524755181463,0.12053885718696096,0.12053885718696096,0.001005118119341339,0.5597220650459199,0.5764947443228802,0.5648435507167102,0.5648435507167102,0.004709970715400788,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1088,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192]",1,64,64,65,64,8192.0,0.015384615384615385,q1k_64q1s8k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.018112000077962875,0.026496000587940216,0.019318400137126445,0.018432000651955605,0.0024249605112359905,0.19770253574610402,0.222811797868348,0.2054811520619412,0.2054811520619412,0.008000944254868043,0.13941746080159492,0.157124211777114,0.14490284788179197,0.14490284788179197,0.005642170080515992,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,32,32,33,32,4096.0,0.030303030303030304,q2k_32q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.02595200017094612,0.03155200183391571,0.02727359998971224,0.026575999334454536,0.0016260948764843166,0.6014684881116659,0.6095472988212,0.6046757700946376,0.6046757700946376,0.0023537435563177303,0.11325152264578045,0.11477269562841545,0.11385542721485654,0.11385542721485654,0.0004431903698023628,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,4112,,,,,,,,False,,,4096,BF16,generic,none,CUDA_EVENT,True,[4096],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,16,16,17,16,4096.0,0.058823529411764705,q4k_16q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014303999952971935,0.03359999880194664,0.01814719969406724,0.015375999733805656,0.005678506740662529,0.06774400174617767,0.07891199737787247,0.07312640026211739,0.07312640026211739,0.003826583051422731,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,2,1024,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512]",1024,512,512,512.0,True,0.0,0.0,0.0,False,0.0,1024.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,2q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01788800023496151,0.028543999418616295,0.021340799890458582,0.019504000432789326,0.003718627037219618,0.09548799693584442,0.1327359974384308,0.10823359936475753,0.10823359936475753,0.013230635292043864,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,4,2048,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512]",2048,512,512,512.0,True,0.0,0.0,0.0,False,0.0,2048.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,4q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.02627200074493885,0.030368000268936157,0.027184000052511693,0.02643200010061264,0.0013545679205210022,0.14364799857139587,0.1597760021686554,0.15008639842271806,0.15008639842271806,0.005367723415171222,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512, 512, 512, 512, 512]",4096,512,512,512.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04150399938225746,0.04726399853825569,0.042950399965047834,0.04224000126123428,0.001720147501765378,0.2433920055627823,0.28963199257850647,0.2549152016639709,0.2549152016639709,0.013450991019687407,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512]",8192,512,512,512.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.025887999683618546,0.03407999873161316,0.028303999826312064,0.026367999613285065,0.0028877495368841042,0.325439989566803,0.3441599905490875,0.33442879617214205,0.33442879617214205,0.005693597811441922,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,2,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[2048, 2048]",4096,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,2q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04137599840760231,0.05084799975156784,0.044828799366950986,0.043087998405098915,0.003560363865797613,0.6110399961471558,0.6421759724617004,0.6204223990440368,0.6204223990440368,0.011214490370794758,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,4,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[2048, 2048, 2048, 2048]",8192,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,4q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014816000126302242,0.02364799939095974,0.016668799985200166,0.01592000015079975,0.0025515238179941247,0.0,0.0,0.0,0.0,0.0,0.0544000007212162,0.09014400094747543,0.06511679962277411,0.06511679962277411,0.013005726711295521,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014527999795973301,0.0208320003002882,0.016604799963533878,0.01566399959847331,0.0021353594266203244,0.0,0.0,0.0,0.0,0.0,0.17871999740600586,0.1961279958486557,0.18568639904260634,0.18568639904260634,0.004566142885274747,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014431999996304512,0.02006400004029274,0.015619200002402068,0.01508800033479929,0.001572984783715635,0.0,0.0,0.0,0.0,0.0,0.2730880081653595,0.29721599817276,0.2815328001976013,0.2815328001976013,0.007016534727616923,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.02755199931561947,0.018678399827331306,0.015919999685138464,0.004323555372382858,0.0,0.0,0.0,0.0,0.0,0.3893119990825653,0.44041600823402405,0.40225600004196166,0.40225600004196166,0.013442998165884852,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014303999952971935,0.02659199945628643,0.016748800035566093,0.015343999955803156,0.0036412341868394082,0.0,0.0,0.0,0.0,0.0,0.7404800057411194,0.9689919948577881,0.8451807916164398,0.8451807916164398,0.08637455920036795,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014751999638974667,0.02160000056028366,0.016752000153064727,0.01521599991247058,0.0025167650350367246,0.0,0.0,0.0,0.0,0.0,0.06412799656391144,0.08956799656152725,0.07242240011692047,0.07242240011692047,0.009797120588901621,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014495999552309513,0.022911999374628067,0.016137600038200618,0.015343999955803156,0.002330911555470513,0.0,0.0,0.0,0.0,0.0,0.3128319978713989,0.3282879889011383,0.3203647971153259,0.3203647971153259,0.004598568238907996,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.03574400022625923,0.01919359974563122,0.01583999954164028,0.006642521913489215,0.0,0.0,0.0,0.0,0.0,0.5050879716873169,0.5311999917030334,0.511932796239853,0.511932796239853,0.007150929647317677,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.02595200017094612,0.016992000117897987,0.015456000342965126,0.003390731937723684,0.0,0.0,0.0,0.0,0.0,0.7385600209236145,0.7681919932365417,0.7483008027076722,0.7483008027076722,0.010214373356208012,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014527999795973301,0.038975998759269714,0.01952639976516366,0.015327999833971262,0.007695927285621061,0.0,0.0,0.0,0.0,0.0,1.4228800535202026,1.5237760543823242,1.4435008168220522,1.4435008168220522,0.0342955291725742,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014495999552309513,0.03907199949026108,0.01793599994853139,0.015248000156134367,0.007167103363234667,0.0,0.0,0.0,0.0,0.0,0.06947200000286102,0.08857599645853043,0.07469440028071403,0.07469440028071403,0.00583249952929747,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014495999552309513,0.02175999991595745,0.016710399929434062,0.015327999833971262,0.002705383977079183,0.0,0.0,0.0,0.0,0.0,0.388480007648468,0.4073280096054077,0.3966591984033584,0.3966591984033584,0.004487315516095964,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,8,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.0144640002399683,0.022048000246286392,0.017011200170964004,0.015392000321298838,0.002776899649845722,0.0,0.0,0.0,0.0,0.0,0.622048020362854,0.6347839832305908,0.6262047946453094,0.6262047946453094,0.004593360122958751,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,16,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014592000283300877,0.02937600016593933,0.01847040019929409,0.015344000421464443,0.005043809142122341,0.0,0.0,0.0,0.0,0.0,0.9105280041694641,0.9304640293121338,0.914108806848526,0.914108806848526,0.00563919268816644,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,32,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01462399959564209,0.039903998374938965,0.01791359977796674,0.015440000221133232,0.007367547649297217,0.0,0.0,0.0,0.0,0.0,1.764799952507019,1.7965760231018066,1.770739197731018,1.770739197731018,0.009082122770320404,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,64,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015168000012636185,0.03187200054526329,0.018553599901497363,0.016127999871969223,0.0050628774268006264,0.16963527081512533,0.1737871320906266,0.17088926838108623,0.17088926838108623,0.0011703114258121128,0.47362872483230506,0.48522089403242025,0.47712993814280913,0.47712993814280913,0.0032675581298664976,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,520,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,8,8,9,8,16384.0,0.1111111111111111,q512_8q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.03868800029158592,0.01921919994056225,0.016959999687969685,0.006691309533030663,0.1589600576212269,0.16117782913137854,0.15969057520605917,0.15969057520605917,0.0006212984448748182,0.5199519263456005,0.5272061673552852,0.5223414198520004,0.5223414198520004,0.002032242112153396,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1040,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,16,16,17,16,16384.0,0.058823529411764705,q1k_16q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.018400000408291817,0.024320000782608986,0.019910399988293647,0.0191040001809597,0.001841532763984352,0.25650752966102763,0.27066608538463055,0.259736894547936,0.259736894547936,0.00462927315947332,0.5278764825612624,0.5570139063374764,0.5345223138928448,0.5345223138928448,0.009526755161797178,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,32,32,33,32,16384.0,0.030303030303030304,q2k_32q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.026048000901937485,0.03728000074625015,0.028636799938976765,0.02705600019544363,0.003313953649011259,0.9270006318443208,0.9712965120641314,0.93480767601568,0.93480767601568,0.012612551450284608,0.8181833128578277,0.8572794566782395,0.8250739157811953,0.8250739157811953,0.01113200873299586,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,4112,,,,,,,,False,,,4096,BF16,generic,none,CUDA_EVENT,True,[4096],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,16,16,17,16,32768.0,0.058823529411764705,q4k_16q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.018079999834299088,0.037696000188589096,0.022092800214886667,0.019024000503122807,0.005986278786633071,0.2839857165542317,0.2876110048757805,0.2849506939696605,0.2849506939696605,0.0009995178425916847,1.0871823008331585,1.101060989333509,1.0908765231323903,1.0908765231323903,0.0038264533900426207,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,32,32,33,32,32768.0,0.030303030303030304,q2k_32q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.015072000212967396,0.025567999109625816,0.01726400014013052,0.015728000551462173,0.0031324465940990903,0.137270464802061,0.13799613818579368,0.13751499486424038,0.13751499486424038,0.00025221761530472904,2.3021855212208884,2.314355908780759,2.3062865750744197,2.3062865750744197,0.004229983070201486,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1088,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,64,64,65,64,32768.0,0.015384615384615385,q1k_64q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014560000039637089,0.022784000262618065,0.016435200069099664,0.015519999898970127,0.0023688301421469523,0.04879999905824661,0.09139200299978256,0.06054079942405224,0.06054079942405224,0.012152608702448775,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,64,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[64],64,64,64,64.0,True,0.0,0.0,0.0,False,0.0,64.0,64,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q64,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014816000126302242,0.02844800055027008,0.017008000146597625,0.015696000307798386,0.003941466294662679,0.047807998955249786,0.07356800138950348,0.05626560002565384,0.05626560002565384,0.00842179125412236,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,128,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[128],128,128,128,128.0,True,0.0,0.0,0.0,False,0.0,128.0,128,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014688000082969666,0.053408000618219376,0.019353600218892097,0.01532800029963255,0.01138302400841626,0.048448000103235245,0.0785600021481514,0.0556256003677845,0.0556256003677845,0.009348274502616908,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,256,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[256],256,256,256,256.0,True,0.0,0.0,0.0,False,0.0,256.0,256,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q256,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015039999969303608,0.03139200061559677,0.01823679991066456,0.016159999649971724,0.004860071238302512,0.055424001067876816,0.08505599945783615,0.0640383992344141,0.0640383992344141,0.00921448636178623,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,512,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,False,0.0,512.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014751999638974667,0.026528000831604004,0.017324799951165915,0.01536000007763505,0.0036004822686428305,0.07660800218582153,0.0942080020904541,0.08209280073642732,0.08209280073642732,0.00515895587669752,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,1024,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[1024],1024,1024,1024,1024.0,True,0.0,0.0,0.0,False,0.0,1024.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.021247999742627144,0.016403199825435876,0.01532800029963255,0.001995897022647934,0.11395200341939926,0.15113599598407745,0.12431039959192276,0.12431039959192276,0.011164431123683732,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,2048,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,2048.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.017184000462293625,0.028672000393271446,0.019865600019693376,0.017823999747633934,0.0035798491315929977,0.3171840012073517,0.3341119885444641,0.3261695951223373,0.3261695951223373,0.005111046452878918,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.024639999493956566,0.030400000512599945,0.02656640000641346,0.02556800004094839,0.0020189846603237303,1.0648640394210815,1.0828479528427124,1.0715327858924866,1.0715327858924866,0.005558639263049851,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014976000413298607,0.021695999428629875,0.017305599898099898,0.01593599934130907,0.0025718046517268054,0.04956800118088722,0.07932800054550171,0.06228480041027069,0.06228480041027069,0.01027160349757827,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,64,448.0,True,FLASH_ATTN,False,vllm020_batch_spec,[64],64,64,64,64.0,True,0.0,0.0,0.0,True,448.0,512.0,64,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q64s512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01539199985563755,0.03580800071358681,0.019686400331556796,0.017136000096797943,0.005918387811304003,0.05158400163054466,0.10713600367307663,0.06364160068333148,0.06364160068333148,0.015832957809696766,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,128,896.0,True,FLASH_ATTN,False,vllm020_batch_spec,[128],128,128,128,128.0,True,0.0,0.0,0.0,True,896.0,1024.0,128,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q128s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015168000012636185,0.02223999984562397,0.016672000009566545,0.015887999907135963,0.0020934945946034563,0.06521599739789963,0.08902399986982346,0.07520959973335266,0.07520959973335266,0.007913840684102929,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,256,1792.0,True,FLASH_ATTN,False,vllm020_batch_spec,[256],256,256,256,256.0,True,0.0,0.0,0.0,True,1792.0,2048.0,256,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q256s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015231999568641186,0.04028800129890442,0.019971200078725816,0.01646399963647127,0.007351305886577286,0.14467200636863708,0.16844800114631653,0.15363519936800005,0.15363519936800005,0.008174435899956223,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,512,3584.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,True,3584.0,4096.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015519999898970127,0.02304000034928322,0.01775679988786578,0.016784000210464,0.0025077243712082584,0.33740800619125366,0.35343998670578003,0.3445120006799698,0.3445120006799698,0.00445648463590358,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,1024,7168.0,True,FLASH_ATTN,False,vllm020_batch_spec,[1024],1024,1024,1024,1024.0,True,0.0,0.0,0.0,True,7168.0,8192.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1ks8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015647999942302704,0.02393599972128868,0.01809599995613098,0.01654400024563074,0.002998393102466254,0.0,0.0,0.0,0.0,0.0,0.0504320003092289,0.0843840017914772,0.060083200410008426,0.060083200410008426,0.00986959318572296,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01603199914097786,0.030047999694943428,0.019510399922728537,0.017680000513792038,0.004065185265963332,0.0,0.0,0.0,0.0,0.0,0.05004800111055374,0.07036799937486649,0.059315200522542,0.059315200522542,0.006537768821329647,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01484800036996603,0.02393599972128868,0.018035199772566558,0.016512000001966953,0.0030667689116777724,0.0,0.0,0.0,0.0,0.0,0.05100800096988678,0.06735999882221222,0.058387200161814694,0.058387200161814694,0.005832787739241381,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01500799972563982,0.026208000257611275,0.017500799987465142,0.01646399963647127,0.0030666103306165714,0.0,0.0,0.0,0.0,0.0,0.05023999884724617,0.07254400104284286,0.057254400476813315,0.057254400476813315,0.006890853653068151,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014688000082969666,0.027327999472618103,0.0179776000790298,0.01648000068962574,0.003783417447531392,0.0,0.0,0.0,0.0,0.0,0.04819199815392494,0.07100799679756165,0.05621119923889638,0.05621119923889638,0.007824391484124725,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,128.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s128,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015647999942302704,0.036448001861572266,0.019136000238358975,0.016704000532627106,0.006076530095624803,0.0,0.0,0.0,0.0,0.0,0.047488000243902206,0.08441600203514099,0.0588383998721838,0.0588383998721838,0.011275940726332522,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015135999768972397,0.033952001482248306,0.02119360016658902,0.018000000156462193,0.007136646614305304,0.0,0.0,0.0,0.0,0.0,0.04931199923157692,0.06992000341415405,0.05755840018391609,0.05755840018391609,0.007297587694315226,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.034272000193595886,0.0204415999352932,0.017311999574303627,0.00695404701803013,0.0,0.0,0.0,0.0,0.0,0.05215999856591225,0.0735040009021759,0.06117440015077591,0.06117440015077591,0.007381118436894922,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015359999611973763,0.03743999823927879,0.02012479966506362,0.01775999926030636,0.006181045509079057,0.0,0.0,0.0,0.0,0.0,0.06355199962854385,0.08508799970149994,0.07019200026988984,0.07019200026988984,0.0062246580442117845,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,1024.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s1k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015552000142633915,0.032607998698949814,0.01959999995306134,0.017487999983131886,0.004882475068460749,0.0,0.0,0.0,0.0,0.0,0.04918399825692177,0.0865280032157898,0.05973760038614274,0.05973760038614274,0.011546847942113974,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015168000012636185,0.02675200067460537,0.017430400010198355,0.016560000367462635,0.003243135094239819,0.0,0.0,0.0,0.0,0.0,0.052671998739242554,0.08367999643087387,0.06238719932734965,0.06238719932734965,0.011207824780630104,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01500799972563982,0.03612799942493439,0.019327999837696553,0.01688000001013279,0.006058251425153085,0.0,0.0,0.0,0.0,0.0,0.06364800035953522,0.07878399640321732,0.06935679838061332,0.06935679838061332,0.004515588328677643,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015104000456631184,0.03711999952793121,0.019670399930328132,0.017152000218629837,0.006165994646522264,0.0,0.0,0.0,0.0,0.0,0.09071999788284302,0.11507199704647064,0.10252480059862136,0.10252480059862136,0.008782544051535657,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,2048.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015039999969303608,0.026528000831604004,0.018396800104528665,0.01601599995046854,0.004279788048986283,0.0,0.0,0.0,0.0,0.0,0.05331199988722801,0.07977599650621414,0.06076480001211167,0.06076480001211167,0.008761579122069606,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015359999611973763,0.02643200010061264,0.01726400004699826,0.015855999663472176,0.003210008417242029,0.0,0.0,0.0,0.0,0.0,0.062431998550891876,0.08246400207281113,0.06970879957079888,0.06970879957079888,0.007016341676068233,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014911999925971031,0.02223999984562397,0.0166015999391675,0.015887999907135963,0.00204913969129354,0.0,0.0,0.0,0.0,0.0,0.10127999633550644,0.1141119971871376,0.10621120035648347,0.10621120035648347,0.004029318311754605,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015359999611973763,0.03363199904561043,0.019305599946528675,0.016671999357640743,0.0055445582307981234,0.0,0.0,0.0,0.0,0.0,0.1319040060043335,0.15839999914169312,0.14040640145540234,0.14040640145540234,0.007998041073596942,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,4096.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s4k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014368000440299511,0.04447999969124794,0.018908800091594458,0.0157279996201396,0.0086814937461721,0.0,0.0,0.0,0.0,0.0,0.06428799778223038,0.08982399851083755,0.07349760085344315,0.07349760085344315,0.008605284512197258,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015104000456631184,0.028672000393271446,0.01890560006722808,0.016080000437796116,0.004797584627815021,0.0,0.0,0.0,0.0,0.0,0.11123199760913849,0.14115199446678162,0.11942399889230729,0.11942399889230729,0.008513394706791027,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014271999709308147,0.024927999824285507,0.016915200091898442,0.015216000378131866,0.003374147499442635,0.0,0.0,0.0,0.0,0.0,0.15887999534606934,0.18111999332904816,0.16934399753808976,0.16934399753808976,0.007415181260761123,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014336000196635723,0.026944000273942947,0.017737600207328796,0.015199999790638685,0.00415598714375255,0.0,0.0,0.0,0.0,0.0,0.2192319929599762,0.23472000658512115,0.22809920012950893,0.22809920012950893,0.004730841376327335,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,8192.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s8k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014688000082969666,0.05052800104022026,0.020320000313222408,0.015520000364631414,0.010604522177924678,0.024270629882498958,0.03340247625954076,0.028446344104128624,0.028446344104128624,0.0032330369099793834,0.023697370291069768,0.03261352722995356,0.027774456325453972,0.027774456325453972,0.0031566742680923386,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,72,,,,,,,,False,,,64,BF16,generic,none,CUDA_EVENT,True,[64],[0],"[512, 512, 512, 512, 512, 512, 512, 512]",1,8,8,9,8,512.0,0.1111111111111111,q64_8q1s512,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.015359999611973763,0.033663999289274216,0.019987199828028678,0.018240000121295452,0.005522929139253732,0.0259194055660947,0.03719755183990719,0.029916029687899703,0.029916029687899703,0.003966666590554318,0.027104595853449518,0.03889844644729374,0.03128396954022015,0.03128396954022015,0.004148046317967881,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,136,,,,,,,,False,,,128,BF16,generic,none,CUDA_EVENT,True,[128],[0],"[1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024]",1,8,8,9,8,1024.0,0.1111111111111111,q128_8q1s1k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014592000283300877,0.03299200162291527,0.019840000104159115,0.016207999549806118,0.006546343008726814,0.02587869595769926,0.03763167265431482,0.030174939058162802,0.030174939058162802,0.0037303581782277364,0.026473304070195713,0.03849632587654989,0.03086826083864964,0.03086826083864964,0.003816069654529222,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,144,,,,,,,,False,,,128,BF16,generic,none,CUDA_EVENT,True,[128],[0],"[1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024]",1,16,16,17,16,1024.0,0.058823529411764705,q128_16q1s1k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.013824000023305416,0.021023999899625778,0.016304000187665223,0.015584000386297703,0.0023236165806545476,0.031810621525966996,0.04156949936878106,0.0346749350032807,0.0346749350032807,0.003130209181143985,0.035677378270900374,0.04662250161636451,0.038889864871740294,0.038889864871740294,0.0035107033960828727,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,272,,,,,,,,False,,,256,BF16,generic,none,CUDA_EVENT,True,[256],[0],"[2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048]",1,16,16,17,16,2048.0,0.058823529411764705,q256_16q1s2k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014527999795973301,0.02175999991595745,0.01569600012153387,0.015008000191301107,0.0020882666168036863,0.038211712107062853,0.07212229256520057,0.04364509673334097,0.04364509673334097,0.009671953074025085,0.0476442859917874,0.08992570455184198,0.05441890342616107,0.05441890342616107,0.01205947791784038,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,288,,,,,,,,False,,,256,BF16,generic,none,CUDA_EVENT,True,[256],[0],"[2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048, 2048]",1,32,32,33,32,2048.0,0.030303030303030304,q256_32q1s2k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014368000440299511,0.033824000507593155,0.017628800217062236,0.014864000026136637,0.005791495403857738,0.05214261250030033,0.06266261508706669,0.05677791295527661,0.05677791295527661,0.0030298223079651514,0.08648138506878382,0.10392938682791132,0.09416928531647478,0.09416928531647478,0.005025126612204489,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,544,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,32,32,33,32,4096.0,0.030303030303030304,q512_32q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014527999795973301,0.031199999153614044,0.017900799959897996,0.015343999955803156,0.004974475661316249,0.06411958891421111,0.07405276123263956,0.06793047918211377,0.06793047918211377,0.0033918658556700006,0.14058441263169497,0.16236323591492058,0.1489399211274489,0.1489399211274489,0.00743678300375353,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,576,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,64,64,65,64,4096.0,0.015384615384615385,q512_64q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014720000326633453,0.03855999931693077,0.018495999928563833,0.015647999942302704,0.006918530056761627,0.09872985549401277,0.10683454583043753,0.10185316839884552,0.10185316839884552,0.0020802254037042074,0.2743261390166379,0.2968454508984596,0.2830044295482752,0.2830044295482752,0.005780016596065092,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1088,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192, 8192]",1,64,64,65,64,8192.0,0.015384615384615385,q1k_64q1s8k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014911999925971031,0.030400000512599945,0.01977920001372695,0.01726400014013052,0.005350883464206169,0.13918872472233365,0.16279522855335207,0.1501912864839173,0.1501912864839173,0.008992185529011513,0.11892328861766266,0.13909276049083735,0.1283239123428725,0.1283239123428725,0.007682951884956939,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,32,32,33,32,4096.0,0.030303030303030304,q2k_32q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.01727999933063984,0.02223999984562397,0.01819519978016615,0.017680000513792038,0.0014397298270620873,0.3728044181625443,0.38295504353701676,0.3761264483787333,0.3761264483787333,0.0033660272332490977,0.07967557017401883,0.08184495665371805,0.08038555277807365,0.08038555277807365,0.0007193856241088455,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,4112,,,,,,,,False,,,4096,BF16,generic,none,CUDA_EVENT,True,[4096],[0],"[4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096, 4096]",1,16,16,17,16,4096.0,0.058823529411764705,q4k_16q1s4k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.01500799972563982,0.034591998904943466,0.01941439984366298,0.01756799966096878,0.005541380584373771,0.05951999872922897,0.08268799632787704,0.06715519949793816,0.06715519949793816,0.006657802116288364,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,2,1024,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512]",1024,512,512,512.0,True,0.0,0.0,0.0,False,0.0,1024.0,1024,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,2q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015104000456631184,0.03283200040459633,0.019180799927562477,0.016543999314308167,0.0052187911233635975,0.07199999690055847,0.08675199747085571,0.07749439924955369,0.07749439924955369,0.004849874741156772,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,4,2048,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512]",2048,512,512,512.0,True,0.0,0.0,0.0,False,0.0,2048.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,4q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01724799908697605,0.02284800074994564,0.01928640007972717,0.018400000408291817,0.0020641390157565697,0.09676799923181534,0.12310399860143663,0.10618879944086074,0.10618879944086074,0.008982738207839057,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512, 512, 512, 512, 512]",4096,512,512,512.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.024224000051617622,0.03315199911594391,0.027436799928545953,0.027328000403940678,0.002679154874546328,0.14716799557209015,0.16412800550460815,0.15470399856567385,0.15470399856567385,0.005423454433987419,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512]",8192,512,512,512.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q512,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01833599992096424,0.04806400090456009,0.026252799853682517,0.022672000341117382,0.00878438440429033,0.1844799965620041,0.2072959989309311,0.19359359890222552,0.19359359890222552,0.007663539820967659,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,2,4096,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[2048, 2048]",4096,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,4096.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,2q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.024480000138282776,0.04979199916124344,0.029081599973142146,0.025679999962449074,0.007227422044689197,0.34147199988365173,0.37968000769615173,0.3583200007677078,0.3583200007677078,0.011804422100726231,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,4,8192,0.0,True,FLASH_ATTN,False,vllm020_batch_spec,"[2048, 2048, 2048, 2048]",8192,2048,2048,2048.0,True,0.0,0.0,0.0,False,0.0,8192.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,4q2k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014720000326633453,0.023744000121951103,0.017753600236028434,0.016368000768125057,0.003061040281616705,0.0,0.0,0.0,0.0,0.0,0.05686400085687637,0.1090880036354065,0.06629760004580021,0.06629760004580021,0.014803727026812366,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.021088000386953354,0.01647359998896718,0.015840000472962856,0.0017964402433206,0.0,0.0,0.0,0.0,0.0,0.08540800213813782,0.09961599856615067,0.09146559983491898,0.09146559983491898,0.00447423706371901,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014560000039637089,0.04156799986958504,0.018512000143527985,0.015584000386297703,0.007849701681877904,0.0,0.0,0.0,0.0,0.0,0.17948800325393677,0.194815993309021,0.1866239994764328,0.1866239994764328,0.004156328241374654,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.029632000252604485,0.019449600111693145,0.016736000776290894,0.005126811425136928,0.0,0.0,0.0,0.0,0.0,0.27529600262641907,0.2898879945278168,0.2825664013624191,0.2825664013624191,0.004542801609674093,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014399999752640724,0.031328000128269196,0.019759999960660933,0.016367999836802483,0.006266303740589241,0.0,0.0,0.0,0.0,0.0,0.39190399646759033,0.4079039990901947,0.3987520009279252,0.3987520009279252,0.0037367727509362725,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,16384.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01539199985563755,0.040031999349594116,0.0212032001465559,0.01649599988013506,0.008068547381446682,0.0,0.0,0.0,0.0,0.0,0.06588800251483917,0.07897599786520004,0.0703904002904892,0.0703904002904892,0.004214826869769361,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014495999552309513,0.022495999932289124,0.016006399970501663,0.01515199989080429,0.002265581884342885,0.0,0.0,0.0,0.0,0.0,0.1231679990887642,0.13526399433612823,0.1288223996758461,0.1288223996758461,0.004328976434897094,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014592000283300877,0.023231999948620796,0.01705600004643202,0.016048000194132328,0.002810904312746391,0.0,0.0,0.0,0.0,0.0,0.3158079981803894,0.33129599690437317,0.32348800003528594,0.32348800003528594,0.00417599076136664,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014527999795973301,0.021888000890612602,0.01708160014823079,0.01609600055962801,0.0025289234621475062,0.0,0.0,0.0,0.0,0.0,0.5103679895401001,0.5200319886207581,0.5132320046424866,0.5132320046424866,0.0035759433157749533,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014688000082969666,0.032416000962257385,0.019353600032627583,0.016176000237464905,0.0061458798960684425,0.0,0.0,0.0,0.0,0.0,0.7400320172309875,0.7516480088233948,0.7449311971664427,0.7449311971664427,0.004573461776168607,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,32768.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.0144640002399683,0.02006400004029274,0.01573119992390275,0.014992000069469213,0.001654446341262991,0.0,0.0,0.0,0.0,0.0,0.07097599655389786,0.09932799637317657,0.07828159928321837,0.07828159928321837,0.009464602895232642,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,[1],1,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01484800036996603,0.02319999970495701,0.016883199848234654,0.015775999519973993,0.0026693906156048403,0.0,0.0,0.0,0.0,0.0,0.14176000654697418,0.1598079949617386,0.15063679963350293,0.15063679963350293,0.005741662969679433,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,8,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1]",8,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,8q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01484800036996603,0.03203200176358223,0.022224000189453363,0.02247999981045723,0.005996176183124981,0.0,0.0,0.0,0.0,0.0,0.39180800318717957,0.41046398878097534,0.4003200054168701,0.4003200054168701,0.005552433764307601,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,16,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",16,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,16q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015072000212967396,0.028192000463604927,0.017152000125497578,0.015536000020802021,0.003847486121602065,0.0,0.0,0.0,0.0,0.0,0.6239359974861145,0.6367359757423401,0.6285343945026398,0.6285343945026398,0.004006084286606002,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,32,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",32,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,32q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014879999682307243,0.026688000187277794,0.01718079997226596,0.015488000120967627,0.0037132536813398722,0.0,0.0,0.0,0.0,0.0,0.9141119718551636,0.9721279740333557,0.927455997467041,0.927455997467041,0.02148767779232235,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,64,0,40960.0,False,FLASH_ATTN,False,vllm020_batch_spec,"[1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1]",64,1,1,1.0,True,0.0,0.0,0.0,False,0,0,0,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,64q1s40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014751999638974667,0.047520000487565994,0.028710400220006704,0.0266720000654459,0.010774688401598903,0.06094816381288764,0.06816969726785131,0.0644060660218149,0.0644060660218149,0.002203106589073183,0.087051838094461,0.09736630405680231,0.09199073574791852,0.09199073574791852,0.0031466818046499644,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,9,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,520,,,,,,,,False,,,512,BF16,generic,none,CUDA_EVENT,True,[512],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,8,8,9,8,16384.0,0.1111111111111111,q512_8q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.014911999925971031,0.02489599958062172,0.017427200078964235,0.01595200039446354,0.003071906489592143,0.19733789497223914,0.2016295616652644,0.19860388994013833,0.19860388994013833,0.001512389485439305,0.44861409134062713,0.45837046456077934,0.45149211526120153,0.45149211526120153,0.0034381598874302305,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1040,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,16,16,17,16,16384.0,0.058823529411764705,q1k_16q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.015039999969303608,0.03046399913728237,0.020643199887126686,0.016944000497460365,0.006320148675595729,0.2157924314537054,0.22215709640166995,0.21769400765743155,0.21769400765743155,0.0019679921844645526,0.4905115822753901,0.5049789194903589,0.4948340005651485,0.4948340005651485,0.0044733865493072344,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384, 16384]",1,32,32,33,32,16384.0,0.030303030303030304,q2k_32q1s16k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.017152000218629837,0.02425600029528141,0.01865920014679432,0.018112000077962875,0.0019454553198986453,0.7430866512973927,0.7578834738720781,0.7455351005236347,0.7455351005236347,0.00416692487556653,0.7369773831645824,0.7516525540362471,0.7394057025273603,0.7394057025273603,0.004132666607961175,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,17,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,4112,,,,,,,,False,,,4096,BF16,generic,none,CUDA_EVENT,True,[4096],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,16,16,17,16,32768.0,0.058823529411764705,q4k_16q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.015263999812304974,0.02470399998128414,0.018009600043296815,0.01657600048929453,0.0031544532774534346,0.25251173919752334,0.25566890792461106,0.2536729054481981,0.2536729054481981,0.0010279562245342983,1.0425282722084215,1.0555630628624373,1.0473223013846877,1.0473223013846877,0.004244053880724073,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,33,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,2080,,,,,,,,False,,,2048,BF16,generic,none,CUDA_EVENT,True,[2048],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,32,32,33,32,32768.0,0.030303030303030304,q2k_32q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.01500799972563982,0.027807999402284622,0.01912960009649396,0.01643200032413006,0.004688164695934997,0.1251379565220268,0.14341821167748953,0.12742497577885914,0.12742497577885914,0.005353812167343766,1.1355340167064276,1.3014137556763312,1.1562870179154463,1.1562870179154463,0.048581869194943325,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,65,0,0,True,FLASH_ATTN,False,true_mixed_fused_projected,,1088,,,,,,,,False,,,1024,BF16,generic,none,CUDA_EVENT,True,[1024],[0],"[32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768, 32768]",1,64,64,65,64,32768.0,0.015384615384615385,q1k_64q1s32k,fused_total_conserving_projection_by_same_tp_pure_prefill_decode_reference_ratio
|
||||
0.0,0.0,0.0,0.0,0.0,0.07446400076150894,0.08246400207281113,0.07736000046133995,0.07713599875569344,0.002301031358835964,11.935359954833984,12.102368354797363,11.96896333694458,11.96896333694458,0.05162549713370252,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,8192,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,8192.0,16384.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.07664000242948532,0.08118399977684021,0.07796800062060356,0.077504001557827,0.0013681081693640953,19.84774398803711,20.15795135498047,19.915702438354494,19.915702438354494,0.0912299324030902,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,8192,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,16384.0,24576.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks24k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.0753600001335144,0.08207999914884567,0.07715519964694977,0.07595199719071388,0.0024238358447475007,27.781503677368164,27.9836483001709,27.844886589050287,27.844886589050287,0.061307010154819624,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,8192,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,24576.0,32768.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04339199885725975,0.050016000866889954,0.04504639990627766,0.04391999915242195,0.002339637813151782,5.1544318199157715,5.269279956817627,5.16938238143921,5.16938238143921,0.03363105774712194,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,4096,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,8192.0,12288.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks12k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04291199892759323,0.058848001062870026,0.045657599717378615,0.04383999854326248,0.004522763233004815,9.256383895874023,9.27734375,9.261776161193849,9.261776161193849,0.0062232312957398745,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,4096,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,16384.0,20480.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks20k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.043296001851558685,0.04982399940490723,0.045123199746012685,0.04387199878692627,0.0023369398894319345,13.365216255187988,13.634464263916016,13.398719978332519,13.398719978332519,0.0787651922310212,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,4096,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,24576.0,28672.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks28k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.02703999914228916,0.0297279991209507,0.027692800015211107,0.02723200060427189,0.000851127720159845,2.387968063354492,2.4014720916748047,2.3920736074447637,2.3920736074447637,0.003552645719470337,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,2048,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,8192.0,10240.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks10k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.02691200003027916,0.029823999851942062,0.027689599990844728,0.027583999559283257,0.0007783369959809023,4.440767765045166,4.4521918296813965,4.444268751144409,4.444268751144409,0.004022325406840951,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,2048,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,16384.0,18432.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks18k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.02703999914228916,0.03888000175356865,0.030262400023639204,0.02801600005477667,0.0037877256209144718,6.500351905822754,6.600607872009277,6.524902391433716,6.524902391433716,0.03321667087212851,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,2048,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,24576.0,26624.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks26k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.07865600287914276,0.08668799698352814,0.08130879923701287,0.07993599772453308,0.0027016201268628523,35.717376708984375,35.843265533447266,35.771837615966795,35.771837615966795,0.04024815379149301,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,40960,1,8192,32768.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,32768.0,40960.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04179200157523155,0.04864000156521797,0.0436256006360054,0.04267200082540512,0.002191476789190783,6.103871822357178,6.18287992477417,6.118390369415283,6.118390369415283,0.022924175047267112,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,8192,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,8192.0,16384.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04163200035691261,0.05987200140953064,0.04625920057296753,0.04403200000524521,0.005544136325859629,10.207136154174805,11.073247909545898,10.326777648925782,10.326777648925782,0.25139335734700863,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,8192,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,16384.0,24576.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks24k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04137599840760231,0.05027199909090996,0.04355199970304966,0.042399998754262924,0.0026572716942034787,14.303423881530762,14.343520164489746,14.311049747467042,14.311049747467042,0.01137511287661936,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,8192,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,24576.0,32768.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.026016000658273697,0.03718400001525879,0.03212799951434135,0.03270399942994118,0.003953184738275612,2.6534719467163086,2.6875839233398438,2.661257576942444,2.661257576942444,0.009372641904054613,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,4096,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,8192.0,12288.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks12k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.025887999683618546,0.03347200155258179,0.027168000116944313,0.02649599965661764,0.0021492000999850562,4.70630407333374,4.7400641441345215,4.714492845535279,4.714492845535279,0.009349196197962609,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,4096,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,16384.0,20480.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks20k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.025631999596953392,0.030719999223947525,0.02736639976501465,0.02711999975144863,0.0013921874495504173,6.759712219238281,6.831999778747559,6.774675178527832,6.774675178527832,0.021507023842020512,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,4096,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,24576.0,28672.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks28k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.017855999991297722,0.024831999093294144,0.01991359982639551,0.018655999563634396,0.0023971416189370203,1.3446400165557861,1.3609600067138672,1.3508928060531615,1.3508928060531615,0.006186954712541735,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,2048,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,8192.0,10240.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks10k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.0180479995906353,0.022752000018954277,0.019507200084626676,0.01896000001579523,0.0015079573553847933,2.5208001136779785,2.5887041091918945,2.534086418151855,2.534086418151855,0.01891953046799999,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,2048,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,16384.0,18432.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks18k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.018144000321626663,0.03222399950027466,0.02052800003439188,0.018864000216126442,0.004114364828004189,3.6922879219055176,3.760576009750366,3.706630396842957,3.706630396842957,0.019888657921309623,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,2048,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,24576.0,26624.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks26k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04169600084424019,0.04822399839758873,0.0439775999635458,0.04334400035440922,0.002101844179466219,18.41494369506836,18.502975463867188,18.445004844665526,18.445004844665526,0.026939913540867878,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,40960,1,8192,32768.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,32768.0,40960.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.023615999147295952,0.03267199918627739,0.026281599886715412,0.0248800003901124,0.003167969318232417,3.1188158988952637,3.1837120056152344,3.133536005020142,3.133536005020142,0.018365009771617643,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,8192,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,8192.0,16384.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks16k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.02380800060927868,0.03283200040459633,0.025670399703085423,0.024639999493956566,0.0026095917641781453,5.1729278564453125,5.243135929107666,5.192828798294067,5.192828798294067,0.01984737475315108,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,8192,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,16384.0,24576.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks24k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.024032000452280045,0.03167999908328056,0.02665280010551214,0.02550400048494339,0.002510131946267113,7.223167896270752,7.257152080535889,7.231532812118529,7.231532812118529,0.009418898846297825,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,8192,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,24576.0,32768.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks32k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01696000061929226,0.023679999634623528,0.01866880003362894,0.017455999739468098,0.002277860345236006,1.478559970855713,1.496127963066101,1.48257919549942,1.48257919549942,0.005278358037488471,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,4096,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,8192.0,12288.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks12k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.017023999243974686,0.030688000842928886,0.02055360022932291,0.018240000121295452,0.004065815389665311,2.651711940765381,2.6691839694976807,2.657548785209656,2.657548785209656,0.005897264516465086,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,4096,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,16384.0,20480.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks20k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.016767999157309532,0.024351999163627625,0.01840319987386465,0.017167999409139156,0.0025234367788448957,3.823551893234253,3.8631999492645264,3.8335999727249135,3.8335999727249135,0.011858923917429712,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,4096,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,24576.0,28672.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks28k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.013952000066637993,0.03622400015592575,0.018214400112628936,0.014719999860972166,0.007277897466521675,0.7039039731025696,0.7163199782371521,0.7077983975410462,0.7077983975410462,0.004048424224033295,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,2048,8192.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,8192.0,10240.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks10k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.015104000456631184,0.03097599931061268,0.020870400313287973,0.02054399996995926,0.005422197456220526,1.2929600477218628,1.313088059425354,1.3006752014160157,1.3006752014160157,0.00790622446009414,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,2048,16384.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,16384.0,18432.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks18k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.03232000023126602,0.019132800027728082,0.01673599984496832,0.0050857949387988245,1.8775999546051025,1.9125759601593018,1.8867840051651,1.8867840051651,0.01196043526228758,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,2048,24576.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,24576.0,26624.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks26k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.023584000766277313,0.03126399964094162,0.02633600030094385,0.025200000032782555,0.0025710957466677864,9.283391952514648,9.48249626159668,9.331705665588379,9.331705665588379,0.06366970422148335,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,40960,1,8192,32768.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,32768.0,40960.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks40k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.07552000135183334,0.0851840004324913,0.07839040011167527,0.07703999802470207,0.0030805904904383677,43.62025451660156,43.89299011230469,43.68390693664551,43.68390693664551,0.08458163192147439,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,40960.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,40960.0,49152.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks48k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.07718399912118912,0.0854400023818016,0.07895359992980958,0.07791999727487564,0.0024425064759109735,59.421791076660156,59.73017501831055,59.5021312713623,59.5021312713623,0.10367381308510447,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,57344.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,57344.0,65536.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks64k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.07875200361013412,0.09055999666452408,0.08208959847688675,0.08087999746203423,0.0033614630055883482,75.2852783203125,75.55481719970703,75.38758392333985,75.38758392333985,0.08870109119325344,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,73728.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,73728.0,81920.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks80k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.07788799703121185,0.09888000041246414,0.0819871999323368,0.08008000254631042,0.005835618489111142,91.13442993164062,91.45164489746094,91.24225158691405,91.24225158691405,0.1009018902177725,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,90112.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,90112.0,98304.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks96k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.07596799731254578,0.08790399879217148,0.07967040091753005,0.07808000221848488,0.0036209252740976605,106.89055633544922,107.34063720703125,106.9560287475586,106.9560287475586,0.13032874360676439,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,106496.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,106496.0,114688.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks112k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.07648000121116638,0.08534400165081024,0.07928640022873878,0.07787200063467026,0.0028305104107121735,122.71724700927734,122.9840316772461,122.8334243774414,122.8334243774414,0.09625274868603513,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,122880.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,122880.0,131072.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks128k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.07664000242948532,0.08774399757385254,0.08081279993057251,0.07972799986600876,0.003487270970977775,130.6565399169922,131.09359741210938,130.79229431152345,130.79229431152345,0.13138911961272914,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,8192,131072.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,131072.0,139264.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks136k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.01532800029963255,0.040063999593257904,0.019459199998527764,0.01601599995046854,0.007496165774486326,9.443936347961426,9.649087905883789,9.489968109130858,9.489968109130858,0.05961646816296781,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,512,130560.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,True,130560.0,131072.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512s128k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.02691200003027916,0.040832001715898514,0.02942080032080412,0.027888000011444092,0.004040048279091587,16.753759384155273,16.90662384033203,16.78766403198242,16.78766403198242,0.04236470233451937,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,2048,65536.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,65536.0,67584.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks66k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04416000097990036,0.04867200180888176,0.04565120078623295,0.04468800127506256,0.0018079971851529544,50.29715347290039,50.50300979614258,50.35054740905761,50.35054740905761,0.0724480602714254,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,4096,98304.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,98304.0,102400.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks100k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.060447998344898224,0.06684800237417221,0.06276160031557083,0.061824001371860504,0.0024045636532566625,96.07142639160156,96.53209686279297,96.1898666381836,96.1898666381836,0.1365794879996899,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,1,147456,1,6144,131072.0,True,FLASH_ATTN,False,vllm020_batch_spec,[6144],6144,6144,6144,6144.0,True,0.0,0.0,0.0,True,131072.0,137216.0,6144,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q6ks134k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04217600077390671,0.05100800096988678,0.04406719990074635,0.042847998440265656,0.002656672983576728,22.5166072845459,22.986656188964844,22.63068161010742,22.63068161010742,0.13840494695459715,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,40960.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,40960.0,49152.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks48k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04182400181889534,0.047200001776218414,0.043644800409674646,0.04262400045990944,0.001927741871473188,30.725727081298828,30.787456512451172,30.74285774230957,30.74285774230957,0.019110324860515015,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,57344.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,57344.0,65536.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks64k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04185599833726883,0.05049600079655647,0.043852799385786054,0.04289599880576134,0.0024573088797598033,38.930206298828125,38.98448181152344,38.94538269042969,38.94538269042969,0.017734743323340796,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,73728.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,73728.0,81920.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks80k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04227200150489807,0.046911999583244324,0.04412479996681214,0.04327999986708164,0.001811858548765738,47.13151931762695,47.245887756347656,47.148198699951166,47.148198699951166,0.033297799765613076,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,90112.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,90112.0,98304.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks96k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04182400181889534,0.06195199862122536,0.04529919996857643,0.043136000633239746,0.005751677082163394,55.338497161865234,55.64672088623047,55.42207336425782,55.42207336425782,0.11010194068839292,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,106496.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,106496.0,114688.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks112k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.04182400181889534,0.054687999188899994,0.04418559968471527,0.042767999693751335,0.0036820249064621804,63.549087524414055,63.80697631835938,63.62084197998047,63.62084197998047,0.08872184973119365,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,122880.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,122880.0,131072.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks128k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.0461760014295578,0.06566400080919266,0.05008000023663044,0.04843199998140335,0.0055350567765421,67.64147186279297,67.91651153564453,67.70619888305666,67.70619888305666,0.0993520482760623,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,8192,131072.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,131072.0,139264.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks136k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.023711999878287315,0.03859199956059456,0.02689919974654913,0.0244159996509552,0.004626699769228834,4.775519847869873,4.829855918884277,4.790303993225097,4.790303993225097,0.01597276061319242,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,512,130560.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,True,130560.0,131072.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512s128k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.024224000051617622,0.0453759990632534,0.027676799893379213,0.025200000032782555,0.006154880946840619,9.579392433166504,9.63871955871582,9.59802885055542,9.59802885055542,0.016811667352984866,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,2048,65536.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,65536.0,67584.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks66k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.02630399912595749,0.03593600168824196,0.02780479993671179,0.027040000073611736,0.0027569987978884785,25.21686363220215,25.383039474487305,25.250255966186526,25.250255966186526,0.04761963064809279,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,4096,98304.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,98304.0,102400.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks100k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.034015998244285583,0.041728001087903976,0.03665280006825924,0.0352960005402565,0.002710617869728372,48.101280212402344,48.49884796142578,48.17219200134278,48.17219200134278,0.11195495368044632,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,2,147456,1,6144,131072.0,True,FLASH_ATTN,False,vllm020_batch_spec,[6144],6144,6144,6144,6144.0,True,0.0,0.0,0.0,True,131072.0,137216.0,6144,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q6ks134k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.024191999807953835,0.039583999663591385,0.027286400087177753,0.025200000032782555,0.004620118437533604,11.32140827178955,11.496224403381348,11.354345512390138,11.354345512390138,0.050117766215439966,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,40960.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,40960.0,49152.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks48k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.024032000452280045,0.03232000023126602,0.026406400091946124,0.02518399991095066,0.002792696117638337,15.42249584197998,15.639776229858397,15.451993656158448,15.451993656158448,0.06304729308178388,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,57344.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,57344.0,65536.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks64k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.024224000051617622,0.03488000109791756,0.026147199980914592,0.02459200005978346,0.003303059079404486,19.5251522064209,19.70569610595703,19.54985942840576,19.54985942840576,0.052248914965338324,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,73728.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,73728.0,81920.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks80k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.024159999564290047,0.03046399913728237,0.0255103999748826,0.024656000547111034,0.00195717586616911,23.630016326904297,23.927040100097656,23.672822570800783,23.672822570800783,0.08715594728873713,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,90112.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,90112.0,98304.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks96k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.02380800060927868,0.03574400022625923,0.02678720001131296,0.024720000103116035,0.003678231234048846,27.736671447753906,27.79840087890625,27.747158622741697,27.747158622741697,0.018952179343788657,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,106496.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,106496.0,114688.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks112k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.024224000051617622,0.042080000042915344,0.02844479978084564,0.02527999971061945,0.006562862750709003,31.83427238464356,32.08835220336914,31.86994876861572,31.86994876861572,0.07444245241186317,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,122880.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,122880.0,131072.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks128k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.02377600036561489,0.03139200061559677,0.026396799832582474,0.025071999989449978,0.0025962810796740263,33.88313674926758,33.96985626220703,33.89708137512208,33.89708137512208,0.02518493598134421,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,8192,131072.0,True,FLASH_ATTN,False,vllm020_batch_spec,[8192],8192,8192,8192,8192.0,True,0.0,0.0,0.0,True,131072.0,139264.0,8192,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q8ks136k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014944000169634819,0.022624000906944275,0.017174400109797715,0.015536000020802021,0.0028331860349340154,3.179744005203247,3.192960023880005,3.182867193222046,3.182867193222046,0.0035239767128429386,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,512,130560.0,True,FLASH_ATTN,False,vllm020_batch_spec,[512],512,512,512,512.0,True,0.0,0.0,0.0,True,130560.0,131072.0,512,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q512s128k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.014655999839305878,0.02985600009560585,0.017318400088697672,0.015568000264465809,0.004479309643844885,4.811935901641846,4.826848030090332,4.8173023700714115,4.8173023700714115,0.00594674080057053,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,2048,65536.0,True,FLASH_ATTN,False,vllm020_batch_spec,[2048],2048,2048,2048,2048.0,True,0.0,0.0,0.0,True,65536.0,67584.0,2048,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q2ks66k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.017184000462293625,0.03916800022125244,0.021356799826025962,0.01935999933630228,0.0062801287963799276,14.374591827392578,14.635007858276367,14.41227512359619,14.41227512359619,0.07569270440104603,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,4096,98304.0,True,FLASH_ATTN,False,vllm020_batch_spec,[4096],4096,4096,4096,4096.0,True,0.0,0.0,0.0,True,98304.0,102400.0,4096,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q4ks100k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
0.0,0.0,0.0,0.0,0.0,0.020479999482631683,0.033215999603271484,0.023340800032019614,0.02092800009995699,0.004304974035052475,24.088096618652344,24.18492889404297,24.11243553161621,24.11243553161621,0.02797845886790172,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,2048,32,4,16,4,147456,1,6144,131072.0,True,FLASH_ATTN,False,vllm020_batch_spec,[6144],6144,6144,6144,6144.0,True,0.0,0.0,0.0,True,131072.0,137216.0,6144,BF16,generic,none,CUDA_EVENT,False,,,,,,,,,,,q6ks134k,measured_FA3_core_plus_measured_KV;reshape_assumed_zero;mean_as_median
|
||||
|
@@ -0,0 +1,37 @@
|
||||
time_stats.emb.min,time_stats.emb.max,time_stats.emb.mean,time_stats.emb.median,time_stats.emb.std,time_stats.input_layernorm.min,time_stats.input_layernorm.max,time_stats.input_layernorm.mean,time_stats.input_layernorm.median,time_stats.input_layernorm.std,time_stats.attn_pre_proj.min,time_stats.attn_pre_proj.max,time_stats.attn_pre_proj.mean,time_stats.attn_pre_proj.median,time_stats.attn_pre_proj.std,time_stats.attn_rope.min,time_stats.attn_rope.max,time_stats.attn_rope.mean,time_stats.attn_rope.median,time_stats.attn_rope.std,time_stats.attn_post_proj.min,time_stats.attn_post_proj.max,time_stats.attn_post_proj.mean,time_stats.attn_post_proj.median,time_stats.attn_post_proj.std,time_stats.post_attention_layernorm.min,time_stats.post_attention_layernorm.max,time_stats.post_attention_layernorm.mean,time_stats.post_attention_layernorm.median,time_stats.post_attention_layernorm.std,n_head,n_kv_head,n_embd,n_expanded_embd,vocab_size,use_gated_mlp,use_qk_norm,attn_output_gate,num_tokens,num_tensor_parallel_workers,padded_n_embd,padded_n_expanded_embd,model_arch,is_step2_mini,share_expert_dim,share_q_dim,measurement_type,profiling_precision,quant_signature
|
||||
0.029184000566601753,0.06780800223350525,0.03157280012965202,0.030736000277101994,0.005874173435341216,0.033215999603271484,0.04825599864125252,0.03443359974771738,0.0337119996547699,0.0031825678429048183,1.438431978225708,1.505568027496338,1.446228802204132,1.4429279565811157,0.014128607642643025,0.538752019405365,0.5440319776535034,0.5416463971138,0.5420799851417542,0.0016099306164432847,1.0073280334472656,1.0163840055465698,1.0098415970802308,1.0081279873847961,0.0032980457197329728,0.04022400081157684,0.041728001087903976,0.04091359991580248,0.04081599973142147,0.0004728505416767937,32,4,2048,768,151936,True,True,False,8192,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
0.01360000018030405,0.06784000247716904,0.017755200061947106,0.0179840000346303,0.00837681421401443,0.01836800016462803,0.03651199862360954,0.019606399815529585,0.018719999119639397,0.0038821958848767424,0.7512000203132629,0.8208960294723511,0.7566704005002975,0.75382399559021,0.014793092586577971,0.28995200991630554,0.29337599873542786,0.29135999977588656,0.29150401055812836,0.000998381071258815,0.5149760246276855,0.5169600248336792,0.5160208016633987,0.5158880054950714,0.0005038652783291977,0.02051199972629547,0.021503999829292297,0.021067200042307378,0.021104000508785248,0.00023542855306902367,32,4,2048,768,151936,True,True,False,4096,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
0.0080960001796484,0.04281599819660187,0.011092799971811474,0.011039999779313803,0.005489135752949302,0.012223999947309494,0.026335999369621277,0.013187199970707298,0.01247999956831336,0.0030236510562153375,0.3928639888763428,0.45372799038887024,0.3976895987987518,0.39528000354766846,0.012905176716136006,0.1547199934720993,0.15884800255298615,0.15712319910526276,0.1573439985513687,0.001165121945135037,0.26633599400520325,0.268095999956131,0.26719200164079665,0.2671840041875839,0.0004242740199415328,0.013024000450968742,0.013887999579310417,0.013489600038155913,0.013520000036805868,0.0002382817554057738,32,4,2048,768,151936,True,True,False,2048,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
0.00825599953532219,0.04755200073122978,0.02398160002194345,0.01961600035429001,0.00874683840888523,0.018880000337958336,0.03868800029158592,0.021590400114655496,0.0208320003002882,0.004057015751746236,0.24316799640655518,0.2710399925708771,0.2521967992186546,0.25065599381923676,0.007998418528894075,0.09644799679517746,0.19120000302791595,0.10407840013504029,0.09963199868798256,0.020043722414992166,0.13600000739097595,0.18892799317836761,0.15760480016469955,0.15760000050067902,0.0112223677907762,0.009664000011980534,0.010015999898314476,0.009836799977347255,0.009824000298976898,9.016971952948177e-05,32,4,2048,768,151936,True,True,False,1024,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
0.017376000061631203,0.044704001396894455,0.026318399980664254,0.027312000282108784,0.007565344034680866,0.01836800016462803,0.03014400042593479,0.021276800055056812,0.020655999891459942,0.002632977855097951,0.1438719928264618,0.17132799327373505,0.152497598528862,0.15012799948453903,0.007947150319625347,0.1430719941854477,0.19305600225925446,0.16630879789590836,0.1685439944267273,0.014628024163894684,0.08899199962615967,0.10467199981212616,0.0943599995225668,0.09374399855732918,0.003472669494074113,0.007615999784320593,0.007935999892652035,0.007769599952735007,0.0077760000713169575,7.680004540222077e-05,32,4,2048,768,151936,True,True,False,512,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
0.016992000862956047,0.0544000007212162,0.02573199989274144,0.026016000658273697,0.00803323401720434,0.018432000651955605,0.02348800003528595,0.020648000109940768,0.02062400057911873,0.0013939985047930988,0.10220800340175629,0.1361600011587143,0.11850560046732425,0.11684799939393997,0.010637754898360304,0.16710400581359863,0.21110400557518005,0.19078560024499894,0.19409599900245667,0.013714352646558832,0.06265600025653839,0.0740479975938797,0.06842879951000214,0.0690080001950264,0.0031292240446560557,0.00979200005531311,0.033440001308918,0.01864320016466081,0.017280000261962414,0.006957998188876383,32,4,2048,768,151936,True,True,False,256,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
0.017152000218629837,0.04396799951791763,0.025907999789342284,0.02598400041460991,0.007714554737915267,0.018400000408291817,0.03667199984192848,0.024132800102233887,0.02112000063061714,0.006120348733354326,0.10678400099277496,0.1363839954137802,0.1193264003843069,0.11583999916911125,0.009838647443214228,0.17017599940299988,0.22748799622058868,0.18853759989142418,0.18433599919080734,0.015728078443174653,0.04569600149989128,0.06652799993753433,0.05192639995366335,0.05151999928057194,0.004516605694976449,0.021856000646948814,0.026335999369621277,0.02384479995816946,0.023599999956786633,0.0011301153013314744,32,4,2048,768,151936,True,True,False,128,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
0.017343999817967415,1.0683200359344482,0.05748240072280168,0.029504000209271908,0.16366824814254827,0.018688000738620758,0.3317759931087494,0.03685439983382821,0.021151999942958355,0.06767422466731164,0.10255999863147736,0.9434880018234253,0.16412640027701855,0.11956800147891045,0.17964606281728834,0.1714559942483902,2.1306240558624268,0.3011296011507511,0.1926399990916252,0.42310127734378766,0.03574400022625923,0.6859520077705383,0.08389280084520578,0.04279999993741512,0.14321647071615612,0.020479999482631683,0.1831360012292862,0.033024000097066165,0.023856000043451786,0.03470987082429836,32,4,2048,768,151936,True,True,False,64,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
0.016095999628305435,0.05142400041222572,0.026363200135529043,0.02676799986511469,0.008637725852473854,0.01849599927663803,0.03577600046992302,0.022193600237369538,0.020848000422120094,0.004615266181181815,0.10540799796581268,0.15014399588108063,0.12211520001292228,0.11896000057458878,0.012772013396624768,0.17315199971199036,0.21478399634361267,0.1881632000207901,0.1873439997434616,0.011761164657572015,0.03481600061058998,0.058079998940229416,0.042200000025331974,0.03969600051641464,0.006461281521769989,0.02143999934196472,0.038816001266241074,0.02466559996828437,0.023856000043451786,0.0035418033748569927,32,4,2048,768,151936,True,True,False,32,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
0.015904000028967857,0.03753599897027016,0.023127999808639287,0.02527999971061945,0.0054230027890305385,0.018400000408291817,0.03001599945127964,0.020488000102341176,0.020096000283956528,0.002585688394137697,0.10355199873447418,0.1438400000333786,0.11536479964852334,0.11124800145626068,0.011136028294919255,0.16502399742603302,0.2072959989309311,0.18646399974822997,0.19075199961662292,0.013296437761247597,0.03142400085926056,0.05215999856591225,0.03888959977775812,0.03750399872660637,0.0057399302067536314,0.020128000527620316,0.04064000025391579,0.023937600292265417,0.023648000322282314,0.004144197400898248,32,4,2048,768,151936,True,True,False,16,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
0.015200000256299973,0.037151999771595,0.02144480012357235,0.02195199951529503,0.005576364091933697,0.017952000722289085,0.0226879995316267,0.019934400077909233,0.019952000118792057,0.0011812499470458758,0.10063999891281128,0.13468800485134125,0.11595199964940547,0.1207519993185997,0.011974301621092394,0.16332800686359406,0.20748800039291382,0.18162400051951408,0.1767839938402176,0.01474120743720334,0.03222399950027466,0.043455999344587326,0.03829439990222454,0.03859200142323971,0.0031378131240041122,0.020959999412298203,0.03587200120091438,0.023795200139284135,0.023423999547958374,0.0031339489695198443,32,4,2048,768,151936,True,True,False,8,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
0.014944000169634819,0.04022400081157684,0.02162720002233982,0.022463999688625336,0.006017219317026815,0.018464000895619392,0.026528000831604004,0.021264000236988066,0.020896000787615776,0.001934095017586082,0.10156799852848053,0.1703999936580658,0.12565439902245998,0.1244799979031086,0.01641776722785409,0.1653759926557541,0.23865599930286407,0.1969360001385212,0.19223999977111816,0.02058903050274278,0.03254399821162224,0.06752000004053116,0.0443536002188921,0.041519999504089355,0.00984829071098989,0.020096000283956528,0.040031999349594116,0.026934400014579297,0.02478400059044361,0.00562071702648593,32,4,2048,768,151936,True,True,False,1,1,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.7531200051307678,0.8461440205574036,0.7618160009384155,0.7576479911804199,0.019464233617982506,0.2922559976577759,0.2985599935054779,0.2953856036067009,0.29576000571250916,0.001575901077689139,0.510047972202301,0.5140479803085327,0.5121696025133133,0.5123839974403381,0.0012263137313476844,,,,,,32,4,2048,768,151936,True,True,False,8192,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.39529600739479065,0.47279998660087585,0.40216960161924364,0.39825600385665894,0.0162749796240819,0.15625600516796112,0.16211199760437012,0.15959519892930984,0.15988799929618835,0.0011493418942396922,0.2635200023651123,0.2642880082130432,0.2639120012521744,0.26392000913619995,0.0001903593401384135,,,,,,32,4,2048,768,151936,True,True,False,4096,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.2531839907169342,0.32419198751449585,0.26924319565296173,0.26049599051475525,0.01657194262368116,0.10051199793815613,0.2375359982252121,0.12411200068891048,0.10311999917030334,0.039466165428540506,0.15881599485874176,0.19289599359035492,0.16896959990262986,0.1640480011701584,0.009162416086418127,,,,,,32,4,2048,768,151936,True,True,False,2048,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.15302400290966034,0.1780800074338913,0.16284480094909667,0.16113600134849548,0.00740355661356144,0.14521600306034088,0.20233599841594696,0.17807039842009545,0.1796799972653389,0.013908448560094403,0.0907519981265068,0.09750399738550186,0.09460479989647866,0.09478399902582169,0.0017965487667361475,,,,,,32,4,2048,768,151936,True,True,False,1024,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.11740799993276596,0.17958399653434753,0.13262080028653145,0.12878400087356567,0.014146060532423056,0.17468799650669098,0.21161599457263947,0.1910431995987892,0.18966399878263474,0.01270903647700971,0.05926400050520897,0.07577600330114365,0.06530559975653887,0.06404799968004227,0.00479458462514625,,,,,,32,4,2048,768,151936,True,True,False,512,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.11260800063610077,0.17209599912166595,0.13610880002379416,0.13809599727392197,0.017675922458993677,0.18729600310325623,0.26822400093078613,0.20815680101513861,0.2078079953789711,0.01700718593658771,0.04499199986457825,0.06537599861621857,0.0507551996037364,0.04787199944257736,0.005727123232692158,,,,,,32,4,2048,768,151936,True,True,False,256,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.11257600039243698,0.23452800512313843,0.13866880126297473,0.13964799791574478,0.026590047697062115,0.1828799992799759,0.24751999974250793,0.20735519900918006,0.20670399814844131,0.01500438131586336,0.03654399886727333,0.05593600124120712,0.0408239996060729,0.03892800025641918,0.004864409450710354,,,,,,32,4,2048,768,151936,True,True,False,128,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.11734399944543839,0.18726399540901184,0.13890240006148816,0.13308800011873245,0.019086167340006257,0.16991999745368958,0.2443840056657791,0.19185120090842248,0.19257599860429764,0.018017536159219673,0.03033600002527237,0.04956800118088722,0.038387199863791466,0.03750400058925152,0.00500897018702887,,,,,,32,4,2048,768,151936,True,True,False,64,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.11420799791812897,0.18115200102329254,0.13938880078494548,0.1393439993262291,0.018141313962922536,0.1693439930677414,0.221343994140625,0.19187839925289155,0.19438399374485016,0.01430752121272052,0.03017600066959858,0.06019200012087822,0.03866560012102127,0.0364960003644228,0.0074199790031205266,,,,,,32,4,2048,768,151936,True,True,False,32,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.11276800185441971,0.16223999857902527,0.1331360016018152,0.13305599987506866,0.01432493740801146,0.17187200486660004,0.23625600337982178,0.19287680014967917,0.1913280040025711,0.01754628831589704,0.029823999851942062,0.07574400305747986,0.03900959976017475,0.036927999928593636,0.009277465083962879,,,,,,32,4,2048,768,151936,True,True,False,16,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.1130559965968132,0.19814400374889374,0.13379519879817964,0.12531199678778648,0.020074427302248836,0.16841599345207214,0.21139200031757355,0.1893615983426571,0.19092799723148346,0.012418720676762711,0.029503999277949333,0.07558400183916092,0.040144000016152856,0.03728000074625015,0.010223239196498205,,,,,,32,4,2048,768,151936,True,True,False,8,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.115167997777462,0.2640640139579773,0.13805920109152794,0.13118399679660797,0.03132849683515479,0.17103999853134155,0.2977280020713806,0.19899839907884598,0.1976960003376007,0.028928046763067948,0.0297279991209507,0.05990400165319443,0.03834720011800528,0.0363520011305809,0.0070885330713495905,,,,,,32,4,2048,768,151936,True,True,False,1,2,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.3940800130367279,0.46483200788497925,0.39923040121793746,0.3957759886980057,0.015108903906233569,0.16022400557994843,0.1634880006313324,0.16203359961509706,0.16228799521923065,0.0010220720912414007,0.26073598861694336,0.2627840042114258,0.26138080209493636,0.2613760083913803,0.00041640363494172775,,,,,,32,4,2048,768,151936,True,True,False,8192,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.2536959946155548,0.27161601185798645,0.25960480123758317,0.2577280104160309,0.005142017404245015,0.09849599748849869,0.10281600058078766,0.10057279989123344,0.10063999891281128,0.0008836232237268523,0.13913600146770477,0.17606399953365326,0.15875840038061143,0.15988799929618835,0.010058952112389172,,,,,,32,4,2048,768,151936,True,True,False,4096,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.15142400562763214,0.1780479997396469,0.16139679849147798,0.1602879986166954,0.0070656055731385115,0.14601600170135498,0.23343999683856964,0.17913119941949845,0.179967999458313,0.023065274309176566,0.09388799965381622,0.12108799815177917,0.10317599996924401,0.10073599964380264,0.007364345618470178,,,,,,32,4,2048,768,151936,True,True,False,2048,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.12198399752378464,0.18406400084495544,0.13770400024950505,0.13232000172138214,0.01623164885687898,0.1773120015859604,0.20688000321388245,0.1870912007987499,0.18535999953746796,0.00825628058448234,0.05766399949789047,0.06739199906587601,0.06187200043350458,0.061216000467538834,0.003004312467037863,,,,,,32,4,2048,768,151936,True,True,False,1024,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.11737599968910217,0.18614399433135986,0.13109439946711063,0.12771200388669968,0.01372168725934724,0.17587199807167053,0.2699519991874695,0.19926720038056372,0.19075199961662292,0.02327478982490538,0.041728001087903976,0.06224000081419945,0.05000480003654957,0.049375999718904495,0.006185850944273961,,,,,,32,4,2048,768,151936,True,True,False,512,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.1159679964184761,0.1610880047082901,0.12988320142030715,0.12494400516152382,0.011865375783184065,0.1767680048942566,0.27379199862480164,0.21035519987344742,0.21275199949741364,0.023206964017550125,0.037087999284267426,0.062144000083208084,0.04439679980278015,0.04283200018107891,0.006155226392511492,,,,,,32,4,2048,768,151936,True,True,False,256,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.12015999853610992,0.158720001578331,0.13573280088603495,0.1343199983239174,0.010666830836014414,0.1737920045852661,0.22127999365329742,0.20261440128087999,0.20321600139141083,0.011153965849067301,0.03587200120091438,0.04944000020623207,0.038265600241720675,0.037328001111745834,0.0030431580113575636,,,,,,32,4,2048,768,151936,True,True,False,128,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.1212799996137619,0.15139199793338776,0.13183839991688728,0.13014400005340576,0.00851207547022104,0.17257599532604218,0.2250880002975464,0.1872655987739563,0.18193599581718445,0.013638284975248818,0.02969600073993206,0.0525440014898777,0.03840640028938651,0.03444799967110157,0.007598511754254174,,,,,,32,4,2048,768,151936,True,True,False,64,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.11270400136709213,0.1624639928340912,0.13508000001311302,0.13180799782276154,0.014243564451606775,0.17052799463272095,0.23715199530124664,0.1953311987221241,0.1966560035943985,0.01500670718978398,0.030239999294281006,0.05270399898290634,0.03807039987295866,0.03742399998009205,0.004906165602021684,,,,,,32,4,2048,768,151936,True,True,False,32,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.12319999933242798,0.17919999361038208,0.14059039913117885,0.14156799763441086,0.016217662982107223,0.1701119989156723,0.21987199783325195,0.18903039917349815,0.1870879977941513,0.014117854507841239,0.030368000268936157,0.06406400352716446,0.04045119984075427,0.03710399940609932,0.00878942205983628,,,,,,32,4,2048,768,151936,True,True,False,16,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.11296000331640244,0.16761599481105804,0.13843199908733367,0.13814399391412735,0.014925286935582404,0.1711679995059967,0.21561600267887115,0.1906527981162071,0.1876479983329773,0.011803992622137974,0.0306560005992651,0.09216000139713287,0.041129599791020155,0.036847999319434166,0.013739721468154687,,,,,,32,4,2048,768,151936,True,True,False,8,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
,,,,,,,,,,0.11929599940776825,0.2008959949016571,0.14535359852015972,0.13971199840307236,0.024769473061576282,0.16803200542926788,0.25123199820518494,0.19573760256171227,0.19366399943828583,0.023196011081705884,0.03049599938094616,0.08089599758386612,0.04739360017701984,0.0453919991850853,0.01286389846889463,,,,,,32,4,2048,768,151936,True,True,False,1,4,2048,768,generic,False,,,CUDA_EVENT,BF16,none
|
||||
|
@@ -0,0 +1,21 @@
|
||||
{
|
||||
"schema": "frontier-profile-v6-code-longctx-v1",
|
||||
"base": "runs/frontier-prefill-kvgrowth-fix-v0/profiles/profile-v5-kvgrowth/attention.csv",
|
||||
"base_sha256": "ff32c38975e68565c85770d85b92bc310ab1d8e34fd02327bef7fb305a9ffae3",
|
||||
"raw_inputs": {
|
||||
"runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/raw/flashattn-code-longctx-tp1.json": "3937c623adcea3d379b82bb7ab63d3b293f917fa59fa6fd6a147581d6a252cb6",
|
||||
"runs/frontier-code-trace-v0/profiles/raw/tp2-r1-v2/raw/flashattn-code-longctx-tp2.json": "e48aa6e6d039bdb68759701c9976f78ad818d6778b6c40b14823a0a4e852c52d",
|
||||
"runs/frontier-code-trace-v0/profiles/raw/tp4-r1-v2/raw/flashattn-code-longctx-tp4.json": "759d33decc09d229d07818df8161ed41180cab575304f335ad4d5e7a917cbc93"
|
||||
},
|
||||
"appended_rows": 33,
|
||||
"max_model_len": 147456,
|
||||
"anchor_checks": [
|
||||
"anchor q1ks8k/TP1: v4=1.1445ms new=1.1426ms rel_diff=0.2%",
|
||||
"anchor q512s4k/TP1: v4=0.3391ms new=0.3367ms rel_diff=0.7%",
|
||||
"anchor q1ks8k/TP2: v4=0.6126ms new=0.6065ms rel_diff=1.0%",
|
||||
"anchor q512s4k/TP2: v4=0.2086ms new=0.2079ms rel_diff=0.3%",
|
||||
"anchor q1ks8k/TP4: v4=0.3445ms new=0.3470ms rel_diff=0.7%",
|
||||
"anchor q512s4k/TP4: v4=0.1536ms new=0.1510ms rel_diff=1.7%"
|
||||
],
|
||||
"output_sha256": "fbcf7e1f95789a6f6d771e24d1fc60958b7daf04eb0db260d27869a19d71d550"
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
time_stats.moe_gating_linear.min,time_stats.moe_gating_linear.max,time_stats.moe_gating_linear.mean,time_stats.moe_gating_linear.median,time_stats.moe_gating_linear.std,time_stats.moe_gating_routing_topk.min,time_stats.moe_gating_routing_topk.max,time_stats.moe_gating_routing_topk.mean,time_stats.moe_gating_routing_topk.median,time_stats.moe_gating_routing_topk.std,time_stats.moe_shuffling.min,time_stats.moe_shuffling.max,time_stats.moe_shuffling.mean,time_stats.moe_shuffling.median,time_stats.moe_shuffling.std,time_stats.moe_grouped_gemm.min,time_stats.moe_grouped_gemm.max,time_stats.moe_grouped_gemm.mean,time_stats.moe_grouped_gemm.median,time_stats.moe_grouped_gemm.std,num_tokens,num_experts,num_experts_per_device,expert_parallel_size,routing_runtime_path,routing_assignment_policy,routing_weight_policy,routing_uses_router_logits,gating_runtime_context,gating_runtime_context_impl,router_topk,hidden_dim,expert_hidden_dim,use_gated,num_tensor_parallel_workers,total_routed_tokens,model_expansion_ratio,tokens_per_expert_avg,tokens_to_experts_ratio,expert_utilization,min_load_ratio,load_imbalance_cv,max_load_ratio,load_entropy,load_gini_coefficient,load_distribution,seed,moe_grouped_gemm_backend,measurement_type,profiling_precision,model_arch,quant_signature,router_median_nonadditivity_ratio,projection_policy
|
||||
0.02502400055527687,0.06092799827456474,0.033839999698102474,0.028672000393271446,0.010315255343709818,0.019360000267624855,0.0352960005402565,0.023424000293016434,0.022064000368118286,0.0038898102289194572,0.0,0.0,0.0,0.0,0.0,0.33926400542259216,0.405023992061615,0.36780479848384856,0.36507199704647064,0.01690507644474779,1,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,8,0.375,0.0625,0.0625,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0233364439829928,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.025280000641942024,0.05132799968123436,0.030641599837690593,0.027343999594449997,0.007562409232763304,0.020160000771284103,0.052960000932216644,0.024145600199699403,0.021424000151455402,0.007450198645599985,0.0,0.0,0.0,0.0,0.0,1.1943039894104004,1.286784052848816,1.228384006023407,1.2273280024528503,0.02832547242381263,8,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,64,0.375,0.5,0.5,0.3984375,0.0,1.346291201783626,4.0,5.59375,0.661865234375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9806430689981738,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024831999093294144,0.046560000628232956,0.030371200107038022,0.02763199992477894,0.005878487205028602,0.020320000126957893,0.04560000076889992,0.02601920012384653,0.02270400058478117,0.00751129965906799,0.0,0.0,0.0,0.0,0.0,1.679744005203247,1.766144037246704,1.7095808148384095,1.7015680074691772,0.02438921262998535,16,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,128,0.375,1.0,1.0,0.625,0.0,1.015504800579495,5.0,6.15516433212955,0.529052734375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9103623678483975,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024288000538945198,0.049375999718904495,0.03086080001667142,0.0267359996214509,0.0070864531384997225,0.020479999482631683,0.030912000685930252,0.02274719988927245,0.021359999664127827,0.0029966813650672505,0.0,0.0,0.0,0.0,0.0,2.1576640605926514,2.2921600341796875,2.2097824096679686,2.188944101333618,0.045572321842012986,32,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,256,0.375,2.0,2.0,0.875,0.0,0.6343057228182637,2.5,6.64370748444639,0.35369873046875,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9610778571819444,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02393599972128868,0.04342399910092354,0.029380799923092126,0.026688000187277794,0.0057374782000526574,0.020031999796628952,0.036607999354600906,0.022487999964505435,0.020911999978125095,0.0038135203062103235,0.0,0.0,0.0,0.0,0.0,2.422368049621582,2.5130879878997803,2.4516672134399413,2.434159994125366,0.03278900287381846,64,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,512,0.375,4.0,4.0,0.984375,0.0,0.4921254921257382,2.25,6.817190042344769,0.272369384765625,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9952941013961014,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.025407999753952026,0.05510399863123894,0.031430399790406224,0.02798399981111288,0.007335050273982725,0.020479999482631683,0.03561599925160408,0.02275839988142252,0.021551999263465405,0.003545718365165811,0.0,0.0,0.0,0.0,0.0,2.2217600345611572,2.289599895477295,2.2571327924728393,2.263375997543335,0.021660416089449488,128,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,1024,0.375,8.0,8.0,1.0,0.125,0.3486861500690843,1.875,6.908192310183997,0.197662353515625,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9273256282883522,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.0244159996509552,0.05395200103521347,0.03188959984108806,0.026031999848783016,0.00943995927387483,0.02051199972629547,0.036607999354600906,0.023247999791055917,0.02147199958562851,0.003910623837069648,0.0,0.0,0.0,0.0,0.0,2.18668794631958,2.318079948425293,2.2218016147613526,2.211087942123413,0.035380897213135316,256,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,2048,0.375,16.0,16.0,1.0,0.4375,0.2525504668006971,1.875,6.953347743053017,0.1410369873046875,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0380599882396606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024512000381946564,0.04620800167322159,0.02884640023112297,0.026320000179111958,0.005516557583355925,0.02054399996995926,0.03846399858593941,0.023401600029319524,0.021263999864459038,0.0044742944198265374,0.0,0.0,0.0,0.0,0.0,2.2291839122772217,2.3929600715637207,2.2908096313476562,2.2804640531539917,0.04479348924786221,512,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,4096,0.375,32.0,32.0,1.0,0.65625,0.15765965680164504,1.5625,6.98229848728205,0.08779525756835938,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9569603278386984,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02412799932062626,0.06940799951553345,0.030817600432783365,0.026800000108778477,0.010047085187652939,0.020128000527620316,0.044096000492572784,0.024606400076299904,0.022304000332951546,0.00622857166823714,0.0,0.0,0.0,0.0,0.0,2.0678720474243164,2.1297600269317627,2.0837119817733765,2.0779199600219727,0.017880044357986735,1024,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,8192,0.375,64.0,64.0,1.0,0.625,0.12169081635504074,1.3125,6.9892029662356325,0.06879425048828125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9524274993623046,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02831999957561493,0.043487999588251114,0.031744000129401685,0.02991999965161085,0.003973415836428909,0.02070399932563305,0.029343999922275543,0.022886400017887353,0.021743999794125557,0.0026873065045088873,0.0,0.0,0.0,0.0,0.0,2.916032075881958,3.0819520950317383,2.9805248022079467,2.9656319618225098,0.05482195799019572,2048,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,16384,0.375,128.0,128.0,1.0,0.796875,0.07935434147688751,1.1796875,6.9954297964750305,0.044734954833984375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8971818172100244,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.03830400109291077,0.06268800050020218,0.043337599746882914,0.040511999279260635,0.005823016247946116,0.023135999217629433,0.03747199848294258,0.025206399988383053,0.02393599972128868,0.003271381256231161,0.0,0.0,0.0,0.0,0.0,4.421599864959717,4.535359859466553,4.486294317245483,4.497056007385254,0.036990243787549344,4096,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,32768,0.375,256.0,256.0,1.0,0.8203125,0.060849326483103046,1.17578125,6.9973188375859685,0.033740997314453125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8113207890716184,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.05660799890756607,0.07932800054550171,0.06249920018017292,0.06039999984204769,0.005636461691480845,0.02956799976527691,0.03747199848294258,0.031126399897038935,0.030287999659776688,0.002157571006631541,0.0,0.0,0.0,0.0,0.0,7.302591800689697,7.402751922607422,7.354758310317993,7.3464319705963135,0.032142662400335566,8192,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,65536,0.375,512.0,512.0,1.0,0.890625,0.0412323087266341,1.08984375,6.998772433185578,0.02334284782409668,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.883909666885606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02502400055527687,0.06092799827456474,0.033839999698102474,0.028672000393271446,0.010315255343709818,0.019360000267624855,0.0352960005402565,0.023424000293016434,0.022064000368118286,0.0038898102289194572,0.0,0.0,0.0,0.0,0.0,0.35280001163482666,0.39692801237106323,0.37662720382213594,0.37196800112724304,0.013401318050665304,1,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,8,0.375,0.0625,0.0625,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0233364439829928,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.025280000641942024,0.05132799968123436,0.030641599837690593,0.027343999594449997,0.007562409232763304,0.020160000771284103,0.052960000932216644,0.024145600199699403,0.021424000151455402,0.007450198645599985,0.0,0.0,0.0,0.0,0.0,0.4692479968070984,0.5523840188980103,0.5134752035140991,0.5100640058517456,0.02291433464135784,8,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,64,0.375,0.5,0.5,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9806430689981738,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024831999093294144,0.046560000628232956,0.030371200107038022,0.02763199992477894,0.005878487205028602,0.020320000126957893,0.04560000076889992,0.02601920012384653,0.02270400058478117,0.00751129965906799,0.0,0.0,0.0,0.0,0.0,0.34652799367904663,0.4119040071964264,0.3789471983909607,0.38550400733947754,0.02073945105803335,16,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,128,0.375,1.0,1.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9103623678483975,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024288000538945198,0.049375999718904495,0.03086080001667142,0.0267359996214509,0.0070864531384997225,0.020479999482631683,0.030912000685930252,0.02274719988927245,0.021359999664127827,0.0029966813650672505,0.0,0.0,0.0,0.0,0.0,0.31462401151657104,0.7456960082054138,0.38617280423641204,0.34545600414276123,0.12230201266253077,32,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,256,0.375,2.0,2.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9610778571819444,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02393599972128868,0.04342399910092354,0.029380799923092126,0.026688000187277794,0.0057374782000526574,0.020031999796628952,0.036607999354600906,0.022487999964505435,0.020911999978125095,0.0038135203062103235,0.0,0.0,0.0,0.0,0.0,0.32521599531173706,0.419871985912323,0.36325119733810424,0.34968000650405884,0.03161798848223672,64,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,512,0.375,4.0,4.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9952941013961014,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.025407999753952026,0.05510399863123894,0.031430399790406224,0.02798399981111288,0.007335050273982725,0.020479999482631683,0.03561599925160408,0.02275839988142252,0.021551999263465405,0.003545718365165811,0.0,0.0,0.0,0.0,0.0,0.289792001247406,0.4663360118865967,0.4091839998960495,0.41655999422073364,0.0446001986506615,128,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,1024,0.375,8.0,8.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9273256282883522,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.0244159996509552,0.05395200103521347,0.03188959984108806,0.026031999848783016,0.00943995927387483,0.02051199972629547,0.036607999354600906,0.023247999791055917,0.02147199958562851,0.003910623837069648,0.0,0.0,0.0,0.0,0.0,0.3761279881000519,0.4416320025920868,0.40686399936676027,0.40540799498558044,0.02257778769899645,256,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,2048,0.375,16.0,16.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0380599882396606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024512000381946564,0.04620800167322159,0.02884640023112297,0.026320000179111958,0.005516557583355925,0.02054399996995926,0.03846399858593941,0.023401600029319524,0.021263999864459038,0.0044742944198265374,0.0,0.0,0.0,0.0,0.0,0.7172480225563049,0.8663039803504944,0.7723807990550995,0.7591840028762817,0.04164772379451242,512,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,4096,0.375,32.0,32.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9569603278386984,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02412799932062626,0.06940799951553345,0.030817600432783365,0.026800000108778477,0.010047085187652939,0.020128000527620316,0.044096000492572784,0.024606400076299904,0.022304000332951546,0.00622857166823714,0.0,0.0,0.0,0.0,0.0,1.0195519924163818,1.2216639518737793,1.1253888130187988,1.1453600525856018,0.06548005322243594,1024,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,8192,0.375,64.0,64.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9524274993623046,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02831999957561493,0.043487999588251114,0.031744000129401685,0.02991999965161085,0.003973415836428909,0.02070399932563305,0.029343999922275543,0.022886400017887353,0.021743999794125557,0.0026873065045088873,0.0,0.0,0.0,0.0,0.0,1.7490559816360474,1.9644800424575806,1.8529024004936219,1.814303994178772,0.08042565617327288,2048,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,16384,0.375,128.0,128.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8971818172100244,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.03830400109291077,0.06268800050020218,0.043337599746882914,0.040511999279260635,0.005823016247946116,0.023135999217629433,0.03747199848294258,0.025206399988383053,0.02393599972128868,0.003271381256231161,0.0,0.0,0.0,0.0,0.0,3.2479360103607178,3.385279893875122,3.296070408821106,3.2800960540771484,0.04525529026442046,4096,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,32768,0.375,256.0,256.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8113207890716184,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.05660799890756607,0.07932800054550171,0.06249920018017292,0.06039999984204769,0.005636461691480845,0.02956799976527691,0.03747199848294258,0.031126399897038935,0.030287999659776688,0.002157571006631541,0.0,0.0,0.0,0.0,0.0,6.344799995422363,6.517856121063232,6.464438438415527,6.478623867034912,0.05116674443145098,8192,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,1,65536,0.375,512.0,512.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.883909666885606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02502400055527687,0.06092799827456474,0.033839999698102474,0.028672000393271446,0.010315255343709818,0.019360000267624855,0.0352960005402565,0.023424000293016434,0.022064000368118286,0.0038898102289194572,0.0,0.0,0.0,0.0,0.0,0.2648000121116638,0.325439989566803,0.28852800130844114,0.28390398621559143,0.01933778377635077,1,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,8,0.375,0.0625,0.0625,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0233364439829928,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.025280000641942024,0.05132799968123436,0.030641599837690593,0.027343999594449997,0.007562409232763304,0.020160000771284103,0.052960000932216644,0.024145600199699403,0.021424000151455402,0.007450198645599985,0.0,0.0,0.0,0.0,0.0,0.7347840070724487,0.862496018409729,0.7769344031810761,0.769216001033783,0.03290485328796285,8,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,64,0.375,0.5,0.5,0.421875,0.0,1.346291201783626,6.0,5.652114648336087,0.636962890625,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9806430689981738,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024831999093294144,0.046560000628232956,0.030371200107038022,0.02763199992477894,0.005878487205028602,0.020320000126957893,0.04560000076889992,0.02601920012384653,0.02270400058478117,0.00751129965906799,0.0,0.0,0.0,0.0,0.0,0.9198399782180786,0.9646080136299133,0.9412063956260681,0.9411839842796326,0.014939365085478117,16,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,128,0.375,1.0,1.0,0.5703125,0.0,1.118033988749895,5.0,6.008641773518898,0.580810546875,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9103623678483975,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024288000538945198,0.049375999718904495,0.03086080001667142,0.0267359996214509,0.0070864531384997225,0.020479999482631683,0.030912000685930252,0.02274719988927245,0.021359999664127827,0.0029966813650672505,0.0,0.0,0.0,0.0,0.0,1.2796800136566162,1.3484159708023071,1.3006976008415223,1.2929120063781738,0.020807998177176254,32,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,256,0.375,2.0,2.0,0.8828125,0.0,0.6959705453537527,3.0,6.60872850615583,0.38055419921875,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9610778571819444,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02393599972128868,0.04342399910092354,0.029380799923092126,0.026688000187277794,0.0057374782000526574,0.020031999796628952,0.036607999354600906,0.022487999964505435,0.020911999978125095,0.0038135203062103235,0.0,0.0,0.0,0.0,0.0,1.3630399703979492,1.4430400133132935,1.3909215927124023,1.3892319798469543,0.022335744492366926,64,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,512,0.375,4.0,4.0,0.984375,0.0,0.5201036555341637,3.0,6.798826509158851,0.28302001953125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9952941013961014,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.025407999753952026,0.05510399863123894,0.031430399790406224,0.02798399981111288,0.007335050273982725,0.020479999482631683,0.03561599925160408,0.02275839988142252,0.021551999263465405,0.003545718365165811,0.0,0.0,0.0,0.0,0.0,1.27948796749115,1.3904000520706177,1.3176063895225525,1.309440016746521,0.038060887827312775,128,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,1024,0.375,8.0,8.0,1.0,0.25,0.3511282039725661,1.875,6.91002266305238,0.1970977783203125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9273256282883522,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.0244159996509552,0.05395200103521347,0.03188959984108806,0.026031999848783016,0.00943995927387483,0.02051199972629547,0.036607999354600906,0.023247999791055917,0.02147199958562851,0.003910623837069648,0.0,0.0,0.0,0.0,0.0,1.264415979385376,1.3145920038223267,1.2791999936103822,1.2753440141677856,0.014130605249568332,256,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,2048,0.375,16.0,16.0,1.0,0.375,0.24692938483248605,1.6875,6.955481130775285,0.13909912109375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0380599882396606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024512000381946564,0.04620800167322159,0.02884640023112297,0.026320000179111958,0.005516557583355925,0.02054399996995926,0.03846399858593941,0.023401600029319524,0.021263999864459038,0.0044742944198265374,0.0,0.0,0.0,0.0,0.0,1.3081920146942139,1.347648024559021,1.3292255997657776,1.329967975616455,0.014558863679016933,512,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,4096,0.375,32.0,32.0,1.0,0.625,0.17143053326165383,1.5625,6.9786675275754035,0.09520339965820312,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9569603278386984,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02412799932062626,0.06940799951553345,0.030817600432783365,0.026800000108778477,0.010047085187652939,0.020128000527620316,0.044096000492572784,0.024606400076299904,0.022304000332951546,0.00622857166823714,0.0,0.0,0.0,0.0,0.0,1.242751955986023,1.3112000226974487,1.2747935891151427,1.266207993030548,0.021093073517695057,1024,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,8192,0.375,64.0,64.0,1.0,0.78125,0.11000099875256815,1.296875,6.991308871213679,0.062183380126953125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9524274993623046,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02831999957561493,0.043487999588251114,0.031744000129401685,0.02991999965161085,0.003973415836428909,0.02070399932563305,0.029343999922275543,0.022886400017887353,0.021743999794125557,0.0026873065045088873,0.0,0.0,0.0,0.0,0.0,1.7388160228729248,1.8077759742736816,1.772764801979065,1.772704005241394,0.021056644077284283,2048,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,16384,0.375,128.0,128.0,1.0,0.78125,0.0864630150197678,1.1796875,6.994552526394139,0.048796653747558594,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8971818172100244,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.03830400109291077,0.06268800050020218,0.043337599746882914,0.040511999279260635,0.005823016247946116,0.023135999217629433,0.03747199848294258,0.025206399988383053,0.02393599972128868,0.003271381256231161,0.0,0.0,0.0,0.0,0.0,2.6563520431518555,2.7063679695129395,2.6785055875778196,2.6791679859161377,0.01639963463052944,4096,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,32768,0.375,256.0,256.0,1.0,0.8671875,0.06127686514721937,1.16015625,6.997291583027146,0.03497934341430664,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8113207890716184,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.05660799890756607,0.07932800054550171,0.06249920018017292,0.06039999984204769,0.005636461691480845,0.02956799976527691,0.03747199848294258,0.031126399897038935,0.030287999659776688,0.002157571006631541,0.0,0.0,0.0,0.0,0.0,4.386879920959473,4.452256202697754,4.4108480453491214,4.406303882598877,0.019768161791937636,8192,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,65536,0.375,512.0,512.0,1.0,0.884765625,0.041723768525324195,1.1171875,6.998746434318934,0.02298593521118164,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.883909666885606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02502400055527687,0.06092799827456474,0.033839999698102474,0.028672000393271446,0.010315255343709818,0.019360000267624855,0.0352960005402565,0.023424000293016434,0.022064000368118286,0.0038898102289194572,0.0,0.0,0.0,0.0,0.0,0.24208000302314758,0.4028480052947998,0.3041536003351212,0.277103990316391,0.05660721484881584,1,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,8,0.375,0.0625,0.0625,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0233364439829928,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.025280000641942024,0.05132799968123436,0.030641599837690593,0.027343999594449997,0.007562409232763304,0.020160000771284103,0.052960000932216644,0.024145600199699403,0.021424000151455402,0.007450198645599985,0.0,0.0,0.0,0.0,0.0,0.2447039932012558,0.30502399802207947,0.26446720361709597,0.26265600323677063,0.016744548364435372,8,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,64,0.375,0.5,0.5,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9806430689981738,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024831999093294144,0.046560000628232956,0.030371200107038022,0.02763199992477894,0.005878487205028602,0.020320000126957893,0.04560000076889992,0.02601920012384653,0.02270400058478117,0.00751129965906799,0.0,0.0,0.0,0.0,0.0,0.2337920069694519,0.2881599962711334,0.26074880361557007,0.264384001493454,0.016469850143940968,16,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,128,0.375,1.0,1.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9103623678483975,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024288000538945198,0.049375999718904495,0.03086080001667142,0.0267359996214509,0.0070864531384997225,0.020479999482631683,0.030912000685930252,0.02274719988927245,0.021359999664127827,0.0029966813650672505,0.0,0.0,0.0,0.0,0.0,0.23369599878787994,0.28591999411582947,0.25465920120477675,0.25385600328445435,0.01593204652619194,32,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,256,0.375,2.0,2.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9610778571819444,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02393599972128868,0.04342399910092354,0.029380799923092126,0.026688000187277794,0.0057374782000526574,0.020031999796628952,0.036607999354600906,0.022487999964505435,0.020911999978125095,0.0038135203062103235,0.0,0.0,0.0,0.0,0.0,0.2295999974012375,0.26556798815727234,0.24674240052700042,0.2497600018978119,0.010345732066199003,64,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,512,0.375,4.0,4.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9952941013961014,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.025407999753952026,0.05510399863123894,0.031430399790406224,0.02798399981111288,0.007335050273982725,0.020479999482631683,0.03561599925160408,0.02275839988142252,0.021551999263465405,0.003545718365165811,0.0,0.0,0.0,0.0,0.0,0.21721599996089935,0.29020801186561584,0.2394208014011383,0.2346400022506714,0.018747330357768585,128,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,1024,0.375,8.0,8.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9273256282883522,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.0244159996509552,0.05395200103521347,0.03188959984108806,0.026031999848783016,0.00943995927387483,0.02051199972629547,0.036607999354600906,0.023247999791055917,0.02147199958562851,0.003910623837069648,0.0,0.0,0.0,0.0,0.0,0.2717759907245636,0.305184006690979,0.28813759982585907,0.28809599578380585,0.01183422600545854,256,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,2048,0.375,16.0,16.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0380599882396606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024512000381946564,0.04620800167322159,0.02884640023112297,0.026320000179111958,0.005516557583355925,0.02054399996995926,0.03846399858593941,0.023401600029319524,0.021263999864459038,0.0044742944198265374,0.0,0.0,0.0,0.0,0.0,0.3917759954929352,0.43772798776626587,0.41130879521369934,0.4131519943475723,0.012992473640805227,512,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,4096,0.375,32.0,32.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9569603278386984,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02412799932062626,0.06940799951553345,0.030817600432783365,0.026800000108778477,0.010047085187652939,0.020128000527620316,0.044096000492572784,0.024606400076299904,0.022304000332951546,0.00622857166823714,0.0,0.0,0.0,0.0,0.0,0.6176639795303345,0.7009919881820679,0.642767995595932,0.6330719888210297,0.024074084919938756,1024,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,8192,0.375,64.0,64.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9524274993623046,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02831999957561493,0.043487999588251114,0.031744000129401685,0.02991999965161085,0.003973415836428909,0.02070399932563305,0.029343999922275543,0.022886400017887353,0.021743999794125557,0.0026873065045088873,0.0,0.0,0.0,0.0,0.0,1.0820800065994263,1.1674879789352417,1.1034304022789,1.0977439880371094,0.0233067292981566,2048,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,16384,0.375,128.0,128.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8971818172100244,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.03830400109291077,0.06268800050020218,0.043337599746882914,0.040511999279260635,0.005823016247946116,0.023135999217629433,0.03747199848294258,0.025206399988383053,0.02393599972128868,0.003271381256231161,0.0,0.0,0.0,0.0,0.0,1.9809919595718384,2.0415360927581787,2.003715181350708,1.992751955986023,0.022066076434645737,4096,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,32768,0.375,256.0,256.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8113207890716184,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.05660799890756607,0.07932800054550171,0.06249920018017292,0.06039999984204769,0.005636461691480845,0.02956799976527691,0.03747199848294258,0.031126399897038935,0.030287999659776688,0.002157571006631541,0.0,0.0,0.0,0.0,0.0,3.790112018585205,3.8651199340820312,3.829139161109924,3.8230879306793213,0.025177464160110564,8192,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,2,65536,0.375,512.0,512.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.883909666885606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02502400055527687,0.06092799827456474,0.033839999698102474,0.028672000393271446,0.010315255343709818,0.019360000267624855,0.0352960005402565,0.023424000293016434,0.022064000368118286,0.0038898102289194572,0.0,0.0,0.0,0.0,0.0,0.212351992726326,0.24383999407291412,0.22760000079870224,0.22723200172185898,0.01050568575837594,1,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,8,0.375,0.0625,0.0625,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0233364439829928,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.025280000641942024,0.05132799968123436,0.030641599837690593,0.027343999594449997,0.007562409232763304,0.020160000771284103,0.052960000932216644,0.024145600199699403,0.021424000151455402,0.007450198645599985,0.0,0.0,0.0,0.0,0.0,0.47494399547576904,0.5184000134468079,0.4920704007148743,0.49169600009918213,0.011991064701471855,8,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,64,0.375,0.5,0.5,0.3984375,0.0,1.3919410907075054,6.0,5.570159765557392,0.667236328125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9806430689981738,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024831999093294144,0.046560000628232956,0.030371200107038022,0.02763199992477894,0.005878487205028602,0.020320000126957893,0.04560000076889992,0.02601920012384653,0.02270400058478117,0.00751129965906799,0.0,0.0,0.0,0.0,0.0,0.6360960006713867,0.7004479765892029,0.6608384013175964,0.6572319865226746,0.020416877242438597,16,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,128,0.375,1.0,1.0,0.625,0.0,1.0307764064044151,4.0,6.138251855282827,0.5382080078125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9103623678483975,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024288000538945198,0.049375999718904495,0.03086080001667142,0.0267359996214509,0.0070864531384997225,0.020479999482631683,0.030912000685930252,0.02274719988927245,0.021359999664127827,0.0029966813650672505,0.0,0.0,0.0,0.0,0.0,0.780896008014679,0.8301439881324768,0.80346559882164,0.8030399978160858,0.016801230312128875,32,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,256,0.375,2.0,2.0,0.859375,0.0,0.6903350635742038,3.0,6.5943747091218174,0.38067626953125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9610778571819444,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02393599972128868,0.04342399910092354,0.029380799923092126,0.026688000187277794,0.0057374782000526574,0.020031999796628952,0.036607999354600906,0.022487999964505435,0.020911999978125095,0.0038135203062103235,0.0,0.0,0.0,0.0,0.0,0.8607040047645569,0.9195200204849243,0.8783008038997651,0.8751039803028107,0.01719059253115595,64,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,512,0.375,4.0,4.0,0.9765625,0.0,0.49410588440130926,2.75,6.814452474347134,0.271270751953125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9952941013961014,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.025407999753952026,0.05510399863123894,0.031430399790406224,0.02798399981111288,0.007335050273982725,0.020479999482631683,0.03561599925160408,0.02275839988142252,0.021551999263465405,0.003545718365165811,0.0,0.0,0.0,0.0,0.0,0.833952009677887,0.894752025604248,0.8619967997074127,0.863215982913971,0.018716378851797198,128,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,1024,0.375,8.0,8.0,1.0,0.25,0.33693529145074724,2.125,6.9186075263155535,0.1867218017578125,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9273256282883522,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.0244159996509552,0.05395200103521347,0.03188959984108806,0.026031999848783016,0.00943995927387483,0.02051199972629547,0.036607999354600906,0.023247999791055917,0.02147199958562851,0.003910623837069648,0.0,0.0,0.0,0.0,0.0,0.834879994392395,0.8871039748191833,0.8651552021503448,0.8638879954814911,0.015262430894400969,256,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,2048,0.375,16.0,16.0,1.0,0.375,0.25567294018677456,1.8125,6.952441049154937,0.14349365234375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0380599882396606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024512000381946564,0.04620800167322159,0.02884640023112297,0.026320000179111958,0.005516557583355925,0.02054399996995926,0.03846399858593941,0.023401600029319524,0.021263999864459038,0.0044742944198265374,0.0,0.0,0.0,0.0,0.0,0.8518080115318298,0.9097599983215332,0.8810272097587586,0.8751040101051331,0.017005819633271906,512,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,4096,0.375,32.0,32.0,1.0,0.625,0.1747801353218523,1.53125,6.978069554482723,0.09820938110351562,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9569603278386984,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02412799932062626,0.06940799951553345,0.030817600432783365,0.026800000108778477,0.010047085187652939,0.020128000527620316,0.044096000492572784,0.024606400076299904,0.022304000332951546,0.00622857166823714,0.0,0.0,0.0,0.0,0.0,0.8470079898834229,0.9010239839553833,0.8694015920162201,0.8716959953308105,0.016138881171190216,1024,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,8192,0.375,64.0,64.0,1.0,0.65625,0.1158122428154187,1.3125,6.9901908183358845,0.06445503234863281,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9524274993623046,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02831999957561493,0.043487999588251114,0.031744000129401685,0.02991999965161085,0.003973415836428909,0.02070399932563305,0.029343999922275543,0.022886400017887353,0.021743999794125557,0.0026873065045088873,0.0,0.0,0.0,0.0,0.0,1.1698240041732788,1.2311359643936157,1.1888479948043824,1.1890720129013062,0.017978797794492758,2048,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,16384,0.375,128.0,128.0,1.0,0.78125,0.08347181893108634,1.1796875,6.994921772573154,0.046871185302734375,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8971818172100244,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.03830400109291077,0.06268800050020218,0.043337599746882914,0.040511999279260635,0.005823016247946116,0.023135999217629433,0.03747199848294258,0.025206399988383053,0.02393599972128868,0.003271381256231161,0.0,0.0,0.0,0.0,0.0,1.7702720165252686,1.8097599744796753,1.7919103980064393,1.7956640124320984,0.012641295676021557,4096,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,32768,0.375,256.0,256.0,1.0,0.78125,0.06866734477822484,1.20703125,6.996602562938728,0.03801727294921875,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8113207890716184,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.05660799890756607,0.07932800054550171,0.06249920018017292,0.06039999984204769,0.005636461691480845,0.02956799976527691,0.03747199848294258,0.031126399897038935,0.030287999659776688,0.002157571006631541,0.0,0.0,0.0,0.0,0.0,2.968672037124634,3.0278079509735107,2.9899007797241213,2.9824799299240112,0.01816414122631667,8192,128,128,1,standard_fused_topk,logit_topk,softmax_renorm,True,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,65536,0.375,512.0,512.0,1.0,0.8984375,0.04399546833977376,1.126953125,6.998607314922362,0.024699926376342773,uniform_random_logits,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.883909666885606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02502400055527687,0.06092799827456474,0.033839999698102474,0.028672000393271446,0.010315255343709818,0.019360000267624855,0.0352960005402565,0.023424000293016434,0.022064000368118286,0.0038898102289194572,0.0,0.0,0.0,0.0,0.0,0.19574399292469025,0.2512960135936737,0.21939200013875962,0.2199999988079071,0.017532156418212565,1,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,8,0.375,0.0625,0.0625,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0233364439829928,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.025280000641942024,0.05132799968123436,0.030641599837690593,0.027343999594449997,0.007562409232763304,0.020160000771284103,0.052960000932216644,0.024145600199699403,0.021424000151455402,0.007450198645599985,0.0,0.0,0.0,0.0,0.0,0.20483200252056122,0.24006399512290955,0.22215040028095245,0.22433599829673767,0.00969639786132892,8,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,64,0.375,0.5,0.5,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9806430689981738,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024831999093294144,0.046560000628232956,0.030371200107038022,0.02763199992477894,0.005878487205028602,0.020320000126957893,0.04560000076889992,0.02601920012384653,0.02270400058478117,0.00751129965906799,0.0,0.0,0.0,0.0,0.0,0.20559999346733093,0.24726399779319763,0.22126719802618028,0.22207999974489212,0.01328093478485499,16,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,128,0.375,1.0,1.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9103623678483975,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024288000538945198,0.049375999718904495,0.03086080001667142,0.0267359996214509,0.0070864531384997225,0.020479999482631683,0.030912000685930252,0.02274719988927245,0.021359999664127827,0.0029966813650672505,0.0,0.0,0.0,0.0,0.0,0.20003199577331543,0.2301120012998581,0.21453119963407516,0.21598400175571442,0.010402239855151332,32,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,256,0.375,2.0,2.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9610778571819444,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02393599972128868,0.04342399910092354,0.029380799923092126,0.026688000187277794,0.0057374782000526574,0.020031999796628952,0.036607999354600906,0.022487999964505435,0.020911999978125095,0.0038135203062103235,0.0,0.0,0.0,0.0,0.0,0.19551999866962433,0.22972799837589264,0.21238719969987868,0.21488000452518463,0.010706095835489097,64,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,512,0.375,4.0,4.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9952941013961014,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.025407999753952026,0.05510399863123894,0.031430399790406224,0.02798399981111288,0.007335050273982725,0.020479999482631683,0.03561599925160408,0.02275839988142252,0.021551999263465405,0.003545718365165811,0.0,0.0,0.0,0.0,0.0,0.19420799612998962,0.2903999984264374,0.2211231991648674,0.21476799994707108,0.025955008886196004,128,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,1024,0.375,8.0,8.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9273256282883522,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.0244159996509552,0.05395200103521347,0.03188959984108806,0.026031999848783016,0.00943995927387483,0.02051199972629547,0.036607999354600906,0.023247999791055917,0.02147199958562851,0.003910623837069648,0.0,0.0,0.0,0.0,0.0,0.2290560007095337,0.289247989654541,0.2543327987194061,0.24939200282096863,0.01663236753503115,256,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,2048,0.375,16.0,16.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,1.0380599882396606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.024512000381946564,0.04620800167322159,0.02884640023112297,0.026320000179111958,0.005516557583355925,0.02054399996995926,0.03846399858593941,0.023401600029319524,0.021263999864459038,0.0044742944198265374,0.0,0.0,0.0,0.0,0.0,0.30831998586654663,0.36953601241111755,0.3324000000953674,0.3288639932870865,0.018617999572156707,512,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,4096,0.375,32.0,32.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9569603278386984,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02412799932062626,0.06940799951553345,0.030817600432783365,0.026800000108778477,0.010047085187652939,0.020128000527620316,0.044096000492572784,0.024606400076299904,0.022304000332951546,0.00622857166823714,0.0,0.0,0.0,0.0,0.0,0.462911993265152,0.5497919917106628,0.4893856018781662,0.4816960096359253,0.023348152887178286,1024,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,8192,0.375,64.0,64.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.9524274993623046,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.02831999957561493,0.043487999588251114,0.031744000129401685,0.02991999965161085,0.003973415836428909,0.02070399932563305,0.029343999922275543,0.022886400017887353,0.021743999794125557,0.0026873065045088873,0.0,0.0,0.0,0.0,0.0,0.7662079930305481,0.8717759847640991,0.788454395532608,0.7744799852371216,0.03174746482604812,2048,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,16384,0.375,128.0,128.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8971818172100244,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.03830400109291077,0.06268800050020218,0.043337599746882914,0.040511999279260635,0.005823016247946116,0.023135999217629433,0.03747199848294258,0.025206399988383053,0.02393599972128868,0.003271381256231161,0.0,0.0,0.0,0.0,0.0,1.363935947418213,1.4143040180206299,1.3812703967094422,1.3798720240592957,0.014770450075530007,4096,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,32768,0.375,256.0,256.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.8113207890716184,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
0.05660799890756607,0.07932800054550171,0.06249920018017292,0.06039999984204769,0.005636461691480845,0.02956799976527691,0.03747199848294258,0.031126399897038935,0.030287999659776688,0.002157571006631541,0.0,0.0,0.0,0.0,0.0,2.5507519245147705,2.680704116821289,2.579859209060669,2.566223978996277,0.03673283558365955,8192,128,128,1,standard_fused_topk,fixed_hotset8,softmax_renorm,False,standalone_legacy,vllm020_replicated_linear,8,2048,768,True,4,65536,0.375,512.0,512.0,0.0625,0.0,3.872983346207417,16.0,3.0,0.9375,hotset8,20260716,FlashInfer CUTLASS,CUDA_EVENT,BF16,generic,none,0.883909666885606,measured_gate+topk+modular_expert;shuffling_zero_because_expert_measurement_includes_prepare_finalize
|
||||
|
@@ -0,0 +1,8 @@
|
||||
3937c623adcea3d379b82bb7ab63d3b293f917fa59fa6fd6a147581d6a252cb6 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/raw/flashattn-code-longctx-tp1.json
|
||||
4e6d6002db037bdc71d40f76d6de979ef47960612c15dd985c04048c25bd1d41 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/aituner.commit
|
||||
420e4a613b1bcdb44b873eaa911f4a23129713ff9d486fc783b31a009a55cc85 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/batch-specs.txt
|
||||
5dd612a7806900e9d12afc27ccef196cc5518aea831aa4ed3b0a53bcfb5746cf runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/max-model-len.txt
|
||||
484da176e7355ab5532d5129a8e78bea0ae84b3aa0a42c7f0bea9689667762ff runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/pip-freeze.txt
|
||||
595b25965f76bf7ac25aa3b60126a437a083619303303aaf63346e993862253a runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/source.sha256
|
||||
4355a46b19d348dc2f57c046f8ef63d4538ebb936000f3c9ee954a27460dd865 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/vllm-allow-long-max-model-len.txt
|
||||
11e8f5af440e3db4c01b519a7b0e30bbcecc3a57bfb53dd86d32e601beb95a32 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/vllm-source.commit
|
||||
@@ -0,0 +1 @@
|
||||
e251046c306a8e03214fe1b2f40caabb98f76b7b
|
||||
@@ -0,0 +1,13 @@
|
||||
q8ks48k
|
||||
q8ks64k
|
||||
q8ks80k
|
||||
q8ks96k
|
||||
q8ks112k
|
||||
q8ks128k
|
||||
q8ks136k
|
||||
q512s128k
|
||||
q2ks66k
|
||||
q4ks100k
|
||||
q6ks134k
|
||||
q1ks8k
|
||||
q512s4k
|
||||
@@ -0,0 +1 @@
|
||||
147456
|
||||
@@ -0,0 +1,179 @@
|
||||
aiohappyeyeballs==2.7.1
|
||||
aiohttp==3.14.1
|
||||
aiosignal==1.4.0
|
||||
annotated-doc==0.0.4
|
||||
annotated-types==0.7.0
|
||||
anthropic==0.117.0
|
||||
anyio==4.14.2
|
||||
apache-tvm-ffi==0.1.9
|
||||
astor==0.8.1
|
||||
attrs==26.1.0
|
||||
blake3==1.0.9
|
||||
cachetools==7.1.4
|
||||
cbor2==6.1.3
|
||||
certifi==2026.6.17
|
||||
cffi==2.1.0
|
||||
charset-normalizer==3.4.9
|
||||
click==8.4.2
|
||||
cloudpickle==3.1.2
|
||||
compressed-tensors==0.15.0.1
|
||||
cryptography==49.0.0
|
||||
cuda-bindings==12.9.7
|
||||
cuda-pathfinder==1.5.6
|
||||
cuda-python==12.9.7
|
||||
cuda-tile==1.5.0
|
||||
cuda-toolkit==12.9.1
|
||||
depyf==0.20.0
|
||||
detect-installer==0.1.0
|
||||
dill==0.4.1
|
||||
diskcache==5.6.3
|
||||
distro==1.9.0
|
||||
dnspython==2.8.0
|
||||
docstring-parser==0.18.0
|
||||
einops==0.8.2
|
||||
email-validator==2.3.0
|
||||
fastapi==0.139.2
|
||||
fastapi-cli==0.0.32
|
||||
fastapi-cloud-cli==0.22.2
|
||||
fastar==0.11.0
|
||||
fastsafetensors==0.3.3
|
||||
filelock==3.31.1
|
||||
flashinfer-cubin==0.6.8.post1
|
||||
flashinfer-python==0.6.8.post1
|
||||
frozenlist==1.8.0
|
||||
fsspec==2026.6.0
|
||||
gguf==0.19.0
|
||||
googleapis-common-protos==1.75.0
|
||||
grpcio==1.82.1
|
||||
h11==0.16.0
|
||||
hf-xet==1.5.2
|
||||
httpcore==1.0.9
|
||||
httptools==0.8.0
|
||||
httpx==0.28.1
|
||||
httpx-sse==0.4.3
|
||||
huggingface-hub==1.24.0
|
||||
idna==3.18
|
||||
ijson==3.5.1
|
||||
interegular==0.3.3
|
||||
jinja2==3.1.6
|
||||
jiter==0.16.0
|
||||
jmespath==1.1.0
|
||||
jsonschema==4.26.0
|
||||
jsonschema-specifications==2025.9.1
|
||||
lark==1.2.2
|
||||
llguidance==1.3.0
|
||||
llvmlite==0.47.0
|
||||
lm-format-enforcer==0.11.3
|
||||
loguru==0.7.3
|
||||
markdown-it-py==4.2.0
|
||||
markupsafe==3.0.3
|
||||
mcp==1.28.1
|
||||
mdurl==0.1.2
|
||||
mistral-common==1.11.6
|
||||
ml-dtypes==0.5.4
|
||||
model-hosting-container-standards==0.1.16
|
||||
mpmath==1.3.0
|
||||
msgspec==0.21.1
|
||||
multidict==6.7.1
|
||||
networkx==3.6.1
|
||||
ninja==1.13.0
|
||||
numba==0.65.0
|
||||
numpy==2.3.5
|
||||
nvidia-cublas-cu12==12.9.1.4
|
||||
nvidia-cuda-cupti-cu12==12.9.79
|
||||
nvidia-cuda-nvrtc-cu12==12.9.86
|
||||
nvidia-cuda-runtime-cu12==12.9.79
|
||||
nvidia-cudnn-cu12==9.17.1.4
|
||||
nvidia-cudnn-frontend==1.18.0
|
||||
nvidia-cufft-cu12==11.4.1.4
|
||||
nvidia-cufile-cu12==1.14.1.1
|
||||
nvidia-curand-cu12==10.3.10.19
|
||||
nvidia-cusolver-cu12==11.7.5.82
|
||||
nvidia-cusparse-cu12==12.5.10.65
|
||||
nvidia-cusparselt-cu12==0.7.1
|
||||
nvidia-cutlass-dsl==4.5.3
|
||||
nvidia-cutlass-dsl-libs-base==4.5.3
|
||||
nvidia-ml-py==13.610.43
|
||||
nvidia-nccl-cu12==2.28.9
|
||||
nvidia-nvjitlink-cu12==12.9.86
|
||||
nvidia-nvshmem-cu12==3.4.5
|
||||
nvidia-nvtx-cu12==12.9.79
|
||||
openai==2.46.0
|
||||
openai-harmony==0.0.8
|
||||
opencv-python-headless==5.0.0.93
|
||||
opentelemetry-api==1.44.0
|
||||
opentelemetry-exporter-otlp==1.44.0
|
||||
opentelemetry-exporter-otlp-proto-common==1.44.0
|
||||
opentelemetry-exporter-otlp-proto-grpc==1.44.0
|
||||
opentelemetry-exporter-otlp-proto-http==1.44.0
|
||||
opentelemetry-proto==1.44.0
|
||||
opentelemetry-sdk==1.44.0
|
||||
opentelemetry-semantic-conventions==0.65b0
|
||||
opentelemetry-semantic-conventions-ai==0.5.1
|
||||
outlines-core==0.2.14
|
||||
packaging==26.2
|
||||
partial-json-parser==0.2.1.1.post7
|
||||
pillow==12.3.0
|
||||
prometheus-client==0.25.0
|
||||
prometheus-fastapi-instrumentator==8.0.2
|
||||
propcache==0.5.2
|
||||
protobuf==7.35.1
|
||||
psutil==7.2.2
|
||||
py-cpuinfo==9.0.0
|
||||
pybase64==1.4.3
|
||||
pycountry==26.2.16
|
||||
pycparser==3.0
|
||||
pydantic==2.13.4
|
||||
pydantic-core==2.46.4
|
||||
pydantic-extra-types==2.11.1
|
||||
pydantic-settings==2.14.2
|
||||
pygments==2.20.0
|
||||
pyjwt==2.13.0
|
||||
python-dotenv==1.2.2
|
||||
python-json-logger==4.1.0
|
||||
python-multipart==0.0.32
|
||||
pyyaml==6.0.3
|
||||
pyzmq==27.1.0
|
||||
quack-kernels==0.5.0
|
||||
referencing==0.37.0
|
||||
regex==2026.7.19
|
||||
requests==2.34.2
|
||||
rich==15.0.0
|
||||
rich-toolkit==0.20.3
|
||||
rignore==0.8.0
|
||||
rpds-py==2026.6.3
|
||||
safetensors==0.8.0
|
||||
sentencepiece==0.2.2
|
||||
sentry-sdk==2.66.0
|
||||
setproctitle==1.3.7
|
||||
setuptools==80.10.2
|
||||
shellingham==1.5.4
|
||||
six==1.17.0
|
||||
sniffio==1.3.1
|
||||
sse-starlette==3.4.5
|
||||
starlette==1.3.1
|
||||
supervisor==4.3.0
|
||||
sympy==1.14.0
|
||||
tabulate==0.10.0
|
||||
tiktoken==0.13.0
|
||||
tilelang==0.1.9
|
||||
tokenizers==0.22.2
|
||||
torch==2.11.0+cu129
|
||||
torch-c-dlpack-ext==0.1.5
|
||||
torchaudio==2.11.0+cu129
|
||||
torchvision==0.26.0+cu129
|
||||
tqdm==4.69.0
|
||||
transformers==5.14.1
|
||||
triton==3.6.0
|
||||
typer==0.27.0
|
||||
typing-extensions==4.16.0
|
||||
typing-inspection==0.4.2
|
||||
urllib3==2.7.0
|
||||
uvicorn==0.51.0
|
||||
uvloop==0.22.1
|
||||
vllm @ https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0%2Bcu129-cp38-abi3-manylinux_2_31_x86_64.whl
|
||||
watchfiles==1.2.0
|
||||
websockets==16.1.1
|
||||
xgrammar==0.2.3
|
||||
yarl==1.24.5
|
||||
z3-solver==4.15.4.0
|
||||
@@ -0,0 +1,2 @@
|
||||
42381f4bd1b67710d4068eb8993240930326a804c6e5167b0b78df3420e47c74 /home/admin/cpfs/wjh/aituner/aituner-code-trace-fbaa909/runs/frontier-code-trace-v0/../frontier-qwen30-vllm020-profile-v1/profile_vllm020_flashattn.py
|
||||
8a7a72b3adb25db6a61a86f9b04d2d19ebbfe3f2d20480852ef2f45376ac3b85 runs/frontier-code-trace-v0/run_flashattn_code_longctx.sh
|
||||
@@ -0,0 +1 @@
|
||||
1
|
||||
@@ -0,0 +1 @@
|
||||
88d34c6409e9fb3c7b8ca0c04756f061d2099eb1
|
||||
@@ -0,0 +1,564 @@
|
||||
{
|
||||
"environment": {
|
||||
"attention_backend": "FLASH_ATTN",
|
||||
"block_size": 16,
|
||||
"dtype": "bfloat16",
|
||||
"gpu": "NVIDIA H20",
|
||||
"model": "/home/admin/cpfs/wjh/models/Qwen/Qwen3-30B-A3B",
|
||||
"profile_kv_update": true,
|
||||
"profile_method": "cuda_event",
|
||||
"torch_cuda": "12.9",
|
||||
"torch_version": "2.11.0+cu129",
|
||||
"vllm_source_commit": "88d34c6409e9fb3c7b8ca0c04756f061d2099eb1",
|
||||
"vllm_version": "0.20.0"
|
||||
},
|
||||
"rows": [
|
||||
{
|
||||
"attention_core_excludes_kv_cache_update": true,
|
||||
"config": {
|
||||
"backend": "FLASH_ATTN",
|
||||
"batch_spec": "q8ks48k",
|
||||
"block_size": 16,
|
||||
"device": "cuda:0",
|
||||
"dtype": "torch.bfloat16",
|
||||
"head_dim": 128,
|
||||
"kv_cache_dtype": "auto",
|
||||
"kv_lora_rank": null,
|
||||
"num_kv_heads": 4,
|
||||
"num_kv_splits": null,
|
||||
"num_layers": 1,
|
||||
"num_q_heads": 32,
|
||||
"prefill_backend": null,
|
||||
"profile_memory": true,
|
||||
"qk_nope_head_dim": null,
|
||||
"qk_rope_head_dim": null,
|
||||
"reorder_batch_threshold": null,
|
||||
"repeats": 10,
|
||||
"use_cuda_graphs": false,
|
||||
"v_head_dim": null,
|
||||
"warmup_iters": 5
|
||||
},
|
||||
"error": null,
|
||||
"kv_cache_update_time": {
|
||||
"max_ms": 0.0851840004324913,
|
||||
"mean_ms": 0.07839040011167527,
|
||||
"median_ms": 0.07703999802470207,
|
||||
"min_ms": 0.07552000135183334,
|
||||
"std_ms": 0.0030805904904383677
|
||||
},
|
||||
"max_time": 0.043892990112304686,
|
||||
"mean_time": 0.04368390693664551,
|
||||
"memory_allocated_mb": 240.0849609375,
|
||||
"memory_reserved_mb": 246.0,
|
||||
"min_time": 0.043620254516601564,
|
||||
"std_time": 8.458163192147439e-05,
|
||||
"tensor_parallel_size": 1,
|
||||
"throughput_tokens_per_sec": 187529.01410308387
|
||||
},
|
||||
{
|
||||
"attention_core_excludes_kv_cache_update": true,
|
||||
"config": {
|
||||
"backend": "FLASH_ATTN",
|
||||
"batch_spec": "q8ks64k",
|
||||
"block_size": 16,
|
||||
"device": "cuda:0",
|
||||
"dtype": "torch.bfloat16",
|
||||
"head_dim": 128,
|
||||
"kv_cache_dtype": "auto",
|
||||
"kv_lora_rank": null,
|
||||
"num_kv_heads": 4,
|
||||
"num_kv_splits": null,
|
||||
"num_layers": 1,
|
||||
"num_q_heads": 32,
|
||||
"prefill_backend": null,
|
||||
"profile_memory": true,
|
||||
"qk_nope_head_dim": null,
|
||||
"qk_rope_head_dim": null,
|
||||
"reorder_batch_threshold": null,
|
||||
"repeats": 10,
|
||||
"use_cuda_graphs": false,
|
||||
"v_head_dim": null,
|
||||
"warmup_iters": 5
|
||||
},
|
||||
"error": null,
|
||||
"kv_cache_update_time": {
|
||||
"max_ms": 0.0854400023818016,
|
||||
"mean_ms": 0.07895359992980958,
|
||||
"median_ms": 0.07791999727487564,
|
||||
"min_ms": 0.07718399912118912,
|
||||
"std_ms": 0.0024425064759109735
|
||||
},
|
||||
"max_time": 0.059730175018310544,
|
||||
"mean_time": 0.0595021312713623,
|
||||
"memory_allocated_mb": 272.0888671875,
|
||||
"memory_reserved_mb": 374.0,
|
||||
"min_time": 0.05942179107666016,
|
||||
"std_time": 0.00010367381308510448,
|
||||
"tensor_parallel_size": 1,
|
||||
"throughput_tokens_per_sec": 137675.74076699864
|
||||
},
|
||||
{
|
||||
"attention_core_excludes_kv_cache_update": true,
|
||||
"config": {
|
||||
"backend": "FLASH_ATTN",
|
||||
"batch_spec": "q8ks80k",
|
||||
"block_size": 16,
|
||||
"device": "cuda:0",
|
||||
"dtype": "torch.bfloat16",
|
||||
"head_dim": 128,
|
||||
"kv_cache_dtype": "auto",
|
||||
"kv_lora_rank": null,
|
||||
"num_kv_heads": 4,
|
||||
"num_kv_splits": null,
|
||||
"num_layers": 1,
|
||||
"num_q_heads": 32,
|
||||
"prefill_backend": null,
|
||||
"profile_memory": true,
|
||||
"qk_nope_head_dim": null,
|
||||
"qk_rope_head_dim": null,
|
||||
"reorder_batch_threshold": null,
|
||||
"repeats": 10,
|
||||
"use_cuda_graphs": false,
|
||||
"v_head_dim": null,
|
||||
"warmup_iters": 5
|
||||
},
|
||||
"error": null,
|
||||
"kv_cache_update_time": {
|
||||
"max_ms": 0.09055999666452408,
|
||||
"mean_ms": 0.08208959847688675,
|
||||
"median_ms": 0.08087999746203423,
|
||||
"min_ms": 0.07875200361013412,
|
||||
"std_ms": 0.0033614630055883482
|
||||
},
|
||||
"max_time": 0.07555481719970703,
|
||||
"mean_time": 0.07538758392333984,
|
||||
"memory_allocated_mb": 304.0927734375,
|
||||
"memory_reserved_mb": 534.0,
|
||||
"min_time": 0.0752852783203125,
|
||||
"std_time": 8.870109119325343e-05,
|
||||
"tensor_parallel_size": 1,
|
||||
"throughput_tokens_per_sec": 108665.10867797918
|
||||
},
|
||||
{
|
||||
"attention_core_excludes_kv_cache_update": true,
|
||||
"config": {
|
||||
"backend": "FLASH_ATTN",
|
||||
"batch_spec": "q8ks96k",
|
||||
"block_size": 16,
|
||||
"device": "cuda:0",
|
||||
"dtype": "torch.bfloat16",
|
||||
"head_dim": 128,
|
||||
"kv_cache_dtype": "auto",
|
||||
"kv_lora_rank": null,
|
||||
"num_kv_heads": 4,
|
||||
"num_kv_splits": null,
|
||||
"num_layers": 1,
|
||||
"num_q_heads": 32,
|
||||
"prefill_backend": null,
|
||||
"profile_memory": true,
|
||||
"qk_nope_head_dim": null,
|
||||
"qk_rope_head_dim": null,
|
||||
"reorder_batch_threshold": null,
|
||||
"repeats": 10,
|
||||
"use_cuda_graphs": false,
|
||||
"v_head_dim": null,
|
||||
"warmup_iters": 5
|
||||
},
|
||||
"error": null,
|
||||
"kv_cache_update_time": {
|
||||
"max_ms": 0.09888000041246414,
|
||||
"mean_ms": 0.0819871999323368,
|
||||
"median_ms": 0.08008000254631042,
|
||||
"min_ms": 0.07788799703121185,
|
||||
"std_ms": 0.005835618489111142
|
||||
},
|
||||
"max_time": 0.09145164489746094,
|
||||
"mean_time": 0.09124225158691406,
|
||||
"memory_allocated_mb": 336.0966796875,
|
||||
"memory_reserved_mb": 726.0,
|
||||
"min_time": 0.09113442993164063,
|
||||
"std_time": 0.0001009018902177725,
|
||||
"tensor_parallel_size": 1,
|
||||
"throughput_tokens_per_sec": 89782.96630697016
|
||||
},
|
||||
{
|
||||
"attention_core_excludes_kv_cache_update": true,
|
||||
"config": {
|
||||
"backend": "FLASH_ATTN",
|
||||
"batch_spec": "q8ks112k",
|
||||
"block_size": 16,
|
||||
"device": "cuda:0",
|
||||
"dtype": "torch.bfloat16",
|
||||
"head_dim": 128,
|
||||
"kv_cache_dtype": "auto",
|
||||
"kv_lora_rank": null,
|
||||
"num_kv_heads": 4,
|
||||
"num_kv_splits": null,
|
||||
"num_layers": 1,
|
||||
"num_q_heads": 32,
|
||||
"prefill_backend": null,
|
||||
"profile_memory": true,
|
||||
"qk_nope_head_dim": null,
|
||||
"qk_rope_head_dim": null,
|
||||
"reorder_batch_threshold": null,
|
||||
"repeats": 10,
|
||||
"use_cuda_graphs": false,
|
||||
"v_head_dim": null,
|
||||
"warmup_iters": 5
|
||||
},
|
||||
"error": null,
|
||||
"kv_cache_update_time": {
|
||||
"max_ms": 0.08790399879217148,
|
||||
"mean_ms": 0.07967040091753005,
|
||||
"median_ms": 0.07808000221848488,
|
||||
"min_ms": 0.07596799731254578,
|
||||
"std_ms": 0.0036209252740976605
|
||||
},
|
||||
"max_time": 0.10734063720703126,
|
||||
"mean_time": 0.1069560287475586,
|
||||
"memory_allocated_mb": 368.1005859375,
|
||||
"memory_reserved_mb": 950.0,
|
||||
"min_time": 0.10689055633544922,
|
||||
"std_time": 0.00013032874360676438,
|
||||
"tensor_parallel_size": 1,
|
||||
"throughput_tokens_per_sec": 76592.22295299545
|
||||
},
|
||||
{
|
||||
"attention_core_excludes_kv_cache_update": true,
|
||||
"config": {
|
||||
"backend": "FLASH_ATTN",
|
||||
"batch_spec": "q8ks128k",
|
||||
"block_size": 16,
|
||||
"device": "cuda:0",
|
||||
"dtype": "torch.bfloat16",
|
||||
"head_dim": 128,
|
||||
"kv_cache_dtype": "auto",
|
||||
"kv_lora_rank": null,
|
||||
"num_kv_heads": 4,
|
||||
"num_kv_splits": null,
|
||||
"num_layers": 1,
|
||||
"num_q_heads": 32,
|
||||
"prefill_backend": null,
|
||||
"profile_memory": true,
|
||||
"qk_nope_head_dim": null,
|
||||
"qk_rope_head_dim": null,
|
||||
"reorder_batch_threshold": null,
|
||||
"repeats": 10,
|
||||
"use_cuda_graphs": false,
|
||||
"v_head_dim": null,
|
||||
"warmup_iters": 5
|
||||
},
|
||||
"error": null,
|
||||
"kv_cache_update_time": {
|
||||
"max_ms": 0.08534400165081024,
|
||||
"mean_ms": 0.07928640022873878,
|
||||
"median_ms": 0.07787200063467026,
|
||||
"min_ms": 0.07648000121116638,
|
||||
"std_ms": 0.0028305104107121735
|
||||
},
|
||||
"max_time": 0.1229840316772461,
|
||||
"mean_time": 0.12283342437744141,
|
||||
"memory_allocated_mb": 400.1044921875,
|
||||
"memory_reserved_mb": 1206.0,
|
||||
"min_time": 0.12271724700927734,
|
||||
"std_time": 9.625274868603512e-05,
|
||||
"tensor_parallel_size": 1,
|
||||
"throughput_tokens_per_sec": 66691.94514049937
|
||||
},
|
||||
{
|
||||
"attention_core_excludes_kv_cache_update": true,
|
||||
"config": {
|
||||
"backend": "FLASH_ATTN",
|
||||
"batch_spec": "q8ks136k",
|
||||
"block_size": 16,
|
||||
"device": "cuda:0",
|
||||
"dtype": "torch.bfloat16",
|
||||
"head_dim": 128,
|
||||
"kv_cache_dtype": "auto",
|
||||
"kv_lora_rank": null,
|
||||
"num_kv_heads": 4,
|
||||
"num_kv_splits": null,
|
||||
"num_layers": 1,
|
||||
"num_q_heads": 32,
|
||||
"prefill_backend": null,
|
||||
"profile_memory": true,
|
||||
"qk_nope_head_dim": null,
|
||||
"qk_rope_head_dim": null,
|
||||
"reorder_batch_threshold": null,
|
||||
"repeats": 10,
|
||||
"use_cuda_graphs": false,
|
||||
"v_head_dim": null,
|
||||
"warmup_iters": 5
|
||||
},
|
||||
"error": null,
|
||||
"kv_cache_update_time": {
|
||||
"max_ms": 0.08774399757385254,
|
||||
"mean_ms": 0.08081279993057251,
|
||||
"median_ms": 0.07972799986600876,
|
||||
"min_ms": 0.07664000242948532,
|
||||
"std_ms": 0.003487270970977775
|
||||
},
|
||||
"max_time": 0.13109359741210938,
|
||||
"mean_time": 0.13079229431152345,
|
||||
"memory_allocated_mb": 416.1064453125,
|
||||
"memory_reserved_mb": 1478.0,
|
||||
"min_time": 0.1306565399169922,
|
||||
"std_time": 0.00013138911961272913,
|
||||
"tensor_parallel_size": 1,
|
||||
"throughput_tokens_per_sec": 62633.65929255852
|
||||
},
|
||||
{
|
||||
"attention_core_excludes_kv_cache_update": true,
|
||||
"config": {
|
||||
"backend": "FLASH_ATTN",
|
||||
"batch_spec": "q512s128k",
|
||||
"block_size": 16,
|
||||
"device": "cuda:0",
|
||||
"dtype": "torch.bfloat16",
|
||||
"head_dim": 128,
|
||||
"kv_cache_dtype": "auto",
|
||||
"kv_lora_rank": null,
|
||||
"num_kv_heads": 4,
|
||||
"num_kv_splits": null,
|
||||
"num_layers": 1,
|
||||
"num_q_heads": 32,
|
||||
"prefill_backend": null,
|
||||
"profile_memory": true,
|
||||
"qk_nope_head_dim": null,
|
||||
"qk_rope_head_dim": null,
|
||||
"reorder_batch_threshold": null,
|
||||
"repeats": 10,
|
||||
"use_cuda_graphs": false,
|
||||
"v_head_dim": null,
|
||||
"warmup_iters": 5
|
||||
},
|
||||
"error": null,
|
||||
"kv_cache_update_time": {
|
||||
"max_ms": 0.040063999593257904,
|
||||
"mean_ms": 0.019459199998527764,
|
||||
"median_ms": 0.01601599995046854,
|
||||
"min_ms": 0.01532800029963255,
|
||||
"std_ms": 0.007496165774486326
|
||||
},
|
||||
"max_time": 0.009649087905883789,
|
||||
"mean_time": 0.009489968109130859,
|
||||
"memory_allocated_mb": 265.0458984375,
|
||||
"memory_reserved_mb": 1478.0,
|
||||
"min_time": 0.009443936347961425,
|
||||
"std_time": 5.961646816296781e-05,
|
||||
"tensor_parallel_size": 1,
|
||||
"throughput_tokens_per_sec": 53951.70922728124
|
||||
},
|
||||
{
|
||||
"attention_core_excludes_kv_cache_update": true,
|
||||
"config": {
|
||||
"backend": "FLASH_ATTN",
|
||||
"batch_spec": "q2ks66k",
|
||||
"block_size": 16,
|
||||
"device": "cuda:0",
|
||||
"dtype": "torch.bfloat16",
|
||||
"head_dim": 128,
|
||||
"kv_cache_dtype": "auto",
|
||||
"kv_lora_rank": null,
|
||||
"num_kv_heads": 4,
|
||||
"num_kv_splits": null,
|
||||
"num_layers": 1,
|
||||
"num_q_heads": 32,
|
||||
"prefill_backend": null,
|
||||
"profile_memory": true,
|
||||
"qk_nope_head_dim": null,
|
||||
"qk_rope_head_dim": null,
|
||||
"reorder_batch_threshold": null,
|
||||
"repeats": 10,
|
||||
"use_cuda_graphs": false,
|
||||
"v_head_dim": null,
|
||||
"warmup_iters": 5
|
||||
},
|
||||
"error": null,
|
||||
"kv_cache_update_time": {
|
||||
"max_ms": 0.040832001715898514,
|
||||
"mean_ms": 0.02942080032080412,
|
||||
"median_ms": 0.027888000011444092,
|
||||
"min_ms": 0.02691200003027916,
|
||||
"std_ms": 0.004040048279091587
|
||||
},
|
||||
"max_time": 0.016906623840332032,
|
||||
"mean_time": 0.01678766403198242,
|
||||
"memory_allocated_mb": 168.04248046875,
|
||||
"memory_reserved_mb": 1478.0,
|
||||
"min_time": 0.016753759384155274,
|
||||
"std_time": 4.236470233451937e-05,
|
||||
"tensor_parallel_size": 1,
|
||||
"throughput_tokens_per_sec": 121994.34037387963
|
||||
},
|
||||
{
|
||||
"attention_core_excludes_kv_cache_update": true,
|
||||
"config": {
|
||||
"backend": "FLASH_ATTN",
|
||||
"batch_spec": "q4ks100k",
|
||||
"block_size": 16,
|
||||
"device": "cuda:0",
|
||||
"dtype": "torch.bfloat16",
|
||||
"head_dim": 128,
|
||||
"kv_cache_dtype": "auto",
|
||||
"kv_lora_rank": null,
|
||||
"num_kv_heads": 4,
|
||||
"num_kv_splits": null,
|
||||
"num_layers": 1,
|
||||
"num_q_heads": 32,
|
||||
"prefill_backend": null,
|
||||
"profile_memory": true,
|
||||
"qk_nope_head_dim": null,
|
||||
"qk_rope_head_dim": null,
|
||||
"reorder_batch_threshold": null,
|
||||
"repeats": 10,
|
||||
"use_cuda_graphs": false,
|
||||
"v_head_dim": null,
|
||||
"warmup_iters": 5
|
||||
},
|
||||
"error": null,
|
||||
"kv_cache_update_time": {
|
||||
"max_ms": 0.04867200180888176,
|
||||
"mean_ms": 0.04565120078623295,
|
||||
"median_ms": 0.04468800127506256,
|
||||
"min_ms": 0.04416000097990036,
|
||||
"std_ms": 0.0018079971851529544
|
||||
},
|
||||
"max_time": 0.05050300979614258,
|
||||
"mean_time": 0.05035054740905761,
|
||||
"memory_allocated_mb": 272.06640625,
|
||||
"memory_reserved_mb": 1478.0,
|
||||
"min_time": 0.05029715347290039,
|
||||
"std_time": 7.24480602714254e-05,
|
||||
"tensor_parallel_size": 1,
|
||||
"throughput_tokens_per_sec": 81349.66173700758
|
||||
},
|
||||
{
|
||||
"attention_core_excludes_kv_cache_update": true,
|
||||
"config": {
|
||||
"backend": "FLASH_ATTN",
|
||||
"batch_spec": "q6ks134k",
|
||||
"block_size": 16,
|
||||
"device": "cuda:0",
|
||||
"dtype": "torch.bfloat16",
|
||||
"head_dim": 128,
|
||||
"kv_cache_dtype": "auto",
|
||||
"kv_lora_rank": null,
|
||||
"num_kv_heads": 4,
|
||||
"num_kv_splits": null,
|
||||
"num_layers": 1,
|
||||
"num_q_heads": 32,
|
||||
"prefill_backend": null,
|
||||
"profile_memory": true,
|
||||
"qk_nope_head_dim": null,
|
||||
"qk_rope_head_dim": null,
|
||||
"reorder_batch_threshold": null,
|
||||
"repeats": 10,
|
||||
"use_cuda_graphs": false,
|
||||
"v_head_dim": null,
|
||||
"warmup_iters": 5
|
||||
},
|
||||
"error": null,
|
||||
"kv_cache_update_time": {
|
||||
"max_ms": 0.06684800237417221,
|
||||
"mean_ms": 0.06276160031557083,
|
||||
"median_ms": 0.061824001371860504,
|
||||
"min_ms": 0.060447998344898224,
|
||||
"std_ms": 0.0024045636532566625
|
||||
},
|
||||
"max_time": 0.09653209686279297,
|
||||
"mean_time": 0.0961898666381836,
|
||||
"memory_allocated_mb": 376.09033203125,
|
||||
"memory_reserved_mb": 1478.0,
|
||||
"min_time": 0.09607142639160156,
|
||||
"std_time": 0.0001365794879996899,
|
||||
"tensor_parallel_size": 1,
|
||||
"throughput_tokens_per_sec": 63873.67208970714
|
||||
},
|
||||
{
|
||||
"attention_core_excludes_kv_cache_update": true,
|
||||
"config": {
|
||||
"backend": "FLASH_ATTN",
|
||||
"batch_spec": "q1ks8k",
|
||||
"block_size": 16,
|
||||
"device": "cuda:0",
|
||||
"dtype": "torch.bfloat16",
|
||||
"head_dim": 128,
|
||||
"kv_cache_dtype": "auto",
|
||||
"kv_lora_rank": null,
|
||||
"num_kv_heads": 4,
|
||||
"num_kv_splits": null,
|
||||
"num_layers": 1,
|
||||
"num_q_heads": 32,
|
||||
"prefill_backend": null,
|
||||
"profile_memory": true,
|
||||
"qk_nope_head_dim": null,
|
||||
"qk_rope_head_dim": null,
|
||||
"reorder_batch_threshold": null,
|
||||
"repeats": 10,
|
||||
"use_cuda_graphs": false,
|
||||
"v_head_dim": null,
|
||||
"warmup_iters": 5
|
||||
},
|
||||
"error": null,
|
||||
"kv_cache_update_time": {
|
||||
"max_ms": 0.0326399989426136,
|
||||
"mean_ms": 0.021955199912190436,
|
||||
"median_ms": 0.020560000091791153,
|
||||
"min_ms": 0.018559999763965607,
|
||||
"std_ms": 0.00393975597463501
|
||||
},
|
||||
"max_time": 0.0011526399850845337,
|
||||
"mean_time": 0.0011425888061523438,
|
||||
"memory_allocated_mb": 34.0205078125,
|
||||
"memory_reserved_mb": 1480.0,
|
||||
"min_time": 0.0011403199434280396,
|
||||
"std_time": 3.498447348453724e-06,
|
||||
"tensor_parallel_size": 1,
|
||||
"throughput_tokens_per_sec": 896210.4253833097
|
||||
},
|
||||
{
|
||||
"attention_core_excludes_kv_cache_update": true,
|
||||
"config": {
|
||||
"backend": "FLASH_ATTN",
|
||||
"batch_spec": "q512s4k",
|
||||
"block_size": 16,
|
||||
"device": "cuda:0",
|
||||
"dtype": "torch.bfloat16",
|
||||
"head_dim": 128,
|
||||
"kv_cache_dtype": "auto",
|
||||
"kv_lora_rank": null,
|
||||
"num_kv_heads": 4,
|
||||
"num_kv_splits": null,
|
||||
"num_layers": 1,
|
||||
"num_q_heads": 32,
|
||||
"prefill_backend": null,
|
||||
"profile_memory": true,
|
||||
"qk_nope_head_dim": null,
|
||||
"qk_rope_head_dim": null,
|
||||
"reorder_batch_threshold": null,
|
||||
"repeats": 10,
|
||||
"use_cuda_graphs": false,
|
||||
"v_head_dim": null,
|
||||
"warmup_iters": 5
|
||||
},
|
||||
"error": null,
|
||||
"kv_cache_update_time": {
|
||||
"max_ms": 0.02783999964594841,
|
||||
"mean_ms": 0.017366399988532066,
|
||||
"median_ms": 0.01611199975013733,
|
||||
"min_ms": 0.015424000099301338,
|
||||
"std_ms": 0.003547472261377056
|
||||
},
|
||||
"max_time": 0.00034601598978042605,
|
||||
"mean_time": 0.0003366623997688293,
|
||||
"memory_allocated_mb": 17.015625,
|
||||
"memory_reserved_mb": 1480.0,
|
||||
"min_time": 0.0003308480083942413,
|
||||
"std_time": 4.813544996936389e-06,
|
||||
"tensor_parallel_size": 1,
|
||||
"throughput_tokens_per_sec": 1520811.3539010207
|
||||
}
|
||||
],
|
||||
"schema_version": "qwen30_vllm020_flashattn_raw.v1"
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
f113ab5a05167f8ac4c1938c4a04fadc11d5dc2bb8c536a69db23638e6ef73d4 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/raw/flashattn-code-longctx-tp1.json
|
||||
4e6d6002db037bdc71d40f76d6de979ef47960612c15dd985c04048c25bd1d41 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/aituner.commit
|
||||
420e4a613b1bcdb44b873eaa911f4a23129713ff9d486fc783b31a009a55cc85 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/batch-specs.txt
|
||||
5dd612a7806900e9d12afc27ccef196cc5518aea831aa4ed3b0a53bcfb5746cf runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/max-model-len.txt
|
||||
484da176e7355ab5532d5129a8e78bea0ae84b3aa0a42c7f0bea9689667762ff runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/pip-freeze.txt
|
||||
595b25965f76bf7ac25aa3b60126a437a083619303303aaf63346e993862253a runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/source.sha256
|
||||
4355a46b19d348dc2f57c046f8ef63d4538ebb936000f3c9ee954a27460dd865 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/vllm-allow-long-max-model-len.txt
|
||||
11e8f5af440e3db4c01b519a7b0e30bbcecc3a57bfb53dd86d32e601beb95a32 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/vllm-source.commit
|
||||
@@ -0,0 +1 @@
|
||||
e251046c306a8e03214fe1b2f40caabb98f76b7b
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user