1445 lines
57 KiB
Python
1445 lines
57 KiB
Python
#!/usr/bin/env python3
|
||
"""Analyze frozen good/bad Frontier selection cases by margin and residual."""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import json
|
||
import math
|
||
import os
|
||
import re
|
||
from dataclasses import dataclass
|
||
from pathlib import Path
|
||
from statistics import fmean
|
||
from typing import Any
|
||
|
||
|
||
HERE = Path(__file__).resolve().parent
|
||
DEFAULT_INPUT_ROOT = HERE / "frozen-inputs"
|
||
DEFAULT_RESULTS_ROOT = HERE / "results"
|
||
|
||
OBJECTIVE_ORDER = {
|
||
"ttft_mean_ms": 0,
|
||
"ttft_p90_ms": 1,
|
||
"tpot_mean_ms": 2,
|
||
"tpot_p90_ms": 3,
|
||
"e2e_mean_ms": 4,
|
||
"e2e_p90_ms": 5,
|
||
}
|
||
|
||
CONFIG_PATTERN = re.compile(
|
||
r"^tp(?P<tp>[0-9]+)(?:_ep(?P<ep>[0-9]+))?_mns(?P<mns>[0-9]+)$"
|
||
)
|
||
INVERSION_AXES = ("tp-axis", "mns-axis", "mixed")
|
||
|
||
|
||
@dataclass(frozen=True)
|
||
class Surface:
|
||
case: str
|
||
objective: str
|
||
values: dict[str, tuple[float, float]]
|
||
expected_configs: tuple[str, ...]
|
||
expected_sim_winner: str
|
||
expected_real_winner: str
|
||
expected_regret: float | None
|
||
sources: tuple[str, ...]
|
||
|
||
|
||
def load_json(path: Path) -> dict[str, Any]:
|
||
with path.open() as handle:
|
||
value = json.load(handle)
|
||
if not isinstance(value, dict):
|
||
raise ValueError(f"expected JSON object: {path}")
|
||
return value
|
||
|
||
|
||
def atomic_write(path: Path, text: str) -> None:
|
||
path.parent.mkdir(parents=True, exist_ok=True)
|
||
temporary = path.with_suffix(path.suffix + ".tmp")
|
||
temporary.write_text(text)
|
||
os.replace(temporary, path)
|
||
|
||
|
||
def positive_float(value: Any, context: str) -> float:
|
||
if not isinstance(value, (int, float)) or isinstance(value, bool):
|
||
raise ValueError(f"non-numeric value for {context}: {value!r}")
|
||
result = float(value)
|
||
if not math.isfinite(result) or result <= 0:
|
||
raise ValueError(f"non-positive or non-finite value for {context}: {value!r}")
|
||
return result
|
||
|
||
|
||
def normalized_objective(metric: str, statistic: str) -> str:
|
||
return f"{metric.removesuffix('_ms')}_{statistic}_ms"
|
||
|
||
|
||
def objective_key(objective: str) -> tuple[int, str]:
|
||
return OBJECTIVE_ORDER.get(objective, len(OBJECTIVE_ORDER)), objective
|
||
|
||
|
||
def config_dimensions(config: str) -> tuple[tuple[int, int | None], int]:
|
||
match = CONFIG_PATTERN.fullmatch(config)
|
||
if match is None:
|
||
raise ValueError(f"unrecognized config name: {config}")
|
||
ep = int(match.group("ep")) if match.group("ep") is not None else None
|
||
return (int(match.group("tp")), ep), int(match.group("mns"))
|
||
|
||
|
||
def pair_axis(left: str, right: str) -> str:
|
||
if left == right:
|
||
return "none"
|
||
left_plan, left_mns = config_dimensions(left)
|
||
right_plan, right_mns = config_dimensions(right)
|
||
plan_differs = left_plan != right_plan
|
||
mns_differs = left_mns != right_mns
|
||
if plan_differs and not mns_differs:
|
||
return "tp-axis"
|
||
if mns_differs and not plan_differs:
|
||
return "mns-axis"
|
||
if plan_differs and mns_differs:
|
||
return "mixed"
|
||
raise ValueError(f"distinct configs have identical dimensions: {left}, {right}")
|
||
|
||
|
||
def load_q30_trace_pd(input_root: Path) -> list[Surface]:
|
||
relative = Path("q30-trace-pd/comparison.json")
|
||
data = load_json(input_root / relative)
|
||
if data.get("schema") != "frontier-qwen30-piecewise-graph-comparison-v1":
|
||
raise ValueError(f"unexpected schema: {relative}")
|
||
configs = tuple(sorted(data["cells"]))
|
||
surfaces = []
|
||
for source_objective, selection in sorted(
|
||
data["selection"].items(),
|
||
key=lambda item: objective_key(
|
||
normalized_objective(*item[0].split(":"))
|
||
),
|
||
):
|
||
metric, statistic = source_objective.split(":")
|
||
objective = normalized_objective(metric, statistic)
|
||
values = {}
|
||
for config in configs:
|
||
cell = data["cells"][config]["metrics"][metric]
|
||
values[config] = (
|
||
positive_float(cell[f"sim_{statistic}_ms"], f"{relative}:{config}:sim"),
|
||
positive_float(cell[f"real_{statistic}_ms"], f"{relative}:{config}:real"),
|
||
)
|
||
surfaces.append(
|
||
Surface(
|
||
case="q30 trace-pd",
|
||
objective=objective,
|
||
values=values,
|
||
expected_configs=configs,
|
||
expected_sim_winner=selection["sim_winner"],
|
||
expected_real_winner=selection["real_winner"],
|
||
expected_regret=None,
|
||
sources=(str(relative),),
|
||
)
|
||
)
|
||
return surfaces
|
||
|
||
|
||
def load_q30_case(
|
||
input_root: Path,
|
||
case: str,
|
||
comparison_relative: Path,
|
||
audit_relative: Path,
|
||
) -> list[Surface]:
|
||
comparison = load_json(input_root / comparison_relative)
|
||
audit = load_json(input_root / audit_relative)
|
||
if comparison.get("schema") != "qwen30-latency-case-frontier-real-comparison-v1":
|
||
raise ValueError(f"unexpected schema: {comparison_relative}")
|
||
if audit.get("schema") != "qwen30-latency-case-real-audit-v1":
|
||
raise ValueError(f"unexpected schema: {audit_relative}")
|
||
configs = tuple(sorted(set(comparison["cells"]) | set(audit["configs"])))
|
||
surfaces = []
|
||
for source_objective, selection in sorted(
|
||
comparison["selection"].items(),
|
||
key=lambda item: objective_key(
|
||
normalized_objective(*item[0].split(":"))
|
||
),
|
||
):
|
||
metric, statistic = source_objective.split(":")
|
||
objective = normalized_objective(metric, statistic)
|
||
values = {}
|
||
for config in configs:
|
||
if config not in comparison["cells"] or config not in audit["configs"]:
|
||
continue
|
||
sim = comparison["cells"][config]["metrics"].get(metric, {}).get(statistic)
|
||
real = (
|
||
audit["configs"][config]["metrics"]
|
||
.get(metric, {})
|
||
.get(f"pooled_{statistic}_ms")
|
||
)
|
||
if sim is None or real is None:
|
||
continue
|
||
values[config] = (
|
||
positive_float(sim, f"{comparison_relative}:{config}:sim"),
|
||
positive_float(real, f"{audit_relative}:{config}:real"),
|
||
)
|
||
surfaces.append(
|
||
Surface(
|
||
case=case,
|
||
objective=objective,
|
||
values=values,
|
||
expected_configs=configs,
|
||
expected_sim_winner=selection["sim_winner"],
|
||
expected_real_winner=selection["real_winner"],
|
||
expected_regret=selection.get("selected_config_real_regret"),
|
||
sources=(str(comparison_relative), str(audit_relative)),
|
||
)
|
||
)
|
||
return surfaces
|
||
|
||
|
||
def load_q235_campaign(
|
||
input_root: Path, campaign: str, relative: Path
|
||
) -> list[Surface]:
|
||
data = load_json(input_root / relative)
|
||
if data.get("schema") != "qwen235-v020-simulator-real-comparison-v1":
|
||
raise ValueError(f"unexpected schema: {relative}")
|
||
surfaces = []
|
||
for case_name in ("fixed-pd", "fixed-po", "trace-pd", "trace-po"):
|
||
case = data["cases"][case_name]
|
||
configs = tuple(sorted(set(case["sim"]) | set(case["real"])))
|
||
for objective, selection in sorted(
|
||
case["comparison"].items(), key=lambda item: objective_key(item[0])
|
||
):
|
||
values = {}
|
||
for config in configs:
|
||
if config not in case["sim"] or config not in case["real"]:
|
||
continue
|
||
sim = case["sim"][config].get(objective)
|
||
real = case["real"][config].get(objective)
|
||
if sim is None or real is None:
|
||
continue
|
||
values[config] = (
|
||
positive_float(sim, f"{relative}:{case_name}:{config}:sim"),
|
||
positive_float(real, f"{relative}:{case_name}:{config}:real"),
|
||
)
|
||
surfaces.append(
|
||
Surface(
|
||
case=f"q235 {campaign} {case_name}",
|
||
objective=objective,
|
||
values=values,
|
||
expected_configs=configs,
|
||
expected_sim_winner=selection["sim_winner"],
|
||
expected_real_winner=selection["real_winner"],
|
||
expected_regret=selection.get("selected_real_regret"),
|
||
sources=(str(relative),),
|
||
)
|
||
)
|
||
return surfaces
|
||
|
||
|
||
def load_surfaces(input_root: Path) -> list[Surface]:
|
||
surfaces = load_q30_trace_pd(input_root)
|
||
q30_specs = (
|
||
(
|
||
"q30 fixed-pd high",
|
||
Path("q30-fixed-hi/fixed-pd/comparison.json"),
|
||
Path("q30-fixed-hi/fixed-pd/real-audit.json"),
|
||
),
|
||
(
|
||
"q30 fixed-po high",
|
||
Path("q30-fixed-hi/fixed-po/comparison.json"),
|
||
Path("q30-fixed-hi/fixed-po/real-audit.json"),
|
||
),
|
||
(
|
||
"q30 fixed-pd low",
|
||
Path("q30-expansion-lo/fixed-pd-comparison.json"),
|
||
Path("q30-expansion-lo/fixed-pd-real-audit.json"),
|
||
),
|
||
(
|
||
"q30 fixed-po low",
|
||
Path("q30-expansion-lo/fixed-po-comparison.json"),
|
||
Path("q30-expansion-lo/fixed-po-real-audit.json"),
|
||
),
|
||
(
|
||
"q30 trace-po low",
|
||
Path("q30-expansion-lo/trace-po-comparison.json"),
|
||
Path("q30-expansion-lo/trace-po-real-audit.json"),
|
||
),
|
||
)
|
||
for case, comparison, audit in q30_specs:
|
||
surfaces.extend(load_q30_case(input_root, case, comparison, audit))
|
||
surfaces.extend(
|
||
load_q235_campaign(
|
||
input_root, "A0", Path("q235-fourcase-a0/comparison.json")
|
||
)
|
||
)
|
||
surfaces.extend(
|
||
load_q235_campaign(
|
||
input_root, "A1", Path("q235-ablation-a1/comparison.json")
|
||
)
|
||
)
|
||
return surfaces
|
||
|
||
|
||
def analyze_surface(surface: Surface) -> tuple[dict[str, Any], list[str]]:
|
||
anomalies = []
|
||
paired_configs = tuple(sorted(surface.values))
|
||
missing_configs = sorted(set(surface.expected_configs) - set(paired_configs))
|
||
if missing_configs:
|
||
anomalies.append(
|
||
f"{surface.case} / {surface.objective}: missing paired sim+real data for "
|
||
+ ", ".join(missing_configs)
|
||
)
|
||
if len(paired_configs) < 2:
|
||
return (
|
||
{
|
||
"case": surface.case,
|
||
"objective": surface.objective,
|
||
"sim_winner": None,
|
||
"real_winner": None,
|
||
"regret": None,
|
||
"config_uniform_scale": None,
|
||
"log_ratio_spread_all": None,
|
||
"log_ratio_spread_real_top3": None,
|
||
"real_relative_margin_best_vs_second": None,
|
||
"sim_winner_real_relative_loss": None,
|
||
"failure": None,
|
||
"h_scale_condition_spread_gt_log_margin": None,
|
||
"h_scale_verdict": "N/A",
|
||
"winner_match": None,
|
||
"winner_deciding_axis": None,
|
||
"winner_pair_m_s_percent": None,
|
||
"winner_pair_delta_s_percent": None,
|
||
"winner_pair_strict_reversal": None,
|
||
"winner_pair_relation": "N/A",
|
||
"n_inv_tp": None,
|
||
"n_inv_mns": None,
|
||
"n_inv_mixed": None,
|
||
"n_tie_tp": None,
|
||
"n_tie_mns": None,
|
||
"n_tie_mixed": None,
|
||
"fragile_success": None,
|
||
"pairwise_vs_real_winner": [],
|
||
"paired_config_count": len(paired_configs),
|
||
"expected_config_count": len(surface.expected_configs),
|
||
"real_top3_configs": [],
|
||
"missing_configs": missing_configs,
|
||
"sources": list(surface.sources),
|
||
},
|
||
anomalies,
|
||
)
|
||
|
||
sim_ranking = sorted(paired_configs, key=lambda config: (surface.values[config][0], config))
|
||
real_ranking = sorted(paired_configs, key=lambda config: (surface.values[config][1], config))
|
||
sim_winner = sim_ranking[0]
|
||
real_winner = real_ranking[0]
|
||
real_best = surface.values[real_winner][1]
|
||
regret = surface.values[sim_winner][1] / real_best - 1.0
|
||
margin = surface.values[real_ranking[1]][1] / real_best - 1.0
|
||
log_ratios = {
|
||
config: math.log(surface.values[config][0] / surface.values[config][1])
|
||
for config in paired_configs
|
||
}
|
||
spread_all = max(log_ratios.values()) - min(log_ratios.values())
|
||
top3 = tuple(real_ranking[:3])
|
||
top3_ratios = [log_ratios[config] for config in top3]
|
||
spread_top3 = max(top3_ratios) - min(top3_ratios)
|
||
scale = math.exp(fmean(log_ratios.values()))
|
||
failure = regret > 0.05
|
||
residual_exceeds_log_margin = spread_top3 > math.log1p(margin)
|
||
h_scale_holds = failure == residual_exceeds_log_margin
|
||
|
||
pairwise = []
|
||
inversion_counts = {axis: 0 for axis in INVERSION_AXES}
|
||
sim_at_real_winner = surface.values[real_winner][0]
|
||
for config in paired_configs:
|
||
sim_value, real_value = surface.values[config]
|
||
real_pair_log = math.log(real_value / real_best)
|
||
sim_pair_log = math.log(sim_value / sim_at_real_winner)
|
||
pair_margin_percent = 100.0 * real_pair_log
|
||
delta_percent = 100.0 * (sim_pair_log - real_pair_log)
|
||
axis = pair_axis(real_winner, config)
|
||
inversion = real_value > real_best and sim_value < sim_at_real_winner
|
||
delta_inversion = (
|
||
pair_margin_percent > 0
|
||
and delta_percent < -pair_margin_percent
|
||
)
|
||
if inversion != delta_inversion:
|
||
anomalies.append(
|
||
f"{surface.case} / {surface.objective} / {config}: direct inversion "
|
||
f"test disagrees with delta < -margin"
|
||
)
|
||
if inversion:
|
||
inversion_counts[axis] += 1
|
||
pairwise.append(
|
||
{
|
||
"config": config,
|
||
"axis_vs_real_winner": axis,
|
||
"real_pair_margin_percent": pair_margin_percent,
|
||
"signed_pairwise_delta_percent": delta_percent,
|
||
"inversion": inversion,
|
||
}
|
||
)
|
||
|
||
winner_pair = next(item for item in pairwise if item["config"] == sim_winner)
|
||
winner_match = sim_winner == real_winner
|
||
if winner_match:
|
||
winner_pair_relation = "match"
|
||
elif winner_pair["inversion"]:
|
||
winner_pair_relation = "reversal"
|
||
elif surface.values[sim_winner][0] == sim_at_real_winner:
|
||
winner_pair_relation = "sim-tie"
|
||
else:
|
||
winner_pair_relation = "non-reversal"
|
||
tie_counts = {axis: 0 for axis in INVERSION_AXES}
|
||
if winner_pair_relation == "sim-tie":
|
||
tie_counts[winner_pair["axis_vs_real_winner"]] += 1
|
||
|
||
if sim_winner != surface.expected_sim_winner:
|
||
anomalies.append(
|
||
f"{surface.case} / {surface.objective}: recomputed sim winner "
|
||
f"{sim_winner} != frozen JSON {surface.expected_sim_winner}"
|
||
)
|
||
if real_winner != surface.expected_real_winner:
|
||
anomalies.append(
|
||
f"{surface.case} / {surface.objective}: recomputed real winner "
|
||
f"{real_winner} != frozen JSON {surface.expected_real_winner}"
|
||
)
|
||
if surface.expected_regret is not None and not math.isclose(
|
||
regret, float(surface.expected_regret), rel_tol=1e-12, abs_tol=1e-12
|
||
):
|
||
anomalies.append(
|
||
f"{surface.case} / {surface.objective}: recomputed regret "
|
||
f"{regret:.12g} != frozen JSON {float(surface.expected_regret):.12g}"
|
||
)
|
||
|
||
return (
|
||
{
|
||
"case": surface.case,
|
||
"objective": surface.objective,
|
||
"sim_winner": sim_winner,
|
||
"real_winner": real_winner,
|
||
"regret": regret,
|
||
"config_uniform_scale": scale,
|
||
"log_ratio_spread_all": spread_all,
|
||
"log_ratio_spread_real_top3": spread_top3,
|
||
"real_relative_margin_best_vs_second": margin,
|
||
"sim_winner_real_relative_loss": regret,
|
||
"failure": failure,
|
||
"h_scale_condition_spread_gt_log_margin": residual_exceeds_log_margin,
|
||
"h_scale_verdict": "HOLDS" if h_scale_holds else "COUNTEREXAMPLE",
|
||
"winner_match": winner_match,
|
||
"winner_deciding_axis": pair_axis(real_winner, sim_winner),
|
||
"winner_pair_m_s_percent": winner_pair["real_pair_margin_percent"],
|
||
"winner_pair_delta_s_percent": winner_pair["signed_pairwise_delta_percent"],
|
||
"winner_pair_strict_reversal": winner_pair["inversion"],
|
||
"winner_pair_relation": winner_pair_relation,
|
||
"n_inv_tp": inversion_counts["tp-axis"],
|
||
"n_inv_mns": inversion_counts["mns-axis"],
|
||
"n_inv_mixed": inversion_counts["mixed"],
|
||
"n_tie_tp": tie_counts["tp-axis"],
|
||
"n_tie_mns": tie_counts["mns-axis"],
|
||
"n_tie_mixed": tie_counts["mixed"],
|
||
"fragile_success": winner_match and margin < 0.01,
|
||
"pairwise_vs_real_winner": pairwise,
|
||
"paired_config_count": len(paired_configs),
|
||
"expected_config_count": len(surface.expected_configs),
|
||
"real_top3_configs": list(top3),
|
||
"missing_configs": missing_configs,
|
||
"sources": list(surface.sources),
|
||
},
|
||
anomalies,
|
||
)
|
||
|
||
|
||
def row_index(rows: list[dict[str, Any]]) -> dict[tuple[str, str], dict[str, Any]]:
|
||
return {(row["case"], row["objective"]): row for row in rows}
|
||
|
||
|
||
def cross_checks(rows: list[dict[str, Any]]) -> tuple[list[dict[str, Any]], list[str]]:
|
||
by_key = row_index(rows)
|
||
checks = []
|
||
anomalies = []
|
||
|
||
def check(name: str, observed: float, expected_percent: float) -> None:
|
||
observed_percent = observed * 100.0
|
||
passed = round(observed_percent, 1) == expected_percent
|
||
checks.append(
|
||
{
|
||
"name": name,
|
||
"observed_percent": observed_percent,
|
||
"expected_percent": expected_percent,
|
||
"passed": passed,
|
||
}
|
||
)
|
||
if not passed:
|
||
anomalies.append(
|
||
f"hard cross-check failed: {name}: observed {observed_percent:.6f}% "
|
||
f"vs expected {expected_percent:.1f}%"
|
||
)
|
||
|
||
check(
|
||
"q30 high-pressure fixed-pd TTFT mean regret",
|
||
by_key[("q30 fixed-pd high", "ttft_mean_ms")]["regret"],
|
||
58.0,
|
||
)
|
||
check(
|
||
"q235 A0 fixed-pd TPOT mean regret",
|
||
by_key[("q235 A0 fixed-pd", "tpot_mean_ms")]["regret"],
|
||
33.0,
|
||
)
|
||
check(
|
||
"q235 A1 fixed-pd TPOT mean regret",
|
||
by_key[("q235 A1 fixed-pd", "tpot_mean_ms")]["regret"],
|
||
33.0,
|
||
)
|
||
trace_pd_rows = [row for row in rows if row["case"] == "q30 trace-pd"]
|
||
trace_passed = len(trace_pd_rows) == 6 and all(
|
||
round(row["regret"] * 100.0, 1) == 0.0 for row in trace_pd_rows
|
||
)
|
||
checks.append(
|
||
{
|
||
"name": "q30 trace-pd all-objective regret",
|
||
"observed_percent": [row["regret"] * 100.0 for row in trace_pd_rows],
|
||
"expected_percent": [0.0] * 6,
|
||
"passed": trace_passed,
|
||
}
|
||
)
|
||
if not trace_passed:
|
||
anomalies.append(
|
||
"hard cross-check failed: q30 trace-pd does not have six objectives all at 0.0% regret"
|
||
)
|
||
|
||
def check_sequence(
|
||
name: str, observed: list[float], expected_percent: list[float]
|
||
) -> None:
|
||
observed_percent = [value * 100.0 for value in observed]
|
||
passed = [round(value, 1) for value in observed_percent] == expected_percent
|
||
checks.append(
|
||
{
|
||
"name": name,
|
||
"observed_percent": observed_percent,
|
||
"expected_percent": expected_percent,
|
||
"passed": passed,
|
||
}
|
||
)
|
||
if not passed:
|
||
anomalies.append(
|
||
f"hard cross-check failed: {name}: observed "
|
||
f"{[round(value, 6) for value in observed_percent]} vs expected "
|
||
f"{expected_percent}"
|
||
)
|
||
|
||
check_sequence(
|
||
"q235 Trace-PO TTFT p90 A0 to A1 regret",
|
||
[
|
||
by_key[("q235 A0 trace-po", "ttft_p90_ms")]["regret"],
|
||
by_key[("q235 A1 trace-po", "ttft_p90_ms")]["regret"],
|
||
],
|
||
[21.2, 0.3],
|
||
)
|
||
fixed_pd_objectives = (
|
||
"tpot_mean_ms",
|
||
"tpot_p90_ms",
|
||
"e2e_mean_ms",
|
||
"e2e_p90_ms",
|
||
)
|
||
fixed_pd_expected = [33.0, 37.2, 30.7, 34.6]
|
||
check_sequence(
|
||
"q235 Fixed-PD TPOT/E2E regrets unchanged from A0 to A1",
|
||
[
|
||
by_key[(f"q235 {campaign} fixed-pd", objective)]["regret"]
|
||
for campaign in ("A0", "A1")
|
||
for objective in fixed_pd_objectives
|
||
],
|
||
fixed_pd_expected + fixed_pd_expected,
|
||
)
|
||
return checks, anomalies
|
||
|
||
|
||
def known_attribution(input_root: Path) -> dict[str, Any]:
|
||
q30_relative = Path("q30-admission-diag/result.json")
|
||
q235_relative = Path("q235-state-diag/state-diagnosis-exact.json")
|
||
q30 = load_json(input_root / q30_relative)
|
||
q235 = load_json(input_root / q235_relative)
|
||
if q30.get("schema") != "qwen30-fixed-pd-ttft-admission-diagnosis-v1":
|
||
raise ValueError(f"unexpected schema: {q30_relative}")
|
||
if q235.get("schema") != "qwen235-fixed-pd-state-diagnosis-v1":
|
||
raise ValueError(f"unexpected schema: {q235_relative}")
|
||
|
||
q30_cells = []
|
||
for config, cell in sorted(q30["cells"].items()):
|
||
q30_cells.append(
|
||
{
|
||
"config": config,
|
||
"sim_tpot_over_real": cell["tpot_overprediction_ratio"],
|
||
"sim_required_slots": cell["simulator"]["required_slots"],
|
||
"mns": cell["mns"],
|
||
"real_required_slots_upper_bound": cell["real"]["required_slots_upper_bound"],
|
||
"sim_first_scheduling_wait_ms": cell["simulator"]["first_scheduling_delay_ms"],
|
||
"sim_queue_free_ttft_ms": cell["simulator"]["queue_free_ttft_ms"],
|
||
"sim_ttft_ms": cell["simulator"]["ttft_ms"],
|
||
"real_ttft_ms": cell["real"]["ttft_ms"],
|
||
"real_waiting_max": cell["real"]["periodic_queue"]["waiting_max"],
|
||
}
|
||
)
|
||
|
||
q235_configs = []
|
||
for config in sorted(q235["simulator"]):
|
||
q235_configs.append(
|
||
{
|
||
"config": config,
|
||
"sim_own_decode_batch_mean": q235["simulator"][config]["decode_batch_size"]["mean"],
|
||
"real_exact_decode_batch_mean": q235["real_iteration_state"][config]["decode_batch_size"]["mean"],
|
||
"real_token_weighted_iteration_ms": q235["real_iteration_state"][config]["decode_token_weighted_iteration_elapsed_ms"],
|
||
"exact_state_matched_sim_ms": q235["exact_state_matched"]["configs"][config]["components_ms"]["total"],
|
||
"exact_state_coverage": q235["exact_state_matched"]["configs"][config]["coverage"],
|
||
}
|
||
)
|
||
|
||
component_order = (
|
||
"total",
|
||
"moe_compute",
|
||
"attention",
|
||
"tp_dp_communication",
|
||
"ep_communication",
|
||
"moe_routing",
|
||
"dense_mlp_compute",
|
||
"pipeline_communication",
|
||
"runtime_overhead",
|
||
)
|
||
q235_components = []
|
||
own = q235["simulator_internal_all_step_tp8_minus_tp4_ms"]
|
||
exact = q235["exact_state_matched"]["tp8_minus_tp4_ms"]
|
||
for component in component_order:
|
||
q235_components.append(
|
||
{
|
||
"component": component,
|
||
"own_composition_tp8_minus_tp4_ms": own[component],
|
||
"exact_state_tp8_minus_tp4_ms": exact[component],
|
||
}
|
||
)
|
||
return {
|
||
"q30_admission": {
|
||
"source": str(q30_relative),
|
||
"verdict": q30["verdict"],
|
||
"cells": q30_cells,
|
||
"contrasts": q30["contrasts"],
|
||
},
|
||
"q235_state": {
|
||
"source": str(q235_relative),
|
||
"verdict": q235["verdict"],
|
||
"configs": q235_configs,
|
||
"components": q235_components,
|
||
"contrast_decomposition": q235["contrast_decomposition"],
|
||
"reference": q235["reference"],
|
||
},
|
||
}
|
||
|
||
|
||
def directional_summary(rows: list[dict[str, Any]]) -> dict[str, Any]:
|
||
case_order = list(dict.fromkeys(row["case"] for row in rows))
|
||
by_case = []
|
||
for case in case_order:
|
||
case_rows = [row for row in rows if row["case"] == case]
|
||
by_case.append(
|
||
{
|
||
"case": case,
|
||
"n_inv_tp": sum(row["n_inv_tp"] for row in case_rows),
|
||
"n_inv_mns": sum(row["n_inv_mns"] for row in case_rows),
|
||
"n_inv_mixed": sum(row["n_inv_mixed"] for row in case_rows),
|
||
"n_tie_tp": sum(row["n_tie_tp"] for row in case_rows),
|
||
"n_tie_mns": sum(row["n_tie_mns"] for row in case_rows),
|
||
"n_tie_mixed": sum(row["n_tie_mixed"] for row in case_rows),
|
||
}
|
||
)
|
||
fragile = [
|
||
{
|
||
"case": row["case"],
|
||
"objective": row["objective"],
|
||
"real_relative_margin_percent": 100.0
|
||
* row["real_relative_margin_best_vs_second"],
|
||
}
|
||
for row in rows
|
||
if row["fragile_success"]
|
||
]
|
||
material_failures = [
|
||
{
|
||
"case": row["case"],
|
||
"objective": row["objective"],
|
||
"winner_deciding_axis": row["winner_deciding_axis"],
|
||
"winner_pair_m_s_percent": row["winner_pair_m_s_percent"],
|
||
"winner_pair_delta_s_percent": row["winner_pair_delta_s_percent"],
|
||
"regret": row["regret"],
|
||
}
|
||
for row in rows
|
||
if row["failure"]
|
||
]
|
||
boundary_winner_ties = [
|
||
{
|
||
"case": row["case"],
|
||
"objective": row["objective"],
|
||
"real_winner": row["real_winner"],
|
||
"sim_winner": row["sim_winner"],
|
||
"winner_deciding_axis": row["winner_deciding_axis"],
|
||
}
|
||
for row in rows
|
||
if row["winner_pair_relation"] == "sim-tie"
|
||
]
|
||
return {
|
||
"axis_inversions_total": {
|
||
"tp-axis": sum(row["n_inv_tp"] for row in rows),
|
||
"mns-axis": sum(row["n_inv_mns"] for row in rows),
|
||
"mixed": sum(row["n_inv_mixed"] for row in rows),
|
||
},
|
||
"axis_inversions_by_case": by_case,
|
||
"axis_ties_total": {
|
||
"tp-axis": sum(row["n_tie_tp"] for row in rows),
|
||
"mns-axis": sum(row["n_tie_mns"] for row in rows),
|
||
"mixed": sum(row["n_tie_mixed"] for row in rows),
|
||
},
|
||
"strict_winner_reversal_count": sum(
|
||
row["winner_pair_relation"] == "reversal" for row in rows
|
||
),
|
||
"sim_tie_winner_mismatch_count": sum(
|
||
row["winner_pair_relation"] == "sim-tie" for row in rows
|
||
),
|
||
"boundary_winner_ties": boundary_winner_ties,
|
||
"material_failures": material_failures,
|
||
"margin_robust_success_count": sum(
|
||
row["winner_match"] and not row["fragile_success"] for row in rows
|
||
),
|
||
"fragile_success_count": len(fragile),
|
||
"fragile_successes": fragile,
|
||
}
|
||
|
||
|
||
def analyze(input_root: Path) -> dict[str, Any]:
|
||
surfaces = load_surfaces(input_root)
|
||
rows = []
|
||
anomalies = []
|
||
for surface in surfaces:
|
||
row, row_anomalies = analyze_surface(surface)
|
||
rows.append(row)
|
||
anomalies.extend(row_anomalies)
|
||
checks, check_anomalies = cross_checks(rows)
|
||
anomalies.extend(check_anomalies)
|
||
counterexamples = [
|
||
{"case": row["case"], "objective": row["objective"]}
|
||
for row in rows
|
||
if row["h_scale_verdict"] == "COUNTEREXAMPLE"
|
||
]
|
||
data_gaps = [
|
||
{
|
||
"case": row["case"],
|
||
"objective": row["objective"],
|
||
"missing_configs": row["missing_configs"],
|
||
}
|
||
for row in rows
|
||
if row["missing_configs"] or row["h_scale_verdict"] == "N/A"
|
||
]
|
||
directional = directional_summary(rows)
|
||
return {
|
||
"schema": "frontier-split-rootcause-decomposition-v2",
|
||
"definitions": {
|
||
"objective_direction": "all objectives are minimized",
|
||
"config_uniform_scale": "geomean_c(sim_c / real_c) over paired configs",
|
||
"log_ratio_spread_all": "max_c(log(sim_c / real_c)) - min_c(log(sim_c / real_c))",
|
||
"log_ratio_spread_real_top3": "log-ratio spread over the three configs with lowest real objective values",
|
||
"real_relative_margin_best_vs_second": "real_second / real_best - 1",
|
||
"regret": "real_sim_winner / real_best - 1, recomputed from per-config values",
|
||
"failure": "regret > 0.05",
|
||
"h_scale_verdict": "HOLDS iff failure equals (top-3 log-ratio spread > log1p(real relative margin))",
|
||
"signed_pairwise_delta_percent": "100 * (log(sim_c / sim_w) - log(real_c / real_w)); negative favors c relative to real",
|
||
"real_pair_margin_percent": "100 * log(real_c / real_w)",
|
||
"pair_inversion": "real_c > real_w and sim_c < sim_w, equivalently delta_c < -m_c",
|
||
"winner_pair_tie": "sim_s == sim_w with real_s > real_w; a decision-boundary winner-label mismatch, not a strict inversion",
|
||
"winner_deciding_axis": "axis difference between real winner w and sim winner s; none for a winner match",
|
||
"fragile_success": "winner match and real best-vs-second relative margin < 0.01",
|
||
},
|
||
"summary": {
|
||
"row_count": len(rows),
|
||
"case_count": len({row["case"] for row in rows}),
|
||
"failure_count": sum(row["failure"] is True for row in rows),
|
||
"winner_mismatch_count": sum(row["winner_match"] is False for row in rows),
|
||
"h_scale_counterexample_count": len(counterexamples),
|
||
"h_scale_counterexamples": counterexamples,
|
||
"data_gap_count": len(data_gaps),
|
||
},
|
||
"directional_summary": directional,
|
||
"rows": rows,
|
||
"cross_checks": checks,
|
||
"data_gaps": data_gaps,
|
||
"anomalies": anomalies,
|
||
"known_attribution": known_attribution(input_root),
|
||
}
|
||
|
||
|
||
def percent(value: float | None, digits: int = 1) -> str:
|
||
if value is None:
|
||
return "N/A"
|
||
return f"{value * 100.0:.{digits}f}%"
|
||
|
||
|
||
def multiple(value: float | None) -> str:
|
||
if value is None:
|
||
return "N/A"
|
||
return f"{value:.3f}x"
|
||
|
||
|
||
def percent_points(value: float | None, signed: bool = False) -> str:
|
||
if value is None:
|
||
return "N/A"
|
||
sign = "+" if signed else ""
|
||
return f"{value:{sign}.1f}%"
|
||
|
||
|
||
def markdown(result: dict[str, Any]) -> str:
|
||
summary = result["summary"]
|
||
lines = [
|
||
"# Frontier good/bad split decomposition",
|
||
"",
|
||
"All 70 rows are recomputed from frozen per-config JSON values. Lower is better for every objective. "
|
||
"`scale` is the geometric mean of `sim/real`; spreads are shown as `100 ×` log-space width, while `real margin` is the ordinary relative difference `100 × (real_second / real_best - 1)`. "
|
||
"H-SCALE compares the log-space spread with `log1p(real margin)`, not directly with the ordinary relative margin. "
|
||
"The real top-3 neighborhood is selected independently for every row. Directional `m_s` and `delta_s` are already expressed in log-percentage points.",
|
||
"",
|
||
f"- Rows / case surfaces: {summary['row_count']} / {summary['case_count']}.",
|
||
f"- Failures (`regret > 5%`): {summary['failure_count']}.",
|
||
f"- Exact winner mismatches: {summary['winner_mismatch_count']}.",
|
||
f"- H-SCALE counterexamples: {summary['h_scale_counterexample_count']}.",
|
||
"",
|
||
"## Unified decomposition",
|
||
"",
|
||
"| case | objective | sim winner | real winner | regret | scale | spread all | spread real top-3 | real margin (relative) | sim-winner real loss | failure | H-SCALE | deciding axis | m_s | delta_s | n_inv_tp | n_inv_mns | n_inv_mixed | fragile success |",
|
||
"|---|---|---|---|---:|---:|---:|---:|---:|---:|---|---|---|---:|---:|---:|---:|---:|---|",
|
||
]
|
||
for row in result["rows"]:
|
||
lines.append(
|
||
"| {case} | {objective} | {sim} | {real} | {regret} | {scale} | "
|
||
"{spread_all} | {spread_top3} | {margin} | {loss} | {failure} | {verdict} | "
|
||
"{axis} | {pair_margin} | {delta} | {n_tp} | {n_mns} | {n_mixed} | {fragile} |".format(
|
||
case=row["case"],
|
||
objective=row["objective"],
|
||
sim=row["sim_winner"] or "N/A",
|
||
real=row["real_winner"] or "N/A",
|
||
regret=percent(row["regret"]),
|
||
scale=multiple(row["config_uniform_scale"]),
|
||
spread_all=percent(row["log_ratio_spread_all"]),
|
||
spread_top3=percent(row["log_ratio_spread_real_top3"]),
|
||
margin=percent(row["real_relative_margin_best_vs_second"]),
|
||
loss=percent(row["sim_winner_real_relative_loss"]),
|
||
failure=(
|
||
"yes" if row["failure"] is True else "no" if row["failure"] is False else "N/A"
|
||
),
|
||
verdict=row["h_scale_verdict"],
|
||
axis=row["winner_deciding_axis"] or "N/A",
|
||
pair_margin=percent_points(row["winner_pair_m_s_percent"]),
|
||
delta=percent_points(row["winner_pair_delta_s_percent"], signed=True),
|
||
n_tp=row["n_inv_tp"] if row["n_inv_tp"] is not None else "N/A",
|
||
n_mns=row["n_inv_mns"] if row["n_inv_mns"] is not None else "N/A",
|
||
n_mixed=(
|
||
row["n_inv_mixed"] if row["n_inv_mixed"] is not None else "N/A"
|
||
),
|
||
fragile=(
|
||
"yes"
|
||
if row["fragile_success"] is True
|
||
else "no"
|
||
if row["fragile_success"] is False
|
||
else "N/A"
|
||
),
|
||
)
|
||
)
|
||
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"## H-SCALE verdict",
|
||
"",
|
||
]
|
||
)
|
||
if summary["h_scale_counterexample_count"] == 0:
|
||
lines.append(
|
||
"H-SCALE holds on every available row: failure occurs exactly when the real-top-3 differential residual exceeds `log1p` of the real best-vs-second relative margin."
|
||
)
|
||
else:
|
||
failure_counterexamples = sum(
|
||
row["failure"] is True and row["h_scale_verdict"] == "COUNTEREXAMPLE"
|
||
for row in result["rows"]
|
||
)
|
||
nonfailure_counterexamples = summary["h_scale_counterexample_count"] - failure_counterexamples
|
||
lines.extend(
|
||
[
|
||
"H-SCALE is not sufficient as stated; the following rows are counterexamples to the biconditional:",
|
||
"",
|
||
f"All {nonfailure_counterexamples} observed counterexamples are non-failures with spread > `log1p(margin)`; "
|
||
f"there are {failure_counterexamples} failures with spread <= `log1p(margin)`. Thus the threshold is necessary for the observed failures but not sufficient to predict them.",
|
||
"",
|
||
"| case | objective | failure | spread real top-3 | real margin | counterexample type |",
|
||
"|---|---|---|---:|---:|---|",
|
||
]
|
||
)
|
||
for row in result["rows"]:
|
||
if row["h_scale_verdict"] != "COUNTEREXAMPLE":
|
||
continue
|
||
kind = (
|
||
"failure with spread <= log1p(margin)"
|
||
if row["failure"]
|
||
else "non-failure with spread > log1p(margin)"
|
||
)
|
||
lines.append(
|
||
f"| {row['case']} | {row['objective']} | "
|
||
f"{'yes' if row['failure'] else 'no'} | "
|
||
f"{percent(row['log_ratio_spread_real_top3'])} | "
|
||
f"{percent(row['real_relative_margin_best_vs_second'])} | {kind} |"
|
||
)
|
||
|
||
directional = result["directional_summary"]
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"## S0b directional conclusions",
|
||
"",
|
||
"Directional pair statistics remove the S0 range statistic's ambiguity by measuring every config against the real winner. "
|
||
"Strict inversions use the frozen-value test `real_c > real_w and sim_c < sim_w`; simulator ties are not counted as inversions.",
|
||
"",
|
||
"### (a) Which config axis decides each material failure?",
|
||
"",
|
||
"| case | objective | real winner → sim winner | deciding axis | m_s | delta_s | -delta_s | regret |",
|
||
"|---|---|---|---|---:|---:|---:|---:|",
|
||
]
|
||
)
|
||
for row in result["rows"]:
|
||
if not row["failure"]:
|
||
continue
|
||
lines.append(
|
||
f"| {row['case']} | {row['objective']} | {row['real_winner']} → "
|
||
f"{row['sim_winner']} | {row['winner_deciding_axis']} | "
|
||
f"{percent_points(row['winner_pair_m_s_percent'])} | "
|
||
f"{percent_points(row['winner_pair_delta_s_percent'], signed=True)} | "
|
||
f"{percent_points(-row['winner_pair_delta_s_percent'])} | "
|
||
f"{percent(row['regret'])} |"
|
||
)
|
||
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"### (b) Where do TP-axis and MNS-axis inversions and boundary ties concentrate?",
|
||
"",
|
||
"`n_inv_*` counts sum strict `(real winner, c)` pair inversions across all objectives in each case; a row can contain more than one inverted pair. "
|
||
"`n_tie_*` counts cover exact simulator ties only for the simulator-selected winner versus the real winner. Ties remain decision-boundary winner-label mismatches, not strict reversals.",
|
||
"",
|
||
"| case | n_inv_tp | n_inv_mns | n_inv_mixed | n_tie_tp | n_tie_mns | n_tie_mixed |",
|
||
"|---|---:|---:|---:|---:|---:|---:|",
|
||
]
|
||
)
|
||
for item in directional["axis_inversions_by_case"]:
|
||
lines.append(
|
||
f"| {item['case']} | {item['n_inv_tp']} | {item['n_inv_mns']} | "
|
||
f"{item['n_inv_mixed']} | {item['n_tie_tp']} | {item['n_tie_mns']} | "
|
||
f"{item['n_tie_mixed']} |"
|
||
)
|
||
totals = directional["axis_inversions_total"]
|
||
tie_totals = directional["axis_ties_total"]
|
||
lines.append(
|
||
f"| **total** | **{totals['tp-axis']}** | **{totals['mns-axis']}** | "
|
||
f"**{totals['mixed']}** | **{tie_totals['tp-axis']}** | "
|
||
f"**{tie_totals['mns-axis']}** | **{tie_totals['mixed']}** |"
|
||
)
|
||
|
||
def concentrated(axis_key: str) -> str:
|
||
ranked = sorted(
|
||
(
|
||
(item[axis_key], item["case"])
|
||
for item in directional["axis_inversions_by_case"]
|
||
if item[axis_key] > 0
|
||
),
|
||
key=lambda item: (-item[0], item[1]),
|
||
)
|
||
return ", ".join(f"{case} ({count})" for count, case in ranked[:5])
|
||
|
||
by_key = row_index(result["rows"])
|
||
q30_ttft_axes = [
|
||
by_key[("q30 fixed-pd high", objective)]["winner_deciding_axis"]
|
||
for objective in ("ttft_mean_ms", "ttft_p90_ms")
|
||
]
|
||
q235_fixed_pd_failures = [
|
||
row
|
||
for row in result["rows"]
|
||
if row["case"] in ("q235 A0 fixed-pd", "q235 A1 fixed-pd")
|
||
and row["failure"]
|
||
]
|
||
q235_fixed_pd_axes = {
|
||
row["winner_deciding_axis"] for row in q235_fixed_pd_failures
|
||
}
|
||
q235_consistency = (
|
||
"The eight Q235 A0/A1 Fixed-PD failures are all tp-axis. This does not conflict with the A0/MNS64 state-drift diagnosis, but does not constitute additional state-drift evidence for A1/MNS128."
|
||
if len(q235_fixed_pd_failures) == 8 and q235_fixed_pd_axes == {"tp-axis"}
|
||
else "The Q235 A0/A1 Fixed-PD failures do not match the expected eight tp-axis cases; no state-drift compatibility claim is made."
|
||
)
|
||
lines.extend(
|
||
[
|
||
"",
|
||
f"- Among strict inversions, TP-axis cases are most concentrated in: {concentrated('n_inv_tp')}.",
|
||
f"- Among strict inversions, MNS-axis cases are most concentrated in: {concentrated('n_inv_mns')}.",
|
||
f"- {q235_consistency}",
|
||
f"- Q30 admission evidence does not map one-to-one to an MNS-axis deciding pair: high-pressure Fixed-PD TTFT mean is **{q30_ttft_axes[0]}** and p90 is **{q30_ttft_axes[1]}**. "
|
||
"This is an axis/mechanism mismatch, not a direct causal contradiction: changing TP changes offered arrival pressure, while the diagnosed nonlinear event is crossing the MNS admission cap. It rules out the simpler claim that an MNS-threshold mechanism must appear as an mns-axis winner pair.",
|
||
f"- All {directional['sim_tie_winner_mismatch_count']} exact simulator winner ties are mns-axis. Including these boundary errors gives MNS **{totals['mns-axis']} + {tie_totals['mns-axis']} = {totals['mns-axis'] + tie_totals['mns-axis']}**, comparable to TP's **{totals['tp-axis']}** strict inversions.",
|
||
"",
|
||
"The exact ties below show that Frontier is completely insensitive to MNS changes within the corresponding case/objective cell:",
|
||
"",
|
||
"| case | objective | real winner ↔ simulator-selected tied winner (MNS pair) |",
|
||
"|---|---|---|",
|
||
]
|
||
)
|
||
for item in directional["boundary_winner_ties"]:
|
||
lines.append(
|
||
f"| {item['case']} | {item['objective']} | {item['real_winner']} ↔ "
|
||
f"{item['sim_winner']} |"
|
||
)
|
||
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"### (c) How many exact-winner successes are margin-robust?",
|
||
"",
|
||
f"Among exact winner matches, **{directional['margin_robust_success_count']}** are margin-robust (`margin >= 1%`) and "
|
||
f"**{directional['fragile_success_count']}** are fragile (`margin < 1%`). "
|
||
f"Separately, {directional['sim_tie_winner_mismatch_count']} winner-label mismatches come from exact simulator ties and sit on the decision boundary rather than being strict inversions.",
|
||
"",
|
||
"| fragile-success case | objective | real margin |",
|
||
"|---|---|---:|",
|
||
]
|
||
)
|
||
for item in directional["fragile_successes"]:
|
||
lines.append(
|
||
f"| {item['case']} | {item['objective']} | "
|
||
f"{item['real_relative_margin_percent']:.1f}% |"
|
||
)
|
||
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"## Cross-checks",
|
||
"",
|
||
"| check | observed | frozen expectation | result |",
|
||
"|---|---:|---:|---|",
|
||
]
|
||
)
|
||
for check in result["cross_checks"]:
|
||
observed = check["observed_percent"]
|
||
expected = check["expected_percent"]
|
||
if isinstance(observed, list):
|
||
observed_text = ", ".join(f"{value:.1f}%" for value in observed)
|
||
expected_text = ", ".join(f"{value:.1f}%" for value in expected)
|
||
else:
|
||
observed_text = f"{observed:.1f}%"
|
||
expected_text = f"{expected:.1f}%"
|
||
lines.append(
|
||
f"| {check['name']} | {observed_text} | {expected_text} | "
|
||
f"{'PASS' if check['passed'] else 'FAIL'} |"
|
||
)
|
||
|
||
lines.extend(["", "## Data gaps", ""])
|
||
if result["data_gaps"]:
|
||
for gap in result["data_gaps"]:
|
||
missing = ", ".join(gap["missing_configs"]) or "insufficient paired configs"
|
||
lines.append(f"- {gap['case']} / {gap['objective']}: N/A because {missing}.")
|
||
else:
|
||
lines.append("No per-config sim/real pairing gaps were found in the requested surfaces.")
|
||
|
||
lines.extend(["", "## 异常", ""])
|
||
if result["anomalies"]:
|
||
lines.extend(f"- {item}" for item in result["anomalies"])
|
||
else:
|
||
lines.append("- 无。重算 winner/regret 与 frozen JSON 一致,硬性交叉核对均通过。")
|
||
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"## Figure",
|
||
"",
|
||
"`margin-vs-residual.png` uses x = `100 × log1p(real best-vs-second relative margin)` and y = `100 ×` real-top-3 log-ratio spread. "
|
||
"Both axes are logarithmic because the real margins span six orders of magnitude. Green means exact winner match; red means winner mismatch. "
|
||
"The diagonal is the H-SCALE threshold, while failure itself is defined by regret, not by exact winner match.",
|
||
"",
|
||
"`decision-pair-axis.png` is a mechanism census, not a hypothesis test: for a strict winner reversal, `-delta_s > m_s` is algebraically equivalent to the mismatch. "
|
||
"The frozen surfaces also contain deterministic winner-label mismatches caused by exact simulator ties; those points lie on `-delta_s = m_s` and are not counted as strict inversions. "
|
||
"Color identifies the winner-deciding config axis and marker area is proportional to regret.",
|
||
"",
|
||
"## A0 vs A1 对照",
|
||
"",
|
||
"A1 repairs the Q235 Trace-PO p90 ranking error but leaves the four large Fixed-PD TPOT/E2E regrets unchanged.",
|
||
"",
|
||
"| case / objective | A0 regret | A1 regret |",
|
||
"|---|---:|---:|",
|
||
]
|
||
)
|
||
by_key = row_index(result["rows"])
|
||
a0_a1_rows = (
|
||
("trace-po / ttft_p90_ms", "trace-po", "ttft_p90_ms"),
|
||
("fixed-pd / tpot_mean_ms", "fixed-pd", "tpot_mean_ms"),
|
||
("fixed-pd / tpot_p90_ms", "fixed-pd", "tpot_p90_ms"),
|
||
("fixed-pd / e2e_mean_ms", "fixed-pd", "e2e_mean_ms"),
|
||
("fixed-pd / e2e_p90_ms", "fixed-pd", "e2e_p90_ms"),
|
||
)
|
||
for label, case, objective in a0_a1_rows:
|
||
a0 = by_key[(f"q235 A0 {case}", objective)]["regret"]
|
||
a1 = by_key[(f"q235 A1 {case}", objective)]["regret"]
|
||
lines.append(f"| {label} | {percent(a0)} | {percent(a1)} |")
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"Thus Trace-PO TTFT p90 changes **21.2% → 0.3%**, whereas Fixed-PD remains **33.0% / 37.2% / 30.7% / 34.6%** in both A0 and A1.",
|
||
"",
|
||
"## 已知分量归因摘要",
|
||
"",
|
||
"### Q30 high-pressure Fixed-PD TTFT: first-scheduling wait vs execution-side TTFT",
|
||
"",
|
||
"`queue-free TTFT` is the execution-side TTFT after removing first-scheduling wait; in the frozen diagnostic, simulated TTFT is their sum (up to floating-point rounding).",
|
||
"",
|
||
"| config | sim TPOT / real | sim slots / MNS | real slots upper / MNS | first-scheduling wait | queue-free TTFT | sim TTFT | real TTFT | real waiting max |",
|
||
"|---|---:|---:|---:|---:|---:|---:|---:|---:|",
|
||
]
|
||
)
|
||
q30 = result["known_attribution"]["q30_admission"]
|
||
for cell in q30["cells"]:
|
||
lines.append(
|
||
f"| {cell['config']} | {cell['sim_tpot_over_real']:.2f}x | "
|
||
f"{cell['sim_required_slots']:.1f}/{cell['mns']} | "
|
||
f"{cell['real_required_slots_upper_bound']:.1f}/{cell['mns']} | "
|
||
f"{cell['sim_first_scheduling_wait_ms']:.1f} ms | "
|
||
f"{cell['sim_queue_free_ttft_ms']:.1f} ms | "
|
||
f"{cell['sim_ttft_ms']:.1f} ms | {cell['real_ttft_ms']:.1f} ms | "
|
||
f"{cell['real_waiting_max']} |"
|
||
)
|
||
contrasts = q30["contrasts"]
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"For TP4/MNS32 minus TP2/MNS64, the observed simulator contrast is "
|
||
f"**{contrasts['observed_sim_ttft_tp4_minus_tp2_ms']:+.1f} ms**; removing first-scheduling wait changes it to "
|
||
f"**{contrasts['queue_free_sim_ttft_tp4_minus_tp2_ms']:+.1f} ms**, aligned with the real contrast "
|
||
f"**{contrasts['real_ttft_tp4_minus_tp2_ms']:+.1f} ms**. This localizes the Q30 high-pressure TTFT reversal to admission queueing, while the upstream service-time overprediction remains unattributed in the frozen data.",
|
||
"",
|
||
"### Q235 Fixed-PD: own-composition vs exact-state contrast",
|
||
"",
|
||
"Positive TP8−TP4 means TP8 is slower and therefore agrees with the real TP4 winner.",
|
||
"",
|
||
"| config | sim own decode batch mean | real exact decode batch mean | real token-weighted iteration | exact-state matched sim | coverage |",
|
||
"|---|---:|---:|---:|---:|---:|",
|
||
]
|
||
)
|
||
q235 = result["known_attribution"]["q235_state"]
|
||
for config in q235["configs"]:
|
||
lines.append(
|
||
f"| {config['config']} | {config['sim_own_decode_batch_mean']:.3f} | "
|
||
f"{config['real_exact_decode_batch_mean']:.3f} | "
|
||
f"{config['real_token_weighted_iteration_ms']:.4f} ms | "
|
||
f"{config['exact_state_matched_sim_ms']:.4f} ms | "
|
||
f"{config['exact_state_coverage'] * 100.0:.1f}% |"
|
||
)
|
||
lines.extend(
|
||
[
|
||
"",
|
||
"| component | own-composition TP8−TP4 | exact-state TP8−TP4 |",
|
||
"|---|---:|---:|",
|
||
]
|
||
)
|
||
for component in q235["components"]:
|
||
lines.append(
|
||
f"| {component['component']} | "
|
||
f"{component['own_composition_tp8_minus_tp4_ms']:+.4f} ms | "
|
||
f"{component['exact_state_tp8_minus_tp4_ms']:+.4f} ms |"
|
||
)
|
||
decomposition = q235["contrast_decomposition"]
|
||
reference = q235["reference"]
|
||
lines.extend(
|
||
[
|
||
"",
|
||
f"The total contrast flips from **{decomposition['simulator_internal_tp8_minus_tp4_ms']:+.4f} ms** on Frontier's own composition to "
|
||
f"**{decomposition['exact_state_matched_tp8_minus_tp4_ms']:+.4f} ms** at exact real state, a "
|
||
f"**{decomposition['state_composition_shift_ms']:+.4f} ms** state-composition shift. "
|
||
f"The exact real iteration contrast is **{reference['exact_iteration_tp8_minus_tp4_ms']:+.4f} ms** and the frozen real TPOT contrast is "
|
||
f"**{reference['observed_real_tp8_minus_tp4_ms']:+.4f} ms**. This supports closed-loop state/composition drift for Q235 Fixed-PD; it does not by itself attribute every remaining conditional execution residual.",
|
||
"",
|
||
]
|
||
)
|
||
return "\n".join(lines)
|
||
|
||
|
||
def render_plot(rows: list[dict[str, Any]], output: Path) -> None:
|
||
os.environ.setdefault(
|
||
"MPLCONFIGDIR", "/tmp/aituner-frontier-split-rootcause-matplotlib"
|
||
)
|
||
import matplotlib
|
||
|
||
matplotlib.use("Agg")
|
||
import matplotlib.pyplot as plt
|
||
|
||
usable = [
|
||
row
|
||
for row in rows
|
||
if row["real_relative_margin_best_vs_second"] is not None
|
||
and row["log_ratio_spread_real_top3"] is not None
|
||
]
|
||
if not usable:
|
||
raise ValueError("no complete rows available for margin-vs-residual plot")
|
||
|
||
plt.rcParams.update(
|
||
{
|
||
"font.family": "DejaVu Sans",
|
||
"font.size": 8,
|
||
"axes.titlesize": 10,
|
||
"axes.labelsize": 9,
|
||
}
|
||
)
|
||
fig, ax = plt.subplots(figsize=(13.0, 9.0), dpi=180)
|
||
colors = {True: "#2a9d8f", False: "#d1495b"}
|
||
labels = {True: "winner match", False: "winner mismatch"}
|
||
for match in (True, False):
|
||
selected = [row for row in usable if row["winner_match"] is match]
|
||
ax.scatter(
|
||
[
|
||
math.log1p(row["real_relative_margin_best_vs_second"]) * 100.0
|
||
for row in selected
|
||
],
|
||
[row["log_ratio_spread_real_top3"] * 100.0 for row in selected],
|
||
s=28,
|
||
color=colors[match],
|
||
edgecolor="white",
|
||
linewidth=0.4,
|
||
alpha=0.88,
|
||
label=labels[match],
|
||
zorder=3,
|
||
)
|
||
|
||
for index, row in enumerate(usable):
|
||
x = math.log1p(row["real_relative_margin_best_vs_second"]) * 100.0
|
||
y = row["log_ratio_spread_real_top3"] * 100.0
|
||
dx = 3 if index % 2 == 0 else -3
|
||
dy = 3 if (index // 2) % 2 == 0 else -5
|
||
ax.annotate(
|
||
row["case"],
|
||
(x, y),
|
||
xytext=(dx, dy),
|
||
textcoords="offset points",
|
||
ha="left" if dx > 0 else "right",
|
||
va="bottom" if dy > 0 else "top",
|
||
fontsize=4.5,
|
||
color="#303030",
|
||
alpha=0.82,
|
||
clip_on=True,
|
||
zorder=4,
|
||
)
|
||
|
||
max_value = max(
|
||
max(
|
||
math.log1p(row["real_relative_margin_best_vs_second"])
|
||
for row in usable
|
||
),
|
||
max(row["log_ratio_spread_real_top3"] for row in usable),
|
||
) * 100.0
|
||
min_value = min(
|
||
min(
|
||
math.log1p(row["real_relative_margin_best_vs_second"])
|
||
for row in usable
|
||
),
|
||
min(row["log_ratio_spread_real_top3"] for row in usable),
|
||
) * 100.0
|
||
lower = min_value / 5.0
|
||
upper = max_value * 2.0
|
||
ax.plot(
|
||
[lower, upper],
|
||
[lower, upper],
|
||
linestyle="--",
|
||
color="#555555",
|
||
linewidth=1.0,
|
||
label="spread = log1p(margin)",
|
||
zorder=2,
|
||
)
|
||
ax.set_xscale("log")
|
||
ax.set_yscale("log")
|
||
ax.set_xlim(left=lower, right=upper)
|
||
ax.set_ylim(bottom=lower, top=upper)
|
||
ax.set_xlabel("Real best-vs-second log margin: 100 × log1p(relative margin)")
|
||
ax.set_ylabel("Top-3 differential residual: 100 × log-ratio spread")
|
||
ax.set_title("Frontier selection: real margin vs config-differential residual")
|
||
ax.grid(True, color="#d8d8d8", linewidth=0.5, alpha=0.7, zorder=1)
|
||
ax.legend(frameon=False, loc="upper left")
|
||
fig.tight_layout()
|
||
output.parent.mkdir(parents=True, exist_ok=True)
|
||
temporary = output.with_suffix(output.suffix + ".tmp")
|
||
fig.savefig(
|
||
temporary,
|
||
format="png",
|
||
dpi=180,
|
||
metadata={"Software": "analyze_split_decomposition.py"},
|
||
)
|
||
plt.close(fig)
|
||
os.replace(temporary, output)
|
||
|
||
|
||
def render_decision_pair_axis_plot(
|
||
rows: list[dict[str, Any]], output: Path
|
||
) -> None:
|
||
os.environ.setdefault(
|
||
"MPLCONFIGDIR", "/tmp/aituner-frontier-split-rootcause-matplotlib"
|
||
)
|
||
import matplotlib
|
||
|
||
matplotlib.use("Agg")
|
||
import matplotlib.pyplot as plt
|
||
from matplotlib.lines import Line2D
|
||
|
||
usable = [
|
||
row
|
||
for row in rows
|
||
if row["winner_match"] is False
|
||
and row["winner_pair_m_s_percent"] > 0
|
||
and -row["winner_pair_delta_s_percent"] > 0
|
||
]
|
||
if not usable:
|
||
raise ValueError("no winner mismatches available for decision-pair-axis plot")
|
||
|
||
plt.rcParams.update(
|
||
{
|
||
"font.family": "DejaVu Sans",
|
||
"font.size": 8,
|
||
"axes.titlesize": 10,
|
||
"axes.labelsize": 9,
|
||
}
|
||
)
|
||
colors = {
|
||
"tp-axis": "#3572b0",
|
||
"mns-axis": "#e07a1f",
|
||
"mixed": "#7b4ab5",
|
||
}
|
||
fig, ax = plt.subplots(figsize=(11.0, 8.0), dpi=180)
|
||
for axis in INVERSION_AXES:
|
||
selected = [row for row in usable if row["winner_deciding_axis"] == axis]
|
||
ax.scatter(
|
||
[row["winner_pair_m_s_percent"] for row in selected],
|
||
[-row["winner_pair_delta_s_percent"] for row in selected],
|
||
s=[1000.0 * row["regret"] for row in selected],
|
||
color=colors[axis],
|
||
edgecolor="white",
|
||
linewidth=0.6,
|
||
alpha=0.82,
|
||
label=axis,
|
||
zorder=3,
|
||
)
|
||
|
||
values = [row["winner_pair_m_s_percent"] for row in usable] + [
|
||
-row["winner_pair_delta_s_percent"] for row in usable
|
||
]
|
||
lower = min(values) / 1.8
|
||
upper = max(values) * 1.8
|
||
ax.plot(
|
||
[lower, upper],
|
||
[lower, upper],
|
||
linestyle="--",
|
||
color="#555555",
|
||
linewidth=1.0,
|
||
label="reversal boundary",
|
||
zorder=2,
|
||
)
|
||
ax.set_xscale("log")
|
||
ax.set_yscale("log")
|
||
ax.set_xlim(lower, upper)
|
||
ax.set_ylim(lower, upper)
|
||
ax.set_xlabel("Winner-pair real log margin $m_s$ (%)")
|
||
ax.set_ylabel("Simulator preference strength $-\\delta_s$ (%)")
|
||
ax.set_title("Winner-deciding pair census by config axis")
|
||
ax.grid(True, color="#d8d8d8", linewidth=0.5, alpha=0.7, zorder=1)
|
||
axis_handles = [
|
||
Line2D(
|
||
[0],
|
||
[0],
|
||
marker="o",
|
||
linestyle="none",
|
||
markerfacecolor=colors[axis],
|
||
markeredgecolor="white",
|
||
markersize=7,
|
||
label=axis,
|
||
)
|
||
for axis in INVERSION_AXES
|
||
]
|
||
axis_handles.append(
|
||
Line2D(
|
||
[0],
|
||
[0],
|
||
linestyle="--",
|
||
color="#555555",
|
||
linewidth=1.0,
|
||
label="reversal boundary",
|
||
)
|
||
)
|
||
axis_legend = ax.legend(
|
||
handles=axis_handles, frameon=False, loc="upper left", title="deciding axis"
|
||
)
|
||
ax.add_artist(axis_legend)
|
||
size_handles = [
|
||
ax.scatter(
|
||
[],
|
||
[],
|
||
s=10.0 * regret_percent,
|
||
facecolor="none",
|
||
edgecolor="#555555",
|
||
linewidth=0.7,
|
||
label=f"{regret_percent}%",
|
||
)
|
||
for regret_percent in (5, 20, 50)
|
||
]
|
||
ax.legend(
|
||
handles=size_handles,
|
||
frameon=False,
|
||
loc="lower right",
|
||
title="regret (marker area)",
|
||
)
|
||
fig.tight_layout()
|
||
output.parent.mkdir(parents=True, exist_ok=True)
|
||
temporary = output.with_suffix(output.suffix + ".tmp")
|
||
fig.savefig(
|
||
temporary,
|
||
format="png",
|
||
dpi=180,
|
||
metadata={"Software": "analyze_split_decomposition.py"},
|
||
)
|
||
plt.close(fig)
|
||
os.replace(temporary, output)
|
||
|
||
|
||
def parse_args() -> argparse.Namespace:
|
||
parser = argparse.ArgumentParser()
|
||
parser.add_argument("--input-root", type=Path, default=DEFAULT_INPUT_ROOT)
|
||
parser.add_argument("--results-root", type=Path, default=DEFAULT_RESULTS_ROOT)
|
||
return parser.parse_args()
|
||
|
||
|
||
def main() -> None:
|
||
args = parse_args()
|
||
result = analyze(args.input_root.resolve())
|
||
results_root = args.results_root.resolve()
|
||
atomic_write(
|
||
results_root / "decomposition.json",
|
||
json.dumps(result, indent=2, sort_keys=True) + "\n",
|
||
)
|
||
atomic_write(results_root / "decomposition.md", markdown(result))
|
||
render_plot(result["rows"], results_root / "margin-vs-residual.png")
|
||
render_decision_pair_axis_plot(
|
||
result["rows"], results_root / "decision-pair-axis.png"
|
||
)
|
||
print(
|
||
json.dumps(
|
||
{
|
||
"status": "PASS" if not result["anomalies"] else "ANOMALY",
|
||
"rows": result["summary"]["row_count"],
|
||
"h_scale_counterexamples": result["summary"]["h_scale_counterexample_count"],
|
||
"anomalies": len(result["anomalies"]),
|
||
},
|
||
sort_keys=True,
|
||
)
|
||
)
|
||
if result["anomalies"]:
|
||
raise SystemExit("analysis anomalies found; see results/decomposition.md")
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|