Add linear_ms SLO rule (length-aware TTFT budget)

threshold_ms = intercept_ms + per_token_ms * input_tokens. Lets the TTFT target
scale with prefill work, e.g. "4s + L_in/8k" => intercept_ms=4000, per_token_ms=0.125
(4s base, +1s per 8k input tokens). slo + spec + test.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-06-15 20:35:23 +08:00
parent 77af4ded2a
commit ed2bbe0323
3 changed files with 46 additions and 0 deletions

View File

@@ -29,6 +29,9 @@ def _rule_threshold_ms(rule: ThresholdRule, prompt_tokens: int | None) -> float:
if rule.kind == "fixed_ms":
assert rule.threshold_ms is not None
return rule.threshold_ms
if rule.kind == "linear_ms":
assert rule.intercept_ms is not None and rule.per_token_ms is not None
return float(rule.intercept_ms) + float(rule.per_token_ms) * float(prompt_tokens or 0)
if rule.kind != "step_ms":
raise ValueError(f"Unsupported threshold rule: {rule.kind}")
prompt = float(prompt_tokens or 0)

View File

@@ -504,6 +504,8 @@ class ThresholdRule:
kind: str
threshold_ms: float | None = None
buckets: list[dict[str, float]] = field(default_factory=list)
intercept_ms: float | None = None
per_token_ms: float | None = None
@classmethod
def from_dict(cls, data: Mapping[str, Any], *, context: str) -> "ThresholdRule":
@@ -515,6 +517,18 @@ class ThresholdRule:
data.get("threshold_ms"), context=f"{context}.threshold_ms"
),
)
if kind == "linear_ms":
# threshold = intercept_ms + per_token_ms * input_tokens
# e.g. "4s + L_in/8k" -> intercept_ms=4000, per_token_ms=0.125
intercept_ms = _require_float(
data.get("intercept_ms"), context=f"{context}.intercept_ms"
)
per_token_ms = _require_float(
data.get("per_token_ms"), context=f"{context}.per_token_ms"
)
if intercept_ms < 0 or per_token_ms < 0:
raise SpecError(f"{context}.intercept_ms/per_token_ms must be >= 0.")
return cls(kind=kind, intercept_ms=intercept_ms, per_token_ms=per_token_ms)
if kind == "step_ms":
raw = data.get("buckets")
if not isinstance(raw, list) or not raw: