KVCache simulator for LLM serving cluster routing research

Discrete-event simulator for evaluating KV cache-aware routing policies in prefill-disaggregated LLM serving clusters. Models a two-tier KV cache hierarchy (L0 GPU HBM + L1 CPU DRAM) with RDMA/PCIe link contention, architecture-derived roofline compute (MoE, MLA, DSA), and a cluster-wide meta-store for prefix-aware routing decisions. Includes 11 routing policies (random, round_robin, least_loaded, least_tokens, ttl_aware, precise, min_pd, cache_load, cache_score, estimated_ttft, prefix_affinity), HuggingFace config.json auto-parsing, built-in GPU hardware presets (H100/H800/H20/A100/B200), and ablation tooling for systematic policy comparison across real Alibaba serving traces. Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
2026-04-14 01:16:02 +08:00
commit ec73a95e05
52 changed files with 6005 additions and 0 deletions
--- a/configs/glm5-8xb200-blk512.yaml
+++ b/configs/glm5-8xb200-blk512.yaml
@@ -0,0 +1,68 @@
+# GLM-5 (zai-org/GLM-5) on 8 x B200 SXM (192GB each).
+# Architecture from HuggingFace config.json — all roofline coefficients
+# are derived automatically.
+
+model:
+  name: glm-5
+  # Core architecture (from HF config.json)
+  num_layers: 78
+  hidden_size: 6144
+  num_attention_heads: 64
+  num_kv_heads: 64             # formalism; MLA overrides KV cache sizing
+  head_dim: 64
+  intermediate_size: 12288     # shared expert FFN width
+  dtype_bytes: 2               # BF16
+  block_size_tokens: 512       # matches bailian-traces blksz_512
+
+  # MoE: 256 routed + 1 shared, 8 active per token
+  moe:
+    num_experts: 256
+    num_active_experts: 8
+    num_shared_experts: 1
+    expert_intermediate_size: 2048   # moe_intermediate_size
+
+  # MLA (Multi-head Latent Attention): compressed KV cache
+  mla:
+    kv_lora_rank: 512
+    q_lora_rank: 2048
+    qk_nope_head_dim: 192
+    qk_rope_head_dim: 64
+    v_head_dim: 256
+
+  # DSA (DeepSeek Sparse Attention): sub-quadratic past dense_window
+  attention:
+    type: dsa
+    dense_window: 4096
+    sparse_stride: 8
+    first_dense_layers: 3
+
+hardware:
+  # Aggregate of 8 x B200 in one tensor-parallel group.
+  gpu_flops:        1.80e16    # 8 * 2.25 PFLOPS BF16 dense
+  gpu_mem_bw:       6.40e13    # 8 * 8 TB/s HBM3e
+  # KV budget after FP8 weights + activations. GLM-5 FP8 ~744GB of 1536GB.
+  hbm_bytes:        500.0e9
+  dram_bytes:       1.5e12     # ~1.5 TB usable CPU DRAM / v6d per node
+  pcie_bw:          128.0e9    # PCIe Gen6 x16
+  pcie_latency_us:  4.0
+  rdma_bw:          50.0e9     # ConnectX-7 400 Gbps
+  rdma_latency_us:  6.0
+  max_batch_slots:  256
+  prefill_chunk_tokens: 4096
+
+cluster:
+  num_instances: 64
+  meta_store:
+    ttl_seconds: 300.0
+  router:
+    mode: min_pd
+    precise_probe_latency_us: 50.0
+    precise_probe_topk: 4
+    load_alpha: 1.0
+
+sim:
+  trace_path: bailian-traces/glm_coder_blksz_512_040915-040917.jsonl
+  max_requests: null
+  output_dir: runs/glm5_8xb200_blk512
+  sample_interval_s: 1.0
+  seed: 42