Freeze code long context profile v6

This commit is contained in:
2026-07-24 00:52:23 +08:00
parent 89d5ebbbc0
commit b80d3f03de
60 changed files with 5353 additions and 1 deletions

View File

@@ -0,0 +1,8 @@
e89bb293d5d8d6d4ec226de86ce6e4cf46daaf6eefb5b3f2809a7d079b948ecf runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/raw/flashattn-code-longctx-tp2.json
4e6d6002db037bdc71d40f76d6de979ef47960612c15dd985c04048c25bd1d41 runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/aituner.commit
420e4a613b1bcdb44b873eaa911f4a23129713ff9d486fc783b31a009a55cc85 runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/batch-specs.txt
5dd612a7806900e9d12afc27ccef196cc5518aea831aa4ed3b0a53bcfb5746cf runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/max-model-len.txt
484da176e7355ab5532d5129a8e78bea0ae84b3aa0a42c7f0bea9689667762ff runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/pip-freeze.txt
595b25965f76bf7ac25aa3b60126a437a083619303303aaf63346e993862253a runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/source.sha256
4355a46b19d348dc2f57c046f8ef63d4538ebb936000f3c9ee954a27460dd865 runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/vllm-allow-long-max-model-len.txt
11e8f5af440e3db4c01b519a7b0e30bbcecc3a57bfb53dd86d32e601beb95a32 runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/vllm-source.commit

View File

@@ -0,0 +1 @@
e251046c306a8e03214fe1b2f40caabb98f76b7b

View File

@@ -0,0 +1,13 @@
q8ks48k
q8ks64k
q8ks80k
q8ks96k
q8ks112k
q8ks128k
q8ks136k
q512s128k
q2ks66k
q4ks100k
q6ks134k
q1ks8k
q512s4k

View File

@@ -0,0 +1,179 @@
aiohappyeyeballs==2.7.1
aiohttp==3.14.1
aiosignal==1.4.0
annotated-doc==0.0.4
annotated-types==0.7.0
anthropic==0.117.0
anyio==4.14.2
apache-tvm-ffi==0.1.9
astor==0.8.1
attrs==26.1.0
blake3==1.0.9
cachetools==7.1.4
cbor2==6.1.3
certifi==2026.6.17
cffi==2.1.0
charset-normalizer==3.4.9
click==8.4.2
cloudpickle==3.1.2
compressed-tensors==0.15.0.1
cryptography==49.0.0
cuda-bindings==12.9.7
cuda-pathfinder==1.5.6
cuda-python==12.9.7
cuda-tile==1.5.0
cuda-toolkit==12.9.1
depyf==0.20.0
detect-installer==0.1.0
dill==0.4.1
diskcache==5.6.3
distro==1.9.0
dnspython==2.8.0
docstring-parser==0.18.0
einops==0.8.2
email-validator==2.3.0
fastapi==0.139.2
fastapi-cli==0.0.32
fastapi-cloud-cli==0.22.2
fastar==0.11.0
fastsafetensors==0.3.3
filelock==3.31.1
flashinfer-cubin==0.6.8.post1
flashinfer-python==0.6.8.post1
frozenlist==1.8.0
fsspec==2026.6.0
gguf==0.19.0
googleapis-common-protos==1.75.0
grpcio==1.82.1
h11==0.16.0
hf-xet==1.5.2
httpcore==1.0.9
httptools==0.8.0
httpx==0.28.1
httpx-sse==0.4.3
huggingface-hub==1.24.0
idna==3.18
ijson==3.5.1
interegular==0.3.3
jinja2==3.1.6
jiter==0.16.0
jmespath==1.1.0
jsonschema==4.26.0
jsonschema-specifications==2025.9.1
lark==1.2.2
llguidance==1.3.0
llvmlite==0.47.0
lm-format-enforcer==0.11.3
loguru==0.7.3
markdown-it-py==4.2.0
markupsafe==3.0.3
mcp==1.28.1
mdurl==0.1.2
mistral-common==1.11.6
ml-dtypes==0.5.4
model-hosting-container-standards==0.1.16
mpmath==1.3.0
msgspec==0.21.1
multidict==6.7.1
networkx==3.6.1
ninja==1.13.0
numba==0.65.0
numpy==2.3.5
nvidia-cublas-cu12==12.9.1.4
nvidia-cuda-cupti-cu12==12.9.79
nvidia-cuda-nvrtc-cu12==12.9.86
nvidia-cuda-runtime-cu12==12.9.79
nvidia-cudnn-cu12==9.17.1.4
nvidia-cudnn-frontend==1.18.0
nvidia-cufft-cu12==11.4.1.4
nvidia-cufile-cu12==1.14.1.1
nvidia-curand-cu12==10.3.10.19
nvidia-cusolver-cu12==11.7.5.82
nvidia-cusparse-cu12==12.5.10.65
nvidia-cusparselt-cu12==0.7.1
nvidia-cutlass-dsl==4.5.3
nvidia-cutlass-dsl-libs-base==4.5.3
nvidia-ml-py==13.610.43
nvidia-nccl-cu12==2.28.9
nvidia-nvjitlink-cu12==12.9.86
nvidia-nvshmem-cu12==3.4.5
nvidia-nvtx-cu12==12.9.79
openai==2.46.0
openai-harmony==0.0.8
opencv-python-headless==5.0.0.93
opentelemetry-api==1.44.0
opentelemetry-exporter-otlp==1.44.0
opentelemetry-exporter-otlp-proto-common==1.44.0
opentelemetry-exporter-otlp-proto-grpc==1.44.0
opentelemetry-exporter-otlp-proto-http==1.44.0
opentelemetry-proto==1.44.0
opentelemetry-sdk==1.44.0
opentelemetry-semantic-conventions==0.65b0
opentelemetry-semantic-conventions-ai==0.5.1
outlines-core==0.2.14
packaging==26.2
partial-json-parser==0.2.1.1.post7
pillow==12.3.0
prometheus-client==0.25.0
prometheus-fastapi-instrumentator==8.0.2
propcache==0.5.2
protobuf==7.35.1
psutil==7.2.2
py-cpuinfo==9.0.0
pybase64==1.4.3
pycountry==26.2.16
pycparser==3.0
pydantic==2.13.4
pydantic-core==2.46.4
pydantic-extra-types==2.11.1
pydantic-settings==2.14.2
pygments==2.20.0
pyjwt==2.13.0
python-dotenv==1.2.2
python-json-logger==4.1.0
python-multipart==0.0.32
pyyaml==6.0.3
pyzmq==27.1.0
quack-kernels==0.5.0
referencing==0.37.0
regex==2026.7.19
requests==2.34.2
rich==15.0.0
rich-toolkit==0.20.3
rignore==0.8.0
rpds-py==2026.6.3
safetensors==0.8.0
sentencepiece==0.2.2
sentry-sdk==2.66.0
setproctitle==1.3.7
setuptools==80.10.2
shellingham==1.5.4
six==1.17.0
sniffio==1.3.1
sse-starlette==3.4.5
starlette==1.3.1
supervisor==4.3.0
sympy==1.14.0
tabulate==0.10.0
tiktoken==0.13.0
tilelang==0.1.9
tokenizers==0.22.2
torch==2.11.0+cu129
torch-c-dlpack-ext==0.1.5
torchaudio==2.11.0+cu129
torchvision==0.26.0+cu129
tqdm==4.69.0
transformers==5.14.1
triton==3.6.0
typer==0.27.0
typing-extensions==4.16.0
typing-inspection==0.4.2
urllib3==2.7.0
uvicorn==0.51.0
uvloop==0.22.1
vllm @ https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0%2Bcu129-cp38-abi3-manylinux_2_31_x86_64.whl
watchfiles==1.2.0
websockets==16.1.1
xgrammar==0.2.3
yarl==1.24.5
z3-solver==4.15.4.0

View File

@@ -0,0 +1,2 @@
42381f4bd1b67710d4068eb8993240930326a804c6e5167b0b78df3420e47c74 /home/admin/cpfs/wjh/aituner/aituner-code-trace-fbaa909/runs/frontier-code-trace-v0/../frontier-qwen30-vllm020-profile-v1/profile_vllm020_flashattn.py
8a7a72b3adb25db6a61a86f9b04d2d19ebbfe3f2d20480852ef2f45376ac3b85 runs/frontier-code-trace-v0/run_flashattn_code_longctx.sh

View File

@@ -0,0 +1 @@
88d34c6409e9fb3c7b8ca0c04756f061d2099eb1

View File

@@ -0,0 +1,564 @@
{
"environment": {
"attention_backend": "FLASH_ATTN",
"block_size": 16,
"dtype": "bfloat16",
"gpu": "NVIDIA H20",
"model": "/home/admin/cpfs/wjh/models/Qwen/Qwen3-30B-A3B",
"profile_kv_update": true,
"profile_method": "cuda_event",
"torch_cuda": "12.9",
"torch_version": "2.11.0+cu129",
"vllm_source_commit": "88d34c6409e9fb3c7b8ca0c04756f061d2099eb1",
"vllm_version": "0.20.0"
},
"rows": [
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks48k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.047520000487565994,
"mean_ms": 0.04390720054507256,
"median_ms": 0.043088000267744064,
"min_ms": 0.041760001331567764,
"std_ms": 0.0020753894539536763
},
"max_time": 0.0226856632232666,
"mean_time": 0.022550320053100585,
"memory_allocated_mb": 120.0849609375,
"memory_reserved_mb": 134.0,
"min_time": 0.02251683235168457,
"std_time": 5.03848780842368e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 363276.440454495
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks64k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.05392000079154968,
"mean_ms": 0.04477120004594326,
"median_ms": 0.04323200136423111,
"min_ms": 0.04150399938225746,
"std_ms": 0.003602053060896233
},
"max_time": 0.03088857650756836,
"mean_time": 0.030766559982299803,
"memory_allocated_mb": 136.0888671875,
"memory_reserved_mb": 198.0,
"min_time": 0.030721696853637695,
"std_time": 5.739627345668801e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 266263.11179127306
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks80k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.04835199937224388,
"mean_ms": 0.04382400028407574,
"median_ms": 0.0427200011909008,
"min_ms": 0.041600000113248825,
"std_ms": 0.002316465883417635
},
"max_time": 0.03911993789672852,
"mean_time": 0.03901147232055664,
"memory_allocated_mb": 152.0927734375,
"memory_reserved_mb": 278.0,
"min_time": 0.03894742584228516,
"std_time": 5.231496369987919e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 209989.51110295113
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks96k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.04902400076389313,
"mean_ms": 0.04363839998841286,
"median_ms": 0.04273599945008755,
"min_ms": 0.04153599962592125,
"std_ms": 0.002234876899495284
},
"max_time": 0.047580448150634766,
"mean_time": 0.0472187198638916,
"memory_allocated_mb": 168.0966796875,
"memory_reserved_mb": 374.0,
"min_time": 0.04714566421508789,
"std_time": 0.00012505866815108255,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 173490.51443185066
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks112k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.04848000034689903,
"mean_ms": 0.04378239996731281,
"median_ms": 0.042527999728918076,
"min_ms": 0.04163200035691261,
"std_ms": 0.002414242667964844
},
"max_time": 0.05583651351928711,
"mean_time": 0.055455712127685554,
"memory_allocated_mb": 184.1005859375,
"memory_reserved_mb": 486.0,
"min_time": 0.05535052871704101,
"std_time": 0.00014272069779462445,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 147721.48234501254
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.04851200059056282,
"mean_ms": 0.04361280016601086,
"median_ms": 0.042128000408411026,
"min_ms": 0.04179200157523155,
"std_ms": 0.0023557503035821635
},
"max_time": 0.06379235076904297,
"mean_time": 0.06362125053405762,
"memory_allocated_mb": 200.1044921875,
"memory_reserved_mb": 614.0,
"min_time": 0.06355436706542969,
"std_time": 7.081883106781061e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 128762.00846782589
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks136k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.05718399956822395,
"mean_ms": 0.04485439993441105,
"median_ms": 0.042527999728918076,
"min_ms": 0.04156799986958504,
"std_ms": 0.004775190073319638
},
"max_time": 0.06783235168457032,
"mean_time": 0.06771669235229492,
"memory_allocated_mb": 208.1064453125,
"memory_reserved_mb": 750.0,
"min_time": 0.06765017700195312,
"std_time": 5.8650788028093655e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 120974.603387024
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.019807999953627586,
"mean_ms": 0.015823999978601934,
"median_ms": 0.015039999969303608,
"min_ms": 0.014303999952971935,
"std_ms": 0.001913656346272671
},
"max_time": 0.004766176223754883,
"mean_time": 0.0047575551986694335,
"memory_allocated_mb": 132.5458984375,
"memory_reserved_mb": 750.0,
"min_time": 0.004751776218414307,
"std_time": 4.190866412679079e-06,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 107618.29944573072
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q2ks66k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.02687999978661537,
"mean_ms": 0.01938559990376234,
"median_ms": 0.018383999355137348,
"min_ms": 0.018015999346971512,
"std_ms": 0.0025744710494052777
},
"max_time": 0.009582624435424805,
"mean_time": 0.009561939239501951,
"memory_allocated_mb": 84.04248046875,
"memory_reserved_mb": 752.0,
"min_time": 0.009554431915283204,
"std_time": 8.291579854680458e-06,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 214182.49464913702
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q4ks100k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03206399828195572,
"mean_ms": 0.028105599991977214,
"median_ms": 0.02675200067460537,
"min_ms": 0.025728000327944756,
"std_ms": 0.0022390414558466644
},
"max_time": 0.02538035202026367,
"mean_time": 0.0252856897354126,
"memory_allocated_mb": 136.06640625,
"memory_reserved_mb": 752.0,
"min_time": 0.025221471786499024,
"std_time": 4.676405035080222e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 161988.85784252718
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q6ks134k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.04089599847793579,
"mean_ms": 0.03524160049855709,
"median_ms": 0.03411199897527695,
"min_ms": 0.03385600075125694,
"std_ms": 0.0022532652248236193
},
"max_time": 0.048259136199951175,
"mean_time": 0.048143408584594725,
"memory_allocated_mb": 188.09033203125,
"memory_reserved_mb": 752.0,
"min_time": 0.048105857849121095,
"std_time": 5.1879732806325464e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 127618.7162611083
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q1ks8k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.018783999606966972,
"mean_ms": 0.015311999991536141,
"median_ms": 0.014943999703973532,
"min_ms": 0.014271999709308147,
"std_ms": 0.0012243293501672883
},
"max_time": 0.0008861759901046753,
"mean_time": 0.0006353183984756471,
"memory_allocated_mb": 17.0205078125,
"memory_reserved_mb": 752.0,
"min_time": 0.0006019840240478516,
"std_time": 8.386037831062328e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 1611790.2495141604
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s4k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.019840000197291374,
"mean_ms": 0.015875199995934963,
"median_ms": 0.01497599994763732,
"min_ms": 0.014431999996304512,
"std_ms": 0.001896142444948042
},
"max_time": 0.00022230400145053863,
"mean_time": 0.00020609600096940998,
"memory_allocated_mb": 8.515625,
"memory_reserved_mb": 752.0,
"min_time": 0.00019760000705718993,
"std_time": 7.520956759601396e-06,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 2484279.1591865686
}
],
"schema_version": "qwen30_vllm020_flashattn_raw.v1"
}