Freeze code long context profile v6

This commit is contained in:
2026-07-24 00:52:23 +08:00
parent 89d5ebbbc0
commit b80d3f03de
60 changed files with 5353 additions and 1 deletions

View File

@@ -0,0 +1,8 @@
3937c623adcea3d379b82bb7ab63d3b293f917fa59fa6fd6a147581d6a252cb6 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/raw/flashattn-code-longctx-tp1.json
4e6d6002db037bdc71d40f76d6de979ef47960612c15dd985c04048c25bd1d41 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/aituner.commit
420e4a613b1bcdb44b873eaa911f4a23129713ff9d486fc783b31a009a55cc85 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/batch-specs.txt
5dd612a7806900e9d12afc27ccef196cc5518aea831aa4ed3b0a53bcfb5746cf runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/max-model-len.txt
484da176e7355ab5532d5129a8e78bea0ae84b3aa0a42c7f0bea9689667762ff runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/pip-freeze.txt
595b25965f76bf7ac25aa3b60126a437a083619303303aaf63346e993862253a runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/source.sha256
4355a46b19d348dc2f57c046f8ef63d4538ebb936000f3c9ee954a27460dd865 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/vllm-allow-long-max-model-len.txt
11e8f5af440e3db4c01b519a7b0e30bbcecc3a57bfb53dd86d32e601beb95a32 runs/frontier-code-trace-v0/profiles/raw/tp1-r1-v2/provenance/vllm-source.commit

View File

@@ -0,0 +1 @@
e251046c306a8e03214fe1b2f40caabb98f76b7b

View File

@@ -0,0 +1,13 @@
q8ks48k
q8ks64k
q8ks80k
q8ks96k
q8ks112k
q8ks128k
q8ks136k
q512s128k
q2ks66k
q4ks100k
q6ks134k
q1ks8k
q512s4k

View File

@@ -0,0 +1,179 @@
aiohappyeyeballs==2.7.1
aiohttp==3.14.1
aiosignal==1.4.0
annotated-doc==0.0.4
annotated-types==0.7.0
anthropic==0.117.0
anyio==4.14.2
apache-tvm-ffi==0.1.9
astor==0.8.1
attrs==26.1.0
blake3==1.0.9
cachetools==7.1.4
cbor2==6.1.3
certifi==2026.6.17
cffi==2.1.0
charset-normalizer==3.4.9
click==8.4.2
cloudpickle==3.1.2
compressed-tensors==0.15.0.1
cryptography==49.0.0
cuda-bindings==12.9.7
cuda-pathfinder==1.5.6
cuda-python==12.9.7
cuda-tile==1.5.0
cuda-toolkit==12.9.1
depyf==0.20.0
detect-installer==0.1.0
dill==0.4.1
diskcache==5.6.3
distro==1.9.0
dnspython==2.8.0
docstring-parser==0.18.0
einops==0.8.2
email-validator==2.3.0
fastapi==0.139.2
fastapi-cli==0.0.32
fastapi-cloud-cli==0.22.2
fastar==0.11.0
fastsafetensors==0.3.3
filelock==3.31.1
flashinfer-cubin==0.6.8.post1
flashinfer-python==0.6.8.post1
frozenlist==1.8.0
fsspec==2026.6.0
gguf==0.19.0
googleapis-common-protos==1.75.0
grpcio==1.82.1
h11==0.16.0
hf-xet==1.5.2
httpcore==1.0.9
httptools==0.8.0
httpx==0.28.1
httpx-sse==0.4.3
huggingface-hub==1.24.0
idna==3.18
ijson==3.5.1
interegular==0.3.3
jinja2==3.1.6
jiter==0.16.0
jmespath==1.1.0
jsonschema==4.26.0
jsonschema-specifications==2025.9.1
lark==1.2.2
llguidance==1.3.0
llvmlite==0.47.0
lm-format-enforcer==0.11.3
loguru==0.7.3
markdown-it-py==4.2.0
markupsafe==3.0.3
mcp==1.28.1
mdurl==0.1.2
mistral-common==1.11.6
ml-dtypes==0.5.4
model-hosting-container-standards==0.1.16
mpmath==1.3.0
msgspec==0.21.1
multidict==6.7.1
networkx==3.6.1
ninja==1.13.0
numba==0.65.0
numpy==2.3.5
nvidia-cublas-cu12==12.9.1.4
nvidia-cuda-cupti-cu12==12.9.79
nvidia-cuda-nvrtc-cu12==12.9.86
nvidia-cuda-runtime-cu12==12.9.79
nvidia-cudnn-cu12==9.17.1.4
nvidia-cudnn-frontend==1.18.0
nvidia-cufft-cu12==11.4.1.4
nvidia-cufile-cu12==1.14.1.1
nvidia-curand-cu12==10.3.10.19
nvidia-cusolver-cu12==11.7.5.82
nvidia-cusparse-cu12==12.5.10.65
nvidia-cusparselt-cu12==0.7.1
nvidia-cutlass-dsl==4.5.3
nvidia-cutlass-dsl-libs-base==4.5.3
nvidia-ml-py==13.610.43
nvidia-nccl-cu12==2.28.9
nvidia-nvjitlink-cu12==12.9.86
nvidia-nvshmem-cu12==3.4.5
nvidia-nvtx-cu12==12.9.79
openai==2.46.0
openai-harmony==0.0.8
opencv-python-headless==5.0.0.93
opentelemetry-api==1.44.0
opentelemetry-exporter-otlp==1.44.0
opentelemetry-exporter-otlp-proto-common==1.44.0
opentelemetry-exporter-otlp-proto-grpc==1.44.0
opentelemetry-exporter-otlp-proto-http==1.44.0
opentelemetry-proto==1.44.0
opentelemetry-sdk==1.44.0
opentelemetry-semantic-conventions==0.65b0
opentelemetry-semantic-conventions-ai==0.5.1
outlines-core==0.2.14
packaging==26.2
partial-json-parser==0.2.1.1.post7
pillow==12.3.0
prometheus-client==0.25.0
prometheus-fastapi-instrumentator==8.0.2
propcache==0.5.2
protobuf==7.35.1
psutil==7.2.2
py-cpuinfo==9.0.0
pybase64==1.4.3
pycountry==26.2.16
pycparser==3.0
pydantic==2.13.4
pydantic-core==2.46.4
pydantic-extra-types==2.11.1
pydantic-settings==2.14.2
pygments==2.20.0
pyjwt==2.13.0
python-dotenv==1.2.2
python-json-logger==4.1.0
python-multipart==0.0.32
pyyaml==6.0.3
pyzmq==27.1.0
quack-kernels==0.5.0
referencing==0.37.0
regex==2026.7.19
requests==2.34.2
rich==15.0.0
rich-toolkit==0.20.3
rignore==0.8.0
rpds-py==2026.6.3
safetensors==0.8.0
sentencepiece==0.2.2
sentry-sdk==2.66.0
setproctitle==1.3.7
setuptools==80.10.2
shellingham==1.5.4
six==1.17.0
sniffio==1.3.1
sse-starlette==3.4.5
starlette==1.3.1
supervisor==4.3.0
sympy==1.14.0
tabulate==0.10.0
tiktoken==0.13.0
tilelang==0.1.9
tokenizers==0.22.2
torch==2.11.0+cu129
torch-c-dlpack-ext==0.1.5
torchaudio==2.11.0+cu129
torchvision==0.26.0+cu129
tqdm==4.69.0
transformers==5.14.1
triton==3.6.0
typer==0.27.0
typing-extensions==4.16.0
typing-inspection==0.4.2
urllib3==2.7.0
uvicorn==0.51.0
uvloop==0.22.1
vllm @ https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0%2Bcu129-cp38-abi3-manylinux_2_31_x86_64.whl
watchfiles==1.2.0
websockets==16.1.1
xgrammar==0.2.3
yarl==1.24.5
z3-solver==4.15.4.0

View File

@@ -0,0 +1,2 @@
42381f4bd1b67710d4068eb8993240930326a804c6e5167b0b78df3420e47c74 /home/admin/cpfs/wjh/aituner/aituner-code-trace-fbaa909/runs/frontier-code-trace-v0/../frontier-qwen30-vllm020-profile-v1/profile_vllm020_flashattn.py
8a7a72b3adb25db6a61a86f9b04d2d19ebbfe3f2d20480852ef2f45376ac3b85 runs/frontier-code-trace-v0/run_flashattn_code_longctx.sh

View File

@@ -0,0 +1 @@
88d34c6409e9fb3c7b8ca0c04756f061d2099eb1

View File

@@ -0,0 +1,564 @@
{
"environment": {
"attention_backend": "FLASH_ATTN",
"block_size": 16,
"dtype": "bfloat16",
"gpu": "NVIDIA H20",
"model": "/home/admin/cpfs/wjh/models/Qwen/Qwen3-30B-A3B",
"profile_kv_update": true,
"profile_method": "cuda_event",
"torch_cuda": "12.9",
"torch_version": "2.11.0+cu129",
"vllm_source_commit": "88d34c6409e9fb3c7b8ca0c04756f061d2099eb1",
"vllm_version": "0.20.0"
},
"rows": [
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks48k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.0851840004324913,
"mean_ms": 0.07839040011167527,
"median_ms": 0.07703999802470207,
"min_ms": 0.07552000135183334,
"std_ms": 0.0030805904904383677
},
"max_time": 0.043892990112304686,
"mean_time": 0.04368390693664551,
"memory_allocated_mb": 240.0849609375,
"memory_reserved_mb": 246.0,
"min_time": 0.043620254516601564,
"std_time": 8.458163192147439e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 187529.01410308387
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks64k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.0854400023818016,
"mean_ms": 0.07895359992980958,
"median_ms": 0.07791999727487564,
"min_ms": 0.07718399912118912,
"std_ms": 0.0024425064759109735
},
"max_time": 0.059730175018310544,
"mean_time": 0.0595021312713623,
"memory_allocated_mb": 272.0888671875,
"memory_reserved_mb": 374.0,
"min_time": 0.05942179107666016,
"std_time": 0.00010367381308510448,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 137675.74076699864
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks80k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.09055999666452408,
"mean_ms": 0.08208959847688675,
"median_ms": 0.08087999746203423,
"min_ms": 0.07875200361013412,
"std_ms": 0.0033614630055883482
},
"max_time": 0.07555481719970703,
"mean_time": 0.07538758392333984,
"memory_allocated_mb": 304.0927734375,
"memory_reserved_mb": 534.0,
"min_time": 0.0752852783203125,
"std_time": 8.870109119325343e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 108665.10867797918
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks96k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.09888000041246414,
"mean_ms": 0.0819871999323368,
"median_ms": 0.08008000254631042,
"min_ms": 0.07788799703121185,
"std_ms": 0.005835618489111142
},
"max_time": 0.09145164489746094,
"mean_time": 0.09124225158691406,
"memory_allocated_mb": 336.0966796875,
"memory_reserved_mb": 726.0,
"min_time": 0.09113442993164063,
"std_time": 0.0001009018902177725,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 89782.96630697016
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks112k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.08790399879217148,
"mean_ms": 0.07967040091753005,
"median_ms": 0.07808000221848488,
"min_ms": 0.07596799731254578,
"std_ms": 0.0036209252740976605
},
"max_time": 0.10734063720703126,
"mean_time": 0.1069560287475586,
"memory_allocated_mb": 368.1005859375,
"memory_reserved_mb": 950.0,
"min_time": 0.10689055633544922,
"std_time": 0.00013032874360676438,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 76592.22295299545
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.08534400165081024,
"mean_ms": 0.07928640022873878,
"median_ms": 0.07787200063467026,
"min_ms": 0.07648000121116638,
"std_ms": 0.0028305104107121735
},
"max_time": 0.1229840316772461,
"mean_time": 0.12283342437744141,
"memory_allocated_mb": 400.1044921875,
"memory_reserved_mb": 1206.0,
"min_time": 0.12271724700927734,
"std_time": 9.625274868603512e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 66691.94514049937
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks136k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.08774399757385254,
"mean_ms": 0.08081279993057251,
"median_ms": 0.07972799986600876,
"min_ms": 0.07664000242948532,
"std_ms": 0.003487270970977775
},
"max_time": 0.13109359741210938,
"mean_time": 0.13079229431152345,
"memory_allocated_mb": 416.1064453125,
"memory_reserved_mb": 1478.0,
"min_time": 0.1306565399169922,
"std_time": 0.00013138911961272913,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 62633.65929255852
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.040063999593257904,
"mean_ms": 0.019459199998527764,
"median_ms": 0.01601599995046854,
"min_ms": 0.01532800029963255,
"std_ms": 0.007496165774486326
},
"max_time": 0.009649087905883789,
"mean_time": 0.009489968109130859,
"memory_allocated_mb": 265.0458984375,
"memory_reserved_mb": 1478.0,
"min_time": 0.009443936347961425,
"std_time": 5.961646816296781e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 53951.70922728124
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q2ks66k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.040832001715898514,
"mean_ms": 0.02942080032080412,
"median_ms": 0.027888000011444092,
"min_ms": 0.02691200003027916,
"std_ms": 0.004040048279091587
},
"max_time": 0.016906623840332032,
"mean_time": 0.01678766403198242,
"memory_allocated_mb": 168.04248046875,
"memory_reserved_mb": 1478.0,
"min_time": 0.016753759384155274,
"std_time": 4.236470233451937e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 121994.34037387963
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q4ks100k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.04867200180888176,
"mean_ms": 0.04565120078623295,
"median_ms": 0.04468800127506256,
"min_ms": 0.04416000097990036,
"std_ms": 0.0018079971851529544
},
"max_time": 0.05050300979614258,
"mean_time": 0.05035054740905761,
"memory_allocated_mb": 272.06640625,
"memory_reserved_mb": 1478.0,
"min_time": 0.05029715347290039,
"std_time": 7.24480602714254e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 81349.66173700758
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q6ks134k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.06684800237417221,
"mean_ms": 0.06276160031557083,
"median_ms": 0.061824001371860504,
"min_ms": 0.060447998344898224,
"std_ms": 0.0024045636532566625
},
"max_time": 0.09653209686279297,
"mean_time": 0.0961898666381836,
"memory_allocated_mb": 376.09033203125,
"memory_reserved_mb": 1478.0,
"min_time": 0.09607142639160156,
"std_time": 0.0001365794879996899,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 63873.67208970714
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q1ks8k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.0326399989426136,
"mean_ms": 0.021955199912190436,
"median_ms": 0.020560000091791153,
"min_ms": 0.018559999763965607,
"std_ms": 0.00393975597463501
},
"max_time": 0.0011526399850845337,
"mean_time": 0.0011425888061523438,
"memory_allocated_mb": 34.0205078125,
"memory_reserved_mb": 1480.0,
"min_time": 0.0011403199434280396,
"std_time": 3.498447348453724e-06,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 896210.4253833097
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s4k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.02783999964594841,
"mean_ms": 0.017366399988532066,
"median_ms": 0.01611199975013733,
"min_ms": 0.015424000099301338,
"std_ms": 0.003547472261377056
},
"max_time": 0.00034601598978042605,
"mean_time": 0.0003366623997688293,
"memory_allocated_mb": 17.015625,
"memory_reserved_mb": 1480.0,
"min_time": 0.0003308480083942413,
"std_time": 4.813544996936389e-06,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 1520811.3539010207
}
],
"schema_version": "qwen30_vllm020_flashattn_raw.v1"
}

View File

@@ -0,0 +1,8 @@
f113ab5a05167f8ac4c1938c4a04fadc11d5dc2bb8c536a69db23638e6ef73d4 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/raw/flashattn-code-longctx-tp1.json
4e6d6002db037bdc71d40f76d6de979ef47960612c15dd985c04048c25bd1d41 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/aituner.commit
420e4a613b1bcdb44b873eaa911f4a23129713ff9d486fc783b31a009a55cc85 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/batch-specs.txt
5dd612a7806900e9d12afc27ccef196cc5518aea831aa4ed3b0a53bcfb5746cf runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/max-model-len.txt
484da176e7355ab5532d5129a8e78bea0ae84b3aa0a42c7f0bea9689667762ff runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/pip-freeze.txt
595b25965f76bf7ac25aa3b60126a437a083619303303aaf63346e993862253a runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/source.sha256
4355a46b19d348dc2f57c046f8ef63d4538ebb936000f3c9ee954a27460dd865 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/vllm-allow-long-max-model-len.txt
11e8f5af440e3db4c01b519a7b0e30bbcecc3a57bfb53dd86d32e601beb95a32 runs/frontier-code-trace-v0/profiles/raw/tp1-r2-v2/provenance/vllm-source.commit

View File

@@ -0,0 +1 @@
e251046c306a8e03214fe1b2f40caabb98f76b7b

View File

@@ -0,0 +1,13 @@
q8ks48k
q8ks64k
q8ks80k
q8ks96k
q8ks112k
q8ks128k
q8ks136k
q512s128k
q2ks66k
q4ks100k
q6ks134k
q1ks8k
q512s4k

View File

@@ -0,0 +1,179 @@
aiohappyeyeballs==2.7.1
aiohttp==3.14.1
aiosignal==1.4.0
annotated-doc==0.0.4
annotated-types==0.7.0
anthropic==0.117.0
anyio==4.14.2
apache-tvm-ffi==0.1.9
astor==0.8.1
attrs==26.1.0
blake3==1.0.9
cachetools==7.1.4
cbor2==6.1.3
certifi==2026.6.17
cffi==2.1.0
charset-normalizer==3.4.9
click==8.4.2
cloudpickle==3.1.2
compressed-tensors==0.15.0.1
cryptography==49.0.0
cuda-bindings==12.9.7
cuda-pathfinder==1.5.6
cuda-python==12.9.7
cuda-tile==1.5.0
cuda-toolkit==12.9.1
depyf==0.20.0
detect-installer==0.1.0
dill==0.4.1
diskcache==5.6.3
distro==1.9.0
dnspython==2.8.0
docstring-parser==0.18.0
einops==0.8.2
email-validator==2.3.0
fastapi==0.139.2
fastapi-cli==0.0.32
fastapi-cloud-cli==0.22.2
fastar==0.11.0
fastsafetensors==0.3.3
filelock==3.31.1
flashinfer-cubin==0.6.8.post1
flashinfer-python==0.6.8.post1
frozenlist==1.8.0
fsspec==2026.6.0
gguf==0.19.0
googleapis-common-protos==1.75.0
grpcio==1.82.1
h11==0.16.0
hf-xet==1.5.2
httpcore==1.0.9
httptools==0.8.0
httpx==0.28.1
httpx-sse==0.4.3
huggingface-hub==1.24.0
idna==3.18
ijson==3.5.1
interegular==0.3.3
jinja2==3.1.6
jiter==0.16.0
jmespath==1.1.0
jsonschema==4.26.0
jsonschema-specifications==2025.9.1
lark==1.2.2
llguidance==1.3.0
llvmlite==0.47.0
lm-format-enforcer==0.11.3
loguru==0.7.3
markdown-it-py==4.2.0
markupsafe==3.0.3
mcp==1.28.1
mdurl==0.1.2
mistral-common==1.11.6
ml-dtypes==0.5.4
model-hosting-container-standards==0.1.16
mpmath==1.3.0
msgspec==0.21.1
multidict==6.7.1
networkx==3.6.1
ninja==1.13.0
numba==0.65.0
numpy==2.3.5
nvidia-cublas-cu12==12.9.1.4
nvidia-cuda-cupti-cu12==12.9.79
nvidia-cuda-nvrtc-cu12==12.9.86
nvidia-cuda-runtime-cu12==12.9.79
nvidia-cudnn-cu12==9.17.1.4
nvidia-cudnn-frontend==1.18.0
nvidia-cufft-cu12==11.4.1.4
nvidia-cufile-cu12==1.14.1.1
nvidia-curand-cu12==10.3.10.19
nvidia-cusolver-cu12==11.7.5.82
nvidia-cusparse-cu12==12.5.10.65
nvidia-cusparselt-cu12==0.7.1
nvidia-cutlass-dsl==4.5.3
nvidia-cutlass-dsl-libs-base==4.5.3
nvidia-ml-py==13.610.43
nvidia-nccl-cu12==2.28.9
nvidia-nvjitlink-cu12==12.9.86
nvidia-nvshmem-cu12==3.4.5
nvidia-nvtx-cu12==12.9.79
openai==2.46.0
openai-harmony==0.0.8
opencv-python-headless==5.0.0.93
opentelemetry-api==1.44.0
opentelemetry-exporter-otlp==1.44.0
opentelemetry-exporter-otlp-proto-common==1.44.0
opentelemetry-exporter-otlp-proto-grpc==1.44.0
opentelemetry-exporter-otlp-proto-http==1.44.0
opentelemetry-proto==1.44.0
opentelemetry-sdk==1.44.0
opentelemetry-semantic-conventions==0.65b0
opentelemetry-semantic-conventions-ai==0.5.1
outlines-core==0.2.14
packaging==26.2
partial-json-parser==0.2.1.1.post7
pillow==12.3.0
prometheus-client==0.25.0
prometheus-fastapi-instrumentator==8.0.2
propcache==0.5.2
protobuf==7.35.1
psutil==7.2.2
py-cpuinfo==9.0.0
pybase64==1.4.3
pycountry==26.2.16
pycparser==3.0
pydantic==2.13.4
pydantic-core==2.46.4
pydantic-extra-types==2.11.1
pydantic-settings==2.14.2
pygments==2.20.0
pyjwt==2.13.0
python-dotenv==1.2.2
python-json-logger==4.1.0
python-multipart==0.0.32
pyyaml==6.0.3
pyzmq==27.1.0
quack-kernels==0.5.0
referencing==0.37.0
regex==2026.7.19
requests==2.34.2
rich==15.0.0
rich-toolkit==0.20.3
rignore==0.8.0
rpds-py==2026.6.3
safetensors==0.8.0
sentencepiece==0.2.2
sentry-sdk==2.66.0
setproctitle==1.3.7
setuptools==80.10.2
shellingham==1.5.4
six==1.17.0
sniffio==1.3.1
sse-starlette==3.4.5
starlette==1.3.1
supervisor==4.3.0
sympy==1.14.0
tabulate==0.10.0
tiktoken==0.13.0
tilelang==0.1.9
tokenizers==0.22.2
torch==2.11.0+cu129
torch-c-dlpack-ext==0.1.5
torchaudio==2.11.0+cu129
torchvision==0.26.0+cu129
tqdm==4.69.0
transformers==5.14.1
triton==3.6.0
typer==0.27.0
typing-extensions==4.16.0
typing-inspection==0.4.2
urllib3==2.7.0
uvicorn==0.51.0
uvloop==0.22.1
vllm @ https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0%2Bcu129-cp38-abi3-manylinux_2_31_x86_64.whl
watchfiles==1.2.0
websockets==16.1.1
xgrammar==0.2.3
yarl==1.24.5
z3-solver==4.15.4.0

View File

@@ -0,0 +1,2 @@
42381f4bd1b67710d4068eb8993240930326a804c6e5167b0b78df3420e47c74 /home/admin/cpfs/wjh/aituner/aituner-code-trace-fbaa909/runs/frontier-code-trace-v0/../frontier-qwen30-vllm020-profile-v1/profile_vllm020_flashattn.py
8a7a72b3adb25db6a61a86f9b04d2d19ebbfe3f2d20480852ef2f45376ac3b85 runs/frontier-code-trace-v0/run_flashattn_code_longctx.sh

View File

@@ -0,0 +1 @@
88d34c6409e9fb3c7b8ca0c04756f061d2099eb1

View File

@@ -0,0 +1,564 @@
{
"environment": {
"attention_backend": "FLASH_ATTN",
"block_size": 16,
"dtype": "bfloat16",
"gpu": "NVIDIA H20",
"model": "/home/admin/cpfs/wjh/models/Qwen/Qwen3-30B-A3B",
"profile_kv_update": true,
"profile_method": "cuda_event",
"torch_cuda": "12.9",
"torch_version": "2.11.0+cu129",
"vllm_source_commit": "88d34c6409e9fb3c7b8ca0c04756f061d2099eb1",
"vllm_version": "0.20.0"
},
"rows": [
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks48k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.08799999952316284,
"mean_ms": 0.08035840019583702,
"median_ms": 0.08003199845552444,
"min_ms": 0.07583999633789062,
"std_ms": 0.003823532942228132
},
"max_time": 0.04384675216674805,
"mean_time": 0.043662386703491214,
"memory_allocated_mb": 240.0849609375,
"memory_reserved_mb": 246.0,
"min_time": 0.04360464096069336,
"std_time": 7.18445670309503e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 187621.4430427591
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks64k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.08611200004816055,
"mean_ms": 0.07973119989037514,
"median_ms": 0.07807999849319458,
"min_ms": 0.07692799717187881,
"std_ms": 0.0030488875049308785
},
"max_time": 0.05954601669311523,
"mean_time": 0.0594769702911377,
"memory_allocated_mb": 272.0888671875,
"memory_reserved_mb": 374.0,
"min_time": 0.05941872024536133,
"std_time": 4.5624842215119615e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 137733.98274828802
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks80k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.09087999910116196,
"mean_ms": 0.08214399963617325,
"median_ms": 0.08115199953317642,
"min_ms": 0.07846400141716003,
"std_ms": 0.003409163930931844
},
"max_time": 0.07549359893798828,
"mean_time": 0.07529304885864258,
"memory_allocated_mb": 304.0927734375,
"memory_reserved_mb": 534.0,
"min_time": 0.07523104095458985,
"std_time": 8.131664708395529e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 108801.5444211843
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks96k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.08684799820184708,
"mean_ms": 0.08159359991550445,
"median_ms": 0.08057600259780884,
"min_ms": 0.0790719985961914,
"std_ms": 0.0025743110833498623
},
"max_time": 0.09123193359375,
"mean_time": 0.09110229187011719,
"memory_allocated_mb": 336.0966796875,
"memory_reserved_mb": 726.0,
"min_time": 0.09105142211914062,
"std_time": 5.304259171893706e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 89920.89915453696
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks112k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.08508799970149994,
"mean_ms": 0.07919999957084656,
"median_ms": 0.0780319981276989,
"min_ms": 0.0764480009675026,
"std_ms": 0.002612344815356456
},
"max_time": 0.10705843353271484,
"mean_time": 0.10692193984985351,
"memory_allocated_mb": 368.1005859375,
"memory_reserved_mb": 950.0,
"min_time": 0.10688256072998047,
"std_time": 5.1132907623503757e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 76616.64211763947
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.08479999750852585,
"mean_ms": 0.07948800027370453,
"median_ms": 0.07824000343680382,
"min_ms": 0.0764160007238388,
"std_ms": 0.0028062370889137406
},
"max_time": 0.12304061126708984,
"mean_time": 0.12278123931884766,
"memory_allocated_mb": 400.1044921875,
"memory_reserved_mb": 1206.0,
"min_time": 0.122716796875,
"std_time": 9.14188387039285e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 66720.29086403332
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks136k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.08329600095748901,
"mean_ms": 0.07985600009560585,
"median_ms": 0.07945600152015686,
"min_ms": 0.07705599814653397,
"std_ms": 0.0021785520776957702
},
"max_time": 0.13100553894042968,
"mean_time": 0.1307180953979492,
"memory_allocated_mb": 416.1064453125,
"memory_reserved_mb": 1478.0,
"min_time": 0.1306175994873047,
"std_time": 0.000107931256378197,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 62669.21174961153
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.02659199945628643,
"mean_ms": 0.01834559990093112,
"median_ms": 0.01649599988013506,
"min_ms": 0.015135999768972397,
"std_ms": 0.00375367086613964
},
"max_time": 0.009480863571166993,
"mean_time": 0.009440905380249024,
"memory_allocated_mb": 265.0458984375,
"memory_reserved_mb": 1478.0,
"min_time": 0.009432319641113282,
"std_time": 1.4279626944428254e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 54232.08679446535
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q2ks66k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.0331839993596077,
"mean_ms": 0.028758399933576585,
"median_ms": 0.027872000820934772,
"min_ms": 0.026623999699950218,
"std_ms": 0.002088649186504863
},
"max_time": 0.01684105682373047,
"mean_time": 0.016775878524780276,
"memory_allocated_mb": 168.04248046875,
"memory_reserved_mb": 1478.0,
"min_time": 0.016756032943725584,
"std_time": 2.3830672818836446e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 122080.04468885626
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q4ks100k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.052480001002550125,
"mean_ms": 0.045238399878144264,
"median_ms": 0.04420800134539604,
"min_ms": 0.04303999990224838,
"std_ms": 0.0027615853312914604
},
"max_time": 0.050535968780517575,
"mean_time": 0.0503309886932373,
"memory_allocated_mb": 272.06640625,
"memory_reserved_mb": 1478.0,
"min_time": 0.05029235076904297,
"std_time": 6.98791171184278e-05,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 81381.27436686649
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q6ks134k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.0740479975938797,
"mean_ms": 0.06334079951047897,
"median_ms": 0.06135999970138073,
"min_ms": 0.060095999389886856,
"std_ms": 0.004142020833679573
},
"max_time": 0.09652816009521484,
"mean_time": 0.09613609313964844,
"memory_allocated_mb": 376.09033203125,
"memory_reserved_mb": 1478.0,
"min_time": 0.09607071685791016,
"std_time": 0.0001328159156966631,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 63909.39967859056
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q1ks8k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.028192000463604927,
"mean_ms": 0.021113600023090838,
"median_ms": 0.01923200022429228,
"min_ms": 0.018624000251293182,
"std_ms": 0.0034253398003137136
},
"max_time": 0.0011505279541015624,
"mean_time": 0.001141315186023712,
"memory_allocated_mb": 34.0205078125,
"memory_reserved_mb": 1480.0,
"min_time": 0.0011371200084686279,
"std_time": 4.555471006662272e-06,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 897210.527415803
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s4k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 4,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 32,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.021727999672293663,
"mean_ms": 0.017212800029665232,
"median_ms": 0.0161920003592968,
"min_ms": 0.015231999568641186,
"std_ms": 0.002282587014061408
},
"max_time": 0.0003445119857788086,
"mean_time": 0.0003361823946237564,
"memory_allocated_mb": 17.015625,
"memory_reserved_mb": 1480.0,
"min_time": 0.00032927998900413513,
"std_time": 4.115851563011574e-06,
"tensor_parallel_size": 1,
"throughput_tokens_per_sec": 1522982.7860944727
}
],
"schema_version": "qwen30_vllm020_flashattn_raw.v1"
}

View File

@@ -0,0 +1,8 @@
e48aa6e6d039bdb68759701c9976f78ad818d6778b6c40b14823a0a4e852c52d runs/frontier-code-trace-v0/profiles/raw/tp2-r1-v2/raw/flashattn-code-longctx-tp2.json
4e6d6002db037bdc71d40f76d6de979ef47960612c15dd985c04048c25bd1d41 runs/frontier-code-trace-v0/profiles/raw/tp2-r1-v2/provenance/aituner.commit
420e4a613b1bcdb44b873eaa911f4a23129713ff9d486fc783b31a009a55cc85 runs/frontier-code-trace-v0/profiles/raw/tp2-r1-v2/provenance/batch-specs.txt
5dd612a7806900e9d12afc27ccef196cc5518aea831aa4ed3b0a53bcfb5746cf runs/frontier-code-trace-v0/profiles/raw/tp2-r1-v2/provenance/max-model-len.txt
484da176e7355ab5532d5129a8e78bea0ae84b3aa0a42c7f0bea9689667762ff runs/frontier-code-trace-v0/profiles/raw/tp2-r1-v2/provenance/pip-freeze.txt
595b25965f76bf7ac25aa3b60126a437a083619303303aaf63346e993862253a runs/frontier-code-trace-v0/profiles/raw/tp2-r1-v2/provenance/source.sha256
4355a46b19d348dc2f57c046f8ef63d4538ebb936000f3c9ee954a27460dd865 runs/frontier-code-trace-v0/profiles/raw/tp2-r1-v2/provenance/vllm-allow-long-max-model-len.txt
11e8f5af440e3db4c01b519a7b0e30bbcecc3a57bfb53dd86d32e601beb95a32 runs/frontier-code-trace-v0/profiles/raw/tp2-r1-v2/provenance/vllm-source.commit

View File

@@ -0,0 +1 @@
e251046c306a8e03214fe1b2f40caabb98f76b7b

View File

@@ -0,0 +1,13 @@
q8ks48k
q8ks64k
q8ks80k
q8ks96k
q8ks112k
q8ks128k
q8ks136k
q512s128k
q2ks66k
q4ks100k
q6ks134k
q1ks8k
q512s4k

View File

@@ -0,0 +1,179 @@
aiohappyeyeballs==2.7.1
aiohttp==3.14.1
aiosignal==1.4.0
annotated-doc==0.0.4
annotated-types==0.7.0
anthropic==0.117.0
anyio==4.14.2
apache-tvm-ffi==0.1.9
astor==0.8.1
attrs==26.1.0
blake3==1.0.9
cachetools==7.1.4
cbor2==6.1.3
certifi==2026.6.17
cffi==2.1.0
charset-normalizer==3.4.9
click==8.4.2
cloudpickle==3.1.2
compressed-tensors==0.15.0.1
cryptography==49.0.0
cuda-bindings==12.9.7
cuda-pathfinder==1.5.6
cuda-python==12.9.7
cuda-tile==1.5.0
cuda-toolkit==12.9.1
depyf==0.20.0
detect-installer==0.1.0
dill==0.4.1
diskcache==5.6.3
distro==1.9.0
dnspython==2.8.0
docstring-parser==0.18.0
einops==0.8.2
email-validator==2.3.0
fastapi==0.139.2
fastapi-cli==0.0.32
fastapi-cloud-cli==0.22.2
fastar==0.11.0
fastsafetensors==0.3.3
filelock==3.31.1
flashinfer-cubin==0.6.8.post1
flashinfer-python==0.6.8.post1
frozenlist==1.8.0
fsspec==2026.6.0
gguf==0.19.0
googleapis-common-protos==1.75.0
grpcio==1.82.1
h11==0.16.0
hf-xet==1.5.2
httpcore==1.0.9
httptools==0.8.0
httpx==0.28.1
httpx-sse==0.4.3
huggingface-hub==1.24.0
idna==3.18
ijson==3.5.1
interegular==0.3.3
jinja2==3.1.6
jiter==0.16.0
jmespath==1.1.0
jsonschema==4.26.0
jsonschema-specifications==2025.9.1
lark==1.2.2
llguidance==1.3.0
llvmlite==0.47.0
lm-format-enforcer==0.11.3
loguru==0.7.3
markdown-it-py==4.2.0
markupsafe==3.0.3
mcp==1.28.1
mdurl==0.1.2
mistral-common==1.11.6
ml-dtypes==0.5.4
model-hosting-container-standards==0.1.16
mpmath==1.3.0
msgspec==0.21.1
multidict==6.7.1
networkx==3.6.1
ninja==1.13.0
numba==0.65.0
numpy==2.3.5
nvidia-cublas-cu12==12.9.1.4
nvidia-cuda-cupti-cu12==12.9.79
nvidia-cuda-nvrtc-cu12==12.9.86
nvidia-cuda-runtime-cu12==12.9.79
nvidia-cudnn-cu12==9.17.1.4
nvidia-cudnn-frontend==1.18.0
nvidia-cufft-cu12==11.4.1.4
nvidia-cufile-cu12==1.14.1.1
nvidia-curand-cu12==10.3.10.19
nvidia-cusolver-cu12==11.7.5.82
nvidia-cusparse-cu12==12.5.10.65
nvidia-cusparselt-cu12==0.7.1
nvidia-cutlass-dsl==4.5.3
nvidia-cutlass-dsl-libs-base==4.5.3
nvidia-ml-py==13.610.43
nvidia-nccl-cu12==2.28.9
nvidia-nvjitlink-cu12==12.9.86
nvidia-nvshmem-cu12==3.4.5
nvidia-nvtx-cu12==12.9.79
openai==2.46.0
openai-harmony==0.0.8
opencv-python-headless==5.0.0.93
opentelemetry-api==1.44.0
opentelemetry-exporter-otlp==1.44.0
opentelemetry-exporter-otlp-proto-common==1.44.0
opentelemetry-exporter-otlp-proto-grpc==1.44.0
opentelemetry-exporter-otlp-proto-http==1.44.0
opentelemetry-proto==1.44.0
opentelemetry-sdk==1.44.0
opentelemetry-semantic-conventions==0.65b0
opentelemetry-semantic-conventions-ai==0.5.1
outlines-core==0.2.14
packaging==26.2
partial-json-parser==0.2.1.1.post7
pillow==12.3.0
prometheus-client==0.25.0
prometheus-fastapi-instrumentator==8.0.2
propcache==0.5.2
protobuf==7.35.1
psutil==7.2.2
py-cpuinfo==9.0.0
pybase64==1.4.3
pycountry==26.2.16
pycparser==3.0
pydantic==2.13.4
pydantic-core==2.46.4
pydantic-extra-types==2.11.1
pydantic-settings==2.14.2
pygments==2.20.0
pyjwt==2.13.0
python-dotenv==1.2.2
python-json-logger==4.1.0
python-multipart==0.0.32
pyyaml==6.0.3
pyzmq==27.1.0
quack-kernels==0.5.0
referencing==0.37.0
regex==2026.7.19
requests==2.34.2
rich==15.0.0
rich-toolkit==0.20.3
rignore==0.8.0
rpds-py==2026.6.3
safetensors==0.8.0
sentencepiece==0.2.2
sentry-sdk==2.66.0
setproctitle==1.3.7
setuptools==80.10.2
shellingham==1.5.4
six==1.17.0
sniffio==1.3.1
sse-starlette==3.4.5
starlette==1.3.1
supervisor==4.3.0
sympy==1.14.0
tabulate==0.10.0
tiktoken==0.13.0
tilelang==0.1.9
tokenizers==0.22.2
torch==2.11.0+cu129
torch-c-dlpack-ext==0.1.5
torchaudio==2.11.0+cu129
torchvision==0.26.0+cu129
tqdm==4.69.0
transformers==5.14.1
triton==3.6.0
typer==0.27.0
typing-extensions==4.16.0
typing-inspection==0.4.2
urllib3==2.7.0
uvicorn==0.51.0
uvloop==0.22.1
vllm @ https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0%2Bcu129-cp38-abi3-manylinux_2_31_x86_64.whl
watchfiles==1.2.0
websockets==16.1.1
xgrammar==0.2.3
yarl==1.24.5
z3-solver==4.15.4.0

View File

@@ -0,0 +1,2 @@
42381f4bd1b67710d4068eb8993240930326a804c6e5167b0b78df3420e47c74 /home/admin/cpfs/wjh/aituner/aituner-code-trace-fbaa909/runs/frontier-code-trace-v0/../frontier-qwen30-vllm020-profile-v1/profile_vllm020_flashattn.py
8a7a72b3adb25db6a61a86f9b04d2d19ebbfe3f2d20480852ef2f45376ac3b85 runs/frontier-code-trace-v0/run_flashattn_code_longctx.sh

View File

@@ -0,0 +1 @@
88d34c6409e9fb3c7b8ca0c04756f061d2099eb1

View File

@@ -0,0 +1,564 @@
{
"environment": {
"attention_backend": "FLASH_ATTN",
"block_size": 16,
"dtype": "bfloat16",
"gpu": "NVIDIA H20",
"model": "/home/admin/cpfs/wjh/models/Qwen/Qwen3-30B-A3B",
"profile_kv_update": true,
"profile_method": "cuda_event",
"torch_cuda": "12.9",
"torch_version": "2.11.0+cu129",
"vllm_source_commit": "88d34c6409e9fb3c7b8ca0c04756f061d2099eb1",
"vllm_version": "0.20.0"
},
"rows": [
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks48k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.05100800096988678,
"mean_ms": 0.04406719990074635,
"median_ms": 0.042847998440265656,
"min_ms": 0.04217600077390671,
"std_ms": 0.002656672983576728
},
"max_time": 0.022986656188964845,
"mean_time": 0.02263068161010742,
"memory_allocated_mb": 120.0849609375,
"memory_reserved_mb": 134.0,
"min_time": 0.022516607284545898,
"std_time": 0.00013840494695459716,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 361986.4457083454
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks64k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.047200001776218414,
"mean_ms": 0.043644800409674646,
"median_ms": 0.04262400045990944,
"min_ms": 0.04182400181889534,
"std_ms": 0.001927741871473188
},
"max_time": 0.030787456512451173,
"mean_time": 0.03074285774230957,
"memory_allocated_mb": 136.0888671875,
"memory_reserved_mb": 198.0,
"min_time": 0.030725727081298827,
"std_time": 1.9110324860515016e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 266468.39629114367
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks80k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.05049600079655647,
"mean_ms": 0.043852799385786054,
"median_ms": 0.04289599880576134,
"min_ms": 0.04185599833726883,
"std_ms": 0.0024573088797598033
},
"max_time": 0.03898448181152344,
"mean_time": 0.038945382690429686,
"memory_allocated_mb": 152.0927734375,
"memory_reserved_mb": 278.0,
"min_time": 0.038930206298828124,
"std_time": 1.7734743323340796e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 210345.85961362437
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks96k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.046911999583244324,
"mean_ms": 0.04412479996681214,
"median_ms": 0.04327999986708164,
"min_ms": 0.04227200150489807,
"std_ms": 0.001811858548765738
},
"max_time": 0.04724588775634766,
"mean_time": 0.04714819869995117,
"memory_allocated_mb": 168.0966796875,
"memory_reserved_mb": 374.0,
"min_time": 0.047131519317626956,
"std_time": 3.329779976561307e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 173750.01009335453
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks112k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.06195199862122536,
"mean_ms": 0.04529919996857643,
"median_ms": 0.043136000633239746,
"min_ms": 0.04182400181889534,
"std_ms": 0.005751677082163394
},
"max_time": 0.05564672088623047,
"mean_time": 0.05542207336425782,
"memory_allocated_mb": 184.1005859375,
"memory_reserved_mb": 486.0,
"min_time": 0.05533849716186524,
"std_time": 0.00011010194068839293,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 147811.14279429129
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.054687999188899994,
"mean_ms": 0.04418559968471527,
"median_ms": 0.042767999693751335,
"min_ms": 0.04182400181889534,
"std_ms": 0.0036820249064621804
},
"max_time": 0.06380697631835938,
"mean_time": 0.06362084197998047,
"memory_allocated_mb": 200.1044921875,
"memory_reserved_mb": 614.0,
"min_time": 0.06354908752441406,
"std_time": 8.872184973119366e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 128762.83533905087
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks136k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.06566400080919266,
"mean_ms": 0.05008000023663044,
"median_ms": 0.04843199998140335,
"min_ms": 0.0461760014295578,
"std_ms": 0.0055350567765421
},
"max_time": 0.06791651153564453,
"mean_time": 0.06770619888305665,
"memory_allocated_mb": 208.1064453125,
"memory_reserved_mb": 750.0,
"min_time": 0.06764147186279297,
"std_time": 9.93520482760623e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 120993.35267881997
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03859199956059456,
"mean_ms": 0.02689919974654913,
"median_ms": 0.0244159996509552,
"min_ms": 0.023711999878287315,
"std_ms": 0.004626699769228834
},
"max_time": 0.004829855918884277,
"mean_time": 0.004790303993225097,
"memory_allocated_mb": 132.5458984375,
"memory_reserved_mb": 750.0,
"min_time": 0.004775519847869873,
"std_time": 1.597276061319242e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 106882.56960813323
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q2ks66k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.0453759990632534,
"mean_ms": 0.027676799893379213,
"median_ms": 0.025200000032782555,
"min_ms": 0.024224000051617622,
"std_ms": 0.006154880946840619
},
"max_time": 0.00963871955871582,
"mean_time": 0.00959802885055542,
"memory_allocated_mb": 84.04248046875,
"memory_reserved_mb": 752.0,
"min_time": 0.009579392433166503,
"std_time": 1.6811667352984867e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 213377.14565022237
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q4ks100k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03593600168824196,
"mean_ms": 0.02780479993671179,
"median_ms": 0.027040000073611736,
"min_ms": 0.02630399912595749,
"std_ms": 0.0027569987978884785
},
"max_time": 0.025383039474487303,
"mean_time": 0.025250255966186526,
"memory_allocated_mb": 136.06640625,
"memory_reserved_mb": 752.0,
"min_time": 0.025216863632202147,
"std_time": 4.761963064809279e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 162216.17735222535
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q6ks134k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.041728001087903976,
"mean_ms": 0.03665280006825924,
"median_ms": 0.0352960005402565,
"min_ms": 0.034015998244285583,
"std_ms": 0.002710617869728372
},
"max_time": 0.048498847961425784,
"mean_time": 0.048172192001342776,
"memory_allocated_mb": 188.09033203125,
"memory_reserved_mb": 752.0,
"min_time": 0.04810128021240234,
"std_time": 0.00011195495368044632,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 127542.46266868527
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q1ks8k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.02191999927163124,
"mean_ms": 0.016563199926167727,
"median_ms": 0.015584000386297703,
"min_ms": 0.014879999682307243,
"std_ms": 0.002306362959594964
},
"max_time": 0.0006153280138969421,
"mean_time": 0.0006064576089382171,
"memory_allocated_mb": 17.0205078125,
"memory_reserved_mb": 752.0,
"min_time": 0.0006022400259971619,
"std_time": 4.680714919723724e-06,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 1688493.9440248988
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s4k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03843199834227562,
"mean_ms": 0.019180799927562477,
"median_ms": 0.01601599995046854,
"min_ms": 0.014911999925971031,
"std_ms": 0.006856655009795906
},
"max_time": 0.00022172799706459046,
"mean_time": 0.00020788159817457198,
"memory_allocated_mb": 8.515625,
"memory_reserved_mb": 752.0,
"min_time": 0.00019884799420833587,
"std_time": 7.012280869814824e-06,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 2462940.464648726
}
],
"schema_version": "qwen30_vllm020_flashattn_raw.v1"
}

View File

@@ -0,0 +1,8 @@
e89bb293d5d8d6d4ec226de86ce6e4cf46daaf6eefb5b3f2809a7d079b948ecf runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/raw/flashattn-code-longctx-tp2.json
4e6d6002db037bdc71d40f76d6de979ef47960612c15dd985c04048c25bd1d41 runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/aituner.commit
420e4a613b1bcdb44b873eaa911f4a23129713ff9d486fc783b31a009a55cc85 runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/batch-specs.txt
5dd612a7806900e9d12afc27ccef196cc5518aea831aa4ed3b0a53bcfb5746cf runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/max-model-len.txt
484da176e7355ab5532d5129a8e78bea0ae84b3aa0a42c7f0bea9689667762ff runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/pip-freeze.txt
595b25965f76bf7ac25aa3b60126a437a083619303303aaf63346e993862253a runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/source.sha256
4355a46b19d348dc2f57c046f8ef63d4538ebb936000f3c9ee954a27460dd865 runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/vllm-allow-long-max-model-len.txt
11e8f5af440e3db4c01b519a7b0e30bbcecc3a57bfb53dd86d32e601beb95a32 runs/frontier-code-trace-v0/profiles/raw/tp2-r2-v2/provenance/vllm-source.commit

View File

@@ -0,0 +1 @@
e251046c306a8e03214fe1b2f40caabb98f76b7b

View File

@@ -0,0 +1,13 @@
q8ks48k
q8ks64k
q8ks80k
q8ks96k
q8ks112k
q8ks128k
q8ks136k
q512s128k
q2ks66k
q4ks100k
q6ks134k
q1ks8k
q512s4k

View File

@@ -0,0 +1,179 @@
aiohappyeyeballs==2.7.1
aiohttp==3.14.1
aiosignal==1.4.0
annotated-doc==0.0.4
annotated-types==0.7.0
anthropic==0.117.0
anyio==4.14.2
apache-tvm-ffi==0.1.9
astor==0.8.1
attrs==26.1.0
blake3==1.0.9
cachetools==7.1.4
cbor2==6.1.3
certifi==2026.6.17
cffi==2.1.0
charset-normalizer==3.4.9
click==8.4.2
cloudpickle==3.1.2
compressed-tensors==0.15.0.1
cryptography==49.0.0
cuda-bindings==12.9.7
cuda-pathfinder==1.5.6
cuda-python==12.9.7
cuda-tile==1.5.0
cuda-toolkit==12.9.1
depyf==0.20.0
detect-installer==0.1.0
dill==0.4.1
diskcache==5.6.3
distro==1.9.0
dnspython==2.8.0
docstring-parser==0.18.0
einops==0.8.2
email-validator==2.3.0
fastapi==0.139.2
fastapi-cli==0.0.32
fastapi-cloud-cli==0.22.2
fastar==0.11.0
fastsafetensors==0.3.3
filelock==3.31.1
flashinfer-cubin==0.6.8.post1
flashinfer-python==0.6.8.post1
frozenlist==1.8.0
fsspec==2026.6.0
gguf==0.19.0
googleapis-common-protos==1.75.0
grpcio==1.82.1
h11==0.16.0
hf-xet==1.5.2
httpcore==1.0.9
httptools==0.8.0
httpx==0.28.1
httpx-sse==0.4.3
huggingface-hub==1.24.0
idna==3.18
ijson==3.5.1
interegular==0.3.3
jinja2==3.1.6
jiter==0.16.0
jmespath==1.1.0
jsonschema==4.26.0
jsonschema-specifications==2025.9.1
lark==1.2.2
llguidance==1.3.0
llvmlite==0.47.0
lm-format-enforcer==0.11.3
loguru==0.7.3
markdown-it-py==4.2.0
markupsafe==3.0.3
mcp==1.28.1
mdurl==0.1.2
mistral-common==1.11.6
ml-dtypes==0.5.4
model-hosting-container-standards==0.1.16
mpmath==1.3.0
msgspec==0.21.1
multidict==6.7.1
networkx==3.6.1
ninja==1.13.0
numba==0.65.0
numpy==2.3.5
nvidia-cublas-cu12==12.9.1.4
nvidia-cuda-cupti-cu12==12.9.79
nvidia-cuda-nvrtc-cu12==12.9.86
nvidia-cuda-runtime-cu12==12.9.79
nvidia-cudnn-cu12==9.17.1.4
nvidia-cudnn-frontend==1.18.0
nvidia-cufft-cu12==11.4.1.4
nvidia-cufile-cu12==1.14.1.1
nvidia-curand-cu12==10.3.10.19
nvidia-cusolver-cu12==11.7.5.82
nvidia-cusparse-cu12==12.5.10.65
nvidia-cusparselt-cu12==0.7.1
nvidia-cutlass-dsl==4.5.3
nvidia-cutlass-dsl-libs-base==4.5.3
nvidia-ml-py==13.610.43
nvidia-nccl-cu12==2.28.9
nvidia-nvjitlink-cu12==12.9.86
nvidia-nvshmem-cu12==3.4.5
nvidia-nvtx-cu12==12.9.79
openai==2.46.0
openai-harmony==0.0.8
opencv-python-headless==5.0.0.93
opentelemetry-api==1.44.0
opentelemetry-exporter-otlp==1.44.0
opentelemetry-exporter-otlp-proto-common==1.44.0
opentelemetry-exporter-otlp-proto-grpc==1.44.0
opentelemetry-exporter-otlp-proto-http==1.44.0
opentelemetry-proto==1.44.0
opentelemetry-sdk==1.44.0
opentelemetry-semantic-conventions==0.65b0
opentelemetry-semantic-conventions-ai==0.5.1
outlines-core==0.2.14
packaging==26.2
partial-json-parser==0.2.1.1.post7
pillow==12.3.0
prometheus-client==0.25.0
prometheus-fastapi-instrumentator==8.0.2
propcache==0.5.2
protobuf==7.35.1
psutil==7.2.2
py-cpuinfo==9.0.0
pybase64==1.4.3
pycountry==26.2.16
pycparser==3.0
pydantic==2.13.4
pydantic-core==2.46.4
pydantic-extra-types==2.11.1
pydantic-settings==2.14.2
pygments==2.20.0
pyjwt==2.13.0
python-dotenv==1.2.2
python-json-logger==4.1.0
python-multipart==0.0.32
pyyaml==6.0.3
pyzmq==27.1.0
quack-kernels==0.5.0
referencing==0.37.0
regex==2026.7.19
requests==2.34.2
rich==15.0.0
rich-toolkit==0.20.3
rignore==0.8.0
rpds-py==2026.6.3
safetensors==0.8.0
sentencepiece==0.2.2
sentry-sdk==2.66.0
setproctitle==1.3.7
setuptools==80.10.2
shellingham==1.5.4
six==1.17.0
sniffio==1.3.1
sse-starlette==3.4.5
starlette==1.3.1
supervisor==4.3.0
sympy==1.14.0
tabulate==0.10.0
tiktoken==0.13.0
tilelang==0.1.9
tokenizers==0.22.2
torch==2.11.0+cu129
torch-c-dlpack-ext==0.1.5
torchaudio==2.11.0+cu129
torchvision==0.26.0+cu129
tqdm==4.69.0
transformers==5.14.1
triton==3.6.0
typer==0.27.0
typing-extensions==4.16.0
typing-inspection==0.4.2
urllib3==2.7.0
uvicorn==0.51.0
uvloop==0.22.1
vllm @ https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0%2Bcu129-cp38-abi3-manylinux_2_31_x86_64.whl
watchfiles==1.2.0
websockets==16.1.1
xgrammar==0.2.3
yarl==1.24.5
z3-solver==4.15.4.0

View File

@@ -0,0 +1,2 @@
42381f4bd1b67710d4068eb8993240930326a804c6e5167b0b78df3420e47c74 /home/admin/cpfs/wjh/aituner/aituner-code-trace-fbaa909/runs/frontier-code-trace-v0/../frontier-qwen30-vllm020-profile-v1/profile_vllm020_flashattn.py
8a7a72b3adb25db6a61a86f9b04d2d19ebbfe3f2d20480852ef2f45376ac3b85 runs/frontier-code-trace-v0/run_flashattn_code_longctx.sh

View File

@@ -0,0 +1 @@
88d34c6409e9fb3c7b8ca0c04756f061d2099eb1

View File

@@ -0,0 +1,564 @@
{
"environment": {
"attention_backend": "FLASH_ATTN",
"block_size": 16,
"dtype": "bfloat16",
"gpu": "NVIDIA H20",
"model": "/home/admin/cpfs/wjh/models/Qwen/Qwen3-30B-A3B",
"profile_kv_update": true,
"profile_method": "cuda_event",
"torch_cuda": "12.9",
"torch_version": "2.11.0+cu129",
"vllm_source_commit": "88d34c6409e9fb3c7b8ca0c04756f061d2099eb1",
"vllm_version": "0.20.0"
},
"rows": [
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks48k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.047520000487565994,
"mean_ms": 0.04390720054507256,
"median_ms": 0.043088000267744064,
"min_ms": 0.041760001331567764,
"std_ms": 0.0020753894539536763
},
"max_time": 0.0226856632232666,
"mean_time": 0.022550320053100585,
"memory_allocated_mb": 120.0849609375,
"memory_reserved_mb": 134.0,
"min_time": 0.02251683235168457,
"std_time": 5.03848780842368e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 363276.440454495
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks64k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.05392000079154968,
"mean_ms": 0.04477120004594326,
"median_ms": 0.04323200136423111,
"min_ms": 0.04150399938225746,
"std_ms": 0.003602053060896233
},
"max_time": 0.03088857650756836,
"mean_time": 0.030766559982299803,
"memory_allocated_mb": 136.0888671875,
"memory_reserved_mb": 198.0,
"min_time": 0.030721696853637695,
"std_time": 5.739627345668801e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 266263.11179127306
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks80k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.04835199937224388,
"mean_ms": 0.04382400028407574,
"median_ms": 0.0427200011909008,
"min_ms": 0.041600000113248825,
"std_ms": 0.002316465883417635
},
"max_time": 0.03911993789672852,
"mean_time": 0.03901147232055664,
"memory_allocated_mb": 152.0927734375,
"memory_reserved_mb": 278.0,
"min_time": 0.03894742584228516,
"std_time": 5.231496369987919e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 209989.51110295113
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks96k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.04902400076389313,
"mean_ms": 0.04363839998841286,
"median_ms": 0.04273599945008755,
"min_ms": 0.04153599962592125,
"std_ms": 0.002234876899495284
},
"max_time": 0.047580448150634766,
"mean_time": 0.0472187198638916,
"memory_allocated_mb": 168.0966796875,
"memory_reserved_mb": 374.0,
"min_time": 0.04714566421508789,
"std_time": 0.00012505866815108255,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 173490.51443185066
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks112k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.04848000034689903,
"mean_ms": 0.04378239996731281,
"median_ms": 0.042527999728918076,
"min_ms": 0.04163200035691261,
"std_ms": 0.002414242667964844
},
"max_time": 0.05583651351928711,
"mean_time": 0.055455712127685554,
"memory_allocated_mb": 184.1005859375,
"memory_reserved_mb": 486.0,
"min_time": 0.05535052871704101,
"std_time": 0.00014272069779462445,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 147721.48234501254
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.04851200059056282,
"mean_ms": 0.04361280016601086,
"median_ms": 0.042128000408411026,
"min_ms": 0.04179200157523155,
"std_ms": 0.0023557503035821635
},
"max_time": 0.06379235076904297,
"mean_time": 0.06362125053405762,
"memory_allocated_mb": 200.1044921875,
"memory_reserved_mb": 614.0,
"min_time": 0.06355436706542969,
"std_time": 7.081883106781061e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 128762.00846782589
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks136k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.05718399956822395,
"mean_ms": 0.04485439993441105,
"median_ms": 0.042527999728918076,
"min_ms": 0.04156799986958504,
"std_ms": 0.004775190073319638
},
"max_time": 0.06783235168457032,
"mean_time": 0.06771669235229492,
"memory_allocated_mb": 208.1064453125,
"memory_reserved_mb": 750.0,
"min_time": 0.06765017700195312,
"std_time": 5.8650788028093655e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 120974.603387024
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.019807999953627586,
"mean_ms": 0.015823999978601934,
"median_ms": 0.015039999969303608,
"min_ms": 0.014303999952971935,
"std_ms": 0.001913656346272671
},
"max_time": 0.004766176223754883,
"mean_time": 0.0047575551986694335,
"memory_allocated_mb": 132.5458984375,
"memory_reserved_mb": 750.0,
"min_time": 0.004751776218414307,
"std_time": 4.190866412679079e-06,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 107618.29944573072
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q2ks66k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.02687999978661537,
"mean_ms": 0.01938559990376234,
"median_ms": 0.018383999355137348,
"min_ms": 0.018015999346971512,
"std_ms": 0.0025744710494052777
},
"max_time": 0.009582624435424805,
"mean_time": 0.009561939239501951,
"memory_allocated_mb": 84.04248046875,
"memory_reserved_mb": 752.0,
"min_time": 0.009554431915283204,
"std_time": 8.291579854680458e-06,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 214182.49464913702
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q4ks100k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03206399828195572,
"mean_ms": 0.028105599991977214,
"median_ms": 0.02675200067460537,
"min_ms": 0.025728000327944756,
"std_ms": 0.0022390414558466644
},
"max_time": 0.02538035202026367,
"mean_time": 0.0252856897354126,
"memory_allocated_mb": 136.06640625,
"memory_reserved_mb": 752.0,
"min_time": 0.025221471786499024,
"std_time": 4.676405035080222e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 161988.85784252718
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q6ks134k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.04089599847793579,
"mean_ms": 0.03524160049855709,
"median_ms": 0.03411199897527695,
"min_ms": 0.03385600075125694,
"std_ms": 0.0022532652248236193
},
"max_time": 0.048259136199951175,
"mean_time": 0.048143408584594725,
"memory_allocated_mb": 188.09033203125,
"memory_reserved_mb": 752.0,
"min_time": 0.048105857849121095,
"std_time": 5.1879732806325464e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 127618.7162611083
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q1ks8k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.018783999606966972,
"mean_ms": 0.015311999991536141,
"median_ms": 0.014943999703973532,
"min_ms": 0.014271999709308147,
"std_ms": 0.0012243293501672883
},
"max_time": 0.0008861759901046753,
"mean_time": 0.0006353183984756471,
"memory_allocated_mb": 17.0205078125,
"memory_reserved_mb": 752.0,
"min_time": 0.0006019840240478516,
"std_time": 8.386037831062328e-05,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 1611790.2495141604
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s4k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 2,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 16,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.019840000197291374,
"mean_ms": 0.015875199995934963,
"median_ms": 0.01497599994763732,
"min_ms": 0.014431999996304512,
"std_ms": 0.001896142444948042
},
"max_time": 0.00022230400145053863,
"mean_time": 0.00020609600096940998,
"memory_allocated_mb": 8.515625,
"memory_reserved_mb": 752.0,
"min_time": 0.00019760000705718993,
"std_time": 7.520956759601396e-06,
"tensor_parallel_size": 2,
"throughput_tokens_per_sec": 2484279.1591865686
}
],
"schema_version": "qwen30_vllm020_flashattn_raw.v1"
}

View File

@@ -0,0 +1,8 @@
759d33decc09d229d07818df8161ed41180cab575304f335ad4d5e7a917cbc93 runs/frontier-code-trace-v0/profiles/raw/tp4-r1-v2/raw/flashattn-code-longctx-tp4.json
4e6d6002db037bdc71d40f76d6de979ef47960612c15dd985c04048c25bd1d41 runs/frontier-code-trace-v0/profiles/raw/tp4-r1-v2/provenance/aituner.commit
420e4a613b1bcdb44b873eaa911f4a23129713ff9d486fc783b31a009a55cc85 runs/frontier-code-trace-v0/profiles/raw/tp4-r1-v2/provenance/batch-specs.txt
5dd612a7806900e9d12afc27ccef196cc5518aea831aa4ed3b0a53bcfb5746cf runs/frontier-code-trace-v0/profiles/raw/tp4-r1-v2/provenance/max-model-len.txt
484da176e7355ab5532d5129a8e78bea0ae84b3aa0a42c7f0bea9689667762ff runs/frontier-code-trace-v0/profiles/raw/tp4-r1-v2/provenance/pip-freeze.txt
595b25965f76bf7ac25aa3b60126a437a083619303303aaf63346e993862253a runs/frontier-code-trace-v0/profiles/raw/tp4-r1-v2/provenance/source.sha256
4355a46b19d348dc2f57c046f8ef63d4538ebb936000f3c9ee954a27460dd865 runs/frontier-code-trace-v0/profiles/raw/tp4-r1-v2/provenance/vllm-allow-long-max-model-len.txt
11e8f5af440e3db4c01b519a7b0e30bbcecc3a57bfb53dd86d32e601beb95a32 runs/frontier-code-trace-v0/profiles/raw/tp4-r1-v2/provenance/vllm-source.commit

View File

@@ -0,0 +1 @@
e251046c306a8e03214fe1b2f40caabb98f76b7b

View File

@@ -0,0 +1,13 @@
q8ks48k
q8ks64k
q8ks80k
q8ks96k
q8ks112k
q8ks128k
q8ks136k
q512s128k
q2ks66k
q4ks100k
q6ks134k
q1ks8k
q512s4k

View File

@@ -0,0 +1,179 @@
aiohappyeyeballs==2.7.1
aiohttp==3.14.1
aiosignal==1.4.0
annotated-doc==0.0.4
annotated-types==0.7.0
anthropic==0.117.0
anyio==4.14.2
apache-tvm-ffi==0.1.9
astor==0.8.1
attrs==26.1.0
blake3==1.0.9
cachetools==7.1.4
cbor2==6.1.3
certifi==2026.6.17
cffi==2.1.0
charset-normalizer==3.4.9
click==8.4.2
cloudpickle==3.1.2
compressed-tensors==0.15.0.1
cryptography==49.0.0
cuda-bindings==12.9.7
cuda-pathfinder==1.5.6
cuda-python==12.9.7
cuda-tile==1.5.0
cuda-toolkit==12.9.1
depyf==0.20.0
detect-installer==0.1.0
dill==0.4.1
diskcache==5.6.3
distro==1.9.0
dnspython==2.8.0
docstring-parser==0.18.0
einops==0.8.2
email-validator==2.3.0
fastapi==0.139.2
fastapi-cli==0.0.32
fastapi-cloud-cli==0.22.2
fastar==0.11.0
fastsafetensors==0.3.3
filelock==3.31.1
flashinfer-cubin==0.6.8.post1
flashinfer-python==0.6.8.post1
frozenlist==1.8.0
fsspec==2026.6.0
gguf==0.19.0
googleapis-common-protos==1.75.0
grpcio==1.82.1
h11==0.16.0
hf-xet==1.5.2
httpcore==1.0.9
httptools==0.8.0
httpx==0.28.1
httpx-sse==0.4.3
huggingface-hub==1.24.0
idna==3.18
ijson==3.5.1
interegular==0.3.3
jinja2==3.1.6
jiter==0.16.0
jmespath==1.1.0
jsonschema==4.26.0
jsonschema-specifications==2025.9.1
lark==1.2.2
llguidance==1.3.0
llvmlite==0.47.0
lm-format-enforcer==0.11.3
loguru==0.7.3
markdown-it-py==4.2.0
markupsafe==3.0.3
mcp==1.28.1
mdurl==0.1.2
mistral-common==1.11.6
ml-dtypes==0.5.4
model-hosting-container-standards==0.1.16
mpmath==1.3.0
msgspec==0.21.1
multidict==6.7.1
networkx==3.6.1
ninja==1.13.0
numba==0.65.0
numpy==2.3.5
nvidia-cublas-cu12==12.9.1.4
nvidia-cuda-cupti-cu12==12.9.79
nvidia-cuda-nvrtc-cu12==12.9.86
nvidia-cuda-runtime-cu12==12.9.79
nvidia-cudnn-cu12==9.17.1.4
nvidia-cudnn-frontend==1.18.0
nvidia-cufft-cu12==11.4.1.4
nvidia-cufile-cu12==1.14.1.1
nvidia-curand-cu12==10.3.10.19
nvidia-cusolver-cu12==11.7.5.82
nvidia-cusparse-cu12==12.5.10.65
nvidia-cusparselt-cu12==0.7.1
nvidia-cutlass-dsl==4.5.3
nvidia-cutlass-dsl-libs-base==4.5.3
nvidia-ml-py==13.610.43
nvidia-nccl-cu12==2.28.9
nvidia-nvjitlink-cu12==12.9.86
nvidia-nvshmem-cu12==3.4.5
nvidia-nvtx-cu12==12.9.79
openai==2.46.0
openai-harmony==0.0.8
opencv-python-headless==5.0.0.93
opentelemetry-api==1.44.0
opentelemetry-exporter-otlp==1.44.0
opentelemetry-exporter-otlp-proto-common==1.44.0
opentelemetry-exporter-otlp-proto-grpc==1.44.0
opentelemetry-exporter-otlp-proto-http==1.44.0
opentelemetry-proto==1.44.0
opentelemetry-sdk==1.44.0
opentelemetry-semantic-conventions==0.65b0
opentelemetry-semantic-conventions-ai==0.5.1
outlines-core==0.2.14
packaging==26.2
partial-json-parser==0.2.1.1.post7
pillow==12.3.0
prometheus-client==0.25.0
prometheus-fastapi-instrumentator==8.0.2
propcache==0.5.2
protobuf==7.35.1
psutil==7.2.2
py-cpuinfo==9.0.0
pybase64==1.4.3
pycountry==26.2.16
pycparser==3.0
pydantic==2.13.4
pydantic-core==2.46.4
pydantic-extra-types==2.11.1
pydantic-settings==2.14.2
pygments==2.20.0
pyjwt==2.13.0
python-dotenv==1.2.2
python-json-logger==4.1.0
python-multipart==0.0.32
pyyaml==6.0.3
pyzmq==27.1.0
quack-kernels==0.5.0
referencing==0.37.0
regex==2026.7.19
requests==2.34.2
rich==15.0.0
rich-toolkit==0.20.3
rignore==0.8.0
rpds-py==2026.6.3
safetensors==0.8.0
sentencepiece==0.2.2
sentry-sdk==2.66.0
setproctitle==1.3.7
setuptools==80.10.2
shellingham==1.5.4
six==1.17.0
sniffio==1.3.1
sse-starlette==3.4.5
starlette==1.3.1
supervisor==4.3.0
sympy==1.14.0
tabulate==0.10.0
tiktoken==0.13.0
tilelang==0.1.9
tokenizers==0.22.2
torch==2.11.0+cu129
torch-c-dlpack-ext==0.1.5
torchaudio==2.11.0+cu129
torchvision==0.26.0+cu129
tqdm==4.69.0
transformers==5.14.1
triton==3.6.0
typer==0.27.0
typing-extensions==4.16.0
typing-inspection==0.4.2
urllib3==2.7.0
uvicorn==0.51.0
uvloop==0.22.1
vllm @ https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0%2Bcu129-cp38-abi3-manylinux_2_31_x86_64.whl
watchfiles==1.2.0
websockets==16.1.1
xgrammar==0.2.3
yarl==1.24.5
z3-solver==4.15.4.0

View File

@@ -0,0 +1,2 @@
42381f4bd1b67710d4068eb8993240930326a804c6e5167b0b78df3420e47c74 /home/admin/cpfs/wjh/aituner/aituner-code-trace-fbaa909/runs/frontier-code-trace-v0/../frontier-qwen30-vllm020-profile-v1/profile_vllm020_flashattn.py
8a7a72b3adb25db6a61a86f9b04d2d19ebbfe3f2d20480852ef2f45376ac3b85 runs/frontier-code-trace-v0/run_flashattn_code_longctx.sh

View File

@@ -0,0 +1 @@
88d34c6409e9fb3c7b8ca0c04756f061d2099eb1

View File

@@ -0,0 +1,564 @@
{
"environment": {
"attention_backend": "FLASH_ATTN",
"block_size": 16,
"dtype": "bfloat16",
"gpu": "NVIDIA H20",
"model": "/home/admin/cpfs/wjh/models/Qwen/Qwen3-30B-A3B",
"profile_kv_update": true,
"profile_method": "cuda_event",
"torch_cuda": "12.9",
"torch_version": "2.11.0+cu129",
"vllm_source_commit": "88d34c6409e9fb3c7b8ca0c04756f061d2099eb1",
"vllm_version": "0.20.0"
},
"rows": [
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks48k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.039583999663591385,
"mean_ms": 0.027286400087177753,
"median_ms": 0.025200000032782555,
"min_ms": 0.024191999807953835,
"std_ms": 0.004620118437533604
},
"max_time": 0.011496224403381348,
"mean_time": 0.011354345512390137,
"memory_allocated_mb": 60.0849609375,
"memory_reserved_mb": 62.0,
"min_time": 0.01132140827178955,
"std_time": 5.011776621543997e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 721485.8831854897
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks64k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03232000023126602,
"mean_ms": 0.026406400091946124,
"median_ms": 0.02518399991095066,
"min_ms": 0.024032000452280045,
"std_ms": 0.002792696117638337
},
"max_time": 0.015639776229858397,
"mean_time": 0.015451993656158448,
"memory_allocated_mb": 68.0888671875,
"memory_reserved_mb": 94.0,
"min_time": 0.01542249584197998,
"std_time": 6.304729308178388e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 530158.1260185833
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks80k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03488000109791756,
"mean_ms": 0.026147199980914592,
"median_ms": 0.02459200005978346,
"min_ms": 0.024224000051617622,
"std_ms": 0.003303059079404486
},
"max_time": 0.01970569610595703,
"mean_time": 0.01954985942840576,
"memory_allocated_mb": 76.0927734375,
"memory_reserved_mb": 134.0,
"min_time": 0.019525152206420898,
"std_time": 5.224891496533832e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 419031.1459783236
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks96k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03046399913728237,
"mean_ms": 0.0255103999748826,
"median_ms": 0.024656000547111034,
"min_ms": 0.024159999564290047,
"std_ms": 0.00195717586616911
},
"max_time": 0.023927040100097656,
"mean_time": 0.02367282257080078,
"memory_allocated_mb": 84.0966796875,
"memory_reserved_mb": 182.0,
"min_time": 0.023630016326904296,
"std_time": 8.715594728873713e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 346050.83426360885
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks112k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03574400022625923,
"mean_ms": 0.02678720001131296,
"median_ms": 0.024720000103116035,
"min_ms": 0.02380800060927868,
"std_ms": 0.003678231234048846
},
"max_time": 0.02779840087890625,
"mean_time": 0.027747158622741696,
"memory_allocated_mb": 92.1005859375,
"memory_reserved_mb": 238.0,
"min_time": 0.027736671447753908,
"std_time": 1.8952179343788658e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 295237.4371509809
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.042080000042915344,
"mean_ms": 0.02844479978084564,
"median_ms": 0.02527999971061945,
"min_ms": 0.024224000051617622,
"std_ms": 0.006562862750709003
},
"max_time": 0.03208835220336914,
"mean_time": 0.03186994876861572,
"memory_allocated_mb": 100.1044921875,
"memory_reserved_mb": 302.0,
"min_time": 0.03183427238464356,
"std_time": 7.444245241186317e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 257044.65543625728
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks136k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03139200061559677,
"mean_ms": 0.026396799832582474,
"median_ms": 0.025071999989449978,
"min_ms": 0.02377600036561489,
"std_ms": 0.0025962810796740263
},
"max_time": 0.033969856262207034,
"mean_time": 0.033897081375122075,
"memory_allocated_mb": 104.1064453125,
"memory_reserved_mb": 370.0,
"min_time": 0.03388313674926758,
"std_time": 2.518493598134421e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 241672.72424853416
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.022624000906944275,
"mean_ms": 0.017174400109797715,
"median_ms": 0.015536000020802021,
"min_ms": 0.014944000169634819,
"std_ms": 0.0028331860349340154
},
"max_time": 0.003192960023880005,
"mean_time": 0.003182867193222046,
"memory_allocated_mb": 66.2958984375,
"memory_reserved_mb": 372.0,
"min_time": 0.003179744005203247,
"std_time": 3.5239767128429387e-06,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 160861.25148115202
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q2ks66k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.02985600009560585,
"mean_ms": 0.017318400088697672,
"median_ms": 0.015568000264465809,
"min_ms": 0.014655999839305878,
"std_ms": 0.004479309643844885
},
"max_time": 0.004826848030090332,
"mean_time": 0.004817302370071412,
"memory_allocated_mb": 42.04248046875,
"memory_reserved_mb": 372.0,
"min_time": 0.004811935901641846,
"std_time": 5.94674080057053e-06,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 425134.20222979283
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q4ks100k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03916800022125244,
"mean_ms": 0.021356799826025962,
"median_ms": 0.01935999933630228,
"min_ms": 0.017184000462293625,
"std_ms": 0.0062801287963799276
},
"max_time": 0.014635007858276367,
"mean_time": 0.014412275123596191,
"memory_allocated_mb": 68.06640625,
"memory_reserved_mb": 372.0,
"min_time": 0.014374591827392579,
"std_time": 7.569270440104603e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 284202.17938345566
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q6ks134k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.033215999603271484,
"mean_ms": 0.023340800032019614,
"median_ms": 0.02092800009995699,
"min_ms": 0.020479999482631683,
"std_ms": 0.004304974035052475
},
"max_time": 0.02418492889404297,
"mean_time": 0.02411243553161621,
"memory_allocated_mb": 96.09033203125,
"memory_reserved_mb": 372.0,
"min_time": 0.024088096618652344,
"std_time": 2.797845886790172e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 254806.28001862322
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q1ks8k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.04992000013589859,
"mean_ms": 0.019958400167524815,
"median_ms": 0.015552000608295202,
"min_ms": 0.014688000082969666,
"std_ms": 0.010290953685254347
},
"max_time": 0.00036675199866294863,
"mean_time": 0.0003469599992036819,
"memory_allocated_mb": 8.5205078125,
"memory_reserved_mb": 372.0,
"min_time": 0.0003364480137825012,
"std_time": 7.875187259592915e-06,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 2951348.865431786
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s4k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.022336000576615334,
"mean_ms": 0.016252800077199935,
"median_ms": 0.014895999804139137,
"min_ms": 0.014271999709308147,
"std_ms": 0.0028271219653727385
},
"max_time": 0.00016867199540138245,
"mean_time": 0.00015103680044412614,
"memory_allocated_mb": 4.265625,
"memory_reserved_mb": 372.0,
"min_time": 0.00014329600334167482,
"std_time": 7.872199103975806e-06,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 3389902.3184710997
}
],
"schema_version": "qwen30_vllm020_flashattn_raw.v1"
}

View File

@@ -0,0 +1,8 @@
6ee8fdf15bef211e7d53fef0c5daf02b17429bc7d88a49cb967c2b2f6134ebe2 runs/frontier-code-trace-v0/profiles/raw/tp4-r2-v2/raw/flashattn-code-longctx-tp4.json
4e6d6002db037bdc71d40f76d6de979ef47960612c15dd985c04048c25bd1d41 runs/frontier-code-trace-v0/profiles/raw/tp4-r2-v2/provenance/aituner.commit
420e4a613b1bcdb44b873eaa911f4a23129713ff9d486fc783b31a009a55cc85 runs/frontier-code-trace-v0/profiles/raw/tp4-r2-v2/provenance/batch-specs.txt
5dd612a7806900e9d12afc27ccef196cc5518aea831aa4ed3b0a53bcfb5746cf runs/frontier-code-trace-v0/profiles/raw/tp4-r2-v2/provenance/max-model-len.txt
484da176e7355ab5532d5129a8e78bea0ae84b3aa0a42c7f0bea9689667762ff runs/frontier-code-trace-v0/profiles/raw/tp4-r2-v2/provenance/pip-freeze.txt
595b25965f76bf7ac25aa3b60126a437a083619303303aaf63346e993862253a runs/frontier-code-trace-v0/profiles/raw/tp4-r2-v2/provenance/source.sha256
4355a46b19d348dc2f57c046f8ef63d4538ebb936000f3c9ee954a27460dd865 runs/frontier-code-trace-v0/profiles/raw/tp4-r2-v2/provenance/vllm-allow-long-max-model-len.txt
11e8f5af440e3db4c01b519a7b0e30bbcecc3a57bfb53dd86d32e601beb95a32 runs/frontier-code-trace-v0/profiles/raw/tp4-r2-v2/provenance/vllm-source.commit

View File

@@ -0,0 +1 @@
e251046c306a8e03214fe1b2f40caabb98f76b7b

View File

@@ -0,0 +1,13 @@
q8ks48k
q8ks64k
q8ks80k
q8ks96k
q8ks112k
q8ks128k
q8ks136k
q512s128k
q2ks66k
q4ks100k
q6ks134k
q1ks8k
q512s4k

View File

@@ -0,0 +1,179 @@
aiohappyeyeballs==2.7.1
aiohttp==3.14.1
aiosignal==1.4.0
annotated-doc==0.0.4
annotated-types==0.7.0
anthropic==0.117.0
anyio==4.14.2
apache-tvm-ffi==0.1.9
astor==0.8.1
attrs==26.1.0
blake3==1.0.9
cachetools==7.1.4
cbor2==6.1.3
certifi==2026.6.17
cffi==2.1.0
charset-normalizer==3.4.9
click==8.4.2
cloudpickle==3.1.2
compressed-tensors==0.15.0.1
cryptography==49.0.0
cuda-bindings==12.9.7
cuda-pathfinder==1.5.6
cuda-python==12.9.7
cuda-tile==1.5.0
cuda-toolkit==12.9.1
depyf==0.20.0
detect-installer==0.1.0
dill==0.4.1
diskcache==5.6.3
distro==1.9.0
dnspython==2.8.0
docstring-parser==0.18.0
einops==0.8.2
email-validator==2.3.0
fastapi==0.139.2
fastapi-cli==0.0.32
fastapi-cloud-cli==0.22.2
fastar==0.11.0
fastsafetensors==0.3.3
filelock==3.31.1
flashinfer-cubin==0.6.8.post1
flashinfer-python==0.6.8.post1
frozenlist==1.8.0
fsspec==2026.6.0
gguf==0.19.0
googleapis-common-protos==1.75.0
grpcio==1.82.1
h11==0.16.0
hf-xet==1.5.2
httpcore==1.0.9
httptools==0.8.0
httpx==0.28.1
httpx-sse==0.4.3
huggingface-hub==1.24.0
idna==3.18
ijson==3.5.1
interegular==0.3.3
jinja2==3.1.6
jiter==0.16.0
jmespath==1.1.0
jsonschema==4.26.0
jsonschema-specifications==2025.9.1
lark==1.2.2
llguidance==1.3.0
llvmlite==0.47.0
lm-format-enforcer==0.11.3
loguru==0.7.3
markdown-it-py==4.2.0
markupsafe==3.0.3
mcp==1.28.1
mdurl==0.1.2
mistral-common==1.11.6
ml-dtypes==0.5.4
model-hosting-container-standards==0.1.16
mpmath==1.3.0
msgspec==0.21.1
multidict==6.7.1
networkx==3.6.1
ninja==1.13.0
numba==0.65.0
numpy==2.3.5
nvidia-cublas-cu12==12.9.1.4
nvidia-cuda-cupti-cu12==12.9.79
nvidia-cuda-nvrtc-cu12==12.9.86
nvidia-cuda-runtime-cu12==12.9.79
nvidia-cudnn-cu12==9.17.1.4
nvidia-cudnn-frontend==1.18.0
nvidia-cufft-cu12==11.4.1.4
nvidia-cufile-cu12==1.14.1.1
nvidia-curand-cu12==10.3.10.19
nvidia-cusolver-cu12==11.7.5.82
nvidia-cusparse-cu12==12.5.10.65
nvidia-cusparselt-cu12==0.7.1
nvidia-cutlass-dsl==4.5.3
nvidia-cutlass-dsl-libs-base==4.5.3
nvidia-ml-py==13.610.43
nvidia-nccl-cu12==2.28.9
nvidia-nvjitlink-cu12==12.9.86
nvidia-nvshmem-cu12==3.4.5
nvidia-nvtx-cu12==12.9.79
openai==2.46.0
openai-harmony==0.0.8
opencv-python-headless==5.0.0.93
opentelemetry-api==1.44.0
opentelemetry-exporter-otlp==1.44.0
opentelemetry-exporter-otlp-proto-common==1.44.0
opentelemetry-exporter-otlp-proto-grpc==1.44.0
opentelemetry-exporter-otlp-proto-http==1.44.0
opentelemetry-proto==1.44.0
opentelemetry-sdk==1.44.0
opentelemetry-semantic-conventions==0.65b0
opentelemetry-semantic-conventions-ai==0.5.1
outlines-core==0.2.14
packaging==26.2
partial-json-parser==0.2.1.1.post7
pillow==12.3.0
prometheus-client==0.25.0
prometheus-fastapi-instrumentator==8.0.2
propcache==0.5.2
protobuf==7.35.1
psutil==7.2.2
py-cpuinfo==9.0.0
pybase64==1.4.3
pycountry==26.2.16
pycparser==3.0
pydantic==2.13.4
pydantic-core==2.46.4
pydantic-extra-types==2.11.1
pydantic-settings==2.14.2
pygments==2.20.0
pyjwt==2.13.0
python-dotenv==1.2.2
python-json-logger==4.1.0
python-multipart==0.0.32
pyyaml==6.0.3
pyzmq==27.1.0
quack-kernels==0.5.0
referencing==0.37.0
regex==2026.7.19
requests==2.34.2
rich==15.0.0
rich-toolkit==0.20.3
rignore==0.8.0
rpds-py==2026.6.3
safetensors==0.8.0
sentencepiece==0.2.2
sentry-sdk==2.66.0
setproctitle==1.3.7
setuptools==80.10.2
shellingham==1.5.4
six==1.17.0
sniffio==1.3.1
sse-starlette==3.4.5
starlette==1.3.1
supervisor==4.3.0
sympy==1.14.0
tabulate==0.10.0
tiktoken==0.13.0
tilelang==0.1.9
tokenizers==0.22.2
torch==2.11.0+cu129
torch-c-dlpack-ext==0.1.5
torchaudio==2.11.0+cu129
torchvision==0.26.0+cu129
tqdm==4.69.0
transformers==5.14.1
triton==3.6.0
typer==0.27.0
typing-extensions==4.16.0
typing-inspection==0.4.2
urllib3==2.7.0
uvicorn==0.51.0
uvloop==0.22.1
vllm @ https://github.com/vllm-project/vllm/releases/download/v0.20.0/vllm-0.20.0%2Bcu129-cp38-abi3-manylinux_2_31_x86_64.whl
watchfiles==1.2.0
websockets==16.1.1
xgrammar==0.2.3
yarl==1.24.5
z3-solver==4.15.4.0

View File

@@ -0,0 +1,2 @@
42381f4bd1b67710d4068eb8993240930326a804c6e5167b0b78df3420e47c74 /home/admin/cpfs/wjh/aituner/aituner-code-trace-fbaa909/runs/frontier-code-trace-v0/../frontier-qwen30-vllm020-profile-v1/profile_vllm020_flashattn.py
8a7a72b3adb25db6a61a86f9b04d2d19ebbfe3f2d20480852ef2f45376ac3b85 runs/frontier-code-trace-v0/run_flashattn_code_longctx.sh

View File

@@ -0,0 +1 @@
88d34c6409e9fb3c7b8ca0c04756f061d2099eb1

View File

@@ -0,0 +1,564 @@
{
"environment": {
"attention_backend": "FLASH_ATTN",
"block_size": 16,
"dtype": "bfloat16",
"gpu": "NVIDIA H20",
"model": "/home/admin/cpfs/wjh/models/Qwen/Qwen3-30B-A3B",
"profile_kv_update": true,
"profile_method": "cuda_event",
"torch_cuda": "12.9",
"torch_version": "2.11.0+cu129",
"vllm_source_commit": "88d34c6409e9fb3c7b8ca0c04756f061d2099eb1",
"vllm_version": "0.20.0"
},
"rows": [
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks48k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.040991999208927155,
"mean_ms": 0.02782719973474741,
"median_ms": 0.025295999832451344,
"min_ms": 0.02412799932062626,
"std_ms": 0.005434033125981124
},
"max_time": 0.011621503829956055,
"mean_time": 0.011469862461090087,
"memory_allocated_mb": 60.0849609375,
"memory_reserved_mb": 62.0,
"min_time": 0.01142630386352539,
"std_time": 5.2644635131708985e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 714219.5495185945
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks64k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03811199963092804,
"mean_ms": 0.02837119996547699,
"median_ms": 0.026176000013947487,
"min_ms": 0.024159999564290047,
"std_ms": 0.004902838828525061
},
"max_time": 0.015688991546630858,
"mean_time": 0.015603910255432129,
"memory_allocated_mb": 68.0888671875,
"memory_reserved_mb": 94.0,
"min_time": 0.01557091236114502,
"std_time": 3.431861179521254e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 524996.6108429873
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks80k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.029759999364614487,
"mean_ms": 0.02612799983471632,
"median_ms": 0.025695999152958393,
"min_ms": 0.024000000208616257,
"std_ms": 0.0019206532069484194
},
"max_time": 0.020114784240722657,
"mean_time": 0.019786701011657717,
"memory_allocated_mb": 76.0927734375,
"memory_reserved_mb": 134.0,
"min_time": 0.019717824935913085,
"std_time": 0.00011854565766692264,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 414015.4538734641
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks96k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.032896000891923904,
"mean_ms": 0.026428800262510776,
"median_ms": 0.02483200002461672,
"min_ms": 0.024159999564290047,
"std_ms": 0.0028081875041272375
},
"max_time": 0.023987648010253906,
"mean_time": 0.023885836791992184,
"memory_allocated_mb": 84.0966796875,
"memory_reserved_mb": 182.0,
"min_time": 0.023849760055541992,
"std_time": 4.082290087440645e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 342964.7481618228
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks112k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03388800099492073,
"mean_ms": 0.02606080025434494,
"median_ms": 0.02440000046044588,
"min_ms": 0.023840000852942467,
"std_ms": 0.003071273491702287
},
"max_time": 0.02814064025878906,
"mean_time": 0.028014582252502435,
"memory_allocated_mb": 92.1005859375,
"memory_reserved_mb": 238.0,
"min_time": 0.027969343185424805,
"std_time": 4.739888751775852e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 292419.138224638
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.02953599952161312,
"mean_ms": 0.025788800045847892,
"median_ms": 0.024800000712275505,
"min_ms": 0.024191999807953835,
"std_ms": 0.002060583463186555
},
"max_time": 0.03218783950805664,
"mean_time": 0.03215217628479004,
"memory_allocated_mb": 100.1044921875,
"memory_reserved_mb": 302.0,
"min_time": 0.03213116836547852,
"std_time": 1.9548439899050275e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 254788.35172583075
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q8ks136k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.032735999673604965,
"mean_ms": 0.02950399983674288,
"median_ms": 0.028351999819278717,
"min_ms": 0.027904000133275986,
"std_ms": 0.0017488521169760844
},
"max_time": 0.03428476715087891,
"mean_time": 0.03422174110412597,
"memory_allocated_mb": 104.1064453125,
"memory_reserved_mb": 370.0,
"min_time": 0.034186080932617186,
"std_time": 3.324003475365458e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 239379.98873506542
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s128k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03446400165557861,
"mean_ms": 0.0248607998713851,
"median_ms": 0.023519999347627163,
"min_ms": 0.0226879995316267,
"std_ms": 0.0035075699446123015
},
"max_time": 0.003272864103317261,
"mean_time": 0.003240902423858643,
"memory_allocated_mb": 66.2958984375,
"memory_reserved_mb": 372.0,
"min_time": 0.003228480100631714,
"std_time": 1.4700871477091224e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 157980.68964705482
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q2ks66k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.03484800085425377,
"mean_ms": 0.02619840018451214,
"median_ms": 0.024639999493956566,
"min_ms": 0.02284800074994564,
"std_ms": 0.0037605668537262792
},
"max_time": 0.005165311813354492,
"mean_time": 0.004916048002243043,
"memory_allocated_mb": 42.04248046875,
"memory_reserved_mb": 372.0,
"min_time": 0.004875135898590088,
"std_time": 8.365277131314812e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 416594.79302593466
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q4ks100k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.023615999147295952,
"mean_ms": 0.018966399878263474,
"median_ms": 0.01788800023496151,
"min_ms": 0.01679999940097332,
"std_ms": 0.0021688374822942504
},
"max_time": 0.0145830717086792,
"mean_time": 0.01451537265777588,
"memory_allocated_mb": 68.06640625,
"memory_reserved_mb": 372.0,
"min_time": 0.014502431869506836,
"std_time": 2.2952325515836057e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 282183.5922900522
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q6ks134k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.028192000463604927,
"mean_ms": 0.021894400380551814,
"median_ms": 0.021088000386953354,
"min_ms": 0.020640000700950623,
"std_ms": 0.0021379161656161685
},
"max_time": 0.024349727630615235,
"mean_time": 0.024299632072448733,
"memory_allocated_mb": 96.09033203125,
"memory_reserved_mb": 372.0,
"min_time": 0.024284608840942384,
"std_time": 2.0842542550131525e-05,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 252843.3344867865
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q1ks8k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.0261439997702837,
"mean_ms": 0.016019200067967178,
"median_ms": 0.01512000011280179,
"min_ms": 0.014368000440299511,
"std_ms": 0.0034013214299104646
},
"max_time": 0.0003514559864997864,
"mean_time": 0.00034397439956665037,
"memory_allocated_mb": 8.5205078125,
"memory_reserved_mb": 372.0,
"min_time": 0.0003375680148601532,
"std_time": 3.760143706773439e-06,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 2976965.731432534
},
{
"attention_core_excludes_kv_cache_update": true,
"config": {
"backend": "FLASH_ATTN",
"batch_spec": "q512s4k",
"block_size": 16,
"device": "cuda:0",
"dtype": "torch.bfloat16",
"head_dim": 128,
"kv_cache_dtype": "auto",
"kv_lora_rank": null,
"num_kv_heads": 1,
"num_kv_splits": null,
"num_layers": 1,
"num_q_heads": 8,
"prefill_backend": null,
"profile_memory": true,
"qk_nope_head_dim": null,
"qk_rope_head_dim": null,
"reorder_batch_threshold": null,
"repeats": 10,
"use_cuda_graphs": false,
"v_head_dim": null,
"warmup_iters": 5
},
"error": null,
"kv_cache_update_time": {
"max_ms": 0.021023999899625778,
"mean_ms": 0.015327999833971262,
"median_ms": 0.014639999717473984,
"min_ms": 0.013856000266969204,
"std_ms": 0.0020222886267191407
},
"max_time": 0.0001605760008096695,
"mean_time": 0.0001493696004152298,
"memory_allocated_mb": 4.265625,
"memory_reserved_mb": 372.0,
"min_time": 0.00014313599467277526,
"std_time": 5.429351067501872e-06,
"tensor_parallel_size": 4,
"throughput_tokens_per_sec": 3427738.9681481416
}
],
"schema_version": "qwen30_vllm020_flashattn_raw.v1"
}