Files
aituner/runs/frontier-multicase-sufficiency-v0/results/community-qwen235b-smoke/profiles/compute/h20/attention_config.yaml

66 lines
1.2 KiB
YAML

attention_backend: FLASHINFER
batch_size_list:
- 1
block_shape: null
block_size: 16
decode_kv_cache_size_list: null
device: h20
disable_ray: true
disable_replicated: false
enable_chunked_prefill_grid_search: false
enable_mixed_prefill: false
enable_true_mixed: false
fixed_chunked_prefill_size: 128
max_batch_size: 1
max_mixed_batch_size: 8
max_model_len: 40960
max_pipeline_parallel_size: 1
max_seq_len: 128
min_batch_size: 1
mixed_batch_size_list: null
mixed_batch_size_max: 32
mixed_batch_size_min: 2
mixed_kv_cache_size_list:
- 0
mixed_mode: both
mixed_num_samples: 3
mixed_profile_strategy: default
mixed_shapes_per_point: 2
mixed_total_tokens_list: null
mixed_total_tokens_max: 1055
mixed_total_tokens_min: 1025
models:
- Qwen3-235B-A22B-FP8
num_gpus: 1
num_tensor_parallel_workers:
- 4
output_dir: /home/admin/cpfs/wjh/frontier-community-qwen235-smoke-20260715/profiles
precision: null
profile_method: cuda_event
profile_only_decode: false
profile_only_prefill: true
skip_confirmation: true
true_mixed_decode_batch_sizes:
- 1
- 2
- 4
- 8
true_mixed_decode_kv_cache_sizes:
- 128
- 256
- 512
- 1024
- 2048
true_mixed_prefill_batch_sizes:
- 1
- 2
- 4
true_mixed_prefill_chunk_sizes:
- 64
- 128
- 256
- 512
- 1024
true_mixed_prefill_kv_cache_size: 0
use_fp8: null