diff --git a/tools/launcher/examples/Qwen/Qwen3.5-4B/specdec_bench_dflash_vllm.yaml b/tools/launcher/examples/Qwen/Qwen3.5-4B/specdec_bench_dflash_vllm.yaml new file mode 100644 index 00000000000..0620b4c78c9 --- /dev/null +++ b/tools/launcher/examples/Qwen/Qwen3.5-4B/specdec_bench_dflash_vllm.yaml @@ -0,0 +1,99 @@ +# SPEED-bench DFlash speculative-decoding run for Qwen3.5-4B via vLLM. +# +# Companion to specdec_bench_mtp_vllm.yaml. This variant exercises the +# z-lab/Qwen3.5-4B-DFlash external draft model. DFlash uses +# vLLM's `speculative_num_draft_tokens`, which the specdec_bench wrapper +# computes from `--block_size`. `--block_size = draft_length + 1` +# (e.g. draft_length=3 → block_size=4; draft_length=7 → block_size=8). +# +# Container: `vllm/vllm-openai:v0.22.1`+ is required. The `dflash` +# speculative method landed in vLLM v0.22.0; older tags reject +# `--speculative_algorithm DFLASH` with "Input should be 'ngram', ..., +# 'mtp'". +# +# Cells in OMNIML-4961 (t0_d3 / t0_d7 / t1_d3 / t1_d7) override per-cell +# knobs (temperature, block_size, save_dir, num_requests) via +# pipeline.task_N.args+=[...] at slurm-invoke time — no per-cell file is +# committed (post-#1564 convention). +# +# Slurm run on cw_dfw: +# uv run slurm.py \ +# --yaml modules/Model-Optimizer/tools/launcher/examples/Qwen/Qwen3.5-4B/specdec_bench_dflash_vllm.yaml \ +# --yes detach=true \ +# pipeline.task_0.args+=["--temperature 0","--block_size 4","--save_dir /scratchspace//qualitative"] \ +# pipeline.task_1.args+=["--temperature 0","--block_size 4","--save_dir /scratchspace//throughput_32k","--num_requests 80","--max_seq_len 65536"] +# +# Reference run: cicd_1780697785 (OMNIML-4962, cw_dfw) produced +# qualitative Average_AL = 5.0395 +# throughput_32k Average_AL = 6.8958 +# See OMNIML-4962 INTERN-ARTIFACTS worklog for the full payload. + +job_name: Qwen3.5-4B_specdec_bench_dflash_vllm + +pipeline: + global_vars: + hf_model: /hf-local/Qwen/Qwen3.5-4B + draft_model: /hf-local/z-lab/Qwen3.5-4B-DFlash + + # task_0: SPEED qualitative split — quality / acceptance-rate numbers. + # tp_size=2 + concurrency=32 trades aa_timing fidelity for ~30x + # wall-clock speedup; acceptance-length (AL) is concurrency-independent + # and is the primary metric we care about for this split. + task_0: + script: common/specdec_bench/run.sh + args: + - --dataset speed + - --dataset_path /hf-local/nvidia/SPEED-Bench-Internal/qualitative + - --engine VLLM + - --speculative_algorithm DFLASH + - --draft_model_dir <> + - --block_size 4 + - --tp_size 2 + - --ep_size 1 + - --concurrency 32 + - --output_length 4096 + - --aa_timing + - --show_progress + - --save_dir /scratchspace/{sweep_name_default}/qualitative + environment: + - HF_MODEL_CKPT: <> + - HF_LOCAL: /hf-local + slurm_config: + _factory_: "slurm_factory" + nodes: 1 + ntasks_per_node: 1 + gpus_per_node: 2 + container: vllm/vllm-openai:v0.22.1 + + # task_1: SPEED throughput_32k split — long-context throughput. + # `--num_requests 80` caps the run at 80 samples (split has 1,536) so + # it fits in the 4h Slurm time-limit; each 32K-input sample takes + # ~60-90s. tp_size=2 doubles the KV-cache budget across 2 GPUs; + # concurrency=8 keeps 8 * 32K = 256K tokens of in-flight KV under that + # doubled budget. + task_1: + script: common/specdec_bench/run.sh + args: + - --dataset speed + - --dataset_path /hf-local/nvidia/SPEED-Bench-Internal/throughput_32k + - --engine VLLM + - --speculative_algorithm DFLASH + - --draft_model_dir <> + - --block_size 4 + - --tp_size 2 + - --ep_size 1 + - --concurrency 8 + - --num_requests 80 + - --output_length 4096 + - --aa_timing + - --show_progress + - --save_dir /scratchspace/{sweep_name_default}/throughput_32k + environment: + - HF_MODEL_CKPT: <> + - HF_LOCAL: /hf-local + slurm_config: + _factory_: "slurm_factory" + nodes: 1 + ntasks_per_node: 1 + gpus_per_node: 2 + container: vllm/vllm-openai:v0.22.1