1# Optional: enable schedule overlapping (experimental, may not be stable)
2# export SGLANG_ENABLE_SPEC_V2=1
3# export SGLANG_ENABLE_DFLASH_SPEC_V2=1
4# export SGLANG_ENABLE_OVERLAP_PLAN_STREAM=1
5
6python -m sglang.launch_server \
7 --model-path openai/gpt-oss-20b \
8 --speculative-algorithm DFLASH \
9 --speculative-draft-model-path z-lab/gpt-oss-20b-DFlash \
10 --tp-size 1 \
11 --dtype bfloat16 \
12 --attention-backend fa3 \
13 --mem-fraction-static 0.75 \
14 --trust-remote-code
1from openai import OpenAI
2
3client = OpenAI(base_url="http://localhost:30000/v1", api_key="EMPTY")
4
5response = client.chat.completions.create(
6 model="openai/gpt-oss-20b",
7 messages=[{"role": "user", "content": "Write a quicksort in Python."}],
8 max_tokens=2048,
9 temperature=0.0,
10)
11print(response.choices[0].message.content)
1uv pip install vllm
2uv pip install -U vllm --torch-backend=auto --extra-index-url https://wheels.vllm.ai/nightly
1vllm serve openai/gpt-oss-20b \
2 --speculative-config '{"method": "dflash", "model": "z-lab/gpt-oss-20b-DFlash", "num_speculative_tokens": 7}' \
3 --attention-backend flash_attn \
4 --max-num-batched-tokens 32768
1from openai import OpenAI
2
3client = OpenAI(base_url="http://localhost:8000/v1", api_key="EMPTY")
4
5response = client.chat.completions.create(
6 model="openai/gpt-oss-20b",
7 messages=[{"role": "user", "content": "Write a quicksort in Python."}],
8 max_tokens=2048,
9 temperature=0.0,
10)
11print(response.choices[0].message.content)
The numbers reported are end-to-end speedup (including prefill time). You can specify different block size during inference by passing --speculative-num-draft-tokens arguments when launch the server.
We are grateful to
Yotta Labs for their compute support in training this draft model.
If you find DFlash useful for your research or applications, please cite our project.
1@misc{chen2026dflash,
2 title = {DFlash: Block Diffusion for Flash Speculative Decoding},
3 author = {Chen, Jian and Liang, Yesheng and Liu, Zhijian},
4 year = {2026},
5 eprint = {2602.06036},
6 archivePrefix = {arXiv},
7 primaryClass = {cs.CL},
8 url = {https://arxiv.org/abs/2602.06036}
9}