Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -99,3 +99,5 @@ proxy_config.yml
# Claude Code local config
CLAUDE.local.md
.worktrees/

.deep_gemm
17 changes: 11 additions & 6 deletions benchmark/profile_restful_api.py
Original file line number Diff line number Diff line change
Expand Up @@ -164,6 +164,7 @@ async def async_request_openai_completions(
assert api_url.endswith('completions'), "OpenAI Completions API URL must end with 'completions'."

prompt = request_func_input.prompt
output_len = request_func_input.output_len

async with aiohttp.ClientSession(timeout=AIOHTTP_TIMEOUT) as session:
payload = {
Expand Down Expand Up @@ -200,10 +201,14 @@ async def async_request_openai_completions(
else:
data = json.loads(chunk)

# NOTE: Some completion API might have a last
# usage summary response without a token so we
# want to check a token was generated
if data['choices'][0]['text']:
# A usage-only final chunk has no choices. Use its
# actual token count, especially when EOS is enabled.
reported_tokens = (data.get('usage') or {}).get('completion_tokens')
if reported_tokens is not None:
output_len = reported_tokens
choices = data.get('choices') or []
text = choices[0].get('text') if choices else None
if text:
timestamp = time.perf_counter()
# First token
if ttft == 0.0:
Expand All @@ -215,12 +220,12 @@ async def async_request_openai_completions(
output.itl.append(timestamp - most_recent_timestamp)

most_recent_timestamp = timestamp
generated_text += data['choices'][0]['text']
generated_text += text

output.generated_text = generated_text
output.success = True
output.latency = latency
output.output_len = request_func_input.output_len
output.output_len = output_len
else:
output.error = response.reason or ''
output.success = False
Expand Down
54 changes: 54 additions & 0 deletions docs/en/advance/spec_decoding.md
Original file line number Diff line number Diff line change
Expand Up @@ -145,6 +145,60 @@ When a DFlash block size is provided, it overrides
`--speculative-num-draft-tokens` by setting the number of newly proposed
tokens to `block_size - 1`.

### DSpark

DSpark uses a parallel DFlash-style draft backbone followed by a lightweight
left-to-right Markov correction. The first LMDeploy implementation uses a
fixed verification window. Draft proposals are deterministic, while target
verification supports both greedy and non-greedy sampling through the common
rejection sampler. It supports external Speculators-format drafts and
DeepSeek-V4 checkpoints that bundle `mtp.*` DSpark weights. For a bundled
checkpoint, leave `model` empty so the target checkpoint is also used as the
draft weight source.

```python
from lmdeploy import PytorchEngineConfig, pipeline
from lmdeploy.messages import SpeculativeConfig

def main():
model = 'deepseek-ai/DeepSeek-V4-Flash-0731'
pipe = pipeline(
model,
backend_config=PytorchEngineConfig(tp=4),
speculative_config=SpeculativeConfig(
method='dspark',
num_speculative_tokens=5,
),
)


if __name__ == '__main__':
main()
```

```shell
lmdeploy serve api_server deepseek-ai/DeepSeek-V4-Flash-0731 \
--backend pytorch \
--tp 4 \
--speculative-algorithm dspark \
--speculative-num-draft-tokens 5
```

DSpark V1 supports CUDA Graph execution, which is the default and recommended
performance path. Set `eager_mode=True` or pass `--eager-mode` only as a
debugging fallback. DSpark V1 supports DP/EP with the `Qwen3DSparkModel` and
`DeepseekV4ForCausalLMDSpark` draft architectures; DFlash supports it with
`DFlashDraftModel`. DP/EP requires microbatch overlap and KV transfer to be
disabled, and does not support PD disaggregation. Prefix caching, draft
KV-cache quantization, guided decoding, and confidence-based dynamic
verification are not supported in this fixed-window version. The
OpenAI-compatible chat endpoint can return output log probabilities when the
server is configured with an appropriate `--logprobs-mode`, such as
`raw_logprobs`. Target-only one-token decoding and fixed-window target
verification can produce different floating-point logits at near ties, so
DSpark V1 does not currently guarantee bitwise-identical greedy output to
target-only execution.

## Guided Decoding with Speculative Decoding

Speculative decoding (MTP) can be combined with [structured output](./structed_output.md) so that the draft tokens proposed by the spec model also respect the grammar constraints (e.g. JSON schema, regex). This significantly improves the acceptance rate compared to running spec decoding without grammar masks.
Expand Down
48 changes: 48 additions & 0 deletions docs/zh_cn/advance/spec_decoding.md
Original file line number Diff line number Diff line change
Expand Up @@ -142,6 +142,54 @@ Qwen/Qwen3.5-35B-A3B \
设置 DFlash block size 后,它会覆盖 `--speculative-num-draft-tokens`,
并将新提出的 token 数设置为 `block_size - 1`。

### DSpark

DSpark 使用 DFlash 风格的并行草稿骨干网络,并通过轻量的从左到右 Markov
校正引入块内依赖。LMDeploy 的首个实现使用固定验证窗口。草稿 token 采用确定性
生成,目标验证则通过通用拒绝采样器同时支持贪心和非贪心采样。该实现支持外部
Speculators 格式草稿模型,以及在同一 checkpoint 中包含 `mtp.*` DSpark
权重的 DeepSeek-V4 模型。对于后一种模型,将 `model` 留空即可复用目标
checkpoint 作为草稿权重来源。

```python
from lmdeploy import PytorchEngineConfig, pipeline
from lmdeploy.messages import SpeculativeConfig

def main():
model = 'deepseek-ai/DeepSeek-V4-Flash-0731'
pipe = pipeline(
model,
backend_config=PytorchEngineConfig(tp=4),
speculative_config=SpeculativeConfig(
method='dspark',
num_speculative_tokens=5,
),
)


if __name__ == '__main__':
main()
```

```shell
lmdeploy serve api_server deepseek-ai/DeepSeek-V4-Flash-0731 \
--backend pytorch \
--tp 4 \
--speculative-algorithm dspark \
--speculative-num-draft-tokens 5
```

DSpark V1 支持 CUDA Graph 执行,该模式是默认且推荐的高性能路径。
仅在调试时使用 `eager_mode=True` 或传入 `--eager-mode` 作为回退方案。
DSpark V1 的 `Qwen3DSparkModel` 和 `DeepseekV4ForCausalLMDSpark` 草稿架构支持
DP/EP;DFlash 的 `DFlashDraftModel` 同样支持。DP/EP 需要关闭 microbatch overlap
和 KV transfer,且不支持 PD 分离。固定窗口版本暂不支持前缀缓存、草稿
KV-cache 量化、引导解码以及基于置信度的动态验证。使用合适的
`--logprobs-mode`(例如 `raw_logprobs`)启动服务后,OpenAI 兼容的 Chat
接口可以返回输出 log probabilities。目标模型的单 token 解码与固定窗口验证在
logits 接近并列时可能产生不同的浮点结果,因此 DSpark V1 目前不保证贪心输出与
仅使用目标模型时逐比特一致。

## 投机解码与结构化输出

投机解码(MTP)可以与[结构化输出](./structed_output.md)结合使用,使草稿模型提出的 token 也遵循语法约束(如 JSON Schema、正则表达式),从而显著提高接受率。
Expand Down
3 changes: 2 additions & 1 deletion lmdeploy/cli/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -835,7 +835,8 @@ def add_spec_group(parser):
spec_group.add_argument('--speculative-algorithm',
type=str,
default=None,
choices=['eagle', 'eagle3', 'deepseek_mtp', 'hy3_mtp', 'qwen3_5_mtp', 'dflash'],
choices=['eagle', 'eagle3', 'deepseek_mtp', 'hy3_mtp', 'qwen3_5_mtp', 'dflash',
'dspark'],
help='The speculative algorithm to use. `None` means speculative decoding is disabled')

spec_group.add_argument('--speculative-draft-model',
Expand Down
19 changes: 19 additions & 0 deletions lmdeploy/pytorch/backends/attention.py
Original file line number Diff line number Diff line change
Expand Up @@ -50,6 +50,13 @@ class V4AttentionMetadata:
cu_seqlens_k: torch.Tensor = None
sum_kv_seqlen: int = None
start_pos: torch.Tensor = None # [bsz] long
causal: bool = True
# Multi-row speculative blocks enter through the decode scheduler. Causal
# target verification may use a packed rectangular-decode executor, while
# non-causal draft blocks retain sparse prefill. Keep this distinction
# after ``is_decoding`` is rewritten so graph metadata can use the fixed
# rectangular token capacity without reading a CUDA scalar.
is_rectangular_decode: bool = False

@classmethod
def from_step_context(cls, attn_metadata, step_ctx, **kwargs) -> 'V4AttentionMetadata':
Expand Down Expand Up @@ -77,6 +84,7 @@ def from_step_context(cls, attn_metadata, step_ctx, **kwargs) -> 'V4AttentionMet
sum_kv_seqlen=step_ctx.sum_kv_seqlen,
cu_seqlens_k=attn_metadata.cu_seqlens_k,
start_pos=(kv_seqlens.to(torch.long) - q_seqlens.to(torch.long)),
causal=kwargs.get('causal', True),
)

def build_indexer_metadata(self):
Expand All @@ -101,6 +109,16 @@ def build_indexer_metadata(self):
class V4AttentionImpl(ABC):
"""DeepSeek-V4 attention implementation contract."""

def build_cache_write_metadata(self, attn_metadata, position_ids: torch.Tensor,
state_ids: torch.Tensor, num_tokens: int):
"""Return KV-only write metadata, or None when unsupported."""
return None

def write_cache(self, kv: torch.Tensor, window_state: torch.Tensor, metadata) -> None:
"""Materialize KV without evaluating attention (when metadata is
supported)."""
raise NotImplementedError

@abstractmethod
def forward(
self,
Expand All @@ -123,6 +141,7 @@ class V4AttentionBuildSpec(BuildSpec[V4AttentionImpl]):
head_dim: int
scale: float
window_size: int
ring_storage_capacity: int
compress_ratio: int


Expand Down
Loading
Loading