Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
39 commits
Select commit Hold shift + click to select a range
870d24e
Create sync.yml
bltcn Dec 6, 2025
a05b815
Update sync.yml
bltcn Dec 6, 2025
eb12bc2
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Dec 8, 2025
2e25797
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Dec 8, 2025
896372e
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Dec 10, 2025
20e97ee
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Dec 10, 2025
820efac
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Dec 10, 2025
45bba7e
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Dec 10, 2025
53b45b1
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Dec 11, 2025
134166a
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Dec 11, 2025
d8dc4c7
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Dec 12, 2025
b15dad6
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Dec 15, 2025
62f2d0d
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Dec 16, 2025
bbac909
Merge branch 'InternLM:main' into main
bltcn Dec 27, 2025
797470d
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Dec 31, 2025
89d6d16
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Dec 31, 2025
67ca368
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Jan 4, 2026
7f0c99a
Merge branch 'InternLM:main' into main
bltcn Jan 9, 2026
ca6e086
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Jan 9, 2026
fb73634
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Jan 12, 2026
481150d
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Jan 14, 2026
2ca429f
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Jan 15, 2026
d79d36f
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Jan 15, 2026
f3d12a1
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Jan 16, 2026
31552fb
Merge branch 'InternLM:main' into main
bltcn Jan 29, 2026
3326421
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Jan 30, 2026
bb8ab9a
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Feb 2, 2026
4e10d3f
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Feb 2, 2026
06f01da
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Feb 3, 2026
4311694
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Feb 3, 2026
7c647ae
Merge branch 'main' of https://github.com/InternLM/lmdeploy
actions-user Feb 4, 2026
b7750b1
Merge branch 'InternLM:main' into main
bltcn Aug 18, 2026
7be6a7d
feat(turbomind): EAGLE3 and Qwen3.5 MTP speculative decoding
lzhangzz Sep 28, 2026
f97f25a
Merge remote-tracking branch 'upstream/main'
bltcn Oct 1, 2026
d6add5b
merge PR #5006: TurboMind speculative decoding (EAGLE3 + Qwen3.5 MTP)
bltcn Oct 3, 2026
ffb4c96
fix(turbomind): normalize MTP speculative method aliases to registere…
bltcn Oct 3, 2026
00feb40
fix(turbomind): GDN state commit kernel missing half_t on pre-SM80 (M…
bltcn Oct 3, 2026
f0a42df
fix(turbomind): keep normalize_spec_method docstring within summary w…
bltcn Oct 3, 2026
ddfab39
fix(tests): use canonical package import for _turbomind + ruff lint f…
bltcn Oct 3, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
41 changes: 41 additions & 0 deletions .github/workflows/sync.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,41 @@
name: Upstream Sync

permissions:
contents: write

on:
schedule:
- cron: "0 * * * *" # 每天 UTC 时间 0点运行一次 (你可以修改这个 cron 表达式)
workflow_dispatch: # 允许你手动点击按钮触发

jobs:
sync_latest_from_upstream:
name: Sync latest commits from upstream repo
runs-on: ubuntu-latest
if: ${{ github.event.repository.fork }} # 只有当这是一个 Fork 仓库时才运行

steps:
# 第一步:检出你的代码
- name: Checkout target repo
uses: actions/checkout@v3

# 第二步:运行同步 Action
- name: Sync upstream changes
id: sync
uses: aormsby/Fork-Sync-With-Upstream-action@v3.4
with:
upstream_sync_repo: InternLM/lmdeploy # 【重要】请修改为源仓库的 用户名/仓库名
upstream_sync_branch: main # 【重要】源仓库的分支名 (main 或 master)
target_sync_branch: main # 你想要同步到的本地分支名
target_repo_token: ${{ secrets.GITHUB_TOKEN }} # 自动生成的 Token,无需修改

# 设置为 true 会在发生冲突时导致 Action 失败,并在测试模式下运行
test_mode: false

# 第三步:如果同步失败(通常是因为有冲突),打印提示
- name: Sync check
if: failure()
run: |
echo "[Error] 由于上游仓库的变更与本地变更冲突,无法自动同步。"
echo "请手动解决冲突。"
exit 1
23 changes: 19 additions & 4 deletions lmdeploy/metrics/stats.py
Original file line number Diff line number Diff line change
Expand Up @@ -285,10 +285,25 @@ def update_from_output(self, outputs: EngineOutput):
"""Update from engine output."""
spec_info = getattr(outputs.req_metrics, 'spec_info', None)
if spec_info:
self.num_drafts += 1
self.num_draft_tokens += spec_info['num_draft_tokens']
self.num_accepted_tokens += spec_info['num_accepted_tokens']
self.num_accepted_tokens_per_pos[:spec_info['num_accepted_tokens']] += 1
if 'num_drafts' in spec_info:
self.num_drafts += \
spec_info['num_drafts']
self.num_draft_tokens += \
spec_info['num_draft_tokens']
self.num_accepted_tokens += \
spec_info['num_accepted_tokens']
self.num_accepted_tokens_per_pos += \
np.asarray(
spec_info[
'num_accepted_tokens_per_pos'])
else:
self.num_drafts += 1
self.num_draft_tokens += \
spec_info['num_draft_tokens']
self.num_accepted_tokens += \
spec_info['num_accepted_tokens']
self.num_accepted_tokens_per_pos[
:spec_info['num_accepted_tokens']] += 1

def update_per_draft(self, num_draft_tokens: int, num_accepted_tokens: int):
"""Update with per draft stats."""
Expand Down
15 changes: 10 additions & 5 deletions lmdeploy/serve/core/async_engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -145,13 +145,12 @@ def __init__(self,
self.session_len = (_get_and_verify_max_len(self.hf_cfg, None)
if backend_config.session_len is None else backend_config.session_len)
backend_config.session_len = self.session_len
if speculative_config is not None and backend == 'turbomind':
logger.warning('speculative decoding is not supported by turbomind ')
# build backend engine
if backend == 'turbomind':
self.engine = self._build_turbomind(model_path=model_path,
backend_config=backend_config,
trust_remote_code=trust_remote_code,
speculative_config=speculative_config,
**kwargs)
elif backend == 'pytorch':
self.engine = self._build_pytorch(model_path=model_path,
Expand All @@ -174,8 +173,9 @@ def __init__(self,
self.backend = backend
self.request_logger = RequestLogger(max_log_len)

self.num_spec_token = 0 if backend == 'turbomind' or speculative_config is None \
else speculative_config.num_speculative_tokens
self.num_spec_token = (0
if speculative_config is None
else speculative_config.num_speculative_tokens)

self.session_mgr = SessionManager()
self.session_mgr.build_request_handle_pool(self.engine, self.backend_config.max_batch_size)
Expand All @@ -201,6 +201,7 @@ def __exit__(self, exc_type, exc_value, traceback):
def _build_turbomind(self,
model_path: str,
backend_config: TurbomindEngineConfig | None = None,
speculative_config: SpeculativeConfig | None = None,
trust_remote_code: bool = False,
**kwargs):
"""Inner build method for turbomind backend."""
Expand All @@ -210,7 +211,11 @@ def _build_turbomind(self,
'TurboMind was requested but its native module is unavailable.'
) from turbomind._import_error
return turbomind.TurboMind.from_pretrained(
model_path, engine_config=backend_config, trust_remote_code=trust_remote_code, **kwargs
model_path,
engine_config=backend_config,
trust_remote_code=trust_remote_code,
speculative_config=speculative_config,
**kwargs
)

def _build_pytorch(self,
Expand Down
5 changes: 5 additions & 0 deletions lmdeploy/turbomind/__init__.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,7 @@
# Copyright (c) OpenMMLab. All rights reserved.

import sys

import torch # noqa: F401

_import_error = None
Expand All @@ -10,6 +12,9 @@
_tm = None
_import_error = error
else:
# Expose the extension under its bare name so submodules that predate the
# lazy-import design can `import _turbomind` without a second load.
sys.modules.setdefault('_turbomind', _tm)
from .turbomind import TurboMind as TurboMind


Expand Down
6 changes: 6 additions & 0 deletions lmdeploy/turbomind/builders/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,11 +7,13 @@
from .attention import AttentionBuilder
from .decoder_layer import DecoderLayerBuilder, DecoderLayerConfig
from .deltanet import DeltaNetBuilder
from .eagle3_weight import Eagle3WeightBuilder, Eagle3WeightConfig
from .ffn import FfnBuilder
from .mla import MLABuilder
from .module_list import ModuleListBuilder, ModuleListConfig
from .moe import MoeBuilder
from .norm import LayerNormBuilder, NormBuilder, make_layer_norm_config, make_norm_config
from .qwen3_5_mtp_weight import Qwen35MtpWeightBuilder, Qwen35MtpWeightConfig
from .text_model import TextModelBuilder
from .vision_model import VisionModelBuilder

Expand All @@ -32,6 +34,8 @@
'DeltaNetBuilder',
'MLABuilder',
'DecoderLayerBuilder',
'Eagle3WeightBuilder',
'Qwen35MtpWeightBuilder',
'ModuleListBuilder',
'NormBuilder',
'LayerNormBuilder',
Expand All @@ -40,6 +44,8 @@
'make_layer_norm_config',
# C++ config re-exports
'DecoderLayerConfig',
'Eagle3WeightConfig',
'Qwen35MtpWeightConfig',
'ModuleListConfig',
# Helper functions
]
13 changes: 13 additions & 0 deletions lmdeploy/turbomind/builders/eagle3_weight.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,13 @@
# Copyright (c) OpenMMLab. All rights reserved.
import _turbomind as _tm

from ._base import Builder

Eagle3WeightConfig = _tm.Eagle3WeightConfig


class Eagle3WeightBuilder(Builder):
"""Builder for the EAGLE3 method weight tree at ModelWeight.spec."""

def add_target_hidden_proj(self, linear):
self._add_linear('target_hidden_proj', linear, split_side=None)
12 changes: 12 additions & 0 deletions lmdeploy/turbomind/builders/qwen3_5_mtp_weight.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,12 @@
# Copyright (c) OpenMMLab. All rights reserved.
import _turbomind as _tm

from ._base import Builder, SplitSide

Qwen35MtpWeightConfig = _tm.Qwen35MtpWeightConfig


class Qwen35MtpWeightBuilder(Builder):

def add_fc(self, linear):
self._add_linear('fc', linear, split_side=SplitSide.OUTPUT)
25 changes: 16 additions & 9 deletions lmdeploy/turbomind/builders/text_model.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,31 +9,38 @@ class TextModelBuilder(Builder):

Constructs a ModelWeight via ``_tm.create_module(ModelWeightConfig)``
on each context (inherited Builder machinery), then attaches it to
externally-owned ``ModelRoot`` sentinel handles as their
``text_model`` child during ``build()``.
externally-owned ``ModelRoot`` sentinel handles as the configured
root child during ``build()``.

Owns ``tok_embeddings`` (Tensor param) and ``output`` (LinearWeight
child) commits on the ModelWeight via ``add_token_embeds`` /
``add_lm_head``.
"""

def __init__(self, config, ctx, *, root_handles,
tp: ParallelGroup, vocab_size):
def __init__(self,
config,
ctx,
*,
root_handles,
tp: ParallelGroup,
vocab_size,
root_child: str = 'text_model'):
super().__init__(config, ctx)
self.tp = tp
self.config.tp_size = tp.size
self._root_handles = root_handles
self._vocab_size = vocab_size
self._root_child = root_child

def build(self) -> BuiltModule:
"""Create ModelWeight via _tm.create_module (via super), then attach
each per-GPU ModelWeight handle to its sentinel root via
add_child_raw."""
"""Build and attach each ModelWeight to the configured root child."""
built = super().build()
for i, (root, text_model) in enumerate(
for i, (root, model_weight) in enumerate(
zip(self._root_handles, built.handles)):
with self._ctx.devices[i]:
root.add_child_raw('text_model', text_model)
root.add_child_raw(
self._root_child,
model_weight)
return built

def add_token_embeds(self, tensor):
Expand Down
77 changes: 53 additions & 24 deletions lmdeploy/turbomind/model_loader.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,12 +15,22 @@ class ModelLoader:
to the C++ runtime.
"""

def __init__(self, model, model_comm, gpu_count, model_path,
data_type, engine_config):
def __init__(self,
model,
model_comm,
gpu_count,
model_path,
data_type,
engine_config,
*,
draft_model=None,
draft_model_path=None):
self.model = model
self.draft_model = draft_model
self.model_comm = model_comm
self.gpu_count = gpu_count
self.model_path = model_path
self.draft_model_path = draft_model_path
self.data_type = data_type
self.engine_config = engine_config
self._bind_runtime()
Expand Down Expand Up @@ -59,33 +69,52 @@ def _bind_runtime(self):
[(e * mlp_tp.size + m) % dense_size
for e, m in zip(ep.ranks, mlp_tp.ranks)])

self.model.bind_runtime(
ctx=ctx,
root_handles=[mc.root(g) for g in range(self.gpu_count)],
attn_tp=attn_tp,
mlp_tp=mlp_tp,
ep=ep,
model_tp=model_tp,
dense_tp=dense_tp,
)
models = (
(self.model,)
if self.draft_model is None
else (self.model, self.draft_model))
for model in models:
model.bind_runtime(
ctx=ctx,
root_handles=[
mc.root(g)
for g in range(self.gpu_count)],
attn_tp=attn_tp,
mlp_tp=mlp_tp,
ep=ep,
model_tp=model_tp,
dense_tp=dense_tp)

def export(self):
ckpt = create_checkpoint(
self.model_path,
mappings=getattr(self.model, '_loader_mappings', []))
@staticmethod
def _export_one(model, model_path):
checkpoint = create_checkpoint(
model_path,
mappings=getattr(
model, '_loader_mappings', []))
try:
self.model.model(Prefix(ckpt))
root = Prefix(checkpoint)
if getattr(model, 'prefix', ''):
root = root + model.prefix
model.model(root)
finally:
ckpt.close()
checkpoint.close()

def export(self):
self._export_one(
self.model, self.model_path)
if self.draft_model is not None:
self._export_one(
self.draft_model,
self.draft_model_path)
torch.cuda.empty_cache()

def export_iter(self):
ckpt = create_checkpoint(
self.model_path,
mappings=getattr(self.model, '_loader_mappings', []))
try:
self.model.model(Prefix(ckpt))
self._export_one(
self.model, self.model_path)
yield -1
if self.draft_model is not None:
self._export_one(
self.draft_model,
self.draft_model_path)
yield -1
finally:
ckpt.close()
torch.cuda.empty_cache()
1 change: 1 addition & 0 deletions lmdeploy/turbomind/models/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,3 +10,4 @@
from .qwen2_vl import Qwen2VLModel # noqa: F401
from .qwen3 import Qwen3TextModel # noqa: F401
from .qwen3_5 import Qwen3_5Model, Qwen3_5TextModel, Qwen3_5VisionModel # noqa: F401
from .qwen3_eagle3 import Qwen3Eagle3TextModel # noqa: F401
Loading
Loading