From 800e9786fe6292c45a1fc1d2c3616f47444b912e Mon Sep 17 00:00:00 2001 From: littlegy <787321726@qq.com> Date: Fri, 28 Aug 2026 10:11:28 +0800 Subject: [PATCH 1/5] improve(autotest): add hard_schema KVV suite and Anthropic inline system tests --- autotest/configs/Qwen/Qwen2.5-7B-Instruct.yml | 2 + autotest/configs/Qwen/Qwen3-8B-FP8.yml | 2 + autotest/configs/Qwen/Qwen3.5-35B-A3B.yml | 8 ++ autotest/configs/README.md | 21 ++++ autotest/configs/deepseek-ai/DeepSeek-V3.yml | 12 ++ .../deepseek-ai/DeepSeek-V4-Flash-0731.yml | 13 +++ autotest/configs/internlm/Intern-S1-Pro.yml | 2 + autotest/configs/internlm/Intern-S1-mini.yml | 44 +++++++ autotest/configs/internlm/Intern-S1.yml | 2 + .../internlm/Intern-S2-Preview-397B-FP8.yml | 2 + .../internlm/Intern-S2-Preview-397B.yml | 2 + .../internlm/Intern-S2-Preview-FP8.yml | 2 + .../configs/internlm/Intern-S2-Preview.yml | 2 + .../internlm/internlm3-8b-instruct.yml | 14 ++- .../meta-llama/Llama-3.1-70B-Instruct.yml | 2 + .../moonshotai/Kimi-K2-Instruct-0905.yml | 12 ++ autotest/configs/zai-org/GLM-4.7-Flash.yml | 2 + autotest/env_paths.yml | 1 + .../test_restful_anthropic_sdk_messages.py | 26 ++++- .../restful/test_restful_anthropic_v1.py | 101 +++++++++++++++-- .../interface/restful/tool_parser/conftest.py | 72 +++++++++++- .../tool_parser/test_tool_call_json_schema.py | 42 +++++++ autotest/utils/anthropic_messages.py | 35 ++++++ autotest/utils/config_utils.py | 10 +- autotest/utils/constant.py | 4 + autotest/utils/run_interface_restful.py | 49 +++++++- autotest/utils/tool_call_json_schema_utils.py | 107 ++++++++++++++++++ 27 files changed, 570 insertions(+), 21 deletions(-) create mode 100644 autotest/interface/restful/tool_parser/test_tool_call_json_schema.py create mode 100644 autotest/utils/tool_call_json_schema_utils.py diff --git a/autotest/configs/Qwen/Qwen2.5-7B-Instruct.yml b/autotest/configs/Qwen/Qwen2.5-7B-Instruct.yml index 615df82a6a..f5a2a74f00 100644 --- a/autotest/configs/Qwen/Qwen2.5-7B-Instruct.yml +++ b/autotest/configs/Qwen/Qwen2.5-7B-Instruct.yml @@ -65,10 +65,12 @@ h: pytorch: suites: - toolcall + - hard_schema extra: tool-call-parser: qwen2d5 turbomind: suites: - toolcall + - hard_schema extra: tool-call-parser: qwen2d5 diff --git a/autotest/configs/Qwen/Qwen3-8B-FP8.yml b/autotest/configs/Qwen/Qwen3-8B-FP8.yml index c3a465695d..232cf33908 100644 --- a/autotest/configs/Qwen/Qwen3-8B-FP8.yml +++ b/autotest/configs/Qwen/Qwen3-8B-FP8.yml @@ -17,6 +17,7 @@ h: pytorch: suites: - toolcall + - hard_schema - reasoning extra: tool-call-parser: qwen3 @@ -24,6 +25,7 @@ h: turbomind: suites: - toolcall + - hard_schema - reasoning extra: tool-call-parser: qwen3 diff --git a/autotest/configs/Qwen/Qwen3.5-35B-A3B.yml b/autotest/configs/Qwen/Qwen3.5-35B-A3B.yml index 85adafd5ec..dcebf64604 100644 --- a/autotest/configs/Qwen/Qwen3.5-35B-A3B.yml +++ b/autotest/configs/Qwen/Qwen3.5-35B-A3B.yml @@ -23,12 +23,14 @@ a100: - logprob - experts - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs enable-return-routed-experts: true tool-call-parser: qwen3coder reasoning-parser: default + enable_thinking: true - suites: - anthropic extra: @@ -39,11 +41,13 @@ a100: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs tool-call-parser: qwen3coder reasoning-parser: default + enable_thinking: true - suites: - anthropic extra: @@ -98,12 +102,14 @@ h: - logprob - experts - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs enable-return-routed-experts: true tool-call-parser: qwen3coder reasoning-parser: default + enable_thinking: true - suites: - anthropic extra: @@ -114,11 +120,13 @@ h: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs tool-call-parser: qwen3coder reasoning-parser: default + enable_thinking: true - suites: - anthropic extra: diff --git a/autotest/configs/README.md b/autotest/configs/README.md index fb006339e0..ca91cbc8ed 100644 --- a/autotest/configs/README.md +++ b/autotest/configs/README.md @@ -189,10 +189,31 @@ is enabled. Suites: - `base` — chat/completions basic cases; generate without logprob/experts + - `logprob` — generate logprob cases + - `experts` — generate routed-experts cases + - `anthropic` — Anthropic Messages HTTP + SDK smoke (`RESTFUL_MODEL_LIST`). Share a profile with `base`/`logprob` when `extra` matches; otherwise its own profile **without** `tool-call-parser` / `reasoning-parser`. + - `toolcall` — `interface/restful/tool_parser/` (requires `tool-call-parser` in yaml `extra`; add `enable-return-routed-experts: true` when toolcall includes `@experts` cases) + +- `hard_schema` — walle MFJS tool-call schema validation (`tool_parser/test_tool_call_json_schema.py`; requires `tool-call-parser`; set `enable_thinking: true` in interface yaml for thinking models; runs full walle case set). Kimi-Vendor-Verifier checkout: `eval_resource/Kimi-Vendor-Verifier` (see `kimi_vendor_verifier_path` in `env_paths.yml`), overridable via `KIMI_VENDOR_VERIFIER_ROOT`; cases override via `WALLE_CASE_DIR`. One representative model per parser (see table below); all `internlm/*` configs with `toolcall` also enable `hard_schema`. + + | `tool-call-parser` | Representative model | + | ------------------ | ------------------------------------------------------- | + | `qwen3coder` | `Qwen/Qwen3.5-35B-A3B` | + | `qwen3` | `Qwen/Qwen3-8B-FP8` | + | `qwen2d5` | `Qwen/Qwen2.5-7B-Instruct` | + | `llama3` | `meta-llama/Llama-3.1-70B-Instruct` | + | `glm47` | `zai-org/GLM-4.7-Flash` | + | `kimi-k2` | `moonshotai/Kimi-K2-Instruct-0905` | + | `deepseek-v32` | `deepseek-ai/DeepSeek-V3` | + | `deepseek-v4` | `deepseek-ai/DeepSeek-V4-Flash-0731` | + | `intern-s1` | `internlm/Intern-S1` (+ all Intern-S1 / Intern-S1-mini) | + | `interns2-preview` | `internlm/Intern-S2-Preview` (+ all Intern-S2 variants) | + | `internlm` | `internlm/internlm3-8b-instruct` | + - `reasoning` — `interface/restful/reasoning_parser/` (requires `reasoning-parser` in yaml `extra`) Notes: diff --git a/autotest/configs/deepseek-ai/DeepSeek-V3.yml b/autotest/configs/deepseek-ai/DeepSeek-V3.yml index ccea5af697..4c2cbf7f13 100644 --- a/autotest/configs/deepseek-ai/DeepSeek-V3.yml +++ b/autotest/configs/deepseek-ai/DeepSeek-V3.yml @@ -10,6 +10,18 @@ h: - nccl test_coverage: - func + interface: + pytorch: + - suites: + - base + - logprob + - toolcall + - hard_schema + extra: + logprobs-mode: raw_logprobs + enable-return-routed-experts: true + tool-call-parser: deepseek-v32 + reasoning-parser: deepseek-v3 - model_type: chat engine_config: tp: 16 diff --git a/autotest/configs/deepseek-ai/DeepSeek-V4-Flash-0731.yml b/autotest/configs/deepseek-ai/DeepSeek-V4-Flash-0731.yml index e599a114b6..bb8ab86cf8 100644 --- a/autotest/configs/deepseek-ai/DeepSeek-V4-Flash-0731.yml +++ b/autotest/configs/deepseek-ai/DeepSeek-V4-Flash-0731.yml @@ -11,6 +11,19 @@ h: test_coverage: - evaluate - func + interface: + pytorch: + - suites: + - base + - logprob + - toolcall + - hard_schema + extra: + logprobs-mode: raw_logprobs + enable-return-routed-experts: true + tool-call-parser: deepseek-v4 + reasoning-parser: deepseek-v4 + enable_thinking: true gen_config: temperature: 1.0 top-p: 1.0 diff --git a/autotest/configs/internlm/Intern-S1-Pro.yml b/autotest/configs/internlm/Intern-S1-Pro.yml index fcb9627787..48a5cbedbd 100644 --- a/autotest/configs/internlm/Intern-S1-Pro.yml +++ b/autotest/configs/internlm/Intern-S1-Pro.yml @@ -21,6 +21,7 @@ h: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs @@ -54,6 +55,7 @@ h: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs diff --git a/autotest/configs/internlm/Intern-S1-mini.yml b/autotest/configs/internlm/Intern-S1-mini.yml index 233df4fd2b..6fdf708f7d 100644 --- a/autotest/configs/internlm/Intern-S1-mini.yml +++ b/autotest/configs/internlm/Intern-S1-mini.yml @@ -20,6 +20,28 @@ a100: - mllm_evaluate - prefix_cache - quantization + interface: + pytorch: + - suites: + - base + - logprob + - toolcall + - hard_schema + extra: + logprobs-mode: raw_logprobs + enable-return-routed-experts: true + tool-call-parser: intern-s1 + reasoning-parser: default + turbomind: + - suites: + - base + - logprob + - toolcall + - hard_schema + extra: + logprobs-mode: raw_logprobs + tool-call-parser: intern-s1 + reasoning-parser: default quantization: turbomind: - kvint4 @@ -45,6 +67,28 @@ h: - func - mllm_evaluate - prefix_cache + interface: + pytorch: + - suites: + - base + - logprob + - toolcall + - hard_schema + extra: + logprobs-mode: raw_logprobs + enable-return-routed-experts: true + tool-call-parser: intern-s1 + reasoning-parser: default + turbomind: + - suites: + - base + - logprob + - toolcall + - hard_schema + extra: + logprobs-mode: raw_logprobs + tool-call-parser: intern-s1 + reasoning-parser: default quantization: turbomind: - kvint8 diff --git a/autotest/configs/internlm/Intern-S1.yml b/autotest/configs/internlm/Intern-S1.yml index 9175a21c71..62ed3570d4 100644 --- a/autotest/configs/internlm/Intern-S1.yml +++ b/autotest/configs/internlm/Intern-S1.yml @@ -52,6 +52,7 @@ h: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs @@ -67,6 +68,7 @@ h: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs diff --git a/autotest/configs/internlm/Intern-S2-Preview-397B-FP8.yml b/autotest/configs/internlm/Intern-S2-Preview-397B-FP8.yml index fd6fa320b3..081605c287 100644 --- a/autotest/configs/internlm/Intern-S2-Preview-397B-FP8.yml +++ b/autotest/configs/internlm/Intern-S2-Preview-397B-FP8.yml @@ -28,6 +28,7 @@ h: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs @@ -71,6 +72,7 @@ h: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs diff --git a/autotest/configs/internlm/Intern-S2-Preview-397B.yml b/autotest/configs/internlm/Intern-S2-Preview-397B.yml index 6637650c95..2558fb52e0 100644 --- a/autotest/configs/internlm/Intern-S2-Preview-397B.yml +++ b/autotest/configs/internlm/Intern-S2-Preview-397B.yml @@ -28,6 +28,7 @@ h: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs @@ -71,6 +72,7 @@ h: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs diff --git a/autotest/configs/internlm/Intern-S2-Preview-FP8.yml b/autotest/configs/internlm/Intern-S2-Preview-FP8.yml index 73d3b45f13..dcac39abc0 100644 --- a/autotest/configs/internlm/Intern-S2-Preview-FP8.yml +++ b/autotest/configs/internlm/Intern-S2-Preview-FP8.yml @@ -25,6 +25,7 @@ h: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs @@ -39,6 +40,7 @@ h: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs diff --git a/autotest/configs/internlm/Intern-S2-Preview.yml b/autotest/configs/internlm/Intern-S2-Preview.yml index f2d3261027..6714babb3a 100644 --- a/autotest/configs/internlm/Intern-S2-Preview.yml +++ b/autotest/configs/internlm/Intern-S2-Preview.yml @@ -64,6 +64,7 @@ h: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs @@ -78,6 +79,7 @@ h: - base - logprob - toolcall + - hard_schema - reasoning extra: logprobs-mode: raw_logprobs diff --git a/autotest/configs/internlm/internlm3-8b-instruct.yml b/autotest/configs/internlm/internlm3-8b-instruct.yml index 95d85abfe4..197eb4c81f 100644 --- a/autotest/configs/internlm/internlm3-8b-instruct.yml +++ b/autotest/configs/internlm/internlm3-8b-instruct.yml @@ -15,16 +15,26 @@ h: - func interface: pytorch: - suites: + - suites: - base - logprob - anthropic extra: logprobs-mode: raw_logprobs + - suites: + - toolcall + - hard_schema + extra: + tool-call-parser: internlm turbomind: - suites: + - suites: - base - logprob - anthropic extra: logprobs-mode: raw_logprobs + - suites: + - toolcall + - hard_schema + extra: + tool-call-parser: internlm diff --git a/autotest/configs/meta-llama/Llama-3.1-70B-Instruct.yml b/autotest/configs/meta-llama/Llama-3.1-70B-Instruct.yml index 5fe3a3d342..f204cf94c4 100644 --- a/autotest/configs/meta-llama/Llama-3.1-70B-Instruct.yml +++ b/autotest/configs/meta-llama/Llama-3.1-70B-Instruct.yml @@ -38,11 +38,13 @@ h: pytorch: suites: - toolcall + - hard_schema extra: tool-call-parser: llama3 turbomind: suites: - toolcall + - hard_schema extra: tool-call-parser: llama3 quantization: diff --git a/autotest/configs/moonshotai/Kimi-K2-Instruct-0905.yml b/autotest/configs/moonshotai/Kimi-K2-Instruct-0905.yml index b6d48b7243..2f102d2183 100644 --- a/autotest/configs/moonshotai/Kimi-K2-Instruct-0905.yml +++ b/autotest/configs/moonshotai/Kimi-K2-Instruct-0905.yml @@ -15,6 +15,18 @@ h: test_coverage: - evaluate - func + interface: + pytorch: + - suites: + - base + - logprob + - toolcall + - hard_schema + extra: + logprobs-mode: raw_logprobs + enable-return-routed-experts: true + tool-call-parser: kimi-k2 + reasoning-parser: default gen_config: temperature: 0.6 deps: diff --git a/autotest/configs/zai-org/GLM-4.7-Flash.yml b/autotest/configs/zai-org/GLM-4.7-Flash.yml index 653ece6284..43d2c857c2 100644 --- a/autotest/configs/zai-org/GLM-4.7-Flash.yml +++ b/autotest/configs/zai-org/GLM-4.7-Flash.yml @@ -32,11 +32,13 @@ h: pytorch: suites: - toolcall + - hard_schema extra: tool-call-parser: glm47 turbomind: suites: - toolcall + - hard_schema extra: tool-call-parser: glm47 - model_type: chat diff --git a/autotest/env_paths.yml b/autotest/env_paths.yml index 8afa48f45c..092cc32e50 100644 --- a/autotest/env_paths.yml +++ b/autotest/env_paths.yml @@ -29,6 +29,7 @@ h: benchmark_path: /mnt/shared-storage-user/llmrazor-share/qa-llm-cicd/cicd-autotest/eval_resource/benchmark_report dataset_path: /mnt/shared-storage-user/llmrazor-share/qa-llm-cicd/cicd-autotest/eval_resource/datasets/ShareGPT_V3_unfiltered_cleaned_split.json prefix_dataset_path: /mnt/shared-storage-user/llmrazor-share/qa-llm-cicd/cicd-autotest/eval_resource/datasets/prefix_cache_test.json + kimi_vendor_verifier_path: /mnt/shared-storage-user/llmrazor-share/qa-llm-cicd/cicd-autotest/eval_resource/Kimi-Vendor-Verifier '3090': env_tag: 3090 device: cuda diff --git a/autotest/interface/restful/test_restful_anthropic_sdk_messages.py b/autotest/interface/restful/test_restful_anthropic_sdk_messages.py index 40cd845091..802ad0128d 100644 --- a/autotest/interface/restful/test_restful_anthropic_sdk_messages.py +++ b/autotest/interface/restful/test_restful_anthropic_sdk_messages.py @@ -7,7 +7,11 @@ pytest.importorskip('anthropic') -from utils.anthropic_messages import get_async_anthropic_client_and_model +from utils.anthropic_messages import ( + ANTHROPIC_SYSTEM_REPLY_OK, + USER_ACKNOWLEDGE, + get_async_anthropic_client_and_model, +) from utils.constant import BACKEND_LIST, RESTFUL_MODEL_LIST @@ -49,6 +53,19 @@ async def _sdk_system_non_stream() -> object: ) +async def _sdk_inline_system_non_stream() -> object: + client, model_name = get_async_anthropic_client_and_model() + return await client.messages.create( + model=model_name, + max_tokens=256, + temperature=0.01, + messages=[ + {'role': 'system', 'content': ANTHROPIC_SYSTEM_REPLY_OK}, + {'role': 'user', 'content': USER_ACKNOWLEDGE}, + ], + ) + + async def _sdk_stream_events_and_final() -> tuple[list, object | None]: client, model_name = get_async_anthropic_client_and_model() stream = await client.messages.create( @@ -96,6 +113,13 @@ def test_sdk_system_message_non_stream(self, backend, model_case): text = _text_from_message(msg) assert len(text) > 0 + def test_sdk_inline_system_message_non_stream(self, backend, model_case): + msg = asyncio.run(_sdk_inline_system_non_stream()) + assert msg.role == 'assistant' + assert msg.stop_reason in ('end_turn', 'max_tokens') + text = _text_from_message(msg).lower() + assert 'ok' in text + def test_sdk_streaming(self, backend, model_case): events, final_msg = asyncio.run(_sdk_stream_events_and_final()) assert len(events) > 0 diff --git a/autotest/interface/restful/test_restful_anthropic_v1.py b/autotest/interface/restful/test_restful_anthropic_v1.py index 9de04911c7..77f3ad4e8b 100644 --- a/autotest/interface/restful/test_restful_anthropic_v1.py +++ b/autotest/interface/restful/test_restful_anthropic_v1.py @@ -8,6 +8,12 @@ import requests from utils.anthropic_messages import ( ANTHROPIC_MESSAGES_HISTORY_THINKING_REPLAY, + ANTHROPIC_SYSTEM_REPLY_ACKNOWLEDGED, + ANTHROPIC_SYSTEM_REPLY_BRIEFLY, + ANTHROPIC_SYSTEM_REPLY_CONFIRMED, + ANTHROPIC_SYSTEM_REPLY_OK, + USER_ACKNOWLEDGE, + USER_ASK_REQUIRED_REPLY, USER_ASK_WEATHER_DALLAS, WEATHER_TOOL_ANTHROPIC, assert_stream_stop_sequence_lifecycle, @@ -15,6 +21,8 @@ assert_success_message_json, assert_warm_yes_answer, build_anthropic_messages_history_tool_result, + build_anthropic_messages_inline_system_history, + build_anthropic_messages_merged_system_prompt, ) from utils.config_utils import get_config from utils.constant import BACKEND_LIST, BASE_URL, RESTFUL_MODEL_LIST @@ -282,7 +290,7 @@ def test_messages_with_system(self, backend, model_case, deployed_model_name: st 'model': deployed_model_name, 'max_tokens': 2048, 'temperature': 0.01, - 'system': 'You reply only with the single word: Acknowledged.', + 'system': ANTHROPIC_SYSTEM_REPLY_ACKNOWLEDGED, 'messages': [{'role': 'user', 'content': 'What is your instruction?'}], }, timeout=120, @@ -333,10 +341,10 @@ def test_messages_system_as_content_blocks(self, backend, model_case, deployed_m 'max_tokens': 256, 'temperature': 0.01, 'system': [ - {'type': 'text', 'text': 'You reply only with the single word: Confirmed.'}, + {'type': 'text', 'text': ANTHROPIC_SYSTEM_REPLY_CONFIRMED}, {'type': 'text', 'text': ' No extra words.'}, ], - 'messages': [{'role': 'user', 'content': 'Acknowledge with your required reply.'}], + 'messages': [{'role': 'user', 'content': USER_ASK_REQUIRED_REPLY}], }, timeout=120, ) @@ -795,7 +803,7 @@ def test_count_tokens_matches_messages_prompt(self, backend, model_case, deploye count_json = { 'model': deployed_model_name, - 'system': 'Reply briefly.', + 'system': ANTHROPIC_SYSTEM_REPLY_BRIEFLY, 'messages': [{'role': 'user', 'content': 'Say hello in one word.'}], } r_count = requests.post( @@ -884,18 +892,95 @@ def test_messages_accepts_system_role_in_messages(self, backend, model_case, dep headers=_anthropic_headers(), json={ 'model': deployed_model_name, - 'max_tokens': 32, + 'max_tokens': 256, 'temperature': 0.01, 'messages': [ - {'role': 'system', 'content': 'Reply with one word: OK.'}, - {'role': 'user', 'content': 'Acknowledge.'}, + {'role': 'system', 'content': ANTHROPIC_SYSTEM_REPLY_OK}, + {'role': 'user', 'content': USER_ACKNOWLEDGE}, ], }, timeout=120, ) assert resp.status_code == 200, resp.text data = assert_success_message_json(resp.json()) - assert len(_assistant_text_from_message_payload(data).strip()) > 0 + text = _assistant_text_from_message_payload(data).lower() + assert 'ok' in text, text[:500] + + def test_messages_inline_system_in_history(self, backend, model_case, deployed_model_name: str): + """Inline ``messages[].role == system`` mid-conversation (merged to + front when the chat template requires system-first).""" + + resp = requests.post( + _MESSAGES_URL, + headers=_anthropic_headers(), + json={ + 'model': deployed_model_name, + 'max_tokens': 256, + 'temperature': 0.01, + 'messages': build_anthropic_messages_inline_system_history(), + }, + timeout=120, + ) + assert resp.status_code == 200, resp.text + data = assert_success_message_json(resp.json()) + text = _assistant_text_from_message_payload(data).lower() + assert 'confirmed' in text, text[:500] + + def test_messages_top_level_and_inline_system(self, backend, model_case, deployed_model_name: str): + """Top-level ``system`` plus inline system messages are accepted + together.""" + + top_level_system, messages = build_anthropic_messages_merged_system_prompt() + resp = requests.post( + _MESSAGES_URL, + headers=_anthropic_headers(), + json={ + 'model': deployed_model_name, + 'max_tokens': 256, + 'temperature': 0.01, + 'system': top_level_system, + 'messages': messages, + }, + timeout=120, + ) + assert resp.status_code == 200, resp.text + data = assert_success_message_json(resp.json()) + text = _assistant_text_from_message_payload(data).lower() + assert 'confirmed' in text or 'acknowledge' in text, text[:500] + + def test_count_tokens_matches_messages_with_inline_system( + self, backend, model_case, deployed_model_name: str): + """``count_tokens`` matches ``/messages`` when system is split across + top-level and ``messages[].role``.""" + + top_level_system, messages = build_anthropic_messages_merged_system_prompt() + count_json = { + 'model': deployed_model_name, + 'system': top_level_system, + 'messages': messages, + } + r_count = requests.post( + _COUNT_TOKENS_URL, + headers=_anthropic_headers(), + json=count_json, + timeout=60, + ) + assert r_count.status_code == 200, r_count.text + counted = _assert_count_tokens_json(r_count.json()) + + r_msg = requests.post( + _MESSAGES_URL, + headers=_anthropic_headers(), + json={ + **count_json, + 'max_tokens': 32, + 'temperature': 0.01, + }, + timeout=120, + ) + assert r_msg.status_code == 200, r_msg.text + data = assert_success_message_json(r_msg.json()) + assert data['usage']['input_tokens'] == counted, (data['usage']['input_tokens'], counted) def test_messages_message_missing_role(self, backend, model_case, deployed_model_name: str): resp = requests.post( diff --git a/autotest/interface/restful/tool_parser/conftest.py b/autotest/interface/restful/tool_parser/conftest.py index 2b5614011f..87f034e4e0 100644 --- a/autotest/interface/restful/tool_parser/conftest.py +++ b/autotest/interface/restful/tool_parser/conftest.py @@ -1,8 +1,17 @@ import os +from functools import cache import pytest from utils.config_utils import model_enables_return_routed_experts from utils.constant import BACKEND_LIST, DEFAULT_MAX_COMPLETION_TOKENS, TOOL_REASONING_MODEL_LIST +from utils.tool_call_json_schema_utils import ( + HARD_SCHEMA_SKIP_NO_PARSER, + HARD_SCHEMA_SKIP_NO_SUITE, + kvv_load_selected_cases, + kvv_validator, + model_has_hard_schema_suite, + model_has_tool_call_parser, +) from utils.tool_reasoning_definitions import ( CONCURRENT_WEATHER_TOOL, DEFAULT_TOOL_CALL_CONCURRENCY, @@ -42,6 +51,13 @@ pytest.mark.parametrize('model_case', TOOL_REASONING_MODEL_LIST), ] +_CLASS_MARKS_HARD_SCHEMA = [ + pytest.mark.order(8), + pytest.mark.hard_schema, + pytest.mark.parametrize('backend', BACKEND_LIST), + pytest.mark.parametrize('model_case', TOOL_REASONING_MODEL_LIST), +] + def _apply_marks(cls): """Apply the shared set of marks to *cls* and return it.""" for m in _CLASS_MARKS: @@ -70,6 +86,13 @@ def _apply_marks_mm(cls): return cls +def _apply_hard_schema_marks(cls): + """Apply hard-schema walle MFJS marks to *cls*.""" + for m in _CLASS_MARKS_HARD_SCHEMA: + cls = m(cls) + return cls + + def is_single_tool_call_only(model_case: str) -> bool: """True when the resolved tool parser cannot delimit multiple tool calls.""" @@ -245,9 +268,48 @@ def _mm_video_tool_call_skip_target(item) -> bool: return 'video' in test_name.lower() +def _hard_schema_case_params(): + try: + selected = kvv_load_selected_cases() + except FileNotFoundError as exc: + return [ + pytest.param( + (None, None, 'missing_cases', 'non-stream'), + marks=pytest.mark.skip(reason=str(exc)), + id='missing_cases', + ), + ] + params = [] + for case, schema, reason in selected: + for mode in kvv_validator().REQUEST_MODES: + params.append( + pytest.param( + (case, schema, reason, mode), + id=f'{case.suite}:{case.line}:{mode}', + ), + ) + return params + + +@cache +def _hard_schema_skip_reason(model_case: str, backend: str) -> str | None: + if not model_has_hard_schema_suite(model_case, backend): + return HARD_SCHEMA_SKIP_NO_SUITE + if not model_has_tool_call_parser(model_case, backend): + return HARD_SCHEMA_SKIP_NO_PARSER + return None + + +def pytest_generate_tests(metafunc): + if 'case_info' not in metafunc.fixturenames: + return + metafunc.parametrize('case_info', _hard_schema_case_params()) + + def pytest_collection_modifyitems(config, items): """Skip parallel-tool tests on single-tool parsers; skip MM tool tests on - text-only models; skip Intern-S1 video MM tool tests.""" + text-only models; skip Intern-S1 video MM tool tests; gate hard_schema by + yaml suite.""" for item in items: callspec = getattr(item, 'callspec', None) if callspec is None: @@ -255,6 +317,14 @@ def pytest_collection_modifyitems(config, items): model_case = callspec.params.get('model_case') if model_case is None: continue + if 'case_info' in getattr(item, 'fixturenames', []): + backend = callspec.params.get('backend') + if backend is None: + continue + skip_reason = _hard_schema_skip_reason(model_case, backend) + if skip_reason is not None: + item.add_marker(pytest.mark.skip(reason=skip_reason)) + continue if _parallel_tool_skip_target(item): if is_single_tool_call_only(model_case): item.add_marker( diff --git a/autotest/interface/restful/tool_parser/test_tool_call_json_schema.py b/autotest/interface/restful/tool_parser/test_tool_call_json_schema.py new file mode 100644 index 0000000000..898178bcf2 --- /dev/null +++ b/autotest/interface/restful/tool_parser/test_tool_call_json_schema.py @@ -0,0 +1,42 @@ +"""Hard-schema tool-call validation (walle MFJS cases via Kimi-Vendor- +Verifier).""" + +from __future__ import annotations + +from typing import Any + +from utils.tool_call_json_schema_utils import ( + HARD_SCHEMA_MAX_TOKENS, + kvv_validator, + resolve_hard_schema_thinking, +) + +from .conftest import _apply_hard_schema_marks, _ToolCallTestBase + + +@_apply_hard_schema_marks +class TestToolCallJsonSchema(_ToolCallTestBase): + def test_tool_call_schema_matches_case_schema( + self, backend, model_case, case_info: tuple[Any, Any, str, str]): + case, schema, selection_reason, mode = case_info + thinking, think_mode = resolve_hard_schema_thinking(model_case, backend) + client, model_name = self._get_client() + mod = kvv_validator() + response = mod.send_tool_schema( + client, + model_name, + schema, + HARD_SCHEMA_MAX_TOKENS, + thinking, + think_mode, + stream=mode == 'stream', + ) + assert response.accepted, ( + f'{case.suite}:{case.line} [{mode}] ({selection_reason}) ' + f'tool schema rejected: {response.message}' + ) + valid, message = mod.validate_arguments(schema, response.arguments) + assert valid, ( + f'{case.suite}:{case.line} [{mode}] ({selection_reason}) ' + f'arguments validation failed: {message}; response: {response.message}' + ) diff --git a/autotest/utils/anthropic_messages.py b/autotest/utils/anthropic_messages.py index dcec9d2c82..668c9a68e5 100644 --- a/autotest/utils/anthropic_messages.py +++ b/autotest/utils/anthropic_messages.py @@ -97,6 +97,17 @@ ASSISTANT_GREETING_AFTER_THINKING = 'Hello — how can I help?' USER_HI = 'Hi.' USER_REPLY_ACK = 'Reply with exactly: ACK' +USER_ACKNOWLEDGE = 'Acknowledge.' +USER_ASK_REQUIRED_REPLY = 'Acknowledge with your required reply.' + +ANTHROPIC_SYSTEM_REPLY_OK = 'Reply with one word: OK.' +ANTHROPIC_SYSTEM_REPLY_CONFIRMED = ( + 'You reply only with the single word: Confirmed.' +) +ANTHROPIC_SYSTEM_REPLY_ACKNOWLEDGED = ( + 'You reply only with the single word: Acknowledged.' +) +ANTHROPIC_SYSTEM_REPLY_BRIEFLY = 'Reply briefly.' ANTHROPIC_SYSTEM_WEATHER = ( 'You are a helpful assistant that can use tools. ' @@ -138,6 +149,30 @@ ] +def build_anthropic_messages_inline_system_history() -> list[dict]: + """Mid-turn ``messages[].role == system`` (Claude Code / beta history).""" + + return [ + {'role': 'user', 'content': 'Say hello in one word.'}, + {'role': 'assistant', 'content': 'Hello.'}, + {'role': 'system', 'content': ANTHROPIC_SYSTEM_REPLY_CONFIRMED}, + {'role': 'user', 'content': USER_ASK_REQUIRED_REPLY}, + ] + + +def build_anthropic_messages_merged_system_prompt() -> tuple[str, list[dict]]: + """Top-level ``system`` plus inline ``messages[].role == system``.""" + + return ( + ANTHROPIC_SYSTEM_REPLY_ACKNOWLEDGED, + [ + {'role': 'user', 'content': 'First question.'}, + {'role': 'system', 'content': ANTHROPIC_SYSTEM_REPLY_CONFIRMED}, + {'role': 'user', 'content': USER_ASK_REQUIRED_REPLY}, + ], + ) + + def build_anthropic_messages_history_tool_result( *, tool_use_id: str = 'toolu_hist_01', diff --git a/autotest/utils/config_utils.py b/autotest/utils/config_utils.py index 6874f380cd..8c409bbe2e 100644 --- a/autotest/utils/config_utils.py +++ b/autotest/utils/config_utils.py @@ -25,9 +25,9 @@ ENGINE_CONFIG_KEY = 'engine_config' TEST_COVERAGE_KEY = 'test_coverage' INTERFACE_KEY = 'interface' -INTERFACE_SUITES = frozenset({'base', 'logprob', 'experts', 'anthropic', 'toolcall', 'reasoning'}) +INTERFACE_SUITES = frozenset({'base', 'logprob', 'experts', 'anthropic', 'toolcall', 'reasoning', 'hard_schema'}) GENERATE_SUITES = frozenset({'base', 'logprob', 'experts'}) -INTERFACE_SUITE_ORDER = ('base', 'logprob', 'experts', 'anthropic', 'toolcall', 'reasoning') +INTERFACE_SUITE_ORDER = ('base', 'logprob', 'experts', 'anthropic', 'toolcall', 'reasoning', 'hard_schema') INTERFACE_BACKENDS_ENV = 'INTERFACE_BACKENDS' @@ -92,8 +92,8 @@ def _entry_has_prefix_cache_accuracy_tuning(entry: dict[str, Any]) -> bool: # ``all``: disable filtering (tests / debug). DEPS_PROFILE_ENV = 'DEPS_PROFILE' EMPTY_DEPS_SELECTOR = '__empty__' -# Autotest-only keys in engine_config.extra (not forwarded to lmdeploy CLI). -CLI_SKIP_EXTRA_KEYS = frozenset() + +CLI_SKIP_EXTRA_KEYS = frozenset({'enable-thinking', 'chat-template-kwargs'}) def get_model_path_from_config(config: dict[str, Any], model_id: str) -> str: @@ -803,6 +803,8 @@ def derive_interface_case_info(profiles: list[str], suites: list[str] | set[str] case_info.append('anthropic_sdk') if 'toolcall' in suite_set: case_info.append('toolcall') + if 'hard_schema' in suite_set: + case_info.append('hard_schema') if 'reasoning' in suite_set: case_info.append('reasoning') return case_info diff --git a/autotest/utils/constant.py b/autotest/utils/constant.py index a5755ed47e..16bdfab6be 100644 --- a/autotest/utils/constant.py +++ b/autotest/utils/constant.py @@ -247,10 +247,14 @@ def _deps_profile_is_legacy() -> bool: 'Qwen/Qwen3.5-397B-A17B-FP8', 'meta-llama/Llama-3.1-70B-Instruct', 'deepseek-ai/DeepSeek-V3', + 'deepseek-ai/DeepSeek-V4-Flash-0731', + 'moonshotai/Kimi-K2-Instruct-0905', 'openai/gpt-oss-20b', 'Qwen/Qwen2.5-7B-Instruct', 'internlm/Intern-S1', + 'internlm/Intern-S1-mini', 'internlm/Intern-S1-Pro', + 'internlm/internlm3-8b-instruct', 'internlm/Intern-S2-Preview', 'internlm/Intern-S2-Preview-FP8', 'internlm/Intern-S2-Preview-397B', diff --git a/autotest/utils/run_interface_restful.py b/autotest/utils/run_interface_restful.py index 48a724b401..3dd13d711e 100644 --- a/autotest/utils/run_interface_restful.py +++ b/autotest/utils/run_interface_restful.py @@ -29,6 +29,12 @@ # Matches historical daily/pr restful ``-n 20``; override via env. INTERFACE_SUITE_WORKERS_ENV = 'INTERFACE_SUITE_WORKERS' _DEFAULT_SUITE_WORKERS = 20 +HARD_SCHEMA_SUITE_WORKERS_ENV = 'HARD_SCHEMA_SUITE_WORKERS' +_DEFAULT_HARD_SCHEMA_SUITE_WORKERS = 4 +HARD_SCHEMA_SUITE_RERUNS = 2 +_HARD_SCHEMA_IGNORE = ( + 'autotest/interface/restful/tool_parser/test_tool_call_json_schema.py' +) def _suite_workers() -> int: @@ -46,6 +52,21 @@ def _suite_workers() -> int: return value +def _hard_schema_workers(default: int) -> int: + raw = os.environ.get(HARD_SCHEMA_SUITE_WORKERS_ENV, '').strip() + if not raw: + return min(default, _DEFAULT_HARD_SCHEMA_SUITE_WORKERS) + try: + value = int(raw) + except ValueError as exc: + raise ValueError( + f'{HARD_SCHEMA_SUITE_WORKERS_ENV} must be an int, got {raw!r}', + ) from exc + if value < 0: + raise ValueError(f'{HARD_SCHEMA_SUITE_WORKERS_ENV} must be >= 0, got {value}') + return value + + def _protocol_model_candidates() -> list[str]: """Model ids used as pytest params in interface protocol suites.""" seen: set[str] = set() @@ -129,6 +150,7 @@ def _pytest_cmd( n_workers: int, log_path: str, reruns: int = 5, + ignore: str | None = None, ) -> int: """Run a nested pytest against one interface suite file.""" cmd = [ @@ -145,6 +167,8 @@ def _pytest_cmd( '-p', 'no:cacheprovider', ] + if ignore: + cmd.extend(['--ignore', ignore]) if m_expr: cmd.extend(['-m', m_expr]) # Concurrent HTTP load against the worker-local api_server (fills GPU). @@ -216,11 +240,12 @@ def _run_interface_suites( k_expr = _pytest_k_expr(model, backend) failures: list[str] = [] - toolcall_marker = f'tool_call and not not_{backend}' + toolcall_marker = f'tool_call and not not_{backend} and not anthropic' if via_proxy: # Exclude return_token_ids / routed_experts / encode(input_ids) cases. - # Exclude Anthropic /v1/messages toolcall cases (proxy returns 404). - toolcall_marker += ' and not experts and not anthropic' + toolcall_marker += ' and not experts' + + hard_schema_marker = f'hard_schema and not not_{backend}' anthropic_marker = f'anthropic and not not_{backend}' @@ -255,6 +280,11 @@ def _run_interface_suites( 'autotest/interface/restful/tool_parser/', toolcall_marker, ), + ( + 'hard_schema', + 'autotest/interface/restful/tool_parser/test_tool_call_json_schema.py', + hard_schema_marker, + ), ( 'reasoning', 'autotest/interface/restful/reasoning_parser/', @@ -265,14 +295,23 @@ def _run_interface_suites( if case_name not in case_info: continue log_path = os.path.join(log_dir, f'log_interface_{case_name}_{case_str}_{port}_{timestamp}.log') + suite_workers = n_workers + suite_reruns = 5 + suite_ignore = None + if case_name == 'toolcall': + suite_ignore = _HARD_SCHEMA_IGNORE + elif case_name == 'hard_schema': + suite_workers = _hard_schema_workers(n_workers) + suite_reruns = HARD_SCHEMA_SUITE_RERUNS rc = _pytest_cmd( rel_path, k_expr=k_expr, m_expr=marker, env=env, - n_workers=n_workers, + n_workers=suite_workers, log_path=log_path, - reruns=5, + reruns=suite_reruns, + ignore=suite_ignore, ) if rc != 0: tail = _read_log_tail(log_path) diff --git a/autotest/utils/tool_call_json_schema_utils.py b/autotest/utils/tool_call_json_schema_utils.py new file mode 100644 index 0000000000..0457cb67c7 --- /dev/null +++ b/autotest/utils/tool_call_json_schema_utils.py @@ -0,0 +1,107 @@ +"""Autotest adapter for Kimi-Vendor-Verifier hard-schema tool-call tests.""" + +from __future__ import annotations + +import os +import sys +from collections.abc import Iterator +from functools import cache, lru_cache +from pathlib import Path +from typing import Any + +from utils.config_utils import ( + build_interface_launch_extra, + get_config, + get_interface_profiles, + iter_model_yaml_entries, +) + +HARD_SCHEMA_SKIP_NO_SUITE = ( + 'model yaml interface profile does not include hard_schema suite' +) +HARD_SCHEMA_SKIP_NO_PARSER = ( + 'hard_schema requires tool-call-parser in interface yaml extra' +) +HARD_SCHEMA_MAX_TOKENS = 8192 + + +def _kvv_root() -> Path: + override = os.environ.get('KIMI_VENDOR_VERIFIER_ROOT', '').strip() + if override: + return Path(override) + path = get_config().get('kimi_vendor_verifier_path', '') + if not path: + raise FileNotFoundError( + 'kimi_vendor_verifier_path missing in env_paths.yml ' + '(or set KIMI_VENDOR_VERIFIER_ROOT)', + ) + return Path(str(path)) + + +@lru_cache(maxsize=1) +def kvv_validator(): + root = str(_kvv_root()) + if root not in sys.path: + sys.path.insert(0, root) + from tests.tool_call_json_schema import validator as mod + + return mod + + +def _case_dir() -> Path: + raw = os.environ.get('WALLE_CASE_DIR', '').strip() + if raw: + return Path(raw) + return _kvv_root() / 'testdata/walle_validator_cases/validator_cases' + + +def kvv_load_selected_cases() -> list[tuple[Any, Any, str]]: + mod = kvv_validator() + cases = mod.load_cases(_case_dir()) + return mod.select_cases( + cases, + selection='all', + requested_cases=set(), + max_cases=None, + ) + + +def _interface_profiles(model_case: str, backend: str) -> Iterator[tuple[dict, dict]]: + for entry in iter_model_yaml_entries(model_case): + for prof in get_interface_profiles(entry, backend): + extra = build_interface_launch_extra( + entry, + backend, + suites=prof.get('suites') or [], + interface_extra=prof.get('extra'), + ) + yield prof, extra + + +def model_has_hard_schema_suite(model_case: str, backend: str) -> bool: + return _model_has_hard_schema_suite(model_case, backend) + + +def model_has_tool_call_parser(model_case: str, backend: str) -> bool: + return _model_has_tool_call_parser(model_case, backend) + + +@cache +def _model_has_hard_schema_suite(model_case: str, backend: str) -> bool: + return any('hard_schema' in (prof.get('suites') or []) for prof, _ in _interface_profiles(model_case, backend)) + + +@cache +def _model_has_tool_call_parser(model_case: str, backend: str) -> bool: + return any(extra.get('tool-call-parser') for _, extra in _interface_profiles(model_case, backend)) + + +def resolve_hard_schema_thinking(model_case: str, backend: str) -> tuple[bool, str]: + for _, extra in _interface_profiles(model_case, backend): + enabled = extra.get('enable_thinking', extra.get('enable-thinking')) + if enabled is True: + return True, 'opensource' + chat_kwargs = extra.get('chat-template-kwargs') or {} + if chat_kwargs.get('enable_thinking') is True: + return True, 'opensource' + return False, 'opensource' From e51feb41072fb179b10c73e3afc3a600b139dea2 Mon Sep 17 00:00:00 2001 From: littlegy <787321726@qq.com> Date: Fri, 28 Aug 2026 10:21:47 +0800 Subject: [PATCH 2/5] consolidate enable_thinking into gen_config only --- autotest/configs/Qwen/Qwen3.5-35B-A3B.yml | 11 +++++++---- autotest/configs/README.md | 2 +- .../deepseek-ai/DeepSeek-V4-Flash-0731.yml | 3 ++- autotest/utils/tool_call_json_schema_utils.py | 15 ++++++++------- 4 files changed, 18 insertions(+), 13 deletions(-) diff --git a/autotest/configs/Qwen/Qwen3.5-35B-A3B.yml b/autotest/configs/Qwen/Qwen3.5-35B-A3B.yml index dcebf64604..f5dea067a0 100644 --- a/autotest/configs/Qwen/Qwen3.5-35B-A3B.yml +++ b/autotest/configs/Qwen/Qwen3.5-35B-A3B.yml @@ -30,7 +30,6 @@ a100: enable-return-routed-experts: true tool-call-parser: qwen3coder reasoning-parser: default - enable_thinking: true - suites: - anthropic extra: @@ -47,11 +46,17 @@ a100: logprobs-mode: raw_logprobs tool-call-parser: qwen3coder reasoning-parser: default - enable_thinking: true - suites: - anthropic extra: logprobs-mode: raw_logprobs + gen_config: + temperature: 1.0 + top-k: 20 + repetition-penalty: 1.0 + top-p: 0.95 + chat-template-kwargs: + enable_thinking: true quantization: turbomind: - kvint8 @@ -109,7 +114,6 @@ h: enable-return-routed-experts: true tool-call-parser: qwen3coder reasoning-parser: default - enable_thinking: true - suites: - anthropic extra: @@ -126,7 +130,6 @@ h: logprobs-mode: raw_logprobs tool-call-parser: qwen3coder reasoning-parser: default - enable_thinking: true - suites: - anthropic extra: diff --git a/autotest/configs/README.md b/autotest/configs/README.md index ca91cbc8ed..742548ddb9 100644 --- a/autotest/configs/README.md +++ b/autotest/configs/README.md @@ -198,7 +198,7 @@ Suites: - `toolcall` — `interface/restful/tool_parser/` (requires `tool-call-parser` in yaml `extra`; add `enable-return-routed-experts: true` when toolcall includes `@experts` cases) -- `hard_schema` — walle MFJS tool-call schema validation (`tool_parser/test_tool_call_json_schema.py`; requires `tool-call-parser`; set `enable_thinking: true` in interface yaml for thinking models; runs full walle case set). Kimi-Vendor-Verifier checkout: `eval_resource/Kimi-Vendor-Verifier` (see `kimi_vendor_verifier_path` in `env_paths.yml`), overridable via `KIMI_VENDOR_VERIFIER_ROOT`; cases override via `WALLE_CASE_DIR`. One representative model per parser (see table below); all `internlm/*` configs with `toolcall` also enable `hard_schema`. +- `hard_schema` — walle MFJS tool-call schema validation (`tool_parser/test_tool_call_json_schema.py`; requires `tool-call-parser`; for thinking models set `gen_config.chat-template-kwargs.enable_thinking: true` (not `interface.extra`); runs full walle case set). Kimi-Vendor-Verifier checkout: `eval_resource/Kimi-Vendor-Verifier` (see `kimi_vendor_verifier_path` in `env_paths.yml`), overridable via `KIMI_VENDOR_VERIFIER_ROOT`; cases override via `WALLE_CASE_DIR`. One representative model per parser (see table below); all `internlm/*` configs with `toolcall` also enable `hard_schema`. | `tool-call-parser` | Representative model | | ------------------ | ------------------------------------------------------- | diff --git a/autotest/configs/deepseek-ai/DeepSeek-V4-Flash-0731.yml b/autotest/configs/deepseek-ai/DeepSeek-V4-Flash-0731.yml index bb8ab86cf8..0c98a7fb45 100644 --- a/autotest/configs/deepseek-ai/DeepSeek-V4-Flash-0731.yml +++ b/autotest/configs/deepseek-ai/DeepSeek-V4-Flash-0731.yml @@ -23,7 +23,8 @@ h: enable-return-routed-experts: true tool-call-parser: deepseek-v4 reasoning-parser: deepseek-v4 - enable_thinking: true gen_config: temperature: 1.0 top-p: 1.0 + chat-template-kwargs: + enable_thinking: true diff --git a/autotest/utils/tool_call_json_schema_utils.py b/autotest/utils/tool_call_json_schema_utils.py index 0457cb67c7..c5085c2fd8 100644 --- a/autotest/utils/tool_call_json_schema_utils.py +++ b/autotest/utils/tool_call_json_schema_utils.py @@ -96,12 +96,13 @@ def _model_has_tool_call_parser(model_case: str, backend: str) -> bool: return any(extra.get('tool-call-parser') for _, extra in _interface_profiles(model_case, backend)) -def resolve_hard_schema_thinking(model_case: str, backend: str) -> tuple[bool, str]: - for _, extra in _interface_profiles(model_case, backend): - enabled = extra.get('enable_thinking', extra.get('enable-thinking')) - if enabled is True: - return True, 'opensource' - chat_kwargs = extra.get('chat-template-kwargs') or {} - if chat_kwargs.get('enable_thinking') is True: +def _gen_config_enable_thinking(entry: dict[str, Any]) -> bool: + chat_kwargs = (entry.get('gen_config') or {}).get('chat-template-kwargs') or {} + return chat_kwargs.get('enable_thinking') is True + + +def resolve_hard_schema_thinking(model_case: str, _backend: str) -> tuple[bool, str]: + for entry in iter_model_yaml_entries(model_case): + if _gen_config_enable_thinking(entry): return True, 'opensource' return False, 'opensource' From 7b159ff7f6fabb20fff4c92a4632c32f61a40c2a Mon Sep 17 00:00:00 2001 From: littlegy <787321726@qq.com> Date: Thu, 17 Sep 2026 16:55:37 +0800 Subject: [PATCH 3/5] fix pr test and add response test --- autotest/configs/internlm/Intern-S1-mini.yml | 2 - .../reasoning_parser/test_reasoning_api.py | 11 +- .../test_restful_anthropic_sdk_messages.py | 4 +- .../restful/test_restful_anthropic_v1.py | 26 +-- .../test_restful_chat_completions_v1.py | 18 +- .../restful/test_restful_responses_v1.py | 211 ++++++++++++++++++ .../test_tool_call_anthropic_sdk.py | 12 +- .../tool_parser/test_tool_call_basic.py | 31 ++- .../tool_parser/test_tool_call_multimodal.py | 20 +- .../tool_parser/test_tool_call_responses.py | 165 ++++++++++++++ autotest/utils/anthropic_messages.py | 1 + autotest/utils/config_utils.py | 3 +- autotest/utils/restful_return_check.py | 66 +++++- autotest/utils/run_interface_restful.py | 33 ++- autotest/utils/tool_reasoning_definitions.py | 10 +- 15 files changed, 534 insertions(+), 79 deletions(-) create mode 100644 autotest/interface/restful/test_restful_responses_v1.py create mode 100644 autotest/interface/restful/tool_parser/test_tool_call_responses.py diff --git a/autotest/configs/internlm/Intern-S1-mini.yml b/autotest/configs/internlm/Intern-S1-mini.yml index 6fdf708f7d..1f4069880f 100644 --- a/autotest/configs/internlm/Intern-S1-mini.yml +++ b/autotest/configs/internlm/Intern-S1-mini.yml @@ -29,7 +29,6 @@ a100: - hard_schema extra: logprobs-mode: raw_logprobs - enable-return-routed-experts: true tool-call-parser: intern-s1 reasoning-parser: default turbomind: @@ -76,7 +75,6 @@ h: - hard_schema extra: logprobs-mode: raw_logprobs - enable-return-routed-experts: true tool-call-parser: intern-s1 reasoning-parser: default turbomind: diff --git a/autotest/interface/restful/reasoning_parser/test_reasoning_api.py b/autotest/interface/restful/reasoning_parser/test_reasoning_api.py index 6a6b50f6cd..48af2d7f45 100644 --- a/autotest/interface/restful/reasoning_parser/test_reasoning_api.py +++ b/autotest/interface/restful/reasoning_parser/test_reasoning_api.py @@ -4,6 +4,7 @@ from utils.restful_return_check import assert_usage from utils.tool_reasoning_definitions import ( CALCULATOR_TOOL, + MESSAGES_HELLO, SEARCH_TOOL, THINK_END_TOKEN, THINK_START_TOKEN, @@ -155,17 +156,17 @@ def test_tool_choice_auto(self, backend, model_case, stream): assert len(content.strip()) > 0 def test_tool_choice_required(self, backend, model_case, stream): - """tool_choice='required': must produce tool call.""" + """tool_choice='required' + Hello/search: must tool-call, not greet.""" try: - r = self._call_api(stream, MESSAGES_REASONING_WEATHER_TOOL, tools=[WEATHER_TOOL], tool_choice='required') + r = self._call_api(stream, MESSAGES_HELLO, tools=[SEARCH_TOOL], tool_choice='required') except BadRequestError as e: pytest.skip(f'tool_choice="required" rejected by server (HTTP 400): {e}') - assert len(r['tool_calls']) >= 1 + assert len(r['tool_calls']) >= 1, r.get('content') assert r['finish_reason'] == 'tool_calls' tc = r['tool_calls'][0] - assert tc['name'] == 'get_current_weather' + assert tc['name'] == 'web_search' parsed = assert_tool_call_dict_fields(tc) - assert 'city' in parsed + assert parsed.get('query'), parsed def test_tool_choice_none(self, backend, model_case, stream): """tool_choice='none': no tool calls, text answer instead.""" diff --git a/autotest/interface/restful/test_restful_anthropic_sdk_messages.py b/autotest/interface/restful/test_restful_anthropic_sdk_messages.py index eb5ac41b74..8f17b3a205 100644 --- a/autotest/interface/restful/test_restful_anthropic_sdk_messages.py +++ b/autotest/interface/restful/test_restful_anthropic_sdk_messages.py @@ -57,8 +57,8 @@ async def _sdk_inline_system_non_stream() -> object: client, model_name = get_async_anthropic_client_and_model() return await client.messages.create( model=model_name, - max_tokens=256, - temperature=0.01, + max_tokens=1024, + extra_body={'temperature': 0.01}, messages=[ {'role': 'system', 'content': ANTHROPIC_SYSTEM_REPLY_OK}, {'role': 'user', 'content': USER_ACKNOWLEDGE}, diff --git a/autotest/interface/restful/test_restful_anthropic_v1.py b/autotest/interface/restful/test_restful_anthropic_v1.py index 2ab64bf803..9e98d11279 100644 --- a/autotest/interface/restful/test_restful_anthropic_v1.py +++ b/autotest/interface/restful/test_restful_anthropic_v1.py @@ -9,7 +9,6 @@ from utils.anthropic_messages import ( ANTHROPIC_MESSAGES_HISTORY_THINKING_REPLAY, ANTHROPIC_SYSTEM_REPLY_ACKNOWLEDGED, - ANTHROPIC_SYSTEM_REPLY_BRIEFLY, ANTHROPIC_SYSTEM_REPLY_CONFIRMED, ANTHROPIC_SYSTEM_REPLY_OK, USER_ACKNOWLEDGE, @@ -792,13 +791,8 @@ def test_count_tokens_matches_messages_prompt(self, backend, model_case): for the same prompt.""" count_json = { -<<<<<<< HEAD - 'model': deployed_model_name, - 'system': ANTHROPIC_SYSTEM_REPLY_BRIEFLY, -======= 'model': deployed_model_name(), 'system': 'Reply briefly.', ->>>>>>> origin/sync-upstream-main 'messages': [{'role': 'user', 'content': 'Say hello in one word.'}], } r_count = requests.post( @@ -886,13 +880,8 @@ def test_messages_accepts_system_role_in_messages(self, backend, model_case): _MESSAGES_URL, headers=_anthropic_headers(), json={ -<<<<<<< HEAD - 'model': deployed_model_name, - 'max_tokens': 256, -======= 'model': deployed_model_name(), - 'max_tokens': 32, ->>>>>>> origin/sync-upstream-main + 'max_tokens': 1024, 'temperature': 0.01, 'messages': [ {'role': 'system', 'content': ANTHROPIC_SYSTEM_REPLY_OK}, @@ -906,7 +895,7 @@ def test_messages_accepts_system_role_in_messages(self, backend, model_case): text = _assistant_text_from_message_payload(data).lower() assert 'ok' in text, text[:500] - def test_messages_inline_system_in_history(self, backend, model_case, deployed_model_name: str): + def test_messages_inline_system_in_history(self, backend, model_case): """Inline ``messages[].role == system`` mid-conversation (merged to front when the chat template requires system-first).""" @@ -914,7 +903,7 @@ def test_messages_inline_system_in_history(self, backend, model_case, deployed_m _MESSAGES_URL, headers=_anthropic_headers(), json={ - 'model': deployed_model_name, + 'model': deployed_model_name(), 'max_tokens': 256, 'temperature': 0.01, 'messages': build_anthropic_messages_inline_system_history(), @@ -926,7 +915,7 @@ def test_messages_inline_system_in_history(self, backend, model_case, deployed_m text = _assistant_text_from_message_payload(data).lower() assert 'confirmed' in text, text[:500] - def test_messages_top_level_and_inline_system(self, backend, model_case, deployed_model_name: str): + def test_messages_top_level_and_inline_system(self, backend, model_case): """Top-level ``system`` plus inline system messages are accepted together.""" @@ -935,7 +924,7 @@ def test_messages_top_level_and_inline_system(self, backend, model_case, deploye _MESSAGES_URL, headers=_anthropic_headers(), json={ - 'model': deployed_model_name, + 'model': deployed_model_name(), 'max_tokens': 256, 'temperature': 0.01, 'system': top_level_system, @@ -948,14 +937,13 @@ def test_messages_top_level_and_inline_system(self, backend, model_case, deploye text = _assistant_text_from_message_payload(data).lower() assert 'confirmed' in text or 'acknowledge' in text, text[:500] - def test_count_tokens_matches_messages_with_inline_system( - self, backend, model_case, deployed_model_name: str): + def test_count_tokens_matches_messages_with_inline_system(self, backend, model_case): """``count_tokens`` matches ``/messages`` when system is split across top-level and ``messages[].role``.""" top_level_system, messages = build_anthropic_messages_merged_system_prompt() count_json = { - 'model': deployed_model_name, + 'model': deployed_model_name(), 'system': top_level_system, 'messages': messages, } diff --git a/autotest/interface/restful/test_restful_chat_completions_v1.py b/autotest/interface/restful/test_restful_chat_completions_v1.py index 4c726a0cdf..050c792fe3 100644 --- a/autotest/interface/restful/test_restful_chat_completions_v1.py +++ b/autotest/interface/restful/test_restful_chat_completions_v1.py @@ -413,7 +413,11 @@ def test_minimum_repetition_penalty(self, backend, model_case, openai_client_and outputs = client.chat.completions.create( model=model_name, messages=[{'role': 'user', 'content': 'Shanghai is'}], - extra_body={'repetition_penalty': 0.0000001, 'min_new_tokens': 100}, + extra_body={ + 'repetition_penalty': 0.0000001, + 'min_new_tokens': 100, + 'chat_template_kwargs': {'enable_thinking': False}, + }, temperature=0.01, max_tokens=200, ) @@ -427,7 +431,11 @@ def test_minimum_repetition_penalty_streaming(self, backend, model_case, openai_ outputs = client.chat.completions.create( model=model_name, messages=[{'role': 'user', 'content': 'Hi, pls intro yourself'}], - extra_body={'repetition_penalty': 0.0000001, 'min_new_tokens': 100}, + extra_body={ + 'repetition_penalty': 0.0000001, + 'min_new_tokens': 100, + 'chat_template_kwargs': {'enable_thinking': False}, + }, temperature=0.01, max_tokens=200, stream=True, @@ -473,7 +481,7 @@ def test_ignore_eos(self, backend, model_case, openai_client_and_model): outputs = client.chat.completions.create( model=model_name, messages=[{'role': 'user', 'content': 'Hi, what is your name?'}], - extra_body={'ignore_eos': True}, + extra_body={'ignore_eos': True, 'chat_template_kwargs': {'enable_thinking': False}}, max_tokens=100, temperature=0.01, ) @@ -489,7 +497,7 @@ def test_ignore_eos_streaming(self, backend, model_case, openai_client_and_model outputs = client.chat.completions.create( model=model_name, messages=[{'role': 'user', 'content': 'Hi, what is your name?'}], - extra_body={'ignore_eos': True}, + extra_body={'ignore_eos': True, 'chat_template_kwargs': {'enable_thinking': False}}, max_tokens=100, temperature=0.01, stream=True, @@ -512,7 +520,7 @@ def test_max_tokens_default_cap_no_overshoot_followup(self, backend, model_case, outputs = client.chat.completions.create( model=model_name, messages=[{'role': 'user', 'content': prompt}], - extra_body={'ignore_eos': True}, + extra_body={'ignore_eos': True, 'chat_template_kwargs': {'enable_thinking': False}}, max_tokens=max_tokens, temperature=0.01, ) diff --git a/autotest/interface/restful/test_restful_responses_v1.py b/autotest/interface/restful/test_restful_responses_v1.py new file mode 100644 index 0000000000..6e06c908a2 --- /dev/null +++ b/autotest/interface/restful/test_restful_responses_v1.py @@ -0,0 +1,211 @@ +"""ETE for ``POST /v1/responses`` (lmdeploy Text V1).""" + +import pytest +import requests +from utils.constant import BACKEND_LIST, BASE_URL, CAPPED_MAX_COMPLETION_TOKENS, RESTFUL_MODEL_LIST +from utils.restful_return_check import ( + assert_responses_batch_return, + assert_responses_error, + get_client_and_model, +) + +_RESPONSES_URL = f'{BASE_URL}/v1/responses' +_MATH_PROMPT = 'What is 13 * 24?' +_QED_INSTRUCTION = 'End the final answer with QED.' +_MATH_PROMPT_QED = f'{_MATH_PROMPT} {_QED_INSTRUCTION}' +_MAX_OUTPUT_TOKENS = CAPPED_MAX_COMPLETION_TOKENS * 4 + + +@pytest.fixture(scope='class') +def openai_client_and_model(): + return get_client_and_model(BASE_URL) + + +def _responses_json(model_name: str, **extra) -> dict: + body = { + 'model': model_name, + 'input': _MATH_PROMPT, + 'max_output_tokens': CAPPED_MAX_COMPLETION_TOKENS, + 'temperature': 0.01, + } + body.update(extra) + return body + + +@pytest.mark.order(8) +@pytest.mark.flaky(reruns=2) +@pytest.mark.parametrize('backend', BACKEND_LIST) +@pytest.mark.parametrize('model_case', RESTFUL_MODEL_LIST) +class TestRestfulOpenAIResponses: + + @pytest.mark.pr_test + def test_return_info(self, backend, model_case, openai_client_and_model): + client, model_name = openai_client_and_model + response = client.responses.create( + model=model_name, + input=_MATH_PROMPT, + max_output_tokens=_MAX_OUTPUT_TOKENS, + temperature=0.01, + ) + output = response.model_dump() + assert_responses_batch_return(output, model_name) + assert '312' in output['output_text'] + + @pytest.mark.pr_test + def test_return_info_streaming(self, backend, model_case, openai_client_and_model): + client, model_name = openai_client_and_model + stream = client.responses.create( + model=model_name, + input=_MATH_PROMPT, + max_output_tokens=_MAX_OUTPUT_TOKENS, + temperature=0.01, + stream=True, + ) + events = list(stream) + assert events[0].type == 'response.created' + assert events[-1].type == 'response.completed' + streaming_text = ''.join(event.delta for event in events if event.type == 'response.output_text.delta') + completed = events[-1].response + assert streaming_text == completed.output_text + assert_responses_batch_return(completed.model_dump(), model_name) + assert '312' in completed.output_text + + def test_instructions(self, backend, model_case, openai_client_and_model): + client, model_name = openai_client_and_model + response = client.responses.create( + model=model_name, + instructions=_QED_INSTRUCTION, + input=_MATH_PROMPT_QED, + max_output_tokens=_MAX_OUTPUT_TOKENS, + temperature=0.01, + ) + output = response.model_dump() + assert_responses_batch_return(output, model_name) + assert output['instructions'] == _QED_INSTRUCTION + assert '312' in output['output_text'] + assert 'QED' in output['output_text'] + + def test_chat_history_input(self, backend, model_case, openai_client_and_model): + client, model_name = openai_client_and_model + response = client.responses.create( + model=model_name, + input=[ + {'role': 'system', 'content': _QED_INSTRUCTION}, + {'role': 'user', 'content': 'What is 5 * 3?'}, + {'role': 'assistant', 'content': '15. QED.'}, + {'role': 'user', 'content': f'Multiply the result by 2. {_QED_INSTRUCTION}'}, + ], + max_output_tokens=_MAX_OUTPUT_TOKENS, + temperature=0.01, + ) + output = response.model_dump() + assert_responses_batch_return(output, model_name) + assert '30' in output['output_text'] + assert 'QED' in output['output_text'] + + def test_input_text_content_parts(self, backend, model_case, openai_client_and_model): + client, model_name = openai_client_and_model + response = client.responses.create( + model=model_name, + input=[{ + 'type': 'message', + 'role': 'user', + 'content': [{'type': 'input_text', 'text': _MATH_PROMPT}], + }], + max_output_tokens=_MAX_OUTPUT_TOKENS, + temperature=0.01, + ) + output = response.model_dump() + assert_responses_batch_return(output, model_name) + assert '312' in output['output_text'] + + def test_max_output_tokens_incomplete(self, backend, model_case, openai_client_and_model): + client, model_name = openai_client_and_model + response = client.responses.create( + model=model_name, + input='Write a long story about a city.', + max_output_tokens=5, + temperature=0.01, + ) + assert_responses_batch_return(response.model_dump(), model_name, status='incomplete') + + def test_unknown_model(self, backend, model_case): + resp = requests.post( + _RESPONSES_URL, + json=_responses_json('definitely-not-a-deployed-model-name'), + timeout=30, + ) + assert_responses_error(resp, status_code=404, error_type='not_found_error', param='model') + + @pytest.mark.parametrize( + 'input_value', + [ + pytest.param(None, id='missing'), + pytest.param([], id='empty-list'), + ], + ) + def test_invalid_input(self, backend, model_case, openai_client_and_model, input_value): + _, model_name = openai_client_and_model + body = _responses_json(model_name) + if input_value is None: + body.pop('input') + else: + body['input'] = input_value + resp = requests.post(_RESPONSES_URL, json=body, timeout=30) + assert_responses_error(resp, status_code=400, error_type='invalid_request_error', param='input') + + @pytest.mark.parametrize( + 'field_name,value', + [ + pytest.param('background', True, id='background'), + pytest.param('previous_response_id', 'resp_not_supported', id='previous_response_id'), + pytest.param('conversation', 'conv_not_supported', id='conversation'), + ], + ) + def test_text_v1_rejects_unsupported_stateful_fields( + self, backend, model_case, openai_client_and_model, field_name, value): + _, model_name = openai_client_and_model + resp = requests.post( + _RESPONSES_URL, + json=_responses_json(model_name, **{field_name: value}), + timeout=30, + ) + assert_responses_error( + resp, + status_code=400, + error_type='invalid_request_error', + param=field_name, + message_substr='not supported by Responses Text V1', + ) + + def test_invalid_temperature(self, backend, model_case, openai_client_and_model): + _, model_name = openai_client_and_model + resp = requests.post( + _RESPONSES_URL, + json=_responses_json(model_name, temperature=-0.1), + timeout=30, + ) + assert_responses_error(resp, status_code=400, error_type='invalid_request_error', param='temperature') + + def test_unsupported_input_item_type(self, backend, model_case, openai_client_and_model): + _, model_name = openai_client_and_model + resp = requests.post( + _RESPONSES_URL, + json=_responses_json(model_name, input=[{'type': 'computer_call', 'call_id': 'call_x'}]), + timeout=30, + ) + assert_responses_error(resp, status_code=400, error_type='invalid_request_error', param='input') + + def test_json_request_required(self, backend, model_case): + resp = requests.post( + _RESPONSES_URL, + headers={'Content-Type': 'text/plain'}, + data='{}', + timeout=30, + ) + assert_responses_error( + resp, + status_code=400, + error_type='invalid_request_error', + message_substr='application/json', + ) diff --git a/autotest/interface/restful/tool_parser/test_tool_call_anthropic_sdk.py b/autotest/interface/restful/tool_parser/test_tool_call_anthropic_sdk.py index d29d2132c4..39a0df01a9 100644 --- a/autotest/interface/restful/tool_parser/test_tool_call_anthropic_sdk.py +++ b/autotest/interface/restful/tool_parser/test_tool_call_anthropic_sdk.py @@ -19,6 +19,7 @@ SEARCH_TOOL_ANTHROPIC, USER_ASK_WEATHER_DALLAS, USER_ASK_WEATHER_DALLAS_VLM, + USER_HELLO, WEATHER_TOOL_ANTHROPIC, WEATHER_TOOL_SINGLE_LOCATION_ANTHROPIC, assert_parallel_weather_tool_inputs, @@ -730,9 +731,8 @@ async def _async_tool_choice_any(log_file: str): model=model_name, max_tokens=_TOOL_MAX_TOKENS, extra_body={'temperature': 0}, - system=ANTHROPIC_SYSTEM_WEATHER, - messages=ANTHROPIC_MESSAGES_ASKING_FOR_WEATHER, - tools=[WEATHER_TOOL_ANTHROPIC, SEARCH_TOOL_ANTHROPIC], + messages=[{'role': 'user', 'content': USER_HELLO}], + tools=[SEARCH_TOOL_ANTHROPIC], tool_choice={'type': 'any'}, ) try: @@ -1136,10 +1136,10 @@ def test_tool_non_stream_tool_choice_force_named(self, backend, model_case): def test_tool_non_stream_tool_choice_any(self, backend, model_case): msg = asyncio.run(_async_tool_choice_any(self._log_file)) - assert msg.stop_reason == 'tool_use' + assert msg.stop_reason == 'tool_use', msg.content tool_blocks = _sdk_tool_use_blocks(msg) - names = {b.name for b in tool_blocks} - assert WEATHER_TOOL_ANTHROPIC['name'] in names, names + assert tool_blocks[0].name == SEARCH_TOOL_ANTHROPIC['name'] + assert tool_blocks[0].input.get('query'), tool_blocks[0].input def test_tool_non_stream_weather_with_user_image_url(self, backend, model_case): model_name = deployed_model_name() diff --git a/autotest/interface/restful/tool_parser/test_tool_call_basic.py b/autotest/interface/restful/tool_parser/test_tool_call_basic.py index 35c129e233..546af94964 100644 --- a/autotest/interface/restful/tool_parser/test_tool_call_basic.py +++ b/autotest/interface/restful/tool_parser/test_tool_call_basic.py @@ -2,6 +2,7 @@ from openai import BadRequestError from utils.constant import CAPPED_MAX_COMPLETION_TOKENS from utils.tool_reasoning_definitions import ( + MESSAGES_HELLO, SEARCH_TOOL, WEATHER_TOOL, assert_arguments_parseable, @@ -215,48 +216,46 @@ def test_tool_choice_none(self, backend, model_case): # -- required ------------------------------------------------------------ def test_tool_choice_required(self, backend, model_case): - """tool_choice='required': model MUST return at least one tool call. - - Only skip when the *server* rejects the request (HTTP error). - """ + """tool_choice='required' + Hello/search: must tool-call, not greet.""" client, model_name = self._get_client() try: response = client.chat.completions.create( model=model_name, - messages=MESSAGES_ASKING_FOR_WEATHER, + messages=MESSAGES_HELLO, temperature=0, max_completion_tokens=CAPPED_MAX_COMPLETION_TOKENS, - tools=[WEATHER_TOOL, SEARCH_TOOL], + tools=[SEARCH_TOOL], tool_choice='required', logprobs=False, ) except BadRequestError as e: pytest.skip(f'tool_choice="required" rejected by server (HTTP 400): {e}') - # Validation MUST fail loudly — never skip on assertion errors choice = response.choices[0] assert choice.message.role == 'assistant' + assert choice.finish_reason == 'tool_calls', ( + f'tool_choice="required" finish_reason={choice.finish_reason!r} ' + f'content={choice.message.content!r}') assert choice.message.tool_calls is not None, ('tool_choice="required" but got no tool_calls') assert len(choice.message.tool_calls) >= 1 for tc in choice.message.tool_calls: assert_tool_call_fields(tc) - assert_arguments_parseable(tc.function.arguments) + assert tc.function.name == 'web_search' + parsed = assert_arguments_parseable(tc.function.arguments) + assert parsed.get('query'), parsed def test_tool_choice_required_streaming(self, backend, model_case): - """tool_choice='required' + streaming: must return tool call chunks. - - Only skip if the server rejects the request, not on parse errors. - """ + """tool_choice='required' + Hello/search streaming: must tool-call.""" client, model_name = self._get_client() try: stream = client.chat.completions.create( model=model_name, - messages=MESSAGES_ASKING_FOR_WEATHER, + messages=MESSAGES_HELLO, temperature=0, max_completion_tokens=CAPPED_MAX_COMPLETION_TOKENS, - tools=[WEATHER_TOOL, SEARCH_TOOL], + tools=[SEARCH_TOOL], tool_choice='required', logprobs=False, stream=True, @@ -266,8 +265,8 @@ def test_tool_choice_required_streaming(self, backend, model_case): r = collect_stream_tool_call(stream) validate_stream_tool_call_result( r, - expected_function_name=None, - **self._parser_validation_kwargs([WEATHER_TOOL, SEARCH_TOOL]), + expected_function_name='web_search', + **self._parser_validation_kwargs([SEARCH_TOOL]), ) # -- specific function --------------------------------------------------- diff --git a/autotest/interface/restful/tool_parser/test_tool_call_multimodal.py b/autotest/interface/restful/tool_parser/test_tool_call_multimodal.py index b7afb5310a..a7d7c78531 100644 --- a/autotest/interface/restful/tool_parser/test_tool_call_multimodal.py +++ b/autotest/interface/restful/tool_parser/test_tool_call_multimodal.py @@ -41,7 +41,6 @@ mm_create_extra_body_for_media_type, mm_dallas_weather_messages, mm_file_to_data_url, - mm_miami_weather_messages, mm_weather_messages_for_media_type, ) @@ -384,7 +383,7 @@ def test_tool_choice_specific_function(self, backend, model_case): def test_tool_choice_required(self, backend, model_case): pose_url = self._require_mm_image(MM_TEST_IMAGE_POSE) - messages = mm_miami_weather_messages(pose_url) + messages = [build_multimodal_user_message('Hello', pose_url)] client, model_name = self._get_client() try: @@ -393,7 +392,7 @@ def test_tool_choice_required(self, backend, model_case): messages=messages, temperature=0, max_completion_tokens=CAPPED_MAX_COMPLETION_TOKENS, - tools=[WEATHER_TOOL, SEARCH_TOOL], + tools=[SEARCH_TOOL], tool_choice='required', logprobs=False, ) @@ -402,15 +401,20 @@ def test_tool_choice_required(self, backend, model_case): choice = response.choices[0] assert choice.message.role == 'assistant' + assert choice.finish_reason == 'tool_calls', ( + f'tool_choice="required" finish_reason={choice.finish_reason!r} ' + f'content={choice.message.content!r}') assert choice.message.tool_calls is not None assert len(choice.message.tool_calls) >= 1 for tc in choice.message.tool_calls: assert_tool_call_fields(tc) - assert_arguments_parseable(tc.function.arguments) + assert tc.function.name == 'web_search' + parsed = assert_arguments_parseable(tc.function.arguments) + assert parsed.get('query'), parsed def test_tool_choice_required_streaming(self, backend, model_case): pose_url = self._require_mm_image(MM_TEST_IMAGE_POSE) - messages = mm_miami_weather_messages(pose_url) + messages = [build_multimodal_user_message('Hello', pose_url)] client, model_name = self._get_client() try: @@ -419,7 +423,7 @@ def test_tool_choice_required_streaming(self, backend, model_case): messages=messages, temperature=0, max_completion_tokens=CAPPED_MAX_COMPLETION_TOKENS, - tools=[WEATHER_TOOL, SEARCH_TOOL], + tools=[SEARCH_TOOL], tool_choice='required', logprobs=False, stream=True, @@ -429,8 +433,8 @@ def test_tool_choice_required_streaming(self, backend, model_case): r = collect_stream_tool_call(stream) validate_stream_tool_call_result( r, - expected_function_name=None, - **self._parser_validation_kwargs([WEATHER_TOOL, SEARCH_TOOL]), + expected_function_name='web_search', + **self._parser_validation_kwargs([SEARCH_TOOL]), ) diff --git a/autotest/interface/restful/tool_parser/test_tool_call_responses.py b/autotest/interface/restful/tool_parser/test_tool_call_responses.py new file mode 100644 index 0000000000..db10f5c33c --- /dev/null +++ b/autotest/interface/restful/tool_parser/test_tool_call_responses.py @@ -0,0 +1,165 @@ +"""Responses Text V1 tool calls (``POST /v1/responses``).""" + +import pytest +import requests +from utils.constant import BASE_URL, CAPPED_MAX_COMPLETION_TOKENS +from utils.restful_return_check import assert_responses_batch_return, assert_responses_error +from utils.tool_reasoning_definitions import MESSAGES_HELLO, SEARCH_TOOL, WEATHER_TOOL, assert_arguments_parseable + +from .conftest import MESSAGES_ASKING_FOR_WEATHER, _apply_marks, _ToolCallTestBase + +_RESPONSES_URL = f'{BASE_URL}/v1/responses' +_WEATHER_NAME = WEATHER_TOOL['function']['name'] +_TOOLS = [ + {'type': 'function', **WEATHER_TOOL['function']}, + {'type': 'function', **SEARCH_TOOL['function']}, +] +_SEARCH_NAME = SEARCH_TOOL['function']['name'] +_SEARCH_TOOL = {'type': 'function', **SEARCH_TOOL['function']} +_CREATE = dict( + input=MESSAGES_ASKING_FOR_WEATHER, + tools=_TOOLS, + temperature=0, + max_output_tokens=CAPPED_MAX_COMPLETION_TOKENS, +) + + +def _assert_weather_function_call(fc): + assert fc['name'] == _WEATHER_NAME + parsed = assert_arguments_parseable(fc['arguments']) + assert 'dallas' in parsed['city'].lower() + assert 'tx' in parsed['state'].lower() + return parsed + + +def _assert_search_function_call(fc): + assert fc['name'] == _SEARCH_NAME + parsed = assert_arguments_parseable(fc['arguments']) + assert parsed.get('query'), parsed + return parsed + + +@pytest.mark.responses +@_apply_marks +class TestToolCallResponses(_ToolCallTestBase): + + def test_non_streaming(self, backend, model_case): + client, model_name = self._get_client() + fcs = assert_responses_batch_return( + client.responses.create( + model=model_name, + tool_choice='required', + input=MESSAGES_HELLO, + tools=[_SEARCH_TOOL], + temperature=0, + max_output_tokens=CAPPED_MAX_COMPLETION_TOKENS, + ).model_dump(), + model_name, + ) + assert fcs + _assert_search_function_call(fcs[0]) + + def test_streaming(self, backend, model_case): + client, model_name = self._get_client() + events = list( + client.responses.create( + model=model_name, + tool_choice='required', + stream=True, + input=MESSAGES_HELLO, + tools=[_SEARCH_TOOL], + temperature=0, + max_output_tokens=CAPPED_MAX_COMPLETION_TOKENS, + )) + assert events[0].type == 'response.created' + assert events[-1].type == 'response.completed' + args_delta = ''.join( + event.delta for event in events if event.type == 'response.function_call_arguments.delta') + fcs = assert_responses_batch_return(events[-1].response.model_dump(), model_name) + assert fcs + assert args_delta == fcs[0]['arguments'] + _assert_search_function_call(fcs[0]) + + def test_named_tool_choice(self, backend, model_case): + client, model_name = self._get_client() + fcs = assert_responses_batch_return( + client.responses.create( + model=model_name, + tool_choice={'type': 'function', 'name': _WEATHER_NAME}, + **_CREATE, + ).model_dump(), + model_name, + ) + assert fcs + _assert_weather_function_call(fcs[0]) + + def test_tool_choice_none(self, backend, model_case): + client, model_name = self._get_client() + output = client.responses.create(model=model_name, tool_choice='none', **_CREATE).model_dump() + fcs = assert_responses_batch_return(output, model_name) + assert not fcs + assert output['output_text'].strip() + + def test_function_call_output_followup(self, backend, model_case): + client, model_name = self._get_client() + fc = assert_responses_batch_return( + client.responses.create( + model=model_name, + tool_choice={'type': 'function', 'name': _WEATHER_NAME}, + **_CREATE, + ).model_dump(), + model_name, + )[0] + _assert_weather_function_call(fc) + output = client.responses.create( + model=model_name, + input=[ + *MESSAGES_ASKING_FOR_WEATHER, + { + 'type': 'function_call', + 'call_id': fc['call_id'], + 'name': fc['name'], + 'arguments': fc['arguments'], + }, + { + 'type': 'function_call_output', + 'call_id': fc['call_id'], + 'output': 'Sunny, 98F in Dallas, TX.', + }, + ], + tools=_TOOLS, + tool_choice='none', + temperature=0, + max_output_tokens=CAPPED_MAX_COMPLETION_TOKENS, + ).model_dump() + fcs = assert_responses_batch_return(output, model_name) + assert not fcs + text = output['output_text'] + assert '98' in text or 'Dallas' in text + + def test_unknown_tool_choice_name(self, backend, model_case): + _, model_name = self._get_client() + resp = requests.post( + _RESPONSES_URL, + json={ + 'model': model_name, + 'input': 'Hi', + 'tools': [_TOOLS[0]], + 'tool_choice': {'type': 'function', 'name': 'missing'}, + }, + timeout=30, + ) + assert_responses_error(resp, status_code=400, error_type='invalid_request_error', param='tool_choice') + + def test_tools_missing_name(self, backend, model_case): + _, model_name = self._get_client() + resp = requests.post( + _RESPONSES_URL, + json={ + 'model': model_name, + 'input': 'Hi', + 'tools': [{'type': 'function'}], + }, + timeout=30, + ) + assert_responses_error(resp, status_code=400, error_type='invalid_request_error', param='tools') diff --git a/autotest/utils/anthropic_messages.py b/autotest/utils/anthropic_messages.py index 8eedf5494d..a4cd3bc30d 100644 --- a/autotest/utils/anthropic_messages.py +++ b/autotest/utils/anthropic_messages.py @@ -89,6 +89,7 @@ # -- Anthropic Messages API prompts (top-level ``system`` + ``messages``) ----- +USER_HELLO = 'Hello' USER_ASK_WEATHER_DALLAS = "What's the weather like in Dallas, TX?" USER_ASK_WEATHER_DALLAS_VLM = f'{USER_ASK_WEATHER_DALLAS} Use tools; ignore any attached image.' USER_FOLLOWUP_WARM_YES = 'In one short phrase, was it warm? Answer yes or no.' diff --git a/autotest/utils/config_utils.py b/autotest/utils/config_utils.py index 8c409bbe2e..fc2c1a0f81 100644 --- a/autotest/utils/config_utils.py +++ b/autotest/utils/config_utils.py @@ -786,7 +786,7 @@ def derive_interface_case_info(profiles: list[str], suites: list[str] | set[str] """Derive REST case groups from model profiles + interface suites. Directory-based suites (toolcall / reasoning) and anthropic protocol files are selected by path in CI; generate - logprob/experts stay in one file and are filtered by pytest marks. + logprob/experts stay in one file and are filtered by pytest marks. Chat models also run ``responses_v1``. """ suite_set = set(suites) case_info: list[str] = [] @@ -797,6 +797,7 @@ def derive_interface_case_info(profiles: list[str], suites: list[str] | set[str] else: if suite_set & GENERATE_SUITES: case_info.append('chat_completions_v1') + case_info.append('responses_v1') case_info.append('generate') if 'anthropic' in suite_set: case_info.append('anthropic_v1') diff --git a/autotest/utils/restful_return_check.py b/autotest/utils/restful_return_check.py index 85cff0ef42..c9abe25d78 100644 --- a/autotest/utils/restful_return_check.py +++ b/autotest/utils/restful_return_check.py @@ -123,6 +123,69 @@ def assert_usage(usage): assert usage.get('completion_tokens') + usage.get('prompt_tokens') == usage.get('total_tokens') +def assert_responses_usage(usage): + assert usage['input_tokens'] > 0 + assert usage['output_tokens'] > 0 + assert usage['total_tokens'] == usage['input_tokens'] + usage['output_tokens'] + assert usage['output_tokens_details']['reasoning_tokens'] >= 0 + + +def assert_responses_batch_return(output, model_name, *, status: str = 'completed'): + """Assert ``POST /v1/responses`` JSON (lmdeploy Text V1). + + Returns ``function_call`` output items (possibly empty). + """ + assert output['id'] + assert output['object'] == 'response' + assert output['model'] == model_name + assert output['status'] == status + assert_responses_usage(output['usage']) + output_items = output['output'] + assert output_items + messages = [] + function_calls = [] + for item in output_items: + assert item['type'] in ('message', 'function_call'), item + if item['type'] == 'message': + assert item['role'] == 'assistant' + content = item['content'] + assert content[0]['type'] == 'output_text' + assert isinstance(content[0]['text'], str) + messages.append(item) + else: + assert item['name'] + assert item['call_id'] + assert isinstance(item['arguments'], str) + function_calls.append(item) + if messages: + assert output['output_text'] == messages[0]['content'][0]['text'] + if status == 'completed': + assert output.get('incomplete_details') is None + if not function_calls: + assert len(output['output_text']) > 0 + else: + assert status == 'incomplete' + assert messages[0]['status'] == 'incomplete' + assert output['incomplete_details']['reason'] == 'max_output_tokens' + return function_calls + + +def assert_responses_error(response: requests.Response, *, status_code: int, error_type: str, + param: str | None = None, message_substr: str | None = None) -> dict: + """Assert nested ``error`` from Responses ``check_request``.""" + assert response.status_code == status_code, response.text[:500] + body = response.json() + err = body['error'] + assert err['message'] + assert err['code'] == status_code + assert err['type'] == error_type + if param is not None: + assert err['param'] == param + if message_substr is not None: + assert message_substr in err['message'] + return body + + def assert_logprobs(logprobs, logprobs_num): assert_logprob_element(logprobs) assert len(logprobs.get('top_logprobs')) >= 0 @@ -239,9 +302,8 @@ def resolve_effective_session_len(config: dict[str, Any], model_id: str) -> int: session_len = int(extra['session-len']) break if session_len is None: - from transformers import AutoConfig - from lmdeploy.utils import _get_and_verify_max_len + from transformers import AutoConfig hf_cfg = AutoConfig.from_pretrained(model_path, trust_remote_code=True) session_len = _get_and_verify_max_len(hf_cfg, None) diff --git a/autotest/utils/run_interface_restful.py b/autotest/utils/run_interface_restful.py index 3dd13d711e..9e2b8ced81 100644 --- a/autotest/utils/run_interface_restful.py +++ b/autotest/utils/run_interface_restful.py @@ -196,7 +196,9 @@ def _run_interface_suites( ``generate`` suite (``/generate`` always emits ``output_ids``), and toolcall tests marked ``experts`` (return_token_ids / routed_experts / encode+input_ids paths). Also skips Anthropic suites and toolcall - tests marked ``anthropic`` (proxy does not expose ``/v1/messages``). + tests marked ``anthropic`` (proxy does not expose ``/v1/messages``), + plus ``responses_v1`` and toolcall tests marked ``responses`` + (proxy does not expose ``/v1/responses``). """ model = run_config['model'] backend = run_config['backend'] @@ -216,13 +218,18 @@ def _run_interface_suites( ) if via_proxy: - # Proxy does not expose Anthropic /v1/messages (404 Not Found). - dropped = [c for c in ('anthropic_v1', 'anthropic_sdk') if c in case_info] + # Proxy does not expose Anthropic /v1/messages or Responses /v1/responses. + dropped = [ + c for c in ('anthropic_v1', 'anthropic_sdk', 'responses_v1') if c in case_info + ] if dropped: - case_info = [c for c in case_info if c not in ('anthropic_v1', 'anthropic_sdk')] + case_info = [ + c for c in case_info + if c not in ('anthropic_v1', 'anthropic_sdk', 'responses_v1') + ] print( f'proxy: skipping {", ".join(dropped)} ' - '(Anthropic Messages API not available via proxy)', + '(Anthropic / Responses API not available via proxy)', flush=True, ) @@ -242,8 +249,9 @@ def _run_interface_suites( toolcall_marker = f'tool_call and not not_{backend} and not anthropic' if via_proxy: - # Exclude return_token_ids / routed_experts / encode(input_ids) cases. - toolcall_marker += ' and not experts' + # Exclude return_token_ids / routed_experts / encode(input_ids) cases, + # and Responses toolcall (proxy has no /v1/responses). + toolcall_marker += ' and not experts and not responses' hard_schema_marker = f'hard_schema and not not_{backend}' @@ -255,6 +263,11 @@ def _run_interface_suites( 'autotest/interface/restful/test_restful_chat_completions_v1.py', f'not not_{backend}', ), + ( + 'responses_v1', + 'autotest/interface/restful/test_restful_responses_v1.py', + f'not not_{backend}', + ), ( 'completions_v1', 'autotest/interface/restful/test_restful_completions_v1.py', @@ -427,9 +440,11 @@ def run_interface_restful_proxy_distributed_test(config, run_config, manager) -> """Run interface suites against LMDeploy proxy (dp/ep multi-node). Skips ``generate``, Anthropic suites (``/v1/messages`` 404 via proxy), - and toolcall ``experts`` / ``anthropic``-marked cases: proxy cannot + Responses suites (``/v1/responses`` 404 via proxy), and toolcall + ``experts`` / ``anthropic`` / ``responses``-marked cases: proxy cannot safely carry large ``/generate`` / encode / return_token_ids / - routed_experts payloads, and does not forward Anthropic Messages. + routed_experts payloads, and does not forward Anthropic Messages or + Responses. One ``ApiServerPerTest`` restart per launch profile. All ranks join each phase; workers sync via a shared done-flag, and also exit when master diff --git a/autotest/utils/tool_reasoning_definitions.py b/autotest/utils/tool_reasoning_definitions.py index dab642eb3b..ced29508a5 100644 --- a/autotest/utils/tool_reasoning_definitions.py +++ b/autotest/utils/tool_reasoning_definitions.py @@ -7,10 +7,6 @@ import aiohttp import requests -from utils.config_utils import get_model_path_from_config -from utils.constant import CAPPED_MAX_COMPLETION_TOKENS, DEFAULT_PORT -from utils.restful_return_check import get_client_and_model - from lmdeploy.serve.openai.protocol import ( ChatCompletionRequest, ChatCompletionResponseStreamChoice, @@ -23,6 +19,9 @@ _normalize_request_messages, _parse_tool_call_arguments_dict, ) +from utils.config_utils import get_model_path_from_config +from utils.constant import CAPPED_MAX_COMPLETION_TOKENS, DEFAULT_PORT +from utils.restful_return_check import get_client_and_model BASE_HTTP_URL = f"http://{os.getenv('MASTER_ADDR', 'localhost')}" PORT = os.getenv('LMDEPLOY_PORT', str(DEFAULT_PORT)) @@ -105,6 +104,9 @@ def get_reasoning_open_close_tags(reasoning_parser_name: str = 'default') -> tup }, } +# Greeting + search + required: model must still tool-call (not reply in text). +MESSAGES_HELLO = [{'role': 'user', 'content': 'Hello'}] + CALCULATOR_TOOL = { 'type': 'function', 'function': { From 26d2dfb840e28e4eec468d10752da837a1f3ad98 Mon Sep 17 00:00:00 2001 From: littlegy <787321726@qq.com> Date: Thu, 17 Sep 2026 17:16:10 +0800 Subject: [PATCH 4/5] fix lint --- autotest/configs/deepseek-ai/DeepSeek-V4-Flash-0731.yml | 3 +++ autotest/utils/restful_return_check.py | 3 ++- autotest/utils/tool_reasoning_definitions.py | 7 ++++--- 3 files changed, 9 insertions(+), 4 deletions(-) diff --git a/autotest/configs/deepseek-ai/DeepSeek-V4-Flash-0731.yml b/autotest/configs/deepseek-ai/DeepSeek-V4-Flash-0731.yml index 0c98a7fb45..1d90554860 100644 --- a/autotest/configs/deepseek-ai/DeepSeek-V4-Flash-0731.yml +++ b/autotest/configs/deepseek-ai/DeepSeek-V4-Flash-0731.yml @@ -4,6 +4,9 @@ h: - model_type: chat engine_config: tp: 4 + extra: + max-batch-size: 256 + cache-max-entry-count: 0.9 backends: - name: pytorch communicators: diff --git a/autotest/utils/restful_return_check.py b/autotest/utils/restful_return_check.py index c9abe25d78..6aa29ef7c9 100644 --- a/autotest/utils/restful_return_check.py +++ b/autotest/utils/restful_return_check.py @@ -302,9 +302,10 @@ def resolve_effective_session_len(config: dict[str, Any], model_id: str) -> int: session_len = int(extra['session-len']) break if session_len is None: - from lmdeploy.utils import _get_and_verify_max_len from transformers import AutoConfig + from lmdeploy.utils import _get_and_verify_max_len + hf_cfg = AutoConfig.from_pretrained(model_path, trust_remote_code=True) session_len = _get_and_verify_max_len(hf_cfg, None) return session_len diff --git a/autotest/utils/tool_reasoning_definitions.py b/autotest/utils/tool_reasoning_definitions.py index ced29508a5..ca0ffdaac0 100644 --- a/autotest/utils/tool_reasoning_definitions.py +++ b/autotest/utils/tool_reasoning_definitions.py @@ -7,6 +7,10 @@ import aiohttp import requests +from utils.config_utils import get_model_path_from_config +from utils.constant import CAPPED_MAX_COMPLETION_TOKENS, DEFAULT_PORT +from utils.restful_return_check import get_client_and_model + from lmdeploy.serve.openai.protocol import ( ChatCompletionRequest, ChatCompletionResponseStreamChoice, @@ -19,9 +23,6 @@ _normalize_request_messages, _parse_tool_call_arguments_dict, ) -from utils.config_utils import get_model_path_from_config -from utils.constant import CAPPED_MAX_COMPLETION_TOKENS, DEFAULT_PORT -from utils.restful_return_check import get_client_and_model BASE_HTTP_URL = f"http://{os.getenv('MASTER_ADDR', 'localhost')}" PORT = os.getenv('LMDEPLOY_PORT', str(DEFAULT_PORT)) From 558d072c68cc7864ad1cc962aabecdda209eaefb Mon Sep 17 00:00:00 2001 From: littlegy <787321726@qq.com> Date: Fri, 18 Sep 2026 19:17:19 +0800 Subject: [PATCH 5/5] update cached_tokens test --- autotest/configs/Qwen/Qwen3.5-27B.yml | 22 ++++++ autotest/configs/Qwen/Qwen3.5-35B-A3B-FP8.yml | 22 ++++++ autotest/configs/Qwen/Qwen3.5-35B-A3B.yml | 27 +++++++- .../configs/Qwen/Qwen3.5-397B-A17B-FP8.yml | 26 +------ autotest/configs/Qwen/Qwen3.5-397B-A17B.yml | 5 +- autotest/configs/Qwen/Qwen3.5-4B.yml | 15 +++++ autotest/configs/README.md | 8 +++ .../internlm/Intern-S2-Preview-397B-FP8.yml | 26 +++++++ .../internlm/Intern-S2-Preview-397B.yml | 26 +++++++ .../internlm/Intern-S2-Preview-FP8.yml | 1 + .../configs/internlm/Intern-S2-Preview.yml | 1 + .../restful/test_restful_generate.py | 43 ++++++------ autotest/tools/pipeline/mllm_case.py | 2 +- .../test_restful_chat_hf_pytorch_mllm.py | 21 ++++++ .../test_restful_chat_hf_turbomind_mllm.py | 20 ++++++ autotest/utils/config_utils.py | 6 +- autotest/utils/restful_return_check.py | 67 +++++++++++++++++++ autotest/utils/run_restful_chat.py | 26 +++++-- 18 files changed, 303 insertions(+), 61 deletions(-) diff --git a/autotest/configs/Qwen/Qwen3.5-27B.yml b/autotest/configs/Qwen/Qwen3.5-27B.yml index 70a5f73e5d..62f82cae6f 100644 --- a/autotest/configs/Qwen/Qwen3.5-27B.yml +++ b/autotest/configs/Qwen/Qwen3.5-27B.yml @@ -82,3 +82,25 @@ h: top-p: 0.95 chat-template-kwargs: enable_thinking: true +- model_type: + - chat + - vl + engine_config: + tp: 2 + extra: + prefix-cache-state-budget: 256 + prefix-cache-decode-state-interval: 1024 + max-prefill-token-num: 64 + backends: + - name: pytorch + communicators: + - nccl + test_coverage: + - prefix_cache + gen_config: + temperature: 1.0 + top-k: 20 + repetition-penalty: 1.0 + top-p: 0.95 + chat-template-kwargs: + enable_thinking: true diff --git a/autotest/configs/Qwen/Qwen3.5-35B-A3B-FP8.yml b/autotest/configs/Qwen/Qwen3.5-35B-A3B-FP8.yml index 03960874c6..4aae71b367 100644 --- a/autotest/configs/Qwen/Qwen3.5-35B-A3B-FP8.yml +++ b/autotest/configs/Qwen/Qwen3.5-35B-A3B-FP8.yml @@ -59,6 +59,28 @@ h: top-p: 0.95 chat-template-kwargs: enable_thinking: true +- model_type: + - chat + - vl + engine_config: + tp: 1 + extra: + prefix-cache-state-budget: 256 + prefix-cache-decode-state-interval: 1024 + max-prefill-token-num: 64 + backends: + - name: pytorch + communicators: + - nccl + test_coverage: + - prefix_cache + gen_config: + temperature: 1.0 + top-k: 20 + repetition-penalty: 1.0 + top-p: 0.95 + chat-template-kwargs: + enable_thinking: true - model_type: - chat - vl diff --git a/autotest/configs/Qwen/Qwen3.5-35B-A3B.yml b/autotest/configs/Qwen/Qwen3.5-35B-A3B.yml index f5dea067a0..79eb292abf 100644 --- a/autotest/configs/Qwen/Qwen3.5-35B-A3B.yml +++ b/autotest/configs/Qwen/Qwen3.5-35B-A3B.yml @@ -146,6 +146,31 @@ h: top-p: 0.95 chat-template-kwargs: enable_thinking: true +- model_type: + - chat + - vl + engine_config: + tp: 2 + extra: + prefix-cache-state-budget: 256 + prefix-cache-decode-state-interval: 1024 + max-prefill-token-num: 64 + backends: + - name: pytorch + communicators: + - nccl + test_coverage: + - prefix_cache + quantization: + pytorch: + - fp8 + gen_config: + temperature: 1.0 + top-k: 20 + repetition-penalty: 1.0 + top-p: 0.95 + chat-template-kwargs: + enable_thinking: true - model_type: - chat - vl @@ -194,8 +219,6 @@ ascend: - func - longtext_benchmark - mllm_evaluate - - prefix_cache - - mllm_evaluate gen_config: temperature: 1.0 top-k: 20 diff --git a/autotest/configs/Qwen/Qwen3.5-397B-A17B-FP8.yml b/autotest/configs/Qwen/Qwen3.5-397B-A17B-FP8.yml index e5182deac8..0625cb7b45 100644 --- a/autotest/configs/Qwen/Qwen3.5-397B-A17B-FP8.yml +++ b/autotest/configs/Qwen/Qwen3.5-397B-A17B-FP8.yml @@ -66,31 +66,6 @@ h: top-p: 0.95 chat-template-kwargs: enable_thinking: true -- model_type: - - chat - - vl - engine_config: - tp: 2 - dp: 4 - ep: 8 - extra: - max-batch-size: 256 - cache-max-entry-count: 0.7 - prefix-cache-decode-state-interval: 1024 - prefix-cache-state-budget: 256 - backends: - - name: turbomind - communicators: - - nccl - test_coverage: - - prefix_cache - gen_config: - temperature: 1.0 - top-k: 20 - repetition-penalty: 1.0 - top-p: 0.95 - chat-template-kwargs: - enable_thinking: true - model_type: - chat - vl @@ -103,6 +78,7 @@ h: cache-max-entry-count: 0.9 prefix-cache-decode-state-interval: 1024 prefix-cache-state-budget: 256 + max-prefill-token-num: 64 backends: - name: pytorch communicators: diff --git a/autotest/configs/Qwen/Qwen3.5-397B-A17B.yml b/autotest/configs/Qwen/Qwen3.5-397B-A17B.yml index 3bc6a2a5b7..c0dafd372c 100644 --- a/autotest/configs/Qwen/Qwen3.5-397B-A17B.yml +++ b/autotest/configs/Qwen/Qwen3.5-397B-A17B.yml @@ -87,10 +87,8 @@ h: cache-max-entry-count: 0.7 prefix-cache-decode-state-interval: 1024 prefix-cache-state-budget: 256 + max-prefill-token-num: 64 backends: - - name: turbomind - communicators: - - nccl - name: pytorch communicators: - nccl @@ -126,7 +124,6 @@ h: - longtext_benchmark - longtext_evaluate - mllm_evaluate - - prefix_cache gen_config: temperature: 1.0 top-k: 20 diff --git a/autotest/configs/Qwen/Qwen3.5-4B.yml b/autotest/configs/Qwen/Qwen3.5-4B.yml index eed9f9f023..c2c44ca594 100644 --- a/autotest/configs/Qwen/Qwen3.5-4B.yml +++ b/autotest/configs/Qwen/Qwen3.5-4B.yml @@ -15,3 +15,18 @@ h: - nccl test_coverage: - func +- model_type: + - chat + - vl + engine_config: + tp: 1 + extra: + prefix-cache-state-budget: 256 + prefix-cache-decode-state-interval: 1024 + max-prefill-token-num: 64 + backends: + - name: pytorch + communicators: + - nccl + test_coverage: + - prefix_cache diff --git a/autotest/configs/README.md b/autotest/configs/README.md index 742548ddb9..e0254e4fb6 100644 --- a/autotest/configs/README.md +++ b/autotest/configs/README.md @@ -102,6 +102,14 @@ Rules: - Keep MTP (speculative decoding) on its own row with `speculative-algorithm` in `engine_config.extra`; use `func` and/or `evaluate` in `test_coverage`. - Use `prefix_cache` in `test_coverage`; do not add `enable-prefix-caching` manually to `engine_config.extra`. +- SSM models (Qwen3.5 / Intern-S2): dedicated row with **only** `prefix_cache` in + `test_coverage`, **pytorch only** (TurboMind has no serve CLI for checkpoint + interval; do not list turbomind on this row). Put `max-prefill-token-num: 64` + and `prefix-cache-state-budget` (plus `prefix-cache-decode-state-interval` if + evaluate also uses the row) in that row's `engine_config.extra`. Do not put + these knobs on `func`/`evaluate` rows. `assert_prefix_cache_hit` uses the same + short prompt as AR (text in `*_llm.py`; image in `*_mllm.py` prefix_cache). + Do not inject extras in Python. - Use `quantization` in `test_coverage` only for runtime weight-quant rows (`awq`, `gptq`, `w8a8`). ## `interface` (REST interface coverage) diff --git a/autotest/configs/internlm/Intern-S2-Preview-397B-FP8.yml b/autotest/configs/internlm/Intern-S2-Preview-397B-FP8.yml index 081605c287..a8a7a4becb 100644 --- a/autotest/configs/internlm/Intern-S2-Preview-397B-FP8.yml +++ b/autotest/configs/internlm/Intern-S2-Preview-397B-FP8.yml @@ -89,3 +89,29 @@ h: top-p: 0.95 chat-template-kwargs: enable_thinking: true +- model_type: + - chat + - vl + engine_config: + tp: 2 + dp: 4 + ep: 8 + extra: + max-batch-size: 256 + cache-max-entry-count: 0.7 + prefix-cache-decode-state-interval: 1024 + prefix-cache-state-budget: 256 + max-prefill-token-num: 64 + backends: + - name: pytorch + communicators: + - nccl + test_coverage: + - prefix_cache + gen_config: + temperature: 1.0 + top-k: 20 + repetition-penalty: 1.0 + top-p: 0.95 + chat-template-kwargs: + enable_thinking: true diff --git a/autotest/configs/internlm/Intern-S2-Preview-397B.yml b/autotest/configs/internlm/Intern-S2-Preview-397B.yml index 2558fb52e0..a6aed609c4 100644 --- a/autotest/configs/internlm/Intern-S2-Preview-397B.yml +++ b/autotest/configs/internlm/Intern-S2-Preview-397B.yml @@ -89,3 +89,29 @@ h: top-p: 0.95 chat-template-kwargs: enable_thinking: true +- model_type: + - chat + - vl + engine_config: + tp: 2 + dp: 4 + ep: 8 + extra: + max-batch-size: 256 + cache-max-entry-count: 0.7 + prefix-cache-decode-state-interval: 1024 + prefix-cache-state-budget: 256 + max-prefill-token-num: 64 + backends: + - name: pytorch + communicators: + - nccl + test_coverage: + - prefix_cache + gen_config: + temperature: 1.0 + top-k: 20 + repetition-penalty: 1.0 + top-p: 0.95 + chat-template-kwargs: + enable_thinking: true diff --git a/autotest/configs/internlm/Intern-S2-Preview-FP8.yml b/autotest/configs/internlm/Intern-S2-Preview-FP8.yml index dcac39abc0..ee5679b52a 100644 --- a/autotest/configs/internlm/Intern-S2-Preview-FP8.yml +++ b/autotest/configs/internlm/Intern-S2-Preview-FP8.yml @@ -66,6 +66,7 @@ h: max-batch-size: 256 prefix-cache-decode-state-interval: 1024 prefix-cache-state-budget: 256 + max-prefill-token-num: 64 backends: - name: pytorch communicators: diff --git a/autotest/configs/internlm/Intern-S2-Preview.yml b/autotest/configs/internlm/Intern-S2-Preview.yml index 6714babb3a..4b58454223 100644 --- a/autotest/configs/internlm/Intern-S2-Preview.yml +++ b/autotest/configs/internlm/Intern-S2-Preview.yml @@ -108,6 +108,7 @@ h: max-batch-size: 256 prefix-cache-decode-state-interval: 1024 prefix-cache-state-budget: 256 + max-prefill-token-num: 64 backends: - name: pytorch communicators: diff --git a/autotest/interface/restful/test_restful_generate.py b/autotest/interface/restful/test_restful_generate.py index b8eb9284a9..077067515c 100644 --- a/autotest/interface/restful/test_restful_generate.py +++ b/autotest/interface/restful/test_restful_generate.py @@ -15,6 +15,7 @@ model_enables_return_routed_experts, ) from utils.constant import BACKEND_LIST, BASE_URL, CAPPED_MAX_COMPLETION_TOKENS, RESTFUL_MODEL_LIST +from utils.restful_return_check import has_repeated_fragment from utils.toolkit import encode_text, parse_sse_stream @@ -913,25 +914,29 @@ def test_min_p_parameter(self): def test_repetition_penalty(self): print(f'\n[Model: {self.model_name}] Running repetition penalty test') prompt = 'Repeat repeat repeat repeat' - base = {'prompt': prompt, 'max_tokens': 10, 'top_k': 0, 'stream': False} - - resp_no_penalty = self._post({**base, 'repetition_penalty': 1.0}) - resp_penalty = self._post({**base, 'repetition_penalty': 1.5}) - - text_no_penalty = resp_no_penalty.json()['text'] - text_penalty = resp_penalty.json()['text'] - - def count_repeats(text): - words = text.lower().split() - return sum(1 for i in range(1, len(words)) if words[i] == words[i - 1]) - - repeats_no_penalty = count_repeats(text_no_penalty) - repeats_penalty = count_repeats(text_penalty) - - assert repeats_penalty <= repeats_no_penalty, ( - f'High penalty coefficient ({1.5}) repetition count ({repeats_penalty}) ' - f'not less than low penalty ({1.0}) count ({repeats_no_penalty}), ' - f'repetition_penalty ineffective') + base = { + 'prompt': prompt, + 'max_tokens': 200, + 'temperature': 0.01, + 'ignore_eos': True, + 'stream': False, + } + data_boost = self._post({**base, 'repetition_penalty': 0.0000001}).json() + data_penalize = self._post({**base, 'repetition_penalty': 1.5}).json() + self._validate_generation_response(data=data_boost, validate_tokens=True) + self._validate_generation_response(data=data_penalize, validate_tokens=True) + + boost_ids = data_boost['output_ids'] + penalize_ids = data_penalize['output_ids'] + assert boost_ids, data_boost + assert penalize_ids, data_penalize + boost_repeat, boost_msg = has_repeated_fragment(data_boost['text']) + assert boost_repeat, boost_msg + uniq_boost = len(set(boost_ids)) / len(boost_ids) + uniq_penalize = len(set(penalize_ids)) / len(penalize_ids) + assert uniq_penalize > uniq_boost, ( + f'uniq_penalize={uniq_penalize} uniq_boost={uniq_boost} ' + f'boost_text={data_boost["text"]!r} penalize_text={data_penalize["text"]!r}') def test_ignore_eos_parameter(self): print(f'\n[Model: {self.model_name}] Running ignore_eos parameter test') diff --git a/autotest/tools/pipeline/mllm_case.py b/autotest/tools/pipeline/mllm_case.py index 8a31c6620d..75ccb16b22 100644 --- a/autotest/tools/pipeline/mllm_case.py +++ b/autotest/tools/pipeline/mllm_case.py @@ -85,7 +85,7 @@ def load_video_sampled_pil(video_path: str, num_frames: int, **kwargs: Any) -> t def _is_video_mixed_whitelist_model(model_path: str) -> bool: """Only run video/mixed-mm cases for selected model families.""" m = model_path.lower() - whitelist = ('qwen3-vl', 'qwen3.5', 'interns2-preview') + whitelist = ('qwen3-vl', 'qwen3.5', 'intern-s2-preview') return any(p in m for p in whitelist) diff --git a/autotest/tools/restful/test_restful_chat_hf_pytorch_mllm.py b/autotest/tools/restful/test_restful_chat_hf_pytorch_mllm.py index f5d8a5f66a..b7c58813ca 100644 --- a/autotest/tools/restful/test_restful_chat_hf_pytorch_mllm.py +++ b/autotest/tools/restful/test_restful_chat_hf_pytorch_mllm.py @@ -3,6 +3,7 @@ from utils.run_restful_chat import run_mllm_test BACKEND = 'pytorch' +_PREFIX_CACHE_EXTRA = {'enable-prefix-caching': None} @pytest.mark.gpu_num_1 @@ -34,3 +35,23 @@ def test_restful_chat_tp8(config, run_config, worker_id): @pytest.mark.parametrize('run_config', get_func_config_list(BACKEND, {'tp': 16}, model_type='vl_model')) def test_restful_chat_tp16(config, run_config, worker_id): run_mllm_test(config, run_config, worker_id) + + +@pytest.mark.gpu_num_1 +@pytest.mark.test_3090 +@pytest.mark.test_ascend +@pytest.mark.parametrize( + 'run_config', + get_func_config_list(BACKEND, {'tp': 1}, model_type='vl_model', extra=_PREFIX_CACHE_EXTRA), +) +def test_restful_chat_pytorch_prefix_cache_tp1(config, run_config, worker_id): + run_mllm_test(config, run_config, worker_id) + + +@pytest.mark.gpu_num_2 +@pytest.mark.parametrize( + 'run_config', + get_func_config_list(BACKEND, {'tp': 2}, model_type='vl_model', extra=_PREFIX_CACHE_EXTRA), +) +def test_restful_chat_pytorch_prefix_cache_tp2(config, run_config, worker_id): + run_mllm_test(config, run_config, worker_id) diff --git a/autotest/tools/restful/test_restful_chat_hf_turbomind_mllm.py b/autotest/tools/restful/test_restful_chat_hf_turbomind_mllm.py index 3478861f0f..4d0b49d2e8 100644 --- a/autotest/tools/restful/test_restful_chat_hf_turbomind_mllm.py +++ b/autotest/tools/restful/test_restful_chat_hf_turbomind_mllm.py @@ -4,6 +4,7 @@ from utils.run_restful_chat import run_mllm_test BACKEND = 'turbomind' +_PREFIX_CACHE_EXTRA = {'enable-prefix-caching': None} @pytest.mark.gpu_num_1 @@ -37,6 +38,25 @@ def test_restful_chat_tp16(config, run_config, worker_id): run_mllm_test(config, run_config, worker_id) +@pytest.mark.gpu_num_1 +@pytest.mark.test_3090 +@pytest.mark.parametrize( + 'run_config', + get_func_config_list(BACKEND, {'tp': 1}, model_type='vl_model', extra=_PREFIX_CACHE_EXTRA), +) +def test_restful_chat_prefix_cache_tp1(config, run_config, worker_id): + run_mllm_test(config, run_config, worker_id) + + +@pytest.mark.gpu_num_2 +@pytest.mark.parametrize( + 'run_config', + get_func_config_list(BACKEND, {'tp': 2}, model_type='vl_model', extra=_PREFIX_CACHE_EXTRA), +) +def test_restful_chat_prefix_cache_tp2(config, run_config, worker_id): + run_mllm_test(config, run_config, worker_id) + + @pytest.mark.gpu_num_1 @pytest.mark.other @pytest.mark.parametrize('run_config', TURBOMIND_FALLBACK_TEST_MLLM_GPU1) diff --git a/autotest/utils/config_utils.py b/autotest/utils/config_utils.py index fc2c1a0f81..e049f6e012 100644 --- a/autotest/utils/config_utils.py +++ b/autotest/utils/config_utils.py @@ -406,7 +406,7 @@ def _entry_matches_deps_profile(entry: dict[str, Any], env_key: str, selector: D def _entry_matches_func(entry: dict[str, Any], func_type: str, extra: dict[str, Any] | None) -> bool: funcs = set(entry.get(TEST_COVERAGE_KEY) or []) extra = extra or {} - if extra.get('enable-prefix-caching') is not None: + if 'enable-prefix-caching' in extra: if 'prefix_cache' not in funcs: return False # evaluate/infer accuracy: only dedicated yaml rows with tuned prefix-cache params @@ -1069,7 +1069,7 @@ def _build_run_config_entry( merged_extra = copy.deepcopy(launch_extra) if extra: merged_extra.update(extra) - if extra and extra.get('enable-prefix-caching') is not None: + if extra and 'enable-prefix-caching' in extra: if 'prefix_cache' in (entry.get(TEST_COVERAGE_KEY) or []): merged_extra['enable-prefix-caching'] = None @@ -1327,7 +1327,7 @@ def get_model_list(config: dict[str, Any], Rows with entry-level ``deps`` are never included (regardless of ``DEPS_PROFILE``). """ parallel_config = parallel_config or {'tp': 1} - if extra and extra.get('enable-prefix-caching') is not None: + if extra and 'enable-prefix-caching' in extra: return _model_ids_for_entries(config, backend, parallel_config, model_type, func_type, extra) if func_type == 'func': return _model_ids_for_entries(config, backend, parallel_config, model_type, 'func', extra) diff --git a/autotest/utils/restful_return_check.py b/autotest/utils/restful_return_check.py index 6aa29ef7c9..251de7a3c9 100644 --- a/autotest/utils/restful_return_check.py +++ b/autotest/utils/restful_return_check.py @@ -282,6 +282,73 @@ def get_client_and_model(base_url: str | None = None) -> tuple[OpenAI, str]: return client, models[0].id +# VL paths where VLAsyncEngine silently clears enable_prefix_caching. +_VL_PREFIX_CACHE_UNSUPPORTED = ( + 'intern-s1-mini', + 'glm-4v', + 'gemma-3-27b', + 'internvl3', +) + +def assert_prefix_cache_hit(http_url: str, image_url: str | None = None): + """Same prompt twice: first cold, second cache hit, same generation. + + Unsupported VL paths return without asserting so callers can still run + functional chat cases on the same server. ``image_url`` adds a vision + part (OpenAI ``image_url``) and requires the reply to name the animal. + """ + client, model_name = get_client_and_model(http_url) + model_l = model_name.lower() + if any(tag in model_l for tag in _VL_PREFIX_CACHE_UNSUPPORTED): + print(f'prefix-cache assert skipped for VL path: {model_name}') + return + # SSM prefix_cache yml sets max-prefill-token-num=64 so this length chunks. + pad = 'Prefix-cache probe sentence. ' * 64 + if image_url: + probe = pad + 'What animal is in the image? Reply with one word.' + expect = 'tiger' + content: str | list[dict[str, Any]] = [ + {'type': 'image_url', 'image_url': {'url': image_url}}, + {'type': 'text', 'text': probe}, + ] + else: + probe = pad + 'Reply with one word: ok' + expect = 'ok' + content = probe + messages = [{'role': 'user', 'content': content}] + create = dict( + model=model_name, + messages=messages, + temperature=0, + max_completion_tokens=64, + extra_body={'chat_template_kwargs': {'enable_thinking': False}}, + ) + first = client.chat.completions.create(**create) + second = client.chat.completions.create(**create) + first_choice = first.choices[0] + second_choice = second.choices[0] + first_text = first_choice.message.content + second_text = second_choice.message.content + first_cached = first.usage.prompt_tokens_details.cached_tokens + second_cached = second.usage.prompt_tokens_details.cached_tokens + prompt_tokens = second.usage.prompt_tokens + first_completion = first.usage.completion_tokens + second_completion = second.usage.completion_tokens + msg = (f'first_cached={first_cached} second_cached={second_cached} ' + f'prompt_tokens={prompt_tokens} ' + f'first_completion={first_completion} second_completion={second_completion} ' + f'first_text={first_text!r} second_text={second_text!r}') + assert first.usage.prompt_tokens == prompt_tokens, msg + assert first_completion == second_completion, msg + assert first_cached == 0, msg + # block_size=64; AR reuses complete blocks; SSM hits last published checkpoint + assert prompt_tokens - 64 <= second_cached <= prompt_tokens, msg + assert first_text, msg + assert expect in first_text.lower(), msg + assert first_text == second_text, msg + assert first_choice.finish_reason == second_choice.finish_reason, msg + + @lru_cache(maxsize=1) def deployed_model_name() -> str: """Single model id exposed by the RESTFUL api_server.""" diff --git a/autotest/utils/run_restful_chat.py b/autotest/utils/run_restful_chat.py index fef1929a29..947cf5897d 100644 --- a/autotest/utils/run_restful_chat.py +++ b/autotest/utils/run_restful_chat.py @@ -19,7 +19,11 @@ resolve_extra_params, ) from utils.constant import DEFAULT_PORT, DEFAULT_SERVER, MM_DEMO_TOMB_USER_PROMPT -from utils.restful_return_check import assert_chat_completions_batch_return, get_client_and_model +from utils.restful_return_check import ( + assert_chat_completions_batch_return, + assert_prefix_cache_hit, + get_client_and_model, +) from utils.rule_condition_assert import assert_result from lmdeploy.serve.parsers.response_parser import _parse_tool_call_arguments_dict @@ -156,6 +160,9 @@ def run_all_step(log_path, case_name, cases_info, port: int = DEFAULT_PORT): if model is None: assert False, 'server not start correctly' + if 'prefix-cache' in case_name: + with allure.step('prefix cache cached_tokens'): + assert_prefix_cache_hit(http_url) for case in cases_info.keys(): if case != 'code_testcase' and 'code' in model.lower(): continue @@ -498,7 +505,7 @@ def _consume_chat_completion_stream(stream_iter) -> tuple[str | None, str]: def _is_video_mixed_whitelist_model(model_name: str) -> bool: """Gate video/mixed VL tests to approved model families.""" m = model_name.lower() - return ('qwen3.5' in m or 'qwen3' in m or 'interns2-preview' in m) + return ('qwen3.5' in m or 'qwen3' in m or 'intern-s2-preview' in m) def run_vl_testcase(log_path, resource_path, port: int = DEFAULT_PORT): @@ -536,10 +543,10 @@ def run_vl_testcase(log_path, resource_path, port: int = DEFAULT_PORT): enable_video_mixed = _is_video_mixed_whitelist_model(model_name) if not enable_video_mixed: file.writelines( - f'[video testcase skipped] only enabled for qwen3/qwen3.5/interns2-preview, current model: {model_name}\n') + f'[video testcase skipped] only enabled for qwen3/qwen3.5/intern-s2-preview, current model: {model_name}\n') file.writelines( f'[mixed image+text+video skipped] only enabled for ' - f'qwen3/qwen3.5/interns2-preview, current model: {model_name}\n') + f'qwen3/qwen3.5/intern-s2-preview, current model: {model_name}\n') file.close() allure.attach.file(restful_log, name=restful_log, attachment_type=allure.attachment_type.TEXT) with assume: @@ -1299,9 +1306,14 @@ def run_mllm_test(config, run_config, worker_id): pid, content = start_openai_service(config, run_config, worker_id) try: if pid > 0: - run_vl_testcase(config.get('log_path'), - config.get('resource_path'), - port=DEFAULT_PORT + get_workerid(worker_id)) + port = DEFAULT_PORT + get_workerid(worker_id) + case_name = get_case_str_by_config(run_config) + if 'prefix-cache' in case_name: + http_url = ':'.join([BASE_HTTP_URL, str(port)]) + image_url = f'{config.get("resource_path")}/{PIC}' + with allure.step('prefix cache cached_tokens (image)'): + assert_prefix_cache_hit(http_url, image_url=image_url) + run_vl_testcase(config.get('log_path'), config.get('resource_path'), port=port) else: assert False, f'Failed to start RESTful API server: {_sanitize_server_log(content)}' finally: