Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
27 changes: 27 additions & 0 deletions common/arg.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -3701,6 +3701,33 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
params.sampling.reasoning_budget_message = value;
}
).set_examples({LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI}).set_env("LLAMA_ARG_THINK_BUDGET_MESSAGE"));
add_opt(common_arg(
{"--reasoning-effort-allow"}, "LIST",
"comma-separated reasoning_effort values to pass to the chat template; any other value in a request (or\n"
"chat_template_kwargs) is replaced by --reasoning-effort-fallback instead of reaching a template that\n"
"raises on it (e.g. \"high\" -> HTTP 500). Empty = pass everything (default)",
[](common_params & params, const std::string & value) {
params.reasoning_effort_allow = string_split<std::string>(value, ',');
}
).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_REASONING_EFFORT_ALLOW"));
add_opt(common_arg(
{"--reasoning-effort-fallback"}, "WORD",
string_format("reasoning_effort used in place of a value --reasoning-effort-allow rejects (default: %s)", params.reasoning_effort_fallback.c_str()),
[](common_params & params, const std::string & value) {
params.reasoning_effort_fallback = value;
}
).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_REASONING_EFFORT_FALLBACK"));
add_opt(common_arg(
{"--reasoning-max-tokens-floor"}, "N",
"with thinking on, raise a client max_tokens below N to N, so a small app cap does not end the request\n"
"inside the thinking block with no answer; --reasoning-budget still bounds the thinking. 0 = off (default)",
[](common_params & params, int value) {
if (value < 0) {
throw std::invalid_argument("invalid value");
}
params.reasoning_max_tokens_floor = value;
}
).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_REASONING_MAX_TOKENS_FLOOR"));
add_opt(common_arg(
{"--reasoning-preserve"},
{"--no-reasoning-preserve"},
Expand Down
6 changes: 6 additions & 0 deletions common/common.h
Original file line number Diff line number Diff line change
Expand Up @@ -643,6 +643,12 @@ struct common_params {
bool force_pure_content_parser = false;
common_reasoning_format reasoning_format = COMMON_REASONING_FORMAT_DEEPSEEK;
int enable_reasoning = -1; // -1 = auto, 0 = disable, 1 = enable

// server: reasoning_effort words passed to the chat template (empty = any); others become the fallback
std::vector<std::string> reasoning_effort_allow;
std::string reasoning_effort_fallback = "medium";
// server: with thinking on, raise a client output cap below this (0 = off)
int32_t reasoning_max_tokens_floor = 0;
bool prefill_assistant = true; // if true, any trailing assistant message will be prefilled into the response
int sleep_idle_seconds = -1; // if >0, server will sleep after this many seconds of idle time

Expand Down
35 changes: 35 additions & 0 deletions tools/server/server-common.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -1321,6 +1321,26 @@ json oaicompat_chat_params_parse(
}
}

// --reasoning-effort-allow: an effort word the chat template does not accept (from the request or from
// chat_template_kwargs) becomes --reasoning-effort-fallback instead of reaching the template. Bonsai 2's
// template raises on "high", which Cline, Kilo and Open WebUI send: HTTP 500 on every request.
if (!opt.reasoning_effort_allow.empty()) {
auto it = inputs.chat_template_kwargs.find("reasoning_effort");
if (it != inputs.chat_template_kwargs.end()) {
std::string effort;
try {
effort = json::parse(it->second).get<std::string>();
} catch (...) {
effort = it->second;
}
const auto & allow = opt.reasoning_effort_allow;
if (std::find(allow.begin(), allow.end(), effort) == allow.end()) {
SRV_INF("reasoning_effort \"%s\" is not in --reasoning-effort-allow: using \"%s\"\n", effort.c_str(), opt.reasoning_effort_fallback.c_str());
it->second = json(opt.reasoning_effort_fallback).dump();
}
}
}

inputs.force_pure_content = opt.force_pure_content;

// Apply chat template to the list of messages
Expand Down Expand Up @@ -1388,6 +1408,21 @@ json oaicompat_chat_params_parse(
}
}

// --reasoning-max-tokens-floor: with thinking on, a client output cap below the floor is raised to it. A
// thinking model spends its first thousands of tokens inside the thinking block; a 256-4096 token app cap
// ends the request there with no answer. The reasoning budget still bounds the thinking.
if (opt.reasoning_max_tokens_floor > 0 && inputs.enable_thinking && !chat_params.thinking_end_tags.empty()) {
for (const char * key : {"n_predict", "max_tokens", "max_completion_tokens"}) {
if (llama_params.contains(key) && llama_params.at(key).is_number_integer()) {
const int v = llama_params.at(key).get<int>();
if (v > 0 && v < opt.reasoning_max_tokens_floor) {
SRV_INF("%s %d raised to %d (thinking on, --reasoning-max-tokens-floor)\n", key, v, opt.reasoning_max_tokens_floor);
llama_params[key] = opt.reasoning_max_tokens_floor;
}
}
}
}

return llama_params;
}

Expand Down
3 changes: 3 additions & 0 deletions tools/server/server-common.h
Original file line number Diff line number Diff line change
Expand Up @@ -310,6 +310,9 @@ struct server_chat_params {
std::string reasoning_budget_message;
std::string media_path;
bool force_pure_content = false;
std::vector<std::string> reasoning_effort_allow; // --reasoning-effort-allow (empty = pass everything)
std::string reasoning_effort_fallback; // --reasoning-effort-fallback
int reasoning_max_tokens_floor = 0; // --reasoning-max-tokens-floor (0 = off)
};

// used by /completions endpoint
Expand Down
5 changes: 4 additions & 1 deletion tools/server/server-context.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -1442,7 +1442,10 @@ struct server_context_impl {
/* reasoning_budget */ params_base.sampling.reasoning_budget_tokens,
/* reasoning_budget_msg */ params_base.sampling.reasoning_budget_message,
/* media_path */ params_base.media_path,
/* force_pure_content */ params_base.force_pure_content_parser
/* force_pure_content */ params_base.force_pure_content_parser,
/* reasoning_effort_allow */ params_base.reasoning_effort_allow,
/* reasoning_effort_fallback */ params_base.reasoning_effort_fallback,
/* reasoning_max_tok_floor */ params_base.reasoning_max_tokens_floor
};

{
Expand Down