Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
61 changes: 61 additions & 0 deletions common/arg.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -2457,6 +2457,18 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
params.kv_mean_center_path = value;
}
).set_env("LLAMA_ARG_KV_MEAN_CENTER"));
add_opt(common_arg(
{"--kv-vram-cells"}, "N",
"tiered KV cache: keep the first N cells of each layer's K/V in VRAM and the rest in pinned system RAM\n"
"mapped into the same device range (CUDA VMM), so the context can exceed VRAM. Positions past N are\n"
"read over PCIe once a sequence is that deep; output is identical to an all-VRAM cache. 0 = off (default)",
[](common_params & params, int value) {
if (value < 0) {
throw std::invalid_argument("invalid value");
}
params.n_kv_vram_cells = value;
}
).set_env("LLAMA_ARG_KV_VRAM_CELLS"));
add_opt(common_arg(
{"--hellaswag"},
"compute HellaSwag score over random tasks from datafile supplied with -f",
Expand Down Expand Up @@ -3701,6 +3713,33 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
params.sampling.reasoning_budget_message = value;
}
).set_examples({LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI}).set_env("LLAMA_ARG_THINK_BUDGET_MESSAGE"));
add_opt(common_arg(
{"--reasoning-effort-allow"}, "LIST",
"comma-separated reasoning_effort values to pass to the chat template; any other value in a request (or\n"
"chat_template_kwargs) is replaced by --reasoning-effort-fallback instead of reaching a template that\n"
"raises on it (e.g. \"high\" -> HTTP 500). Empty = pass everything (default)",
[](common_params & params, const std::string & value) {
params.reasoning_effort_allow = string_split<std::string>(value, ',');
}
).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_REASONING_EFFORT_ALLOW"));
add_opt(common_arg(
{"--reasoning-effort-fallback"}, "WORD",
string_format("reasoning_effort used in place of a value --reasoning-effort-allow rejects (default: %s)", params.reasoning_effort_fallback.c_str()),
[](common_params & params, const std::string & value) {
params.reasoning_effort_fallback = value;
}
).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_REASONING_EFFORT_FALLBACK"));
add_opt(common_arg(
{"--reasoning-max-tokens-floor"}, "N",
"with thinking on, raise a client max_tokens below N to N, so a small app cap does not end the request\n"
"inside the thinking block with no answer; --reasoning-budget still bounds the thinking. 0 = off (default)",
[](common_params & params, int value) {
if (value < 0) {
throw std::invalid_argument("invalid value");
}
params.reasoning_max_tokens_floor = value;
}
).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_REASONING_MAX_TOKENS_FLOOR"));
add_opt(common_arg(
{"--reasoning-preserve"},
{"--no-reasoning-preserve"},
Expand Down Expand Up @@ -4128,6 +4167,28 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
params.speculative.draft.n_depth_max = value;
}
).set_spec().set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_SPEC_DRAFT_DEPTH_MAX"));
add_opt(common_arg(
{"--spec-draft-window"}, "N",
string_format("draft (MTP) context keeps only the last N rows, so its cache stays small and a draft pass costs\n"
"the same at any depth; 0 = full history (default: %d)", params.speculative.draft.n_window),
[](common_params & params, int value) {
if (value < 0) {
throw std::invalid_argument("invalid value");
}
params.speculative.draft.n_window = value;
}
).set_spec().set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_SPEC_DRAFT_WINDOW"));
add_opt(common_arg(
{"--spec-draft-n-max-tail"}, "N",
string_format("draft size once the sequence reaches --kv-vram-cells, where a wider verify reads the host tail\n"
"once for all columns; 0 = --spec-draft-n-max (default: %d)", params.speculative.draft.n_max_tail),
[](common_params & params, int value) {
if (value < 0) {
throw std::invalid_argument("invalid value");
}
params.speculative.draft.n_max_tail = value;
}
).set_spec().set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_SPEC_DRAFT_N_MAX_TAIL"));

add_opt(common_arg(
{"--spec-draft-p-split", "--draft-p-split"}, "P",
Expand Down
1 change: 1 addition & 0 deletions common/common.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -1764,6 +1764,7 @@ struct llama_context_params common_context_params_to_llama(const common_params &
// note: params (and therefore params.kv_mean_center_path) is kept alive by the caller for
// at least as long as it takes to call llama_init_from_model() with the returned cparams
cparams.path_kv_mean_center = params.kv_mean_center_path.empty() ? nullptr : params.kv_mean_center_path.c_str();
cparams.n_kv_vram_cells = (uint32_t) std::max(0, params.n_kv_vram_cells);

return cparams;
}
Expand Down
27 changes: 23 additions & 4 deletions common/common.h
Original file line number Diff line number Diff line change
Expand Up @@ -329,12 +329,22 @@ struct common_params_speculative_draft {
float p_split = 0.1f; // speculative decoding split probability
float p_min = 0.0f; // minimum speculative decoding probability (greedy)

// stop drafting once the sequence is this long (0 = never). Deep in the context a step is bound
// by reading the KV cache, and the draft passes plus the multi-column verify add to that read
// without shortening it: on a 4070 with Bonsai 2 27B the draft is +85% at zero depth, breaks
// even near 24k tokens and costs 30% at 64k. Past the cutoff the slot decodes one token per step.
// stop drafting once the sequence is this long (0 = never). Measured when a deep verify batch ran on the
// vector FA kernel and the draft context grew with the conversation (Bonsai 2 27B on a 4070: +85% at zero
// depth, even near 24k, -30% at 64k). With quantized-KV decode on the MMA FA kernel and n_window below,
// drafting pays at every depth (64k: 48 -> 90 tok/s), so the server default is 0.
int32_t n_depth_max = 0;

// draft (MTP) context window (0 = full history): it keeps only the last n_window rows and is sized for
// them, so its cells are reused, its cache stays small and on the device, and a draft pass costs the same
// at any depth. The MTP head predicts the next few tokens from recent context: 16k rows accept as many
// drafts as the full history at 131k.
int32_t n_window = 0;

// draft size once the sequence reaches the tiered-KV line (--kv-vram-cells; 0 = n_max). Past it a step
// is bound by reading the host tail over PCIe and a wider verify batch reads it once for all columns.
int32_t n_max_tail = 0;

bool backend_sampling = true; // offload draft sampling to the backend (default: on)

common_params_model mparams;
Expand Down Expand Up @@ -588,6 +598,9 @@ struct common_params {
// only takes effect when cache_type_k == GGML_TYPE_Q4_0; see docs/kv-mean-center.md
std::string kv_mean_center_path = "";

// tiered KV cache: cells past this many live in pinned host memory (0 = all in device memory)
int32_t n_kv_vram_cells = 0;

common_conversation_mode conversation_mode = COMMON_CONVERSATION_MODE_AUTO;

// multimodal models (see tools/mtmd)
Expand Down Expand Up @@ -643,6 +656,12 @@ struct common_params {
bool force_pure_content_parser = false;
common_reasoning_format reasoning_format = COMMON_REASONING_FORMAT_DEEPSEEK;
int enable_reasoning = -1; // -1 = auto, 0 = disable, 1 = enable

// server: reasoning_effort words passed to the chat template (empty = any); others become the fallback
std::vector<std::string> reasoning_effort_allow;
std::string reasoning_effort_fallback = "medium";
// server: with thinking on, raise a client output cap below this (0 = off)
int32_t reasoning_max_tokens_floor = 0;
bool prefill_assistant = true; // if true, any trailing assistant message will be prefilled into the response
int sleep_idle_seconds = -1; // if >0, server will sleep after this many seconds of idle time

Expand Down
23 changes: 21 additions & 2 deletions common/speculative.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -2688,7 +2688,8 @@ struct common_speculative_impl_draft_mtp : public common_speculative_impl {

result.push_back(id);

if (params.n_max <= (int) result.size()) {
// the caller's per-draft cap (the server's depth-dependent draft size) as well as the global one
if (params.n_max <= (int) result.size() || (dp.n_max > 0 && dp.n_max <= (int) result.size())) {
drafting[seq_id] = false;
n_drafting--;
continue;
Expand Down Expand Up @@ -3440,8 +3441,26 @@ common_speculative_init_result::common_speculative_init_result(
cparams.ctx_type = LLAMA_CONTEXT_TYPE_MTP;
}

// the draft context holds as many tokens per sequence as the target context
// the draft context holds as many tokens per sequence as the target context, unless the MTP draft only
// ever sees a bounded part of the sequence: the last --spec-draft-window rows (the server drops older ones
// before each batch), or rows up to --spec-draft-depth-max (never fed past it). Sizing it for that keeps
// its cache small; with the window its cells are reused and its attention span stays that short.
cparams.n_ctx = llama_n_ctx(ctx_tgt);
if (spec_mtp) {
const int32_t window = params.speculative.draft.n_window;
const int32_t depth_max = params.speculative.draft.n_depth_max;
const int32_t span = window > 0 ? window : depth_max;
if (span > 0) {
const uint32_t n_seq = std::max<uint32_t>(1, cparams.n_seq_max);
const uint32_t need = (uint32_t) span + 2*std::max<uint32_t>(cparams.n_batch, cparams.n_ubatch) + 256;
const uint32_t capped = GGML_PAD(need, 256) * n_seq;
if (capped < cparams.n_ctx) {
LOG_INF("%s: draft context sized to %u tokens (%s %d, target %u)\n", __func__, capped,
window > 0 ? "spec-draft-window" : "spec-draft-depth-max", span, cparams.n_ctx);
cparams.n_ctx = capped;
}
}
}

// n_rs_seq stays as common_context_params_to_llama set it: the draft context needs the same rollback window as the target, with n_rs_seq == 0 its seq_rm fails silently on partial acceptance and keeps stale positions
cparams.ctx_other = ctx_tgt;
Expand Down
4 changes: 4 additions & 0 deletions ggml/include/ggml-cuda.h
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,10 @@ GGML_BACKEND_API bool ggml_backend_is_cuda(ggml_backend_t backend);
// device buffer
GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_cuda_buffer_type(int device);

// device buffer whose buffers keep vram_frac of each of n_parts equal parts in VRAM and the remainder in
// pinned host memory mapped into the same device address range (CUDA VMM). nullptr without VMM.
GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_cuda_tier_buffer_type(int device, double vram_frac, int n_parts, const char * tag);

// conduct allreduce operation between devices
GGML_BACKEND_API bool ggml_backend_cuda_allreduce_tensor(ggml_backend_t * backends, struct ggml_tensor ** tensors, size_t n_backends);

Expand Down
7 changes: 7 additions & 0 deletions ggml/src/ggml-cuda/fattn-common.cuh
Original file line number Diff line number Diff line change
Expand Up @@ -1112,6 +1112,13 @@ void launch_fattn(
int max_blocks_per_sm = 1; // Max. number of active blocks limited by occupancy.
CUDA_CHECK(cudaOccupancyMaxActiveBlocksPerMultiprocessor(&max_blocks_per_sm, fattn_kernel, block_dim.x * block_dim.y * block_dim.z, nbytes_shared));
GGML_ASSERT(max_blocks_per_sm > 0);
// Batch-invariant mode: the KV split must not depend on which template instance runs. Occupancy differs
// between the 1-query and the multi-query instances (registers, shared memory), so size the split from a
// fixed blocks-per-SM instead; that only shifts work between waves.
// (Stream-k launches keep the real occupancy: their instance is fixed per batch size range already.)
if (ggml_cuda_batch_invariant() && !stream_k) {
max_blocks_per_sm = 4;
}
int parallel_blocks = max_blocks_per_sm;

const int ntiles_KV = (K->ne[1] + nbatch_fa - 1) / nbatch_fa; // Max. number of parallel blocks limited by KV cache length.
Expand Down
47 changes: 47 additions & 0 deletions ggml/src/ggml-cuda/fattn.cu
Original file line number Diff line number Diff line change
Expand Up @@ -475,6 +475,23 @@ static best_fattn_kernel ggml_cuda_get_best_fattn_kernel(const int device, const

// If Turing tensor cores are available, use them:
if (turing_mma_available(cc) && Q->ne[0] != 40 && Q->ne[0] != 72) {
// Quantized-KV decode with wide GQA (up to 8 queries: single-token decode and speculative verify):
// the vector kernel runs one block per Q head, so every K/V row is fetched gqa_ratio times (6x for
// Qwen3.5/Bonsai 2: 24 Q heads over 4 KV heads). The MMA kernel packs the GQA heads of one KV head
// into one tile and reads q4_0/q8_0 K/V in place once. Same rule the F16 path already uses.
// RTX 4070, Bonsai 2 27B, q8_0 K/V, MTP draft: 4k 78.1 -> 86.9 tok/s, 16k 77.7 -> 111.6, 32k 54.5 -> 103.6.
// GGML_CUDA_FA_MMA_DECODE_MIN_KV = shortest KV length that takes this route (default 256, 0 = never).
{
static const int min_kv = [] {
const char * e = getenv("GGML_CUDA_FA_MMA_DECODE_MIN_KV");
return e ? atoi(e) : 256;
}();
if (min_kv > 0 && ggml_is_quantized(K->type) && K->type == V->type &&
ggml_cuda_fattn_mma_kv_native_supported(dst) &&
gqa_opt_applies && gqa_ratio > 4 && Q->ne[1] <= 8 && Q->ne[3] == 1 && K->ne[1] >= min_kv) {
return BEST_FATTN_KERNEL_MMA_F16;
}
}
if (can_use_vector_kernel) {
// batch-invariant mode: the same (vector) kernel for 1 to 8 queries, so a token verified in a
// speculative batch attends with the same arithmetic as a token decoded alone
Expand Down Expand Up @@ -595,8 +612,31 @@ size_t ggml_cuda_flash_attn_ext_get_alloc_size(int device, const ggml_tensor * d
return f16_extra.end - (uintptr_t) dst->data;
}

// ggml-cuda.cu: host-tail staging of tiered buffers (ggml_backend_cuda_tier_buffer_type)
void * ggml_cuda_tier_stage(const void * ptr, size_t nbytes, cudaStream_t stream);

void ggml_cuda_flash_attn_ext(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
ggml_cuda_set_device(ctx.device);

// K/V in a tiered buffer whose host tail this op reaches: copy the used host rows into the VRAM staging
// buffer with the copy engine and point K/V at the all-VRAM alias of the same range for this op. Prefill
// kernels read each K/V row once per query tile, which for host rows would be PCIe traffic every time;
// decode reads each row once, but DMA moves it faster than SMs reading host memory from inside the
// kernel (RTX 4070, 180k context, 86k cells in host memory: +28% decode). Nothing is copied for ops
// that stay below the tier line. The staged bytes are the same bytes, so results are unchanged.
ggml_tensor * K = dst->src[1];
ggml_tensor * V = dst->src[2];
void * K_data = K ? K->data : nullptr;
void * V_data = V ? V->data : nullptr;
if (K && V && V != K) {
if (void * a = ggml_cuda_tier_stage(K->data, ggml_nbytes(K), ctx.stream())) {
K->data = a;
}
if (void * a = ggml_cuda_tier_stage(V->data, ggml_nbytes(V), ctx.stream())) {
V->data = a;
}
}

switch (ggml_cuda_get_best_fattn_kernel(ggml_cuda_get_device(), dst)) {
case BEST_FATTN_KERNEL_NONE:
GGML_ABORT("fatal error");
Expand All @@ -610,6 +650,13 @@ void ggml_cuda_flash_attn_ext(ggml_backend_cuda_context & ctx, ggml_tensor * dst
ggml_cuda_flash_attn_ext_mma_f16(ctx, dst);
break;
}

if (K) {
K->data = K_data;
}
if (V) {
V->data = V_data;
}
}

bool ggml_cuda_flash_attn_ext_supported(int device, const ggml_tensor * dst) {
Expand Down
Loading