Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion src/content/docs-es/models.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -47,7 +47,7 @@ OpenAI y la misma `base URL`.
tag="305B MoE"
leftLabel="generación de texto, chat y visión"
rightLabel="capacidades"
description="Modelo MoE de 305B parámetros, servido en su variante Vision-Exp: acepta imágenes de entrada. Contexto de 1M tokens. Tool calling y razonamiento. Cuota de 3B tokens al mes por miembro."
description="Modelo MoE de 305B parámetros, servido en su variante Vision-Exp: acepta imágenes de entrada. Contexto de 1M tokens. Tool calling y razonamiento. Cuota de 3B tokens al mes por miembro. Salida estructurada: response_format json_object es compatible (el prompt debe contener la palabra JSON; si no, se rechaza con un 400), json_schema no y se rechaza con un 400. Para salida restringida a un esquema, usa qwen3.6 o gemma4."
specs={[
{ label: 'Tipo', value: 'MoE (305B total)' },
{ label: 'Cuantización', value: 'FP8' },
Expand All @@ -60,6 +60,7 @@ OpenAI y la misma `base URL`.
'Tool calling',
'Modo razonamiento',
'Visión (entrada de imagen)',
'Modo JSON (<code>json_object</code>, el prompt debe mencionar JSON) · <code>json_schema</code> no soportado (400)',
'Contexto de 1M tokens',
'Generación en streaming (SSE)',
]}
Expand Down
3 changes: 2 additions & 1 deletion src/content/docs/models.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -47,7 +47,7 @@ with the same `base URL`.
tag="305B MoE"
leftLabel="text generation, chat & vision"
rightLabel="capabilities"
description="305B parameter MoE model, served as the Vision-Exp variant: it takes images as input. 1M token context. Tool calling and reasoning. 3B token monthly quota per member."
description="305B parameter MoE model, served as the Vision-Exp variant: it takes images as input. 1M token context. Tool calling and reasoning. 3B token monthly quota per member. Structured output: response_format json_object is supported (the prompt must contain the word JSON, otherwise it is rejected with a 400), json_schema is not and is rejected with a 400. For schema-constrained output, use qwen3.6 or gemma4."
specs={[
{ label: 'Type', value: 'MoE (305B total)' },
{ label: 'Quantization', value: 'FP8' },
Expand All @@ -60,6 +60,7 @@ with the same `base URL`.
'Tool calling',
'Reasoning mode (adaptive — not level-adjustable)',
'Vision (image input)',
'JSON mode (<code>json_object</code>, the prompt must mention JSON) · <code>json_schema</code> not supported (400)',
'1M token context',
'Streaming generation (SSE)',
]}
Expand Down
5 changes: 3 additions & 2 deletions src/data/openapi.json
Original file line number Diff line number Diff line change
Expand Up @@ -1734,7 +1734,8 @@
"properties": {
"name": {
"type": "string",
"description": "The name of the function to call. Up to 64 characters; letters, digits, underscores and dashes."
"pattern": "^[a-zA-Z0-9_-]{1,64}$",
"description": "The name of the function to call. Up to 64 characters; letters, digits, underscores and dashes. A name outside that pattern is rejected with a `400` before the request reaches the model, on every model."
},
"description": {
"type": "string",
Expand Down Expand Up @@ -1795,7 +1796,7 @@
}
},
"ResponseFormat": {
"description": "Force structured output. `json_object` guarantees syntactically valid JSON; `json_schema` (with `strict: true`) constrains the output to a schema. Works on `qwen3.6` and `gemma4`.",
"description": "Force structured output. `json_object` guarantees syntactically valid JSON; `json_schema` (with `strict: true`) constrains the output to a schema. `json_schema` works on `qwen3.6` and `gemma4`. On `deepseek-v4-flash` it is rejected with a `400` before the request reaches the model; `json_object` still works there, as long as the prompt contains the word JSON (otherwise `400`).",
"oneOf": [
{
"type": "object",
Expand Down
72 changes: 70 additions & 2 deletions src/lib/openapiSpec.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -394,10 +394,19 @@ describe('openapi.json: the model it puts in front of a reader', () => {
}
});

/** The structured-output example is allowed to differ, but not to go stale. */
/**
* The structured-output example is allowed to differ, but not to go stale.
* The ResponseFormat description now also names a model that REJECTS
* `json_schema`, so "the description mentions the model" is no longer
* enough: the model has to be in the sentence that says where it works.
*/
it('keeps the structured-output example on a model that supports it', () => {
const model = chat.requestBody.content['application/json'].examples.json_schema.value.model;
expect(spec.components.schemas.ResponseFormat.description).toContain(`\`${model}\``);
const works = /`json_schema` works on (.+?)\.(?:\s|$)/.exec(
spec.components.schemas.ResponseFormat.description,
);
expect(works, 'ResponseFormat no longer says where json_schema works').not.toBeNull();
expect(works![1]).toContain(`\`${model}\``);
});
});

Expand Down Expand Up @@ -611,3 +620,62 @@ describe('getting-started guides: /usage parity', () => {
}
});
});

/**
* THE GATEWAY REJECTS TWO REQUEST SHAPES WITH A 400 BEFORE ROUTING: a tool
* name outside ^[a-zA-Z0-9_-]{1,64}$ (any model) and `response_format`
* `json_schema` on `deepseek-v4-flash` (`json_object` still works there).
* A member who hits either should find it in the reference and on the model
* card, in both locales, not learn it from the error.
*/
describe('documented 400s: tool names and structured output', () => {
const schemas = spec.components.schemas as any;
const TOOL_NAME = '^[a-zA-Z0-9_-]{1,64}$';

it('publishes the tool-name pattern the gateway enforces', () => {
const name = schemas.Tool.properties.function.properties.name;
expect(name.pattern).toBe(TOOL_NAME);
expect(name.description).toContain('Up to 64 characters; letters, digits, underscores and dashes.');
expect(name.description).toMatch(/rejected with a `400`/);
expect(name.description).toMatch(/every model/);
});

it('the published pattern accepts and rejects what the description says', () => {
const re = new RegExp(TOOL_NAME);
for (const ok of ['get_weather', 'search-web', 'A1', 'x'.repeat(64)]) {
expect(re.test(ok), ok).toBe(true);
}
for (const bad of ['', 'x'.repeat(65), 'files.read', 'mcp:search', 'with space', 'ñandú']) {
expect(re.test(bad), bad).toBe(false);
}
});

it('says json_schema is rejected on deepseek-v4-flash and json_object works there', () => {
const description: string = schemas.ResponseFormat.description;
expect(description).toContain(
'On `deepseek-v4-flash` it is rejected with a `400` before the request reaches the model; `json_object` still works there, as long as the prompt contains the word JSON (otherwise `400`).',
);
const works = /`json_schema` works on (.+?)\.(?:\s|$)/.exec(description)!;
expect(works[1]).not.toContain('deepseek-v4-flash');
});

it('states the same on the deepseek-v4-flash model card, in both locales', () => {
const card = (locale: string) => {
const page = readFileSync(
resolve(dirname(fileURLToPath(import.meta.url)), `../content/docs${locale}/models.mdx`),
'utf-8',
);
const start = page.indexOf('id="deepseek-v4-flash"');
expect(start, `${locale || 'en'}: no deepseek-v4-flash card`).toBeGreaterThan(-1);
return page.slice(start, page.indexOf('/>', start));
};
const en = card('');
expect(en).toContain('json_object is supported (the prompt must contain the word JSON, otherwise it is rejected with a 400), json_schema is not and is rejected with a 400');
expect(en).toContain('use qwen3.6 or gemma4');
expect(en).toContain('<code>json_schema</code> not supported (400)');
const es = card('-es');
expect(es).toContain('json_object es compatible (el prompt debe contener la palabra JSON; si no, se rechaza con un 400), json_schema no y se rechaza con un 400');
expect(es).toContain('usa qwen3.6 o gemma4');
expect(es).toContain('<code>json_schema</code> no soportado (400)');
});
});
Loading