diff --git a/src/content/docs-es/choose-a-model.md b/src/content/docs-es/choose-a-model.md index ab4e2ba..1cc24ce 100644 --- a/src/content/docs-es/choose-a-model.md +++ b/src/content/docs-es/choose-a-model.md @@ -28,7 +28,7 @@ Si no sabes cuál coger, busca en la primera columna lo que quieres hacer. | Mover un agente de código en sesiones largas | `glm5.3` | Está pensado para eso. Necesita el tier premium | | Lo mismo, pero sin el tier premium | `glm5.3-flash` | Mismo contexto de 1M y cuota generosa | | Que conteste rápido | `qwen3.8-flash` | Menos profundidad, mucha menos espera | -| Pasarle un audio al modelo directamente | `mimo-v2.5` | Es el único que oye | +| Pasarle un audio al modelo directamente | `mimo-v2.5` o `mimo-v2.6-flash` | Los dos oyen audio de forma nativa | | Describir o analizar una imagen | `deepseek-v4-flash` | Cualquiera menos `glm5.3` sirve; este es el mejor | | Probar cosas sin gastar cuota | `gemma4` | No tiene contador de tokens | | Montar un buscador o un RAG | `qwen3-embedding` y después `rerank` | Primero recuperas por similitud, luego reordenas por relevancia | @@ -45,6 +45,7 @@ Si no sabes cuál coger, busca en la primera columna lo que quieres hacer. | `glm5.3-flash` | Agentes de código, sin premium | 1M | texto · imagen | 2B tokens/mes | | `qwen3.8-flash` | Respuestas rápidas | 262K | texto · imagen | 500M tokens/mes | | `mimo-v2.5` | Audio de entrada, omnimodal | 1M | texto · imagen · audio | 1.0B tokens/mes | +| `mimo-v2.6-flash` | El MiMo más nuevo, omnimodal | 1M | texto · imagen · audio | 1.0B tokens/mes | | `gemma4` | Tareas cortas y pruebas | 262K | texto · imagen | sin contador | | `qwen3.6` | Generación anterior | 262K | texto · imagen | sin contador | | `qwen3-embedding` | Vectores de 4096 dimensiones | - | texto | sin contador | diff --git a/src/content/docs-es/models.mdx b/src/content/docs-es/models.mdx index 52b2966..785710a 100644 --- a/src/content/docs-es/models.mdx +++ b/src/content/docs-es/models.mdx @@ -142,6 +142,29 @@ OpenAI y la misma `base URL`. ]} /> +max_tokens ≥ 300)', + 'Visión (entrada de imagen)', + 'Audio (entrada de audio)', + 'Contexto de 1M tokens', + 'Generación en streaming (SSE)', + ]} +/> + diff --git a/src/content/docs-es/vscode.mdx b/src/content/docs-es/vscode.mdx index 25114ef..e6a7e1b 100644 --- a/src/content/docs-es/vscode.mdx +++ b/src/content/docs-es/vscode.mdx @@ -85,6 +85,15 @@ Se abre un fichero `chatLanguageModels.json`. Déjalo así: "maxInputTokens": 1015808, "maxOutputTokens": 32768 }, + { + "id": "mimo-v2.6-flash", + "name": "Xiaomi MiMo V2.6 Flash", + "url": "https://api.nan.builders/v1/chat/completions", + "toolCalling": true, + "vision": true, + "maxInputTokens": 1015808, + "maxOutputTokens": 32768 + }, { "id": "gemma4", "name": "Gemma 4", diff --git a/src/content/docs/choose-a-model.md b/src/content/docs/choose-a-model.md index c430002..0b05aef 100644 --- a/src/content/docs/choose-a-model.md +++ b/src/content/docs/choose-a-model.md @@ -28,7 +28,7 @@ If you do not know which one to pick, look for what you want to do in the first | Drive a coding agent through long sessions | `glm5.3` | It is built for that. Needs the premium tier | | The same, but without the premium tier | `glm5.3-flash` | Same 1M context and a generous quota | | Get an answer fast | `qwen3.8-flash` | Less depth, much less waiting | -| Hand the model an audio file directly | `mimo-v2.5` | It is the only one that hears | +| Hand the model an audio file directly | `mimo-v2.5` or `mimo-v2.6-flash` | Both hear audio natively | | Describe or analyze an image | `deepseek-v4-flash` | Any of them except `glm5.3` will do; this is the best | | Try things without spending quota | `gemma4` | It has no token counter | | Build a search engine or a RAG | `qwen3-embedding` and then `rerank` | First you retrieve by similarity, then you reorder by relevance | @@ -45,6 +45,7 @@ If you do not know which one to pick, look for what you want to do in the first | `glm5.3-flash` | Coding agents, without premium | 1M | text · image | 2B tokens/month | | `qwen3.8-flash` | Fast answers | 262K | text · image | 500M tokens/month | | `mimo-v2.5` | Audio input, omnimodal | 1M | text · image · audio | 1.0B tokens/month | +| `mimo-v2.6-flash` | The newest MiMo, omnimodal | 1M | text · image · audio | 1.0B tokens/month | | `gemma4` | Short tasks and testing | 262K | text · image | no counter | | `qwen3.6` | Previous generation | 262K | text · image | no counter | | `qwen3-embedding` | 4096-dimension vectors | - | text | no counter | diff --git a/src/content/docs/models.mdx b/src/content/docs/models.mdx index 21143c3..20f0d87 100644 --- a/src/content/docs/models.mdx +++ b/src/content/docs/models.mdx @@ -142,6 +142,29 @@ with the same `base URL`. ]} /> +max_tokens ≥ 300)', + 'Vision (image input)', + 'Audio (audio input)', + '1M token context', + 'Streaming generation (SSE)', + ]} +/> + diff --git a/src/content/docs/vscode.mdx b/src/content/docs/vscode.mdx index 6409df8..a8b6a6f 100644 --- a/src/content/docs/vscode.mdx +++ b/src/content/docs/vscode.mdx @@ -85,6 +85,15 @@ A `chatLanguageModels.json` file opens. Leave it like this: "maxInputTokens": 1015808, "maxOutputTokens": 32768 }, + { + "id": "mimo-v2.6-flash", + "name": "Xiaomi MiMo V2.6 Flash", + "url": "https://api.nan.builders/v1/chat/completions", + "toolCalling": true, + "vision": true, + "maxInputTokens": 1015808, + "maxOutputTokens": 32768 + }, { "id": "gemma4", "name": "Gemma 4", diff --git a/src/data/modelos.json b/src/data/modelos.json index 19b2644..2c6c056 100644 --- a/src/data/modelos.json +++ b/src/data/modelos.json @@ -27,6 +27,13 @@ "frontier": true, "mostUsed": true }, + { + "id": "mimo-v2.6-flash", + "by": "Xiaomi", + "specs": "omnimodal · 1M context · vision · audio · tool calling · reasoning", + "cuota": "1.0B tokens/mes", + "frontier": true + }, { "id": "glm5.3", "by": "Z.ai", diff --git a/src/data/openapi.json b/src/data/openapi.json index b658085..1b65680 100644 --- a/src/data/openapi.json +++ b/src/data/openapi.json @@ -3,7 +3,7 @@ "info": { "title": "NaN API", "version": "1.0.0", - "description": "Open models on a shared EU inference cluster. Zero logs.\n\nThe NaN API is OpenAI-compatible: predictable, resource-oriented URLs, JSON request and response bodies, and standard HTTP verbs and status codes. Point any OpenAI SDK at our base URL and your existing code keeps working. Change the base URL and the API key, and that's it.\n\nOne schema across every model, so you only learn the API once. Change the `model` field to switch models; everything else stays the same.\n\n- Base URL: `https://api.nan.builders/v1`\n- OpenAPI spec: this document. Import it into Postman, Insomnia, or your own tooling.\n\nIf you use the [Helmcode](https://helmcode.com) enterprise service, the base URL is `https://api.helmcode.com/v1` instead. Every other endpoint is identical.\n\n## Authentication\n\nEvery request authenticates with an API key, sent as a Bearer token:\n\n```\nAuthorization: Bearer $NAN_API_KEY\n```\n\nYou must be a NaN community member. Generate your key from user settings, under \"API Keys\", on the [platform](https://cloud.nan.builders/). The key is personal and non-transferable. Keep it secret: never embed one in client-side code or commit it to source control. Requests must go over HTTPS; calls over plain HTTP fail.\n\n## Making requests\n\nThe API is OpenAI-compatible, so point an official OpenAI SDK at our base URL and change nothing else:\n\n```python\nfrom openai import OpenAI\n\nclient = OpenAI(\n api_key=\"$NAN_API_KEY\",\n base_url=\"https://api.nan.builders/v1\",\n)\n\nresp = client.chat.completions.create(\n model=\"deepseek-v4-flash\",\n messages=[{\"role\": \"user\", \"content\": \"Hello\"}],\n)\nprint(resp.choices[0].message.content)\n```\n\n## Streaming\n\nChat responses can stream token-by-token. Set `\"stream\": true` on `/chat/completions` and the response arrives as Server-Sent Events: each event is a `data:` line carrying a `chat.completion.chunk`, with the new text in `choices[0].delta.content`. A final `data: [DONE]` line ends the stream. Only `/chat/completions` streams incrementally; `/responses` currently emits a single terminal event.\n\n## Rate limits\n\n{{RATE_LIMITS}}\n\nImage endpoints run on their own budget, separate from the model endpoints: 20 requests per minute and 100 requests per month. Exceed any limit and you get a `429`.\n\n## Errors\n\nNaN uses conventional HTTP status codes: `2xx` on success, `4xx` for a problem with the request (a missing parameter, an invalid key, an unavailable model) and `5xx` for a server-side error. Every error returns a JSON body in the OpenAI shape:\n\n```json\n{\n \"error\": {\n \"message\": \"The model 'foo' does not exist.\",\n \"type\": \"invalid_request_error\",\n \"param\": \"model\",\n \"code\": \"model_not_found\"\n }\n}\n```\n\n`message` is human-readable, `param` names the offending field when applicable, and `code` is a short machine-readable string you can branch on.\n\n| Status | Meaning | `code` |\n| --- | --- | --- |\n| `400` | Invalid or malformed parameter (`param` says which); or content blocked by the safety filter. | `invalid_request_error` · `content_policy_violation` |\n| `401` | Missing or invalid API key, or a key whose tier does not reach the requested model (`glm5.3`): \"This API key does not have access to the requested model\", `type: auth_error`. Measured 2026-09-12. | `invalid_api_key` |\n| `402` | The token allowance is spent on a model that carries one. Not retryable: the counter returns to zero when that model's quota period does, the calendar month for the models counted per month and your billing period for `glm5.3`. | `monthly_cap_reached` |\n| `403` | Your tier can't access this endpoint. Image generation requires inference membership. A model your tier cannot reach answers `401`, not this. | `tier_restricted` |\n| `404` | The requested model doesn't exist. | `model_not_found` |\n| `429` | Rate limit hit (`rpm_limit`, `max_parallel_requests`), the rolling 4h token budget of `glm5.3`, or a quota exhausted. | `rate_limit_exceeded` · `insufficient_quota` · `quota_exceeded` |\n| `500` | Something went wrong on our side (includes upstream model errors). | (none) |\n| `524` | Timeout, typical with large audio files on `/audio/transcriptions`. | (none) |\n\nRetry `429` and `5xx` responses with exponential backoff. Don't retry `400`, `401`, `403`, or `404` blindly: they'll fail the same way every time until you change the request. `402` cannot be fixed by repetition either: it clears when that model's quota period resets.\n\n## Model catalog\n\nEvery endpoint takes a `model` id. Capabilities vary by model:\n\n| Model | Use for | Capabilities |\n| --- | --- | --- |\n| `deepseek-v4-flash` | Chat, vision, reasoning | Streaming, tool calling, reasoning, image input, 1M-token context. 3B tokens/month per member |\n| `mimo-v2.5` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `qwen3.8-flash` | Chat, vision, agents | Streaming, tool calling, reasoning (on by default), vision, 262K-token context. 500M tokens/month per member |\n| `glm5.3-flash` | Chat, vision, agents | Streaming, tool calling, reasoning, vision, 1M-token context. 2B tokens/month per member |\n| `qwen3.6` | Chat, agents | Streaming, tool calling, vision, reasoning (opt-out, returns `reasoning_content`) |\n| `gemma4` | Chat, vision, agents | Streaming, tool calling, vision, reasoning (opt-in) |\n| `glm5.3` | Coding, long-horizon agents | Streaming, tool calling, reasoning trace, text-only input, 1M-token context. Premium tier only |\n| `qwen3-embedding` | Embeddings | 4096-dimension vectors |\n| `rerank` | RAG reranking | Qwen3-Reranker-8B, 100+ languages |\n| `kokoro` | Text-to-speech | Multiple voices and audio formats |\n| `whisper` | Speech-to-text | Transcription with word/segment timestamps |\n| `flux-2-klein` | Image generation | Text-to-image and image-to-image |\n\n`glm5.3` is served only to keys on the GLM 5.3 premium tier; every other model is available to any inference member. Call [List models](#tag/Models) for the exact set available to your key.\n\n## Versioning & compatibility\n\nThe API tracks the OpenAI API surface, so OpenAI SDKs and tools work against `https://api.nan.builders/v1` unchanged. This reference documents the stable public `/v1` endpoints, and we add capabilities without breaking existing fields.", + "description": "Open models on a shared EU inference cluster. Zero logs.\n\nThe NaN API is OpenAI-compatible: predictable, resource-oriented URLs, JSON request and response bodies, and standard HTTP verbs and status codes. Point any OpenAI SDK at our base URL and your existing code keeps working. Change the base URL and the API key, and that's it.\n\nOne schema across every model, so you only learn the API once. Change the `model` field to switch models; everything else stays the same.\n\n- Base URL: `https://api.nan.builders/v1`\n- OpenAPI spec: this document. Import it into Postman, Insomnia, or your own tooling.\n\nIf you use the [Helmcode](https://helmcode.com) enterprise service, the base URL is `https://api.helmcode.com/v1` instead. Every other endpoint is identical.\n\n## Authentication\n\nEvery request authenticates with an API key, sent as a Bearer token:\n\n```\nAuthorization: Bearer $NAN_API_KEY\n```\n\nYou must be a NaN community member. Generate your key from user settings, under \"API Keys\", on the [platform](https://cloud.nan.builders/). The key is personal and non-transferable. Keep it secret: never embed one in client-side code or commit it to source control. Requests must go over HTTPS; calls over plain HTTP fail.\n\n## Making requests\n\nThe API is OpenAI-compatible, so point an official OpenAI SDK at our base URL and change nothing else:\n\n```python\nfrom openai import OpenAI\n\nclient = OpenAI(\n api_key=\"$NAN_API_KEY\",\n base_url=\"https://api.nan.builders/v1\",\n)\n\nresp = client.chat.completions.create(\n model=\"deepseek-v4-flash\",\n messages=[{\"role\": \"user\", \"content\": \"Hello\"}],\n)\nprint(resp.choices[0].message.content)\n```\n\n## Streaming\n\nChat responses can stream token-by-token. Set `\"stream\": true` on `/chat/completions` and the response arrives as Server-Sent Events: each event is a `data:` line carrying a `chat.completion.chunk`, with the new text in `choices[0].delta.content`. A final `data: [DONE]` line ends the stream. Only `/chat/completions` streams incrementally; `/responses` currently emits a single terminal event.\n\n## Rate limits\n\n{{RATE_LIMITS}}\n\nImage endpoints run on their own budget, separate from the model endpoints: 20 requests per minute and 100 requests per month. Exceed any limit and you get a `429`.\n\n## Errors\n\nNaN uses conventional HTTP status codes: `2xx` on success, `4xx` for a problem with the request (a missing parameter, an invalid key, an unavailable model) and `5xx` for a server-side error. Every error returns a JSON body in the OpenAI shape:\n\n```json\n{\n \"error\": {\n \"message\": \"The model 'foo' does not exist.\",\n \"type\": \"invalid_request_error\",\n \"param\": \"model\",\n \"code\": \"model_not_found\"\n }\n}\n```\n\n`message` is human-readable, `param` names the offending field when applicable, and `code` is a short machine-readable string you can branch on.\n\n| Status | Meaning | `code` |\n| --- | --- | --- |\n| `400` | Invalid or malformed parameter (`param` says which); or content blocked by the safety filter. | `invalid_request_error` · `content_policy_violation` |\n| `401` | Missing or invalid API key, or a key whose tier does not reach the requested model (`glm5.3`): \"This API key does not have access to the requested model\", `type: auth_error`. Measured 2026-09-12. | `invalid_api_key` |\n| `402` | The token allowance is spent on a model that carries one. Not retryable: the counter returns to zero when that model's quota period does, the calendar month for the models counted per month and your billing period for `glm5.3`. | `monthly_cap_reached` |\n| `403` | Your tier can't access this endpoint. Image generation requires inference membership. A model your tier cannot reach answers `401`, not this. | `tier_restricted` |\n| `404` | The requested model doesn't exist. | `model_not_found` |\n| `429` | Rate limit hit (`rpm_limit`, `max_parallel_requests`), the rolling 4h token budget of `glm5.3`, or a quota exhausted. | `rate_limit_exceeded` · `insufficient_quota` · `quota_exceeded` |\n| `500` | Something went wrong on our side (includes upstream model errors). | (none) |\n| `524` | Timeout, typical with large audio files on `/audio/transcriptions`. | (none) |\n\nRetry `429` and `5xx` responses with exponential backoff. Don't retry `400`, `401`, `403`, or `404` blindly: they'll fail the same way every time until you change the request. `402` cannot be fixed by repetition either: it clears when that model's quota period resets.\n\n## Model catalog\n\nEvery endpoint takes a `model` id. Capabilities vary by model:\n\n| Model | Use for | Capabilities |\n| --- | --- | --- |\n| `deepseek-v4-flash` | Chat, vision, reasoning | Streaming, tool calling, reasoning, image input, 1M-token context. 3B tokens/month per member |\n| `mimo-v2.5` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `mimo-v2.6-flash` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `qwen3.8-flash` | Chat, vision, agents | Streaming, tool calling, reasoning (on by default), vision, 262K-token context. 500M tokens/month per member |\n| `glm5.3-flash` | Chat, vision, agents | Streaming, tool calling, reasoning, vision, 1M-token context. 2B tokens/month per member |\n| `qwen3.6` | Chat, agents | Streaming, tool calling, vision, reasoning (opt-out, returns `reasoning_content`) |\n| `gemma4` | Chat, vision, agents | Streaming, tool calling, vision, reasoning (opt-in) |\n| `glm5.3` | Coding, long-horizon agents | Streaming, tool calling, reasoning trace, text-only input, 1M-token context. Premium tier only |\n| `qwen3-embedding` | Embeddings | 4096-dimension vectors |\n| `rerank` | RAG reranking | Qwen3-Reranker-8B, 100+ languages |\n| `kokoro` | Text-to-speech | Multiple voices and audio formats |\n| `whisper` | Speech-to-text | Transcription with word/segment timestamps |\n| `flux-2-klein` | Image generation | Text-to-image and image-to-image |\n\n`glm5.3` is served only to keys on the GLM 5.3 premium tier; every other model is available to any inference member. Call [List models](#tag/Models) for the exact set available to your key.\n\n## Versioning & compatibility\n\nThe API tracks the OpenAI API surface, so OpenAI SDKs and tools work against `https://api.nan.builders/v1` unchanged. This reference documents the stable public `/v1` endpoints, and we add capabilities without breaking existing fields.", "contact": { "name": "NaN", "url": "https://nan.builders" @@ -152,13 +152,13 @@ "properties": { "model": { "type": "string", - "description": "Model id. See [List models](#tag/Models) for what your key can use. Chat models are `deepseek-v4-flash`, `mimo-v2.5`, `qwen3.8-flash`, `glm5.3-flash`, `qwen3.6`, `gemma4` and `glm5.3`.\n\n`glm5.3` requires a key on the GLM 5.3 premium tier; other keys get `401` `auth_error` (\"This API key does not have access to the requested model\"), and do not see it in [List models](#tag/Models) either. The rest are available to every inference member.", + "description": "Model id. See [List models](#tag/Models) for what your key can use. Chat models are `deepseek-v4-flash`, `mimo-v2.5`, `mimo-v2.6-flash`, `qwen3.8-flash`, `glm5.3-flash`, `qwen3.6`, `gemma4` and `glm5.3`.\n\n`glm5.3` requires a key on the GLM 5.3 premium tier; other keys get `401` `auth_error` (\"This API key does not have access to the requested model\"), and do not see it in [List models](#tag/Models) either. The rest are available to every inference member.", "example": "deepseek-v4-flash" }, "messages": { "type": "array", "minItems": 1, - "description": "The conversation so far, oldest first. `content` is a string, or an array of parts (`text` + `image_url`) for vision input on `deepseek-v4-flash`, `mimo-v2.5`, `qwen3.8-flash`, `glm5.3-flash`, `qwen3.6` and `gemma4`. `glm5.3` is text only.", + "description": "The conversation so far, oldest first. `content` is a string, or an array of parts (`text` + `image_url`) for vision input on `deepseek-v4-flash`, `mimo-v2.5`, `mimo-v2.6-flash`, `qwen3.8-flash`, `glm5.3-flash`, `qwen3.6` and `gemma4`. `glm5.3` is text only.", "items": { "$ref": "#/components/schemas/Message" }, diff --git a/src/lib/__fixtures__/ratelimits.expected.md b/src/lib/__fixtures__/ratelimits.expected.md index 59263f4..fbf4c60 100644 --- a/src/lib/__fixtures__/ratelimits.expected.md +++ b/src/lib/__fixtures__/ratelimits.expected.md @@ -10,6 +10,7 @@ - deepseek-v4-flash: 7 (base plan) · 10 (premium plan) - qwen3.8-flash: 7 (base plan) · 10 (premium plan) - mimo-v2.5: 5 +- mimo-v2.6-flash: 5 - qwen3.6: 5 - gemma4: 5 @@ -28,6 +29,7 @@ Audio, embedding and rerank endpoints have no concurrency limit. - deepseek-v4-flash: 1.5M tpm - mimo-v2.5: 1.5M tpm +- mimo-v2.6-flash: 1.5M tpm - qwen3.6: 1.5M tpm - gemma4: 1.5M tpm diff --git a/src/lib/mdxToText.test.ts b/src/lib/mdxToText.test.ts index a5c6ca0..3a49748 100644 --- a/src/lib/mdxToText.test.ts +++ b/src/lib/mdxToText.test.ts @@ -146,6 +146,7 @@ describe('mdxToText rate limits', () => { expect(out).toContain('- deepseek-v4-flash: 7 (base plan) · 10 (premium plan)'); expect(out).toContain('- qwen3.8-flash: 7 (base plan) · 10 (premium plan)'); expect(out).toContain('- mimo-v2.5: 5'); + expect(out).toContain('- mimo-v2.6-flash: 5'); expect(out).toContain('- qwen3.6: 5'); expect(out).toContain('- gemma4: 5'); // glm5.3 is gated by the window, not by a per-minute rate, and the docs diff --git a/src/lib/modelCatalog.ts b/src/lib/modelCatalog.ts index 15b11ba..dc19449 100644 --- a/src/lib/modelCatalog.ts +++ b/src/lib/modelCatalog.ts @@ -126,8 +126,21 @@ export const MODELS: ModelSpec[] = [ quota: { kind: 'monthly', label: { en: '1.0B tokens / mo', es: '1.0B tokens/mes' } }, endpoint: '/chat/completions', bestFor: { - en: 'Passing audio straight to the model. The only omnimodal one', - es: 'Pasarle audio directamente al modelo. El único omnimodal', + en: 'Passing audio straight to the model. Omnimodal, now alongside V2.6 Flash', + es: 'Pasarle audio directamente al modelo. Omnimodal, ahora junto a V2.6 Flash', + }, + }, + { + id: 'mimo-v2.6-flash', + by: 'Xiaomi', + kind: 'chat', + contextTokens: 1_000_000, + inputs: ['text', 'image', 'audio'], + quota: { kind: 'monthly', label: { en: '1.0B tokens / mo', es: '1.0B tokens/mes' } }, + endpoint: '/chat/completions', + bestFor: { + en: 'The newest MiMo, omnimodal like V2.5: text, image and audio in one model', + es: 'El MiMo más nuevo, omnimodal como V2.5: texto, imagen y audio en un modelo', }, }, { diff --git a/src/lib/openapiSpec.test.ts b/src/lib/openapiSpec.test.ts index 0e25e7b..8c25ad9 100644 --- a/src/lib/openapiSpec.test.ts +++ b/src/lib/openapiSpec.test.ts @@ -41,6 +41,7 @@ const PUBLIC_SURFACE: Array<[string, string]> = [ const NAN_MODELS = [ 'deepseek-v4-flash', 'mimo-v2.5', + 'mimo-v2.6-flash', 'qwen3.8-flash', 'glm5.3-flash', 'qwen3.6', @@ -242,7 +243,7 @@ describe('openapi.json: rate limits come from the single source of truth', () => expect(description).toContain( '| `glm5.3`, `glm5.3-flash`, `deepseek-v4-flash`, `qwen3.8-flash` | 7 (base plan) · 10 (premium plan) |', ); - expect(description).toContain('| `mimo-v2.5`, `qwen3.6`, `gemma4` | 5 |'); + expect(description).toContain('| `mimo-v2.5`, `mimo-v2.6-flash`, `qwen3.6`, `gemma4` | 5 |'); }); it('names the endpoints the per-model concurrency table does not cover', () => { diff --git a/src/lib/rateLimits.test.ts b/src/lib/rateLimits.test.ts index 689f3fc..d68b833 100644 --- a/src/lib/rateLimits.test.ts +++ b/src/lib/rateLimits.test.ts @@ -112,5 +112,6 @@ describe('per-model concurrency', () => { expect(premiumConcurrency(DEFAULT_RATE_LIMITS, 'glm5.3', 5)).toBe(10); // A model without a tier variant falls back to the flat default. expect(premiumConcurrency(DEFAULT_RATE_LIMITS, 'mimo-v2.5', 5)).toBe(5); + expect(premiumConcurrency(DEFAULT_RATE_LIMITS, 'mimo-v2.6-flash', 5)).toBe(5); }); }); diff --git a/src/lib/rateLimits.ts b/src/lib/rateLimits.ts index 6dc01be..fb05ce5 100644 --- a/src/lib/rateLimits.ts +++ b/src/lib/rateLimits.ts @@ -102,6 +102,7 @@ export const DEFAULT_RATE_LIMITS: RateLimitsConfig = { tokensPerMinuteByModel: [ { model: 'deepseek-v4-flash', label: '1.5M tpm' }, { model: 'mimo-v2.5', label: '1.5M tpm' }, + { model: 'mimo-v2.6-flash', label: '1.5M tpm' }, { model: 'qwen3.6', label: '1.5M tpm' }, { model: 'gemma4', label: '1.5M tpm' }, ], @@ -121,6 +122,7 @@ export const DEFAULT_RATE_LIMITS: RateLimitsConfig = { { model: 'deepseek-v4-flash', maxParallel: 5, tierMaxParallel: { inference: 7, premium: 10 } }, { model: 'qwen3.8-flash', maxParallel: 5, tierMaxParallel: { inference: 7, premium: 10 } }, { model: 'mimo-v2.5', maxParallel: 5 }, + { model: 'mimo-v2.6-flash', maxParallel: 5 }, { model: 'qwen3.6', maxParallel: 5 }, { model: 'gemma4', maxParallel: 5 }, ], diff --git a/src/tests/lib/docsClientConfigs.test.ts b/src/tests/lib/docsClientConfigs.test.ts index 2f10eff..4a2c0e8 100644 --- a/src/tests/lib/docsClientConfigs.test.ts +++ b/src/tests/lib/docsClientConfigs.test.ts @@ -149,6 +149,7 @@ const EXPECTED_MODELS: Record = { 'deepseek-v4-flash': { context: 1_048_575, output: 32_768 }, 'qwen3.8-flash': { context: 262_144, output: 32_768 }, 'mimo-v2.5': { context: 1_048_576, output: 32_768 }, + 'mimo-v2.6-flash': { context: 1_048_576, output: 32_768 }, 'glm5.3-flash': { context: 1_048_576, output: 32_768 }, };