From 468c37f2d76a2fc29f1ed59e56be14d7a74225cc Mon Sep 17 00:00:00 2001 From: barckcode Date: Mon, 28 Sep 2026 16:57:29 +0200 Subject: [PATCH] feat(models): retire mimo-v2.5 from the site catalogue mimo-v2.5 was retired from the community on 2026-09-25 and now routes through a model_group_alias onto mimo-v2.6-flash (one-week window). Drop it from every member-facing enumeration: home table, /docs/models cards (en/es), choose-a-model, the opencode/vscode/pi guides (counts 8 -> 7), the OpenAPI reference, and the per-model limit tables/fixture. The examples now point at the surviving mimo-v2.6-flash id. Internal metering keeps the legacy name capped for the window (cloud-api and the hooks are unchanged in this repo). --- src/content/docs-es/choose-a-model.md | 5 ++-- src/content/docs-es/examples.md | 8 +++--- src/content/docs-es/models.mdx | 29 +-------------------- src/content/docs-es/opencode.mdx | 7 +---- src/content/docs-es/pi.mdx | 12 ++------- src/content/docs-es/vscode.mdx | 9 ------- src/content/docs/choose-a-model.md | 5 ++-- src/content/docs/examples.md | 8 +++--- src/content/docs/models.mdx | 29 +-------------------- src/content/docs/opencode.mdx | 7 +---- src/content/docs/pi.mdx | 12 ++------- src/content/docs/vscode.mdx | 9 ------- src/data/modelos.json | 10 +------ src/data/openapi.json | 8 +++--- src/lib/__fixtures__/ratelimits.expected.md | 2 -- src/lib/mdxToText.test.ts | 1 - src/lib/modelCatalog.ts | 17 ++---------- src/lib/openapiSpec.test.ts | 3 +-- src/lib/rateLimits.test.ts | 1 - src/lib/rateLimits.ts | 2 -- src/tests/lib/docsClientConfigs.test.ts | 12 ++++----- 21 files changed, 34 insertions(+), 162 deletions(-) diff --git a/src/content/docs-es/choose-a-model.md b/src/content/docs-es/choose-a-model.md index 7db9d77..b58cb86 100644 --- a/src/content/docs-es/choose-a-model.md +++ b/src/content/docs-es/choose-a-model.md @@ -28,7 +28,7 @@ Si no sabes cuál coger, busca en la primera columna lo que quieres hacer. | Mover un agente de código en sesiones largas | `glm5.3` | Está pensado para eso. Necesita el tier premium | | Lo mismo, pero sin el tier premium | `glm5.3-flash` | Mismo contexto de 1M y cuota generosa | | Que conteste rápido | `qwen3.8-flash` | Menos profundidad, mucha menos espera | -| Pasarle un audio al modelo directamente | `mimo-v2.5` o `mimo-v2.6-flash` | Los dos oyen audio de forma nativa | +| Pasarle un audio al modelo directamente | `mimo-v2.6-flash` | Oye audio de forma nativa | | Describir o analizar una imagen | `deepseek-v4-flash` | Cualquiera menos `glm5.3` sirve; este es el mejor | | Probar cosas sin gastar cuota | `gemma4` | No tiene contador de tokens | | Montar un buscador o un RAG | `qwen3-embedding` y después `rerank` | Primero recuperas por similitud, luego reordenas por relevancia | @@ -45,7 +45,6 @@ Si no sabes cuál coger, busca en la primera columna lo que quieres hacer. | `glm5.3` | Agentes de código y tareas largas | 1M | texto | 3B tokens/periodo de facturación | | `glm5.3-flash` | Agentes de código, sin premium | 1M | texto · imagen | 2B tokens/mes | | `qwen3.8-flash` | Respuestas rápidas | 262K | texto · imagen | 500M tokens/mes | -| `mimo-v2.5` | Audio de entrada, omnimodal | 1M | texto · imagen · audio | 1.0B tokens/mes | | `mimo-v2.6-flash` | El MiMo más nuevo, omnimodal | 1M | texto · imagen · audio | 1.0B tokens/mes | | `gemma4` | Tareas cortas y pruebas | 262K | texto · imagen | sin contador | | `qwen3.6` | Generación anterior | 262K | texto · imagen | sin contador | @@ -67,7 +66,7 @@ Las fichas completas, con parámetros, licencias y modos de razonamiento, están - **El id no es el nombre comercial.** El modelo que en su casa se llama "GLM 5.3 Flash" aquí es `glm5.3-flash`, en minúsculas, sin espacios y con el punto de la versión. - **`-flash` significa rápido**, no pequeño ni peor: son variantes optimizadas para latencia. -- **El punto de la versión cuenta.** `qwen3.6` y `qwen3.8-flash` son modelos distintos, y `mimo-v2.5` lleva el punto donde lo lleva. +- **El punto de la versión cuenta.** `qwen3.6` y `qwen3.8-flash` son modelos distintos, y `mimo-v2.6-flash` lleva el punto donde lo lleva. - **Los ids no cambian de significado.** Cuando servimos una variante nueva de un modelo mantenemos su id si la API es la misma. `deepseek-v4-flash`, por ejemplo, pasó a leer imágenes sin cambiar de nombre. - **Los ids viejos no se apagan de golpe.** `qwen3.6` sigue respondiendo para que las configuraciones que ya lo nombran no se rompan, pero no es lo que te conviene si empiezas hoy. diff --git a/src/content/docs-es/examples.md b/src/content/docs-es/examples.md index 62612a4..facb411 100644 --- a/src/content/docs-es/examples.md +++ b/src/content/docs-es/examples.md @@ -343,7 +343,7 @@ console.log(result.language); // "en" console.log(result.duration); // 5.2 ``` -## model: mimo-v2.5 +## model: mimo-v2.6-flash omnimodal: chat, visión y audio @@ -354,7 +354,7 @@ curl https://api.nan.builders/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-your-key-here" \ -d '{ - "model": "mimo-v2.5", + "model": "mimo-v2.6-flash", "messages": [{"role": "user", "content": "Hello, how are you?"}], "max_tokens": 500 }' @@ -369,7 +369,7 @@ curl https://api.nan.builders/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-your-key-here" \ -d '{ - "model": "mimo-v2.5", + "model": "mimo-v2.6-flash", "messages": [{ "role": "user", "content": [ @@ -392,7 +392,7 @@ client = OpenAI( ) response = client.chat.completions.create( - model="mimo-v2.5", + model="mimo-v2.6-flash", messages=[{ "role": "user", "content": [ diff --git a/src/content/docs-es/models.mdx b/src/content/docs-es/models.mdx index 0b08ab8..08eed24 100644 --- a/src/content/docs-es/models.mdx +++ b/src/content/docs-es/models.mdx @@ -115,33 +115,6 @@ OpenAI y la misma `base URL`. ]} /> -max_tokens ≥ 300)', - 'Visión (entrada de imagen)', - 'Audio (entrada de audio)', - 'Contexto de 1M tokens', - 'Generación en streaming (SSE)', - ]} -/> - diff --git a/src/content/docs-es/vscode.mdx b/src/content/docs-es/vscode.mdx index e6a7e1b..4bcea22 100644 --- a/src/content/docs-es/vscode.mdx +++ b/src/content/docs-es/vscode.mdx @@ -76,15 +76,6 @@ Se abre un fichero `chatLanguageModels.json`. Déjalo así: "maxInputTokens": 229376, "maxOutputTokens": 32768 }, - { - "id": "mimo-v2.5", - "name": "Xiaomi MiMo V2.5", - "url": "https://api.nan.builders/v1/chat/completions", - "toolCalling": true, - "vision": true, - "maxInputTokens": 1015808, - "maxOutputTokens": 32768 - }, { "id": "mimo-v2.6-flash", "name": "Xiaomi MiMo V2.6 Flash", diff --git a/src/content/docs/choose-a-model.md b/src/content/docs/choose-a-model.md index da1ba27..6c02691 100644 --- a/src/content/docs/choose-a-model.md +++ b/src/content/docs/choose-a-model.md @@ -28,7 +28,7 @@ If you do not know which one to pick, look for what you want to do in the first | Drive a coding agent through long sessions | `glm5.3` | It is built for that. Needs the premium tier | | The same, but without the premium tier | `glm5.3-flash` | Same 1M context and a generous quota | | Get an answer fast | `qwen3.8-flash` | Less depth, much less waiting | -| Hand the model an audio file directly | `mimo-v2.5` or `mimo-v2.6-flash` | Both hear audio natively | +| Hand the model an audio file directly | `mimo-v2.6-flash` | It hears audio natively | | Describe or analyze an image | `deepseek-v4-flash` | Any of them except `glm5.3` will do; this is the best | | Try things without spending quota | `gemma4` | It has no token counter | | Build a search engine or a RAG | `qwen3-embedding` and then `rerank` | First you retrieve by similarity, then you reorder by relevance | @@ -45,7 +45,6 @@ If you do not know which one to pick, look for what you want to do in the first | `glm5.3` | Coding agents and long tasks | 1M | text | 3B tokens/billing period | | `glm5.3-flash` | Coding agents, without premium | 1M | text · image | 2B tokens/month | | `qwen3.8-flash` | Fast answers | 262K | text · image | 500M tokens/month | -| `mimo-v2.5` | Audio input, omnimodal | 1M | text · image · audio | 1.0B tokens/month | | `mimo-v2.6-flash` | The newest MiMo, omnimodal | 1M | text · image · audio | 1.0B tokens/month | | `gemma4` | Short tasks and testing | 262K | text · image | no counter | | `qwen3.6` | Previous generation | 262K | text · image | no counter | @@ -67,7 +66,7 @@ The full spec sheets, with parameters, licenses and reasoning modes, are in [Mod - **The id is not the commercial name.** The model its makers call "GLM 5.3 Flash" is `glm5.3-flash` here, lowercase, no spaces, and with the version dot. - **`-flash` means fast**, not small or worse: these are variants optimized for latency. -- **The version dot counts.** `qwen3.6` and `qwen3.8-flash` are different models, and `mimo-v2.5` carries its dot where it carries it. +- **The version dot counts.** `qwen3.6` and `qwen3.8-flash` are different models, and `mimo-v2.6-flash` carries its dot where it carries it. - **Ids do not change meaning.** When we serve a new variant of a model we keep its id if the API is the same. `deepseek-v4-flash`, for instance, started reading images without changing its name. - **Old ids are not switched off overnight.** `qwen3.6` still answers so that configurations already naming it do not break, but it is not what you want if you are starting today. diff --git a/src/content/docs/examples.md b/src/content/docs/examples.md index 6be92f2..339a28a 100644 --- a/src/content/docs/examples.md +++ b/src/content/docs/examples.md @@ -343,7 +343,7 @@ console.log(result.language); // "en" console.log(result.duration); // 5.2 ``` -## model: mimo-v2.5 +## model: mimo-v2.6-flash omnimodal — chat, vision, and audio @@ -354,7 +354,7 @@ curl https://api.nan.builders/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-your-key-here" \ -d '{ - "model": "mimo-v2.5", + "model": "mimo-v2.6-flash", "messages": [{"role": "user", "content": "Hello, how are you?"}], "max_tokens": 500 }' @@ -369,7 +369,7 @@ curl https://api.nan.builders/v1/chat/completions \ -H "Content-Type: application/json" \ -H "Authorization: Bearer sk-your-key-here" \ -d '{ - "model": "mimo-v2.5", + "model": "mimo-v2.6-flash", "messages": [{ "role": "user", "content": [ @@ -392,7 +392,7 @@ client = OpenAI( ) response = client.chat.completions.create( - model="mimo-v2.5", + model="mimo-v2.6-flash", messages=[{ "role": "user", "content": [ diff --git a/src/content/docs/models.mdx b/src/content/docs/models.mdx index b2b2160..ed1185f 100644 --- a/src/content/docs/models.mdx +++ b/src/content/docs/models.mdx @@ -115,33 +115,6 @@ with the same `base URL`. ]} /> -max_tokens ≥ 300)', - 'Vision (image input)', - 'Audio (audio input)', - '1M token context', - 'Streaming generation (SSE)', - ]} -/> - diff --git a/src/content/docs/vscode.mdx b/src/content/docs/vscode.mdx index a8b6a6f..e17833a 100644 --- a/src/content/docs/vscode.mdx +++ b/src/content/docs/vscode.mdx @@ -76,15 +76,6 @@ A `chatLanguageModels.json` file opens. Leave it like this: "maxInputTokens": 229376, "maxOutputTokens": 32768 }, - { - "id": "mimo-v2.5", - "name": "Xiaomi MiMo V2.5", - "url": "https://api.nan.builders/v1/chat/completions", - "toolCalling": true, - "vision": true, - "maxInputTokens": 1015808, - "maxOutputTokens": 32768 - }, { "id": "mimo-v2.6-flash", "name": "Xiaomi MiMo V2.6 Flash", diff --git a/src/data/modelos.json b/src/data/modelos.json index 41e02d4..f7ff973 100644 --- a/src/data/modelos.json +++ b/src/data/modelos.json @@ -19,14 +19,6 @@ "frontier": true, "mostUsed": true }, - { - "id": "mimo-v2.5", - "by": "Xiaomi", - "specs": "omnimodal · 310B-15B MoE · FP8 · 1M context · vision · audio · tool calling · reasoning", - "cuota": "1.0B tokens/mes", - "frontier": true, - "mostUsed": true - }, { "id": "mimo-v2.6-flash", "by": "Xiaomi", @@ -140,4 +132,4 @@ ] } ] -} \ No newline at end of file +} diff --git a/src/data/openapi.json b/src/data/openapi.json index aa85c38..f8a85b7 100644 --- a/src/data/openapi.json +++ b/src/data/openapi.json @@ -3,7 +3,7 @@ "info": { "title": "NaN API", "version": "1.0.0", - "description": "Open models on a shared EU inference cluster. Zero logs.\n\nThe NaN API is OpenAI-compatible: predictable, resource-oriented URLs, JSON request and response bodies, and standard HTTP verbs and status codes. Point any OpenAI SDK at our base URL and your existing code keeps working. Change the base URL and the API key, and that's it.\n\nOne schema across every model, so you only learn the API once. Change the `model` field to switch models; everything else stays the same.\n\n- Base URL: `https://api.nan.builders/v1`\n- OpenAPI spec: this document. Import it into Postman, Insomnia, or your own tooling.\n\nIf you use the [Helmcode](https://helmcode.com) enterprise service, the base URL is `https://api.helmcode.com/v1` instead. Every other endpoint is identical.\n\n## Authentication\n\nEvery request authenticates with an API key, sent as a Bearer token:\n\n```\nAuthorization: Bearer $NAN_API_KEY\n```\n\nYou must be a NaN community member. Generate your key from user settings, under \"API Keys\", on the [platform](https://cloud.nan.builders/). The key is personal and non-transferable. Keep it secret: never embed one in client-side code or commit it to source control. Requests must go over HTTPS; calls over plain HTTP fail.\n\n## Making requests\n\nThe API is OpenAI-compatible, so point an official OpenAI SDK at our base URL and change nothing else:\n\n```python\nfrom openai import OpenAI\n\nclient = OpenAI(\n api_key=\"$NAN_API_KEY\",\n base_url=\"https://api.nan.builders/v1\",\n)\n\nresp = client.chat.completions.create(\n model=\"deepseek-v4-flash\",\n messages=[{\"role\": \"user\", \"content\": \"Hello\"}],\n)\nprint(resp.choices[0].message.content)\n```\n\n## Streaming\n\nChat responses can stream token-by-token. Set `\"stream\": true` on `/chat/completions` and the response arrives as Server-Sent Events: each event is a `data:` line carrying a `chat.completion.chunk`, with the new text in `choices[0].delta.content`. A final `data: [DONE]` line ends the stream. Only `/chat/completions` streams incrementally; `/responses` currently emits a single terminal event.\n\n## Rate limits\n\n{{RATE_LIMITS}}\n\nImage endpoints run on their own budget, separate from the model endpoints: 20 requests per minute and 100 requests per month. The usage endpoint is metered separately too: {{USAGE_RATE_LIMIT}} requests per minute per member. Exceed any limit and you get a `429`.\n\n## Errors\n\nNaN uses conventional HTTP status codes: `2xx` on success, `4xx` for a problem with the request (a missing parameter, an invalid key, an unavailable model) and `5xx` for a server-side error. Every error returns a JSON body in the OpenAI shape:\n\n```json\n{\n \"error\": {\n \"message\": \"The model 'foo' does not exist.\",\n \"type\": \"invalid_request_error\",\n \"param\": \"model\",\n \"code\": \"model_not_found\"\n }\n}\n```\n\n`message` is human-readable, `param` names the offending field when applicable, and `code` is a short machine-readable string you can branch on.\n\n| Status | Meaning | `code` |\n| --- | --- | --- |\n| `400` | Invalid or malformed parameter (`param` says which); or content blocked by the safety filter. | `invalid_request_error` · `content_policy_violation` |\n| `401` | Missing or invalid API key, or a key whose tier does not reach the requested model (`glm5.3`): \"This API key does not have access to the requested model\", `type: auth_error`. Measured 2026-09-12. | `invalid_api_key` |\n| `402` | The token allowance is spent on a model that carries one. Not retryable: the counter returns to zero when that model's quota period does, the calendar month for the models counted per month and your billing period for `glm5.3`. | `monthly_cap_reached` |\n| `403` | Your tier can't access this endpoint. Image generation requires inference membership. A model your tier cannot reach answers `401`, not this. | `tier_restricted` |\n| `404` | The requested model doesn't exist. | `model_not_found` |\n| `429` | Rate limit hit (`rpm_limit`, `max_parallel_requests`), the rolling 4h token budget of `glm5.3`, or a quota exhausted. | `rate_limit_exceeded` · `insufficient_quota` · `quota_exceeded` |\n| `500` | Something went wrong on our side (includes upstream model errors). | (none) |\n| `524` | Timeout, typical with large audio files on `/audio/transcriptions`. | (none) |\n\nRetry `429` and `5xx` responses with exponential backoff. Don't retry `400`, `401`, `403`, or `404` blindly: they'll fail the same way every time until you change the request. `402` cannot be fixed by repetition either: it clears when that model's quota period resets.\n\n## Model catalog\n\nEvery endpoint takes a `model` id. Capabilities vary by model:\n\n| Model | Use for | Capabilities |\n| --- | --- | --- |\n| `deepseek-v4-flash` | Chat, vision, reasoning | Streaming, tool calling, reasoning, image input, 1M-token context. 3B tokens/month per member |\n| `mimo-v2.5` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `mimo-v2.6-flash` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `qwen3.8-flash` | Chat, vision, agents | Streaming, tool calling, reasoning (on by default), vision, 262K-token context. 500M tokens/month per member |\n| `glm5.3-flash` | Chat, vision, agents | Streaming, tool calling, reasoning, vision, 1M-token context. 2B tokens/month per member |\n| `qwen3.6` | Chat, agents | Streaming, tool calling, vision, reasoning (opt-out, returns `reasoning_content`) |\n| `gemma4` | Chat, vision, agents | Streaming, tool calling, vision, reasoning (opt-in) |\n| `glm5.3` | Coding, long-horizon agents | Streaming, tool calling, reasoning trace, text-only input, 1M-token context. Premium tier only |\n| `qwen3-embedding` | Embeddings | 4096-dimension vectors |\n| `rerank` | RAG reranking | Qwen3-Reranker-8B, 100+ languages |\n| `kokoro` | Text-to-speech | Multiple voices and audio formats |\n| `whisper` | Speech-to-text | Transcription with word/segment timestamps |\n| `flux-2-klein` | Image generation | Text-to-image and image-to-image |\n| `qwen-image-2.1` | Image generation (text→image) | 512-1280 px, 1-4 per request, seed 0-2147483647. 100 images/month per member (shared pool with flux-2-klein) |\n\n`glm5.3` is served only to keys on the GLM 5.3 premium tier; every other model is available to any inference member. Call [List models](#tag/Models) for the exact set available to your key.\n\n## Versioning & compatibility\n\nThe API tracks the OpenAI API surface, so OpenAI SDKs and tools work against `https://api.nan.builders/v1` unchanged. This reference documents the stable public `/v1` endpoints, and we add capabilities without breaking existing fields.", + "description": "Open models on a shared EU inference cluster. Zero logs.\n\nThe NaN API is OpenAI-compatible: predictable, resource-oriented URLs, JSON request and response bodies, and standard HTTP verbs and status codes. Point any OpenAI SDK at our base URL and your existing code keeps working. Change the base URL and the API key, and that's it.\n\nOne schema across every model, so you only learn the API once. Change the `model` field to switch models; everything else stays the same.\n\n- Base URL: `https://api.nan.builders/v1`\n- OpenAPI spec: this document. Import it into Postman, Insomnia, or your own tooling.\n\nIf you use the [Helmcode](https://helmcode.com) enterprise service, the base URL is `https://api.helmcode.com/v1` instead. Every other endpoint is identical.\n\n## Authentication\n\nEvery request authenticates with an API key, sent as a Bearer token:\n\n```\nAuthorization: Bearer $NAN_API_KEY\n```\n\nYou must be a NaN community member. Generate your key from user settings, under \"API Keys\", on the [platform](https://cloud.nan.builders/). The key is personal and non-transferable. Keep it secret: never embed one in client-side code or commit it to source control. Requests must go over HTTPS; calls over plain HTTP fail.\n\n## Making requests\n\nThe API is OpenAI-compatible, so point an official OpenAI SDK at our base URL and change nothing else:\n\n```python\nfrom openai import OpenAI\n\nclient = OpenAI(\n api_key=\"$NAN_API_KEY\",\n base_url=\"https://api.nan.builders/v1\",\n)\n\nresp = client.chat.completions.create(\n model=\"deepseek-v4-flash\",\n messages=[{\"role\": \"user\", \"content\": \"Hello\"}],\n)\nprint(resp.choices[0].message.content)\n```\n\n## Streaming\n\nChat responses can stream token-by-token. Set `\"stream\": true` on `/chat/completions` and the response arrives as Server-Sent Events: each event is a `data:` line carrying a `chat.completion.chunk`, with the new text in `choices[0].delta.content`. A final `data: [DONE]` line ends the stream. Only `/chat/completions` streams incrementally; `/responses` currently emits a single terminal event.\n\n## Rate limits\n\n{{RATE_LIMITS}}\n\nImage endpoints run on their own budget, separate from the model endpoints: 20 requests per minute and 100 requests per month. The usage endpoint is metered separately too: {{USAGE_RATE_LIMIT}} requests per minute per member. Exceed any limit and you get a `429`.\n\n## Errors\n\nNaN uses conventional HTTP status codes: `2xx` on success, `4xx` for a problem with the request (a missing parameter, an invalid key, an unavailable model) and `5xx` for a server-side error. Every error returns a JSON body in the OpenAI shape:\n\n```json\n{\n \"error\": {\n \"message\": \"The model 'foo' does not exist.\",\n \"type\": \"invalid_request_error\",\n \"param\": \"model\",\n \"code\": \"model_not_found\"\n }\n}\n```\n\n`message` is human-readable, `param` names the offending field when applicable, and `code` is a short machine-readable string you can branch on.\n\n| Status | Meaning | `code` |\n| --- | --- | --- |\n| `400` | Invalid or malformed parameter (`param` says which); or content blocked by the safety filter. | `invalid_request_error` · `content_policy_violation` |\n| `401` | Missing or invalid API key, or a key whose tier does not reach the requested model (`glm5.3`): \"This API key does not have access to the requested model\", `type: auth_error`. Measured 2026-09-12. | `invalid_api_key` |\n| `402` | The token allowance is spent on a model that carries one. Not retryable: the counter returns to zero when that model's quota period does, the calendar month for the models counted per month and your billing period for `glm5.3`. | `monthly_cap_reached` |\n| `403` | Your tier can't access this endpoint. Image generation requires inference membership. A model your tier cannot reach answers `401`, not this. | `tier_restricted` |\n| `404` | The requested model doesn't exist. | `model_not_found` |\n| `429` | Rate limit hit (`rpm_limit`, `max_parallel_requests`), the rolling 4h token budget of `glm5.3`, or a quota exhausted. | `rate_limit_exceeded` · `insufficient_quota` · `quota_exceeded` |\n| `500` | Something went wrong on our side (includes upstream model errors). | (none) |\n| `524` | Timeout, typical with large audio files on `/audio/transcriptions`. | (none) |\n\nRetry `429` and `5xx` responses with exponential backoff. Don't retry `400`, `401`, `403`, or `404` blindly: they'll fail the same way every time until you change the request. `402` cannot be fixed by repetition either: it clears when that model's quota period resets.\n\n## Model catalog\n\nEvery endpoint takes a `model` id. Capabilities vary by model:\n\n| Model | Use for | Capabilities |\n| --- | --- | --- |\n| `deepseek-v4-flash` | Chat, vision, reasoning | Streaming, tool calling, reasoning, image input, 1M-token context. 3B tokens/month per member |\n| `mimo-v2.6-flash` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `qwen3.8-flash` | Chat, vision, agents | Streaming, tool calling, reasoning (on by default), vision, 262K-token context. 500M tokens/month per member |\n| `glm5.3-flash` | Chat, vision, agents | Streaming, tool calling, reasoning, vision, 1M-token context. 2B tokens/month per member |\n| `qwen3.6` | Chat, agents | Streaming, tool calling, vision, reasoning (opt-out, returns `reasoning_content`) |\n| `gemma4` | Chat, vision, agents | Streaming, tool calling, vision, reasoning (opt-in) |\n| `glm5.3` | Coding, long-horizon agents | Streaming, tool calling, reasoning trace, text-only input, 1M-token context. Premium tier only |\n| `qwen3-embedding` | Embeddings | 4096-dimension vectors |\n| `rerank` | RAG reranking | Qwen3-Reranker-8B, 100+ languages |\n| `kokoro` | Text-to-speech | Multiple voices and audio formats |\n| `whisper` | Speech-to-text | Transcription with word/segment timestamps |\n| `flux-2-klein` | Image generation | Text-to-image and image-to-image |\n| `qwen-image-2.1` | Image generation (text→image) | 512-1280 px, 1-4 per request, seed 0-2147483647. 100 images/month per member (shared pool with flux-2-klein) |\n\n`glm5.3` is served only to keys on the GLM 5.3 premium tier; every other model is available to any inference member. Call [List models](#tag/Models) for the exact set available to your key.\n\n## Versioning & compatibility\n\nThe API tracks the OpenAI API surface, so OpenAI SDKs and tools work against `https://api.nan.builders/v1` unchanged. This reference documents the stable public `/v1` endpoints, and we add capabilities without breaking existing fields.", "contact": { "name": "NaN", "url": "https://nan.builders" @@ -156,13 +156,13 @@ "properties": { "model": { "type": "string", - "description": "Model id. See [List models](#tag/Models) for what your key can use. Chat models are `deepseek-v4-flash`, `mimo-v2.5`, `mimo-v2.6-flash`, `qwen3.8-flash`, `glm5.3-flash`, `qwen3.6`, `gemma4` and `glm5.3`.\n\n`glm5.3` requires a key on the GLM 5.3 premium tier; other keys get `401` `auth_error` (\"This API key does not have access to the requested model\"), and do not see it in [List models](#tag/Models) either. The rest are available to every inference member.", + "description": "Model id. See [List models](#tag/Models) for what your key can use. Chat models are `deepseek-v4-flash`, `mimo-v2.6-flash`, `qwen3.8-flash`, `glm5.3-flash`, `qwen3.6`, `gemma4` and `glm5.3`.\n\n`glm5.3` requires a key on the GLM 5.3 premium tier; other keys get `401` `auth_error` (\"This API key does not have access to the requested model\"), and do not see it in [List models](#tag/Models) either. The rest are available to every inference member.", "example": "deepseek-v4-flash" }, "messages": { "type": "array", "minItems": 1, - "description": "The conversation so far, oldest first. `content` is a string, or an array of parts (`text` + `image_url`) for vision input on `deepseek-v4-flash`, `mimo-v2.5`, `mimo-v2.6-flash`, `qwen3.8-flash`, `glm5.3-flash`, `qwen3.6` and `gemma4`. `glm5.3` is text only.", + "description": "The conversation so far, oldest first. `content` is a string, or an array of parts (`text` + `image_url`) for vision input on `deepseek-v4-flash`, `mimo-v2.6-flash`, `qwen3.8-flash`, `glm5.3-flash`, `qwen3.6` and `gemma4`. `glm5.3` is text only.", "items": { "$ref": "#/components/schemas/Message" }, @@ -2563,4 +2563,4 @@ "description": "NaN Docs", "url": "https://nan.builders/docs" } -} \ No newline at end of file +} diff --git a/src/lib/__fixtures__/ratelimits.expected.md b/src/lib/__fixtures__/ratelimits.expected.md index fbf4c60..193eb39 100644 --- a/src/lib/__fixtures__/ratelimits.expected.md +++ b/src/lib/__fixtures__/ratelimits.expected.md @@ -9,7 +9,6 @@ - glm5.3-flash: 7 (base plan) · 10 (premium plan) - deepseek-v4-flash: 7 (base plan) · 10 (premium plan) - qwen3.8-flash: 7 (base plan) · 10 (premium plan) -- mimo-v2.5: 5 - mimo-v2.6-flash: 5 - qwen3.6: 5 - gemma4: 5 @@ -28,7 +27,6 @@ Audio, embedding and rerank endpoints have no concurrency limit. **tokens / min per model** - deepseek-v4-flash: 1.5M tpm -- mimo-v2.5: 1.5M tpm - mimo-v2.6-flash: 1.5M tpm - qwen3.6: 1.5M tpm - gemma4: 1.5M tpm diff --git a/src/lib/mdxToText.test.ts b/src/lib/mdxToText.test.ts index 3a49748..0cfe40f 100644 --- a/src/lib/mdxToText.test.ts +++ b/src/lib/mdxToText.test.ts @@ -145,7 +145,6 @@ describe('mdxToText rate limits', () => { expect(out).toContain('- glm5.3-flash: 7 (base plan) · 10 (premium plan)'); expect(out).toContain('- deepseek-v4-flash: 7 (base plan) · 10 (premium plan)'); expect(out).toContain('- qwen3.8-flash: 7 (base plan) · 10 (premium plan)'); - expect(out).toContain('- mimo-v2.5: 5'); expect(out).toContain('- mimo-v2.6-flash: 5'); expect(out).toContain('- qwen3.6: 5'); expect(out).toContain('- gemma4: 5'); diff --git a/src/lib/modelCatalog.ts b/src/lib/modelCatalog.ts index f74ca9f..ce96aff 100644 --- a/src/lib/modelCatalog.ts +++ b/src/lib/modelCatalog.ts @@ -117,19 +117,6 @@ export const MODELS: ModelSpec[] = [ es: 'Respuestas rápidas, cuando importa más la latencia que la profundidad', }, }, - { - id: 'mimo-v2.5', - by: 'Xiaomi', - kind: 'chat', - contextTokens: 1_000_000, - inputs: ['text', 'image', 'audio'], - quota: { kind: 'monthly', label: { en: '1.0B tokens / mo', es: '1.0B tokens/mes' } }, - endpoint: '/chat/completions', - bestFor: { - en: 'Passing audio straight to the model. Omnimodal, now alongside V2.6 Flash', - es: 'Pasarle audio directamente al modelo. Omnimodal, ahora junto a V2.6 Flash', - }, - }, { id: 'mimo-v2.6-flash', by: 'Xiaomi', @@ -139,8 +126,8 @@ export const MODELS: ModelSpec[] = [ quota: { kind: 'monthly', label: { en: '1.0B tokens / mo', es: '1.0B tokens/mes' } }, endpoint: '/chat/completions', bestFor: { - en: 'The newest MiMo, omnimodal like V2.5: text, image and audio in one model', - es: 'El MiMo más nuevo, omnimodal como V2.5: texto, imagen y audio en un modelo', + en: 'The newest MiMo, omnimodal: text, image and audio in one model', + es: 'El MiMo más nuevo, omnimodal: texto, imagen y audio en un modelo', }, }, { diff --git a/src/lib/openapiSpec.test.ts b/src/lib/openapiSpec.test.ts index 102ff37..a77706b 100644 --- a/src/lib/openapiSpec.test.ts +++ b/src/lib/openapiSpec.test.ts @@ -41,7 +41,6 @@ const PUBLIC_SURFACE: Array<[string, string]> = [ /** NaN's real catalogue (src/data/modelos.json + the API reference). */ const NAN_MODELS = [ 'deepseek-v4-flash', - 'mimo-v2.5', 'mimo-v2.6-flash', 'qwen3.8-flash', 'glm5.3-flash', @@ -294,7 +293,7 @@ describe('openapi.json: rate limits come from the single source of truth', () => expect(description).toContain( '| `glm5.3`, `glm5.3-flash`, `deepseek-v4-flash`, `qwen3.8-flash` | 7 (base plan) · 10 (premium plan) |', ); - expect(description).toContain('| `mimo-v2.5`, `mimo-v2.6-flash`, `qwen3.6`, `gemma4` | 5 |'); + expect(description).toContain('| `mimo-v2.6-flash`, `qwen3.6`, `gemma4` | 5 |'); }); it('names the endpoints the per-model concurrency table does not cover', () => { diff --git a/src/lib/rateLimits.test.ts b/src/lib/rateLimits.test.ts index d68b833..82f25e3 100644 --- a/src/lib/rateLimits.test.ts +++ b/src/lib/rateLimits.test.ts @@ -111,7 +111,6 @@ describe('per-model concurrency', () => { it('resolves the premium number the premium card publishes', () => { expect(premiumConcurrency(DEFAULT_RATE_LIMITS, 'glm5.3', 5)).toBe(10); // A model without a tier variant falls back to the flat default. - expect(premiumConcurrency(DEFAULT_RATE_LIMITS, 'mimo-v2.5', 5)).toBe(5); expect(premiumConcurrency(DEFAULT_RATE_LIMITS, 'mimo-v2.6-flash', 5)).toBe(5); }); }); diff --git a/src/lib/rateLimits.ts b/src/lib/rateLimits.ts index dd80b22..425525f 100644 --- a/src/lib/rateLimits.ts +++ b/src/lib/rateLimits.ts @@ -120,7 +120,6 @@ export const DEFAULT_RATE_LIMITS: RateLimitsConfig = { perKey: { requestsPerMinute: 60, maxParallel: 5, tierMaxParallel: { inference: 7, premium: 10 } }, tokensPerMinuteByModel: [ { model: 'deepseek-v4-flash', label: '1.5M tpm' }, - { model: 'mimo-v2.5', label: '1.5M tpm' }, { model: 'mimo-v2.6-flash', label: '1.5M tpm' }, { model: 'qwen3.6', label: '1.5M tpm' }, { model: 'gemma4', label: '1.5M tpm' }, @@ -140,7 +139,6 @@ export const DEFAULT_RATE_LIMITS: RateLimitsConfig = { { model: 'glm5.3-flash', maxParallel: 5, tierMaxParallel: { inference: 7, premium: 10 } }, { model: 'deepseek-v4-flash', maxParallel: 5, tierMaxParallel: { inference: 7, premium: 10 } }, { model: 'qwen3.8-flash', maxParallel: 5, tierMaxParallel: { inference: 7, premium: 10 } }, - { model: 'mimo-v2.5', maxParallel: 5 }, { model: 'mimo-v2.6-flash', maxParallel: 5 }, { model: 'qwen3.6', maxParallel: 5 }, { model: 'gemma4', maxParallel: 5 }, diff --git a/src/tests/lib/docsClientConfigs.test.ts b/src/tests/lib/docsClientConfigs.test.ts index 4a2c0e8..ed8fea4 100644 --- a/src/tests/lib/docsClientConfigs.test.ts +++ b/src/tests/lib/docsClientConfigs.test.ts @@ -66,7 +66,7 @@ const here = dirname(fileURLToPath(import.meta.url)); * deepseek-v4-flash 1048575 declared glm5.3 1048576 declared * qwen3.8-flash 262144 declared glm5.3-flash 1048576 declared * qwen3.6 262144 --max-model-len - * gemma4 262144 model card mimo-v2.5 1048576 model card + * gemma4 262144 model card mimo-v2.6-flash 1048576 model card * * EVERY DISAGREEMENT WITH `modelRateLimits`, since the previous version of this * list claimed to be complete and was not: @@ -75,7 +75,8 @@ const here = dirname(fileURLToPath(import.meta.url)); * declares 262144 and the model card calls 262K "the model's native * window". Tracked as helmcode/nan#53; it also inflates that model's ITPM * fourfold. - * * mimo-v2.5 -- 1048576 here against 1_050_000 there. 1,424 tokens. + * * mimo-v2.6-flash -- 1048576 here against 1_050_000 there. 1,424 tokens + * (inherited verbatim from the retired mimo-v2.5 row). * * deepseek-v4-flash -- 1048575 here against 1_048_576 there. ONE token, * and the odd number is the real declaration, corroborated at * `litellm-community/values.yaml:199`. @@ -91,7 +92,7 @@ const here = dirname(fileURLToPath(import.meta.url)); * cornered): * * deepseek-v4-flash 1048575 glm5.3-flash 1048575 - * qwen3.8-flash 131072 mimo-v2.5 131072 + * qwen3.8-flash 131072 mimo-v2.6-flash 131072 * gemma4 262130 qwen3.6 262131 * * Two things follow, and they are why these values stay where they are. @@ -103,7 +104,7 @@ const here = dirname(fileURLToPath(import.meta.url)); * 1048575. So 32768 and 65536 are a BUDGET this site recommends, not a limit * anything enforces, and a member who raises them is not doing anything wrong. * - * TWO MODELS DO HAVE A REAL CAP: qwen3.8-flash and mimo-v2.5 refuse anything + * TWO MODELS DO HAVE A REAL CAP: qwen3.8-flash and mimo-v2.6-flash refuse anything * over 131072, well below their windows, because the upstream that serves them * enforces its own. That figure is a fact about the endpoint and is the one * models.dev already publishes for them. @@ -148,7 +149,6 @@ const EXPECTED_MODELS: Record = { gemma4: { context: 262_144, output: 65_536 }, 'deepseek-v4-flash': { context: 1_048_575, output: 32_768 }, 'qwen3.8-flash': { context: 262_144, output: 32_768 }, - 'mimo-v2.5': { context: 1_048_576, output: 32_768 }, 'mimo-v2.6-flash': { context: 1_048_576, output: 32_768 }, 'glm5.3-flash': { context: 1_048_576, output: 32_768 }, }; @@ -451,7 +451,7 @@ describe.each(LOCALES)('models.json published in %s', (locale) => { /** * Pi's schema for `input` is `("text" | "image")[]`. A third value does not * fail the one model: Pi refuses the whole file, with every other provider - * in it. So mimo-v2.5 is published without its audio, and the page says so. + * in it. So mimo-v2.6-flash is published without its audio, and the page says so. */ test('no model declares an input outside Pi schema', () => { for (const m of piModels(locale).providers.nan.models as any[]) {