diff --git a/src/components/docs/ApiReference.astro b/src/components/docs/ApiReference.astro
index 820b360..3ba0a45 100644
--- a/src/components/docs/ApiReference.astro
+++ b/src/components/docs/ApiReference.astro
@@ -40,12 +40,12 @@ const t = {
en: {
title: 'API Reference · NaN Docs',
description:
- 'Interactive reference for the NaN API: OpenAI-compatible endpoints for chat, embeddings, rerank, audio, images and web search.',
+ 'Interactive reference for the NaN API: OpenAI-compatible endpoints for chat, embeddings, rerank, audio and images.',
},
es: {
title: 'Referencia API · NaN Docs',
description:
- 'Referencia interactiva de la API de NaN: endpoints compatibles con OpenAI para chat, embeddings, rerank, audio, imágenes y búsqueda web.',
+ 'Referencia interactiva de la API de NaN: endpoints compatibles con OpenAI para chat, embeddings, rerank, audio e imágenes.',
},
}[lang];
@@ -162,8 +162,7 @@ const CSP = [
// "Generate MCP" is a hosted-product feature of Scalar: it queries
// their registry at api.scalar.com on every load. Turned off here and
// not merely by hiding the button with CSS, because hiding it does not
- // stop the request. Nothing to do with NaN's /mcp endpoint, which IS
- // documented: that is a spec route, this is a Scalar integration.
+ // stop the request.
mcp: { disabled: true },
// "Ask AI" is Scalar's hosted assistant. Hiding its button is not
// enough: the component mounts anyway and queries api.scalar.com on
diff --git a/src/content/docs-es/agent-setup.mdx b/src/content/docs-es/agent-setup.mdx
index edc1c00..ebd74b0 100644
--- a/src/content/docs-es/agent-setup.mdx
+++ b/src/content/docs-es/agent-setup.mdx
@@ -115,12 +115,6 @@ Cada página de esta sección tiene la misma forma: qué necesitas, la configura
kind: 'Terminal',
note: 'Configura OpenCode, Codex, Pi, droid y Hermes por ti, y te enseña tu consumo.',
},
- {
- name: 'Servidor MCP',
- href: '/es/docs/mcp',
- kind: 'Herramientas',
- note: 'La búsqueda web de NaN dentro de cualquier agente compatible con MCP.',
- },
]}
/>
diff --git a/src/content/docs-es/agents.md b/src/content/docs-es/agents.md
index 54655ce..ddf0133 100644
--- a/src/content/docs-es/agents.md
+++ b/src/content/docs-es/agents.md
@@ -1,7 +1,7 @@
---
title: Agentes
description: "Despliega agentes de IA en una microVM aislada con QEMU: Hermes, terminal web, subida de ficheros y observabilidad."
-order: 20
+order: 19
group: Guías
---
@@ -9,9 +9,6 @@ group: Guías
NaN Cloud te permite desplegar agentes de IA en tu propia **microVM**: una máquina virtual ligera con QEMU y KVM, con su propio kernel, su propio sistema de ficheros y acceso root completo. Aislada del host y del resto de miembros. El primer tipo de agente disponible es **Hermes**.
-> **¿Usas un agente que alojas tú?**
-> Si ejecutas tu propio agente compatible con MCP en otro sitio, puedes enchufarle nuestras herramientas (como la búsqueda web) directamente, con la misma API key, a través de nuestro [servidor MCP](/es/docs/api#tag/mcp) remoto.
-
## Arquitectura
Cada agente corre dentro de su propia microVM de QEMU. En vez de compartir el kernel del host (como haría un contenedor normal), arranca con su propio kernel de Linux. La VM monta un disco ext4 de 20 GiB sobre un volumen persistente en modo bloque. Todo lo que hagas dentro (`apt install`, `pip install`, cambios en `/etc`, ficheros que subas) vive en ese disco y sobrevive a los reinicios.
diff --git a/src/content/docs-es/apps.md b/src/content/docs-es/apps.md
index 5140472..09529d0 100644
--- a/src/content/docs-es/apps.md
+++ b/src/content/docs-es/apps.md
@@ -1,7 +1,7 @@
---
title: Apps
description: Despliega tus apps desde GitHub a NaN Cloud en minutos.
-order: 21
+order: 20
group: Guías
---
diff --git a/src/content/docs-es/claude-code.mdx b/src/content/docs-es/claude-code.mdx
index 48699d3..b2a94b2 100644
--- a/src/content/docs-es/claude-code.mdx
+++ b/src/content/docs-es/claude-code.mdx
@@ -198,7 +198,3 @@ En cualquiera de los dos, evita `glm5.3` para tareas que impliquen leer capturas
- **El comportamiento del agente depende del modelo.** Claude Code está afinado contra los modelos de Anthropic, y un modelo abierto puede usar peor sus herramientas. Es esperable, y no es un fallo del clúster. Esto solo afecta al camino de la pasarela: delegando, Claude Code sigue siendo Claude Code.
- **Las funciones atadas a la cuenta de Anthropic no viajan.** Lo que dependa de la infraestructura de Anthropic no funciona contra otra base URL.
- **El contexto se llena rápido.** Una sesión de agente consume tokens mucho más deprisa que un chat. Si usas `glm5.3`, vigila la ventana móvil de 4 horas descrita en [Elige tu modelo](/es/docs/choose-a-model).
-
-## Las herramientas de NaN, aparte
-
-Independientemente del camino que elijas, puedes darle a Claude Code la búsqueda web del clúster con una sola orden. Está en [Servidor MCP](/es/docs/mcp).
diff --git a/src/content/docs-es/examples.md b/src/content/docs-es/examples.md
index 5a06d90..41d7e2a 100644
--- a/src/content/docs-es/examples.md
+++ b/src/content/docs-es/examples.md
@@ -1,7 +1,7 @@
---
title: Ejemplos
description: Fragmentos de código para conectarte a la API de NaN con Python, Node.js, curl y más.
-order: 19
+order: 18
group: Guías
---
@@ -474,52 +474,6 @@ console.log(image.data[0].url);
Las imágenes necesitan membresía de inferencia, `403` si no la tienes, y van por su propio presupuesto: 20 peticiones por minuto y 100 al mes, que no toca tu cuota de tokens.
-## tool: web search
-
-búsqueda web autenticada para agentes: `POST /v1/search`
-
-### curl
-
-```bash
-curl https://api.nan.builders/v1/search \
- -H "Content-Type: application/json" \
- -H "Authorization: Bearer sk-your-key-here" \
- -d '{
- "query": "latest go release",
- "count": 5,
- "freshness": "pw"
- }'
-# → {"results":[{"title":...,"url":...,"snippet":...,"source":...}],"cached":false}
-```
-
-### python
-
-```python
-import os
-from openai import OpenAI
-
-client = OpenAI(
- api_key=os.environ["NAN_API_KEY"],
- base_url="https://api.nan.builders/v1"
-)
-
-# /search is not part of the standard OpenAI client, but we can invoke it with client.post().
-response = client.post(
- path="/search",
- cast_to=object,
- body={
- "query": "latest go release",
- "count": 5,
- "freshness": "pw",
- },
-)
-
-for r in response["results"]:
- print(r["title"], "-", r["url"])
-```
-
-También funciona con `requests` a pelo o con cualquier cliente HTTP: manda el cuerpo JSON con tu key en el Bearer.
-
## Conectar tu editor o tu agente
Las configuraciones de Cursor, Claude Code, Codex, Cline, OpenCode, Zed y el resto están en [Configurar tu agente](/es/docs/agent-setup), con una página por herramienta.
diff --git a/src/content/docs-es/gentle-ai.mdx b/src/content/docs-es/gentle-ai.mdx
index 640e652..1d6a94e 100644
--- a/src/content/docs-es/gentle-ai.mdx
+++ b/src/content/docs-es/gentle-ai.mdx
@@ -131,5 +131,4 @@ Conviene pasar `sync` después de cada `upgrade`: lo primero actualiza el progra
## Siguientes pasos
- [Configurar tu agente](/es/docs/agent-setup): conecta primero tu herramienta al clúster.
-- [Servidor MCP](/es/docs/mcp): añade además la búsqueda web de NaN a tu agente.
- [Elige tu modelo](/es/docs/choose-a-model): qué modelo pedir para cada tarea.
diff --git a/src/content/docs-es/getting-started.mdx b/src/content/docs-es/getting-started.mdx
index 7e16d23..75ea971 100644
--- a/src/content/docs-es/getting-started.mdx
+++ b/src/content/docs-es/getting-started.mdx
@@ -109,7 +109,7 @@ const resp = await client.chat.completions.create({
console.log(resp.choices[0].message.content);
```
-Tienes más ejemplos, incluidos embeddings, voz, imágenes y búsqueda web, en [Ejemplos](/es/docs/examples).
+Tienes más ejemplos, incluidos embeddings, voz e imágenes, en [Ejemplos](/es/docs/examples).
### Elige el modelo
@@ -153,7 +153,7 @@ El detalle completo, endpoint a endpoint, está en la [referencia de la API](/es
Dos límites van por API key y se aplican a todo lo que llames: peticiones por minuto y peticiones a la vez.
-Encima de esos, algunos modelos llevan el suyo: un techo de tokens por minuto en los modelos de chat grandes, y uno de peticiones por minuto en `rerank`. `glm5.3` no se rige por minuto en absoluto, sino por una ventana móvil de tokens más una asignación por periodo de facturación. La búsqueda web va por un presupuesto separado del de los modelos.
+Encima de esos, algunos modelos llevan el suyo: un techo de tokens por minuto en los modelos de chat grandes, y uno de peticiones por minuto en `rerank`. `glm5.3` no se rige por minuto en absoluto, sino por una ventana móvil de tokens más una asignación por periodo de facturación. Los endpoints de imágenes van por un presupuesto separado del de los modelos.
Las cifras vigentes están al final de [Modelos](/es/docs/models), que es donde se publican para que no haya dos versiones de la misma cifra dando vueltas.
diff --git a/src/content/docs-es/intro.md b/src/content/docs-es/intro.md
index 2e916cc..1e8c728 100644
--- a/src/content/docs-es/intro.md
+++ b/src/content/docs-es/intro.md
@@ -28,7 +28,6 @@ Los límites van por API key — un tope de peticiones por minuto y un máximo d
- [Elige tu modelo](/es/docs/choose-a-model): qué modelo pedir para cada tarea y cómo se escribe su id.
- [Configurar tu agente](/es/docs/agent-setup): Claude Code, Codex, Cursor, Cline, OpenCode, Zed y compañía.
- [CLI de NaN](/es/docs/nan-cli): la herramienta oficial de terminal, que además configura varias de ellas por ti.
-- [Servidor MCP](/es/docs/mcp): la búsqueda web de NaN dentro de tu agente.
- [Referencia de la API](/es/docs/api): todos los endpoints, campo a campo.
- [Modelos](/es/docs/models): las fichas técnicas y los límites.
- [Ejemplos](/es/docs/examples): fragmentos en Python, Node.js y curl.
diff --git a/src/content/docs-es/mcp.mdx b/src/content/docs-es/mcp.mdx
deleted file mode 100644
index e8f3a39..0000000
--- a/src/content/docs-es/mcp.mdx
+++ /dev/null
@@ -1,124 +0,0 @@
----
-title: Servidor MCP
-description: Enchufa la búsqueda web de NaN en cualquier agente compatible con MCP.
-order: 16
-group: Configurar tu agente
----
-
-import Details from '../../components/docs/Details.astro';
-
-# Servidor MCP.
-
-Además de los modelos, NaN expone sus propias herramientas a través de un servidor **MCP** ([Model Context Protocol](https://modelcontextprotocol.io)). Sirve para lo contrario que el resto de esta sección: aquí no le das modelos a tu agente, le das **capacidades**.
-
-Hoy la herramienta disponible es la **búsqueda web**. El registro irá creciendo, así que pregúntale al servidor qué tiene en vez de fiarte de esta frase.
-
-| Campo | Valor |
-|---|---|
-| URL | `https://api.nan.builders/mcp` |
-| Autenticación | `Authorization: Bearer sk-tu-clave` |
-| Transporte | HTTP, sin estado |
-
-> **Esta URL no lleva `/v1`**
-> El servidor MCP vive en la raíz del dominio, no debajo de `/v1` como el resto de la API. Es `https://api.nan.builders/mcp`, sin nada más.
-
-La clave es la misma que usas para los modelos. No hay que generar otra.
-
-## Configuración genérica
-
-Casi todos los clientes MCP usan este mismo fichero, con este mismo bloque:
-
-```json
-{
- "mcpServers": {
- "nan": {
- "url": "https://api.nan.builders/mcp",
- "headers": {
- "Authorization": "Bearer sk-tu-clave"
- }
- }
- }
-}
-```
-
-Dónde va ese fichero depende del cliente:
-
-| Cliente | Dónde |
-|---|---|
-| Cursor | `.cursor/mcp.json` en el proyecto, o `~/.cursor/mcp.json` |
-| Cline | El panel de MCP Servers dentro de la extensión |
-| Zed | El bloque `context_servers` de `~/.config/zed/settings.json` |
-| OpenCode | El bloque `mcp` de tu `opencode.json` |
-
-## Claude Code
-
-Claude Code lo añade por línea de comandos:
-
-```bash
-claude mcp add --transport http nan https://api.nan.builders/mcp \
- --header "Authorization: Bearer sk-tu-clave"
-```
-
-Añade `--scope user` si lo quieres disponible en todos tus proyectos y no solo en el actual. Dentro de una sesión, `/mcp` te enseña los servidores conectados y las herramientas que ofrecen.
-
-Esto es independiente de los modelos: puedes usar la búsqueda web de NaN desde Claude Code aunque los modelos te los sirva otro sitio.
-
-## Comprueba que funciona
-
-Sin cliente de por medio, preguntándole al servidor directamente qué herramientas tiene:
-
-```bash
-curl https://api.nan.builders/mcp \
- -H "Authorization: Bearer $NAN_API_KEY" \
- -H "Content-Type: application/json" \
- -d '{ "jsonrpc": "2.0", "id": 1, "method": "tools/list" }'
-```
-
-La respuesta trae la lista de herramientas con sus argumentos. Ahí verás siempre el conjunto real, que es más fiable que cualquier lista escrita a mano.
-
-Y una búsqueda de verdad:
-
-```bash
-curl https://api.nan.builders/mcp \
- -H "Authorization: Bearer $NAN_API_KEY" \
- -H "Content-Type: application/json" \
- -d '{
- "jsonrpc": "2.0",
- "id": 2,
- "method": "tools/call",
- "params": {
- "name": "web_search",
- "arguments": { "query": "kubernetes 1.34 release", "count": 5 }
- }
- }'
-```
-
-## La herramienta `web_search`
-
-Acepta los mismos argumentos que el endpoint [`POST /v1/search`](/es/docs/api):
-
-| Argumento | Qué hace |
-|---|---|
-| `query` | La búsqueda. Es el único obligatorio |
-| `count` | Cuántos resultados, de 1 a 20. Por defecto 5 |
-| `freshness` | Filtro de antigüedad: `pd` día, `pw` semana, `pm` mes, `py` año |
-| `fetch_content` | Con `true`, además del resumen trae el texto de las páginas. Tarda más |
-
-Las búsquedas salen por NaN, así que tu clave nunca habla con un buscador externo y no necesitas darte de alta en ninguno.
-
-## Límites
-
-La búsqueda web tiene su propio presupuesto, separado del de los modelos: **20 peticiones por minuto, 3 a la vez y 500 búsquedas al día** por clave. Buscar no gasta tu cuota de chat ni al revés.
-
-Da igual si la llamada entra por MCP o por `POST /v1/search`: cuenta lo mismo en el mismo contador. Una búsqueda repetida en los 15 minutos siguientes se sirve de una caché corta y llega marcada con `cached: true`, pero sigue contando.
-
-Si te pasas, la respuesta es un `429` con una cabecera `Retry-After` que te dice cuánto esperar.
-
-
-
-- **El servidor no guarda estado.** Cada petición es independiente y lleva su propia autenticación. No hay sesión que mantener abierta.
-- **Algunos clientes no reenvían las cabeceras que declaras** en todas las fases de la conexión. Si el cliente se conecta pero luego falla al llamar a una herramienta con un error de autenticación, suele ser eso, y no tu clave.
-- **La lista de herramientas cambia.** Usa `tools/list` antes de dar por hecho que una herramienta existe.
-
-
-
diff --git a/src/content/docs-es/models.mdx b/src/content/docs-es/models.mdx
index d230870..52b2966 100644
--- a/src/content/docs-es/models.mdx
+++ b/src/content/docs-es/models.mdx
@@ -1,7 +1,7 @@
---
title: Modelos
description: Especificaciones técnicas, capacidades y parámetros de los modelos del clúster compartido.
-order: 18
+order: 17
group: Referencia
---
diff --git a/src/content/docs/agent-setup.mdx b/src/content/docs/agent-setup.mdx
index c2d6c15..2db131c 100644
--- a/src/content/docs/agent-setup.mdx
+++ b/src/content/docs/agent-setup.mdx
@@ -115,12 +115,6 @@ Yours is not here? If it accepts an OpenAI base URL and an API key, it works. Co
kind: 'Terminal',
note: 'It configures OpenCode, Codex, Pi, droid and Hermes for you, and shows you your usage.',
},
- {
- name: 'MCP server',
- href: '/docs/mcp',
- kind: 'Tools',
- note: 'NaN\'s web search inside any MCP-compatible agent.',
- },
]}
/>
diff --git a/src/content/docs/agents.md b/src/content/docs/agents.md
index cc3f17d..a4beabd 100644
--- a/src/content/docs/agents.md
+++ b/src/content/docs/agents.md
@@ -1,7 +1,7 @@
---
title: Agents
description: "Deploy AI agents in an isolated microVM with QEMU: Hermes, web terminal, file uploads, and observability."
-order: 20
+order: 19
group: Guides
---
@@ -9,9 +9,6 @@ group: Guides
NaN Cloud lets you deploy AI agents in your own **microVM**: a lightweight virtual machine with QEMU + KVM, its own kernel, its own filesystem, and full root access. Isolated from the host and from other members. The first available agent type is **Hermes**.
-> **Using an agent you host yourself?**
-> If you run your own MCP-compatible agent elsewhere, you can plug our tools (such as web search) straight into it with the same API key via our remote [MCP server](/docs/api#tag/mcp).
-
## Architecture
Each agent runs inside its own QEMU microVM. Instead of sharing the host kernel (like a regular container), it starts with its own Linux kernel. The VM mounts a 20 GiB ext4 disk on a block-mode persistent volume. Everything you do inside — `apt install`, `pip install`, edits to `/etc`, files you upload — lives on that disk and survives restarts.
diff --git a/src/content/docs/apps.md b/src/content/docs/apps.md
index f5f4d36..1dd74bf 100644
--- a/src/content/docs/apps.md
+++ b/src/content/docs/apps.md
@@ -1,7 +1,7 @@
---
title: Apps
description: Deploy your apps from GitHub to NaN Cloud in minutes.
-order: 21
+order: 20
group: Guides
---
diff --git a/src/content/docs/claude-code.mdx b/src/content/docs/claude-code.mdx
index da5f3ee..5fc5cbb 100644
--- a/src/content/docs/claude-code.mdx
+++ b/src/content/docs/claude-code.mdx
@@ -198,7 +198,3 @@ On either path, avoid `glm5.3` for tasks that involve reading screenshots: it do
- **Agent behavior depends on the model.** Claude Code is tuned against Anthropic's models, and an open model may use its tools less well. That is to be expected, and it is not a cluster failure. This only affects the gateway path: when delegating, Claude Code is still Claude Code.
- **Features tied to the Anthropic account do not travel.** Anything that depends on Anthropic's infrastructure does not work against another base URL.
- **Context fills up fast.** An agent session eats tokens far quicker than a chat. If you use `glm5.3`, watch the rolling 4-hour window described in [Choose your model](/docs/choose-a-model).
-
-## NaN's tools, separately
-
-Whichever path you choose, you can give Claude Code the cluster's web search with a single command. It is in [MCP server](/docs/mcp).
diff --git a/src/content/docs/examples.md b/src/content/docs/examples.md
index 776ec1c..b7a9cc2 100644
--- a/src/content/docs/examples.md
+++ b/src/content/docs/examples.md
@@ -1,7 +1,7 @@
---
title: Examples
description: Code snippets to connect to the NaN API with Python, Node.js, curl, and more.
-order: 19
+order: 18
group: Guides
---
@@ -474,52 +474,6 @@ console.log(image.data[0].url);
Images need inference membership, `403` otherwise, and they run on their own budget: 20 requests per minute and 100 per month, which does not touch your token quota.
-## tool: web search
-
-authenticated web search for agents — `POST /v1/search`
-
-### curl
-
-```bash
-curl https://api.nan.builders/v1/search \
- -H "Content-Type: application/json" \
- -H "Authorization: Bearer sk-your-key-here" \
- -d '{
- "query": "latest go release",
- "count": 5,
- "freshness": "pw"
- }'
-# → {"results":[{"title":...,"url":...,"snippet":...,"source":...}],"cached":false}
-```
-
-### python
-
-```python
-import os
-from openai import OpenAI
-
-client = OpenAI(
- api_key=os.environ["NAN_API_KEY"],
- base_url="https://api.nan.builders/v1"
-)
-
-# /search is not part of the standard OpenAI client, but we can invoke it with client.post().
-response = client.post(
- path="/search",
- cast_to=object,
- body={
- "query": "latest go release",
- "count": 5,
- "freshness": "pw",
- },
-)
-
-for r in response["results"]:
- print(r["title"], "-", r["url"])
-```
-
-Also works with raw `requests` or any HTTP client — send the JSON body with your Bearer key.
-
## Connect your editor or your agent
The configurations for Cursor, Claude Code, Codex, Cline, OpenCode, Zed and the rest are in [Set up your agent](/docs/agent-setup), with one page per tool.
diff --git a/src/content/docs/gentle-ai.mdx b/src/content/docs/gentle-ai.mdx
index 052576a..d687e59 100644
--- a/src/content/docs/gentle-ai.mdx
+++ b/src/content/docs/gentle-ai.mdx
@@ -131,5 +131,4 @@ It is worth running `sync` after every `upgrade`: the first updates the program,
## Next steps
- [Set up your agent](/docs/agent-setup): connect your tool to the cluster first.
-- [MCP server](/docs/mcp): add NaN's web search to your agent as well.
- [Choose your model](/docs/choose-a-model): which model to ask for each task.
diff --git a/src/content/docs/getting-started.mdx b/src/content/docs/getting-started.mdx
index 089d691..6d649ce 100644
--- a/src/content/docs/getting-started.mdx
+++ b/src/content/docs/getting-started.mdx
@@ -109,7 +109,7 @@ const resp = await client.chat.completions.create({
console.log(resp.choices[0].message.content);
```
-There are more examples, including embeddings, speech, images and web search, in [Examples](/docs/examples).
+There are more examples, including embeddings, speech and images, in [Examples](/docs/examples).
### Choose your model
@@ -153,7 +153,7 @@ The full detail, endpoint by endpoint, is in the [API reference](/docs/api).
Two limits are per API key and apply to everything you call: requests per minute, and requests at once.
-On top of those, some models carry one of their own: a tokens-per-minute ceiling on the big chat models, and a requests-per-minute one on `rerank`. `glm5.3` is not gated per minute at all, but by a rolling token window plus an allowance per billing period. Web search runs on a budget separate from the models'.
+On top of those, some models carry one of their own: a tokens-per-minute ceiling on the big chat models, and a requests-per-minute one on `rerank`. `glm5.3` is not gated per minute at all, but by a rolling token window plus an allowance per billing period. Image endpoints run on a budget separate from the models'.
The current figures are at the end of [Models](/docs/models), which is where they are published so that no two versions of the same number go around.
diff --git a/src/content/docs/intro.md b/src/content/docs/intro.md
index 8166ab8..fe7eba1 100644
--- a/src/content/docs/intro.md
+++ b/src/content/docs/intro.md
@@ -28,7 +28,6 @@ Limits are per API key — a cap on requests per minute and a maximum number of
- [Choose your model](/docs/choose-a-model): which model to ask for each task, and how its id is spelled.
- [Set up your agent](/docs/agent-setup): Claude Code, Codex, Cursor, Cline, OpenCode, Zed and company.
- [NaN CLI](/docs/nan-cli): the official terminal tool, which also configures several of them for you.
-- [MCP server](/docs/mcp): NaN's web search inside your agent.
- [API reference](/docs/api): every endpoint, field by field.
- [Models](/docs/models): the spec sheets and the limits.
- [Examples](/docs/examples): snippets in Python, Node.js and curl.
diff --git a/src/content/docs/mcp.mdx b/src/content/docs/mcp.mdx
deleted file mode 100644
index 924be65..0000000
--- a/src/content/docs/mcp.mdx
+++ /dev/null
@@ -1,123 +0,0 @@
----
-title: MCP server
-description: Plug NaN's web search into any MCP-compatible agent.
-order: 16
-group: Set up your agent
----
-
-import Details from '../../components/docs/Details.astro';
-
-# MCP server.
-
-On top of the models, NaN exposes its own tools through an **MCP** server ([Model Context Protocol](https://modelcontextprotocol.io)). It is for the opposite of the rest of this section: here you are not giving your agent models, you are giving it **capabilities**.
-
-Today the available tool is **web search**. The registry will grow, so ask the server what it has rather than trusting this sentence.
-
-| Field | Value |
-|---|---|
-| URL | `https://api.nan.builders/mcp` |
-| Authentication | `Authorization: Bearer sk-your-key` |
-| Transport | HTTP, stateless |
-
-> **This URL has no `/v1`**
-> The MCP server lives at the root of the domain, not under `/v1` like the rest of the API. It is `https://api.nan.builders/mcp`, nothing else.
-
-The key is the same one you use for the models. There is no second key to generate.
-
-## Generic configuration
-
-Almost every MCP client uses this same file, with this same block:
-
-```json
-{
- "mcpServers": {
- "nan": {
- "url": "https://api.nan.builders/mcp",
- "headers": {
- "Authorization": "Bearer sk-your-key"
- }
- }
- }
-}
-```
-
-Where that file goes depends on the client:
-
-| Client | Where |
-|---|---|
-| Cursor | `.cursor/mcp.json` in the project, or `~/.cursor/mcp.json` |
-| Cline | The MCP Servers panel inside the extension |
-| Zed | The `context_servers` block of `~/.config/zed/settings.json` |
-| OpenCode | The `mcp` block of your `opencode.json` |
-
-## Claude Code
-
-Claude Code adds it from the command line:
-
-```bash
-claude mcp add --transport http nan https://api.nan.builders/mcp \
- --header "Authorization: Bearer sk-your-key"
-```
-
-Add `--scope user` if you want it available in every project and not only the current one. Inside a session, `/mcp` shows you the connected servers and the tools they offer.
-
-This is independent of the models: you can use NaN's web search from Claude Code even if your models come from somewhere else.
-
-## Check that it works
-
-With no client in between, by asking the server directly which tools it has:
-
-```bash
-curl https://api.nan.builders/mcp \
- -H "Authorization: Bearer $NAN_API_KEY" \
- -H "Content-Type: application/json" \
- -d '{ "jsonrpc": "2.0", "id": 1, "method": "tools/list" }'
-```
-
-The answer carries the list of tools with their arguments. That is where you always see the real set, which is more reliable than any hand-written list.
-
-And a real search:
-
-```bash
-curl https://api.nan.builders/mcp \
- -H "Authorization: Bearer $NAN_API_KEY" \
- -H "Content-Type: application/json" \
- -d '{
- "jsonrpc": "2.0",
- "id": 2,
- "method": "tools/call",
- "params": {
- "name": "web_search",
- "arguments": { "query": "kubernetes 1.34 release", "count": 5 }
- }
- }'
-```
-
-## The `web_search` tool
-
-It takes the same arguments as the [`POST /v1/search`](/docs/api) endpoint:
-
-| Argument | What it does |
-|---|---|
-| `query` | The search. The only mandatory one |
-| `count` | How many results, from 1 to 20. Default 5 |
-| `freshness` | Age filter: `pd` day, `pw` week, `pm` month, `py` year |
-| `fetch_content` | With `true`, it brings the text of the pages as well as the summary. It takes longer |
-
-Searches go out through NaN, so your key never talks to an external search engine and you do not need to sign up for one.
-
-## Limits
-
-Web search has its own budget, separate from the models': **20 requests per minute, 3 at once and 500 searches a day** per key. Searching does not spend your chat quota, nor the other way around.
-
-It makes no difference whether the call comes in through MCP or through `POST /v1/search`: it counts the same, on the same counter. A search repeated within the next 15 minutes is served from a short cache and arrives marked `cached: true`, but it still counts.
-
-If you go over, the answer is a `429` with a `Retry-After` header telling you how long to wait.
-
-
-
-- **The server keeps no state.** Every request is independent and carries its own authentication. There is no session to keep open.
-- **Some clients do not forward the headers you declare** at every phase of the connection. If the client connects but then fails to call a tool with an authentication error, that is usually why, and not your key.
-- **The tool list changes.** Use `tools/list` before assuming a tool exists.
-
-
diff --git a/src/content/docs/models.mdx b/src/content/docs/models.mdx
index 8e1c146..21143c3 100644
--- a/src/content/docs/models.mdx
+++ b/src/content/docs/models.mdx
@@ -1,7 +1,7 @@
---
title: Models
description: Technical specifications, capabilities, and parameters of the shared cluster models.
-order: 18
+order: 17
group: Reference
---
diff --git a/src/data/openapi.json b/src/data/openapi.json
index 790fa90..b658085 100644
--- a/src/data/openapi.json
+++ b/src/data/openapi.json
@@ -3,7 +3,7 @@
"info": {
"title": "NaN API",
"version": "1.0.0",
- "description": "Open models on a shared EU inference cluster. Zero logs.\n\nThe NaN API is OpenAI-compatible: predictable, resource-oriented URLs, JSON request and response bodies, and standard HTTP verbs and status codes. Point any OpenAI SDK at our base URL and your existing code keeps working. Change the base URL and the API key, and that's it.\n\nOne schema across every model, so you only learn the API once. Change the `model` field to switch models; everything else stays the same.\n\n- Base URL: `https://api.nan.builders/v1`\n- OpenAPI spec: this document. Import it into Postman, Insomnia, or your own tooling.\n\nIf you use the [Helmcode](https://helmcode.com) enterprise service, the base URL is `https://api.helmcode.com/v1` instead. Every other endpoint is identical.\n\n## Authentication\n\nEvery request authenticates with an API key, sent as a Bearer token:\n\n```\nAuthorization: Bearer $NAN_API_KEY\n```\n\nYou must be a NaN community member. Generate your key from user settings, under \"API Keys\", on the [platform](https://cloud.nan.builders/). The key is personal and non-transferable. Keep it secret: never embed one in client-side code or commit it to source control. Requests must go over HTTPS; calls over plain HTTP fail.\n\n## Making requests\n\nThe API is OpenAI-compatible, so point an official OpenAI SDK at our base URL and change nothing else:\n\n```python\nfrom openai import OpenAI\n\nclient = OpenAI(\n api_key=\"$NAN_API_KEY\",\n base_url=\"https://api.nan.builders/v1\",\n)\n\nresp = client.chat.completions.create(\n model=\"deepseek-v4-flash\",\n messages=[{\"role\": \"user\", \"content\": \"Hello\"}],\n)\nprint(resp.choices[0].message.content)\n```\n\n## Streaming\n\nChat responses can stream token-by-token. Set `\"stream\": true` on `/chat/completions` and the response arrives as Server-Sent Events: each event is a `data:` line carrying a `chat.completion.chunk`, with the new text in `choices[0].delta.content`. A final `data: [DONE]` line ends the stream. Only `/chat/completions` streams incrementally; `/responses` currently emits a single terminal event.\n\n## Rate limits\n\n{{RATE_LIMITS}}\n\nWeb search runs on its own budget, separate from the model endpoints: 20 requests per minute, 3 concurrent, and 500 searches per day per key. Image endpoints have their own too: 20 requests per minute and 100 requests per month. Exceed any limit and you get a `429`.\n\n## Errors\n\nNaN uses conventional HTTP status codes: `2xx` on success, `4xx` for a problem with the request (a missing parameter, an invalid key, an unavailable model) and `5xx` for a server-side error. Every error returns a JSON body in the OpenAI shape:\n\n```json\n{\n \"error\": {\n \"message\": \"The model 'foo' does not exist.\",\n \"type\": \"invalid_request_error\",\n \"param\": \"model\",\n \"code\": \"model_not_found\"\n }\n}\n```\n\n`message` is human-readable, `param` names the offending field when applicable, and `code` is a short machine-readable string you can branch on.\n\n| Status | Meaning | `code` |\n| --- | --- | --- |\n| `400` | Invalid or malformed parameter (`param` says which); or content blocked by the safety filter. | `invalid_request_error` · `content_policy_violation` |\n| `401` | Missing or invalid API key, or a key whose tier does not reach the requested model (`glm5.3`): \"This API key does not have access to the requested model\", `type: auth_error`. Measured 2026-09-12. | `invalid_api_key` |\n| `402` | The token allowance is spent on a model that carries one. Not retryable: the counter returns to zero when that model's quota period does, the calendar month for the models counted per month and your billing period for `glm5.3`. | `monthly_cap_reached` |\n| `403` | Your tier can't access this endpoint. Image generation requires inference membership. A model your tier cannot reach answers `401`, not this. | `tier_restricted` |\n| `404` | The requested model doesn't exist. | `model_not_found` |\n| `429` | Rate limit hit (`rpm_limit`, `max_parallel_requests`), the rolling 4h token budget of `glm5.3`, or a quota exhausted. | `rate_limit_exceeded` · `insufficient_quota` · `quota_exceeded` |\n| `500` | Something went wrong on our side (includes upstream model errors). | (none) |\n| `503` | Web search is temporarily unavailable; retry shortly. | `search_unavailable` |\n| `524` | Timeout, typical with large audio files on `/audio/transcriptions`. | (none) |\n\nRetry `429` and `5xx` responses with exponential backoff. Don't retry `400`, `401`, `403`, or `404` blindly: they'll fail the same way every time until you change the request. `402` cannot be fixed by repetition either: it clears when that model's quota period resets.\n\n## Model catalog\n\nEvery endpoint takes a `model` id. Capabilities vary by model:\n\n| Model | Use for | Capabilities |\n| --- | --- | --- |\n| `deepseek-v4-flash` | Chat, vision, reasoning | Streaming, tool calling, reasoning, image input, 1M-token context. 3B tokens/month per member |\n| `mimo-v2.5` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `qwen3.8-flash` | Chat, vision, agents | Streaming, tool calling, reasoning (on by default), vision, 262K-token context. 500M tokens/month per member |\n| `glm5.3-flash` | Chat, vision, agents | Streaming, tool calling, reasoning, vision, 1M-token context. 2B tokens/month per member |\n| `qwen3.6` | Chat, agents | Streaming, tool calling, vision, reasoning (opt-out, returns `reasoning_content`) |\n| `gemma4` | Chat, vision, agents | Streaming, tool calling, vision, reasoning (opt-in) |\n| `glm5.3` | Coding, long-horizon agents | Streaming, tool calling, reasoning trace, text-only input, 1M-token context. Premium tier only |\n| `qwen3-embedding` | Embeddings | 4096-dimension vectors |\n| `rerank` | RAG reranking | Qwen3-Reranker-8B, 100+ languages |\n| `kokoro` | Text-to-speech | Multiple voices and audio formats |\n| `whisper` | Speech-to-text | Transcription with word/segment timestamps |\n| `flux-2-klein` | Image generation | Text-to-image and image-to-image |\n\n`glm5.3` is served only to keys on the GLM 5.3 premium tier; every other model is available to any inference member. Call [List models](#tag/Models) for the exact set available to your key.\n\n## Versioning & compatibility\n\nThe API tracks the OpenAI API surface, so OpenAI SDKs and tools work against `https://api.nan.builders/v1` unchanged. This reference documents the stable public `/v1` endpoints, and we add capabilities without breaking existing fields.",
+ "description": "Open models on a shared EU inference cluster. Zero logs.\n\nThe NaN API is OpenAI-compatible: predictable, resource-oriented URLs, JSON request and response bodies, and standard HTTP verbs and status codes. Point any OpenAI SDK at our base URL and your existing code keeps working. Change the base URL and the API key, and that's it.\n\nOne schema across every model, so you only learn the API once. Change the `model` field to switch models; everything else stays the same.\n\n- Base URL: `https://api.nan.builders/v1`\n- OpenAPI spec: this document. Import it into Postman, Insomnia, or your own tooling.\n\nIf you use the [Helmcode](https://helmcode.com) enterprise service, the base URL is `https://api.helmcode.com/v1` instead. Every other endpoint is identical.\n\n## Authentication\n\nEvery request authenticates with an API key, sent as a Bearer token:\n\n```\nAuthorization: Bearer $NAN_API_KEY\n```\n\nYou must be a NaN community member. Generate your key from user settings, under \"API Keys\", on the [platform](https://cloud.nan.builders/). The key is personal and non-transferable. Keep it secret: never embed one in client-side code or commit it to source control. Requests must go over HTTPS; calls over plain HTTP fail.\n\n## Making requests\n\nThe API is OpenAI-compatible, so point an official OpenAI SDK at our base URL and change nothing else:\n\n```python\nfrom openai import OpenAI\n\nclient = OpenAI(\n api_key=\"$NAN_API_KEY\",\n base_url=\"https://api.nan.builders/v1\",\n)\n\nresp = client.chat.completions.create(\n model=\"deepseek-v4-flash\",\n messages=[{\"role\": \"user\", \"content\": \"Hello\"}],\n)\nprint(resp.choices[0].message.content)\n```\n\n## Streaming\n\nChat responses can stream token-by-token. Set `\"stream\": true` on `/chat/completions` and the response arrives as Server-Sent Events: each event is a `data:` line carrying a `chat.completion.chunk`, with the new text in `choices[0].delta.content`. A final `data: [DONE]` line ends the stream. Only `/chat/completions` streams incrementally; `/responses` currently emits a single terminal event.\n\n## Rate limits\n\n{{RATE_LIMITS}}\n\nImage endpoints run on their own budget, separate from the model endpoints: 20 requests per minute and 100 requests per month. Exceed any limit and you get a `429`.\n\n## Errors\n\nNaN uses conventional HTTP status codes: `2xx` on success, `4xx` for a problem with the request (a missing parameter, an invalid key, an unavailable model) and `5xx` for a server-side error. Every error returns a JSON body in the OpenAI shape:\n\n```json\n{\n \"error\": {\n \"message\": \"The model 'foo' does not exist.\",\n \"type\": \"invalid_request_error\",\n \"param\": \"model\",\n \"code\": \"model_not_found\"\n }\n}\n```\n\n`message` is human-readable, `param` names the offending field when applicable, and `code` is a short machine-readable string you can branch on.\n\n| Status | Meaning | `code` |\n| --- | --- | --- |\n| `400` | Invalid or malformed parameter (`param` says which); or content blocked by the safety filter. | `invalid_request_error` · `content_policy_violation` |\n| `401` | Missing or invalid API key, or a key whose tier does not reach the requested model (`glm5.3`): \"This API key does not have access to the requested model\", `type: auth_error`. Measured 2026-09-12. | `invalid_api_key` |\n| `402` | The token allowance is spent on a model that carries one. Not retryable: the counter returns to zero when that model's quota period does, the calendar month for the models counted per month and your billing period for `glm5.3`. | `monthly_cap_reached` |\n| `403` | Your tier can't access this endpoint. Image generation requires inference membership. A model your tier cannot reach answers `401`, not this. | `tier_restricted` |\n| `404` | The requested model doesn't exist. | `model_not_found` |\n| `429` | Rate limit hit (`rpm_limit`, `max_parallel_requests`), the rolling 4h token budget of `glm5.3`, or a quota exhausted. | `rate_limit_exceeded` · `insufficient_quota` · `quota_exceeded` |\n| `500` | Something went wrong on our side (includes upstream model errors). | (none) |\n| `524` | Timeout, typical with large audio files on `/audio/transcriptions`. | (none) |\n\nRetry `429` and `5xx` responses with exponential backoff. Don't retry `400`, `401`, `403`, or `404` blindly: they'll fail the same way every time until you change the request. `402` cannot be fixed by repetition either: it clears when that model's quota period resets.\n\n## Model catalog\n\nEvery endpoint takes a `model` id. Capabilities vary by model:\n\n| Model | Use for | Capabilities |\n| --- | --- | --- |\n| `deepseek-v4-flash` | Chat, vision, reasoning | Streaming, tool calling, reasoning, image input, 1M-token context. 3B tokens/month per member |\n| `mimo-v2.5` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `qwen3.8-flash` | Chat, vision, agents | Streaming, tool calling, reasoning (on by default), vision, 262K-token context. 500M tokens/month per member |\n| `glm5.3-flash` | Chat, vision, agents | Streaming, tool calling, reasoning, vision, 1M-token context. 2B tokens/month per member |\n| `qwen3.6` | Chat, agents | Streaming, tool calling, vision, reasoning (opt-out, returns `reasoning_content`) |\n| `gemma4` | Chat, vision, agents | Streaming, tool calling, vision, reasoning (opt-in) |\n| `glm5.3` | Coding, long-horizon agents | Streaming, tool calling, reasoning trace, text-only input, 1M-token context. Premium tier only |\n| `qwen3-embedding` | Embeddings | 4096-dimension vectors |\n| `rerank` | RAG reranking | Qwen3-Reranker-8B, 100+ languages |\n| `kokoro` | Text-to-speech | Multiple voices and audio formats |\n| `whisper` | Speech-to-text | Transcription with word/segment timestamps |\n| `flux-2-klein` | Image generation | Text-to-image and image-to-image |\n\n`glm5.3` is served only to keys on the GLM 5.3 premium tier; every other model is available to any inference member. Call [List models](#tag/Models) for the exact set available to your key.\n\n## Versioning & compatibility\n\nThe API tracks the OpenAI API surface, so OpenAI SDKs and tools work against `https://api.nan.builders/v1` unchanged. This reference documents the stable public `/v1` endpoints, and we add capabilities without breaking existing fields.",
"contact": {
"name": "NaN",
"url": "https://nan.builders"
@@ -53,14 +53,6 @@
{
"name": "Images",
"description": "Text-to-image and image-to-image."
- },
- {
- "name": "Search",
- "description": "Authenticated web search for agents."
- },
- {
- "name": "MCP",
- "description": "Remote MCP server for agents and MCP clients."
}
],
"paths": {
@@ -1398,182 +1390,6 @@
}
]
}
- },
- "/search": {
- "post": {
- "operationId": "search",
- "tags": [
- "Search"
- ],
- "summary": "Web search",
- "description": "An authenticated web search tool for agents. Give it a query and it returns ranked results (title, URL, snippet, source), drawn from a hybrid of upstream search providers. Results come back through the NaN API, so your key never talks to a third-party search provider directly, and you never hold a provider key.\n\nSet `fetch_content: true` to also extract the readable main text of the top results (adds latency). You supply only `query`; when `fetch_content` is `true` the server fetches page content **only** for the URLs the search returned, through a hardened anti-SSRF fetcher, so you cannot pass an arbitrary URL to fetch.\n\nLimits are per API key and separate from the model endpoints, so searching does not consume your chat RPM budget or vice versa: 20 requests per minute, 3 concurrent, and 500 searches per day. Exceeding the per-minute or concurrency limit returns `429` `rate_limit_exceeded`; exhausting the daily quota returns `429` `insufficient_quota`. A `429` carries a `Retry-After` header. Repeated identical queries within ~15 minutes are served from a short-lived cache (`cached: true`); cached hits still count toward your rate and quota.\n\n## Use it as an agent tool\n\nDrop this OpenAI-style function schema into your model call's `tools` array. When the model emits a `web_search` tool call, invoke `POST /v1/search` with the arguments and feed the JSON response back as the tool result.\n\n```json\n{\n \"type\": \"function\",\n \"function\": {\n \"name\": \"web_search\",\n \"description\": \"Search the public web and return relevant results (title, URL, snippet, and optional page content). Use it when the answer may depend on current events, recent developments, prices, release/version numbers, or facts you are not confident are up to date.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"query\": { \"type\": \"string\", \"description\": \"The web search query.\" },\n \"count\": { \"type\": \"integer\", \"description\": \"Number of results to return (1-20).\", \"minimum\": 1, \"maximum\": 20, \"default\": 5 },\n \"freshness\": { \"type\": \"string\", \"description\": \"Restrict results by recency: 'pd' (past day), 'pw' (past week), 'pm' (past month), 'py' (past year), or a 'YYYY-MM-DDtoYYYY-MM-DD' date range. Omit for no time filter.\" },\n \"fetch_content\": { \"type\": \"boolean\", \"description\": \"When true, also fetch and include the readable text of the top results (slower). Default false returns snippets only.\", \"default\": false }\n },\n \"required\": [\"query\"]\n }\n }\n}\n```",
- "requestBody": {
- "required": true,
- "content": {
- "application/json": {
- "schema": {
- "type": "object",
- "required": [
- "query"
- ],
- "properties": {
- "query": {
- "type": "string",
- "description": "The search query.",
- "example": "latest go release"
- },
- "count": {
- "type": "integer",
- "minimum": 1,
- "maximum": 20,
- "default": 5,
- "description": "Number of results to return, 1–20 (values outside the range are clamped).",
- "example": 5
- },
- "freshness": {
- "type": "string",
- "description": "Recency filter: `pd` (past day), `pw` (past week), `pm` (past month), `py` (past year), or a `YYYY-MM-DDtoYYYY-MM-DD` date range. Omit for no time filter.",
- "example": "pw"
- },
- "fetch_content": {
- "type": "boolean",
- "default": false,
- "description": "When `true`, also fetch and include the readable main text of the top results (`content`). Slower; defaults to snippets only.",
- "example": false
- }
- }
- },
- "example": {
- "query": "latest go release",
- "count": 5,
- "freshness": "pw",
- "fetch_content": false
- }
- }
- }
- },
- "responses": {
- "200": {
- "description": "The search results.",
- "content": {
- "application/json": {
- "schema": {
- "$ref": "#/components/schemas/SearchResponse"
- },
- "example": {
- "results": [
- {
- "title": "Go 1.23 is released",
- "url": "https://go.dev/blog/go1.23",
- "snippet": "The latest Go release adds ...",
- "source": "primary"
- }
- ],
- "cached": false
- }
- }
- }
- },
- "400": {
- "$ref": "#/components/responses/BadRequest"
- },
- "401": {
- "$ref": "#/components/responses/Unauthorized"
- },
- "429": {
- "$ref": "#/components/responses/RateLimited"
- },
- "503": {
- "description": "Search is temporarily unavailable (`search_unavailable`); retry shortly.",
- "content": {
- "application/json": {
- "schema": {
- "$ref": "#/components/schemas/Error"
- }
- }
- }
- }
- },
- "x-codeSamples": [
- {
- "lang": "curl",
- "label": "cURL",
- "source": "curl https://api.nan.builders/v1/search \\\n -H \"Authorization: Bearer $NAN_API_KEY\" \\\n -H \"Content-Type: application/json\" \\\n -d '{ \"query\": \"latest go release\", \"count\": 5, \"freshness\": \"pw\" }'"
- },
- {
- "lang": "python",
- "label": "Python",
- "source": "from openai import OpenAI\n\nclient = OpenAI(\n api_key=\"$NAN_API_KEY\",\n base_url=\"https://api.nan.builders/v1\",\n)\n\n# search isn't part of the OpenAI client, so call it directly:\nresp = client.post(\n \"/search\",\n cast_to=object,\n body={\"query\": \"latest go release\", \"count\": 5, \"freshness\": \"pw\"},\n)\nfor r in resp[\"results\"]:\n print(r[\"title\"], r[\"url\"])"
- },
- {
- "lang": "javascript",
- "label": "Node.js",
- "source": "const res = await fetch(\"https://api.nan.builders/v1/search\", {\n method: \"POST\",\n headers: {\n Authorization: `Bearer ${process.env.NAN_API_KEY}`,\n \"Content-Type\": \"application/json\",\n },\n body: JSON.stringify({ query: \"latest go release\", count: 5, freshness: \"pw\" }),\n});\nconst data = await res.json();\nfor (const r of data.results) console.log(r.title, r.url);"
- }
- ]
- }
- },
- "/mcp": {
- "post": {
- "operationId": "mcpJsonRpc",
- "tags": [
- "MCP"
- ],
- "summary": "MCP server (JSON-RPC)",
- "description": "Remote [Model Context Protocol](https://modelcontextprotocol.io) server, so our tools can be used inside any MCP-compatible agent or client with the same `sk-` key as the REST API.\n\nTransport is streamable HTTP and stateless; the protocol is JSON-RPC 2.0 with the methods `initialize`, `tools/list`, `tools/call` and `ping`. Today the server exposes a single tool, `web_search`, with the same arguments as [Web search](#tag/Search); it is a growing registry, so use `tools/list` to discover the current set.\n\nMCP calls share the **same** per-key rate limit, daily quota and concurrency as the equivalent REST endpoint, so there is no separate budget. A `web_search` tool call over MCP counts exactly like a `POST /v1/search` request.\n\nNote this endpoint lives at the host root, `https://api.nan.builders/mcp`, not under `/v1`.",
- "requestBody": {
- "required": true,
- "content": {
- "application/json": {
- "schema": {
- "$ref": "#/components/schemas/JsonRpcRequest"
- },
- "example": {
- "jsonrpc": "2.0",
- "id": 1,
- "method": "tools/call",
- "params": {
- "name": "web_search",
- "arguments": {
- "query": "kubernetes 1.34 release",
- "count": 5
- }
- }
- }
- }
- }
- },
- "responses": {
- "200": {
- "description": "A JSON-RPC 2.0 response. Protocol-level failures are reported in `error` with HTTP `200`.",
- "content": {
- "application/json": {
- "schema": {
- "$ref": "#/components/schemas/JsonRpcResponse"
- }
- }
- }
- },
- "401": {
- "$ref": "#/components/responses/Unauthorized"
- },
- "429": {
- "$ref": "#/components/responses/RateLimited"
- }
- },
- "x-codeSamples": [
- {
- "lang": "curl",
- "label": "cURL",
- "source": "curl https://api.nan.builders/mcp \\\n -H \"Authorization: Bearer $NAN_API_KEY\" \\\n -H \"Content-Type: application/json\" \\\n -d '{\n \"jsonrpc\": \"2.0\",\n \"id\": 1,\n \"method\": \"tools/call\",\n \"params\": {\n \"name\": \"web_search\",\n \"arguments\": { \"query\": \"kubernetes 1.34 release\", \"count\": 5 }\n }\n }'"
- },
- {
- "lang": "json",
- "label": "MCP client config",
- "source": "{\n \"mcpServers\": {\n \"nan\": {\n \"url\": \"https://api.nan.builders/mcp\",\n \"headers\": {\n \"Authorization\": \"Bearer sk-your-key-here\"\n }\n }\n }\n}"
- }
- ]
- }
}
},
"components": {
@@ -2338,129 +2154,6 @@
"$ref": "#/components/schemas/Usage"
}
}
- },
- "SearchResult": {
- "type": "object",
- "description": "One web search result.",
- "properties": {
- "title": {
- "type": "string",
- "description": "The result title (plain text)."
- },
- "url": {
- "type": "string",
- "description": "The result URL."
- },
- "snippet": {
- "type": "string",
- "description": "A short excerpt from the page."
- },
- "content": {
- "type": "string",
- "description": "The readable main text of the page. Present only when `fetch_content` was `true` and the fetch succeeded."
- },
- "source": {
- "type": "string",
- "description": "Which tier served this result: `primary` or `fallback`."
- }
- }
- },
- "SearchResponse": {
- "type": "object",
- "description": "The web search results.",
- "properties": {
- "results": {
- "type": "array",
- "items": {
- "$ref": "#/components/schemas/SearchResult"
- },
- "description": "The ranked results, most relevant first."
- },
- "cached": {
- "type": "boolean",
- "description": "`true` when the results came from the short-lived (~15 min) exact-query cache."
- }
- }
- },
- "JsonRpcRequest": {
- "type": "object",
- "required": [
- "jsonrpc",
- "method"
- ],
- "properties": {
- "jsonrpc": {
- "type": "string",
- "enum": [
- "2.0"
- ],
- "description": "Always `2.0`."
- },
- "id": {
- "description": "Request id, echoed back in the response.",
- "oneOf": [
- {
- "type": "string"
- },
- {
- "type": "integer"
- }
- ]
- },
- "method": {
- "type": "string",
- "enum": [
- "initialize",
- "tools/list",
- "tools/call",
- "ping"
- ],
- "description": "The JSON-RPC method to invoke."
- },
- "params": {
- "type": "object",
- "additionalProperties": true,
- "description": "Method arguments. For `tools/call`: `name` (the tool) and `arguments` (its input)."
- }
- }
- },
- "JsonRpcResponse": {
- "type": "object",
- "properties": {
- "jsonrpc": {
- "type": "string",
- "enum": [
- "2.0"
- ]
- },
- "id": {
- "oneOf": [
- {
- "type": "string"
- },
- {
- "type": "integer"
- }
- ]
- },
- "result": {
- "type": "object",
- "additionalProperties": true,
- "description": "Present on success. Shape depends on the method."
- },
- "error": {
- "type": "object",
- "properties": {
- "code": {
- "type": "integer"
- },
- "message": {
- "type": "string"
- }
- },
- "description": "Present on a protocol-level failure."
- }
- }
}
},
"responses": {
diff --git a/src/lib/apiDoc.ts b/src/lib/apiDoc.ts
index a66f896..039b2f2 100644
--- a/src/lib/apiDoc.ts
+++ b/src/lib/apiDoc.ts
@@ -24,7 +24,7 @@ export const API_DOC_SLUG = 'api';
export const API_DOC_META = {
title: 'API',
description: 'Public API endpoint reference. OpenAI-compatible.',
- order: 17,
+ order: 16,
/**
* The nav group, per locale.
*
@@ -62,7 +62,7 @@ let cache: { key: string; text: string } | null = null;
*
* Memoised on the rate-limit values rather than unconditionally: the spec is
* static within a deployment, but the limits come from the env, so a config
- * change has to produce different text. Walking 12 endpoints and 23 schemas on
+ * change has to produce different text. Walking 10 endpoints and 19 schemas on
* every manifest request would otherwise be repeated work for an identical
* result.
*/
diff --git a/src/lib/docsNav.test.ts b/src/lib/docsNav.test.ts
index 528e08f..a3b4b6f 100644
--- a/src/lib/docsNav.test.ts
+++ b/src/lib/docsNav.test.ts
@@ -84,7 +84,7 @@ describe('slugifyAnchor', () => {
['Rate limits', 'rate-limits'],
['Versioning & compatibility', 'versioning-compatibility'],
['Model catalog', 'model-catalog'],
- ['MCP', 'mcp'],
+ ['Images', 'images'],
])('%s -> %s', (input, expected) => {
expect(slugifyAnchor(input)).toBe(expected);
});
@@ -118,14 +118,11 @@ describe('apiSearchHeadings', () => {
expect(tagSlugs.sort()).toEqual((spec.tags ?? []).map((t) => t.name).sort());
});
- /** agents.md links here by hand; if the tag is renamed, that link dies. */
- it('keeps the MCP anchor that agents.md points at', () => {
- expect(headings.some((h) => h.slug === 'tag/mcp')).toBe(true);
- });
-
it('takes only level-2 headings, not the level-3 ones nested under them', () => {
- // "Use it as an agent tool" is a `#####` inside an operation description.
- expect(headings.some((h) => h.text === 'Use it as an agent tool')).toBe(false);
+ const nested = apiSearchHeadings({
+ info: { description: '## Rate limits\n\n### Per model\n\n#### Per key' },
+ });
+ expect(nested.map((h) => h.text)).toEqual(['Rate limits']);
});
});
diff --git a/src/lib/openapiSpec.test.ts b/src/lib/openapiSpec.test.ts
index 229bb02..0e25e7b 100644
--- a/src/lib/openapiSpec.test.ts
+++ b/src/lib/openapiSpec.test.ts
@@ -17,13 +17,13 @@ import { DEFAULT_RATE_LIMITS, formatTokens, getRateLimitsConfig } from './rateLi
* this mostly watches for is anything from there creeping back in.
*
* The endpoint surface was checked against the real backend by probing each
- * route: the 12 listed here answer 401 (they exist and want auth) while
+ * route: the 10 listed here answer 401 (they exist and want auth) while
* /v1/moderations, /v1/batches and /v1/files answer 404 (not enabled on NaN).
*/
const raw = JSON.stringify(spec);
-/** The 12 public routes verified against api.nan.builders. */
+/** The 10 public routes verified against api.nan.builders. */
const PUBLIC_SURFACE: Array<[string, string]> = [
['/models', 'get'],
['/chat/completions', 'post'],
@@ -35,8 +35,6 @@ const PUBLIC_SURFACE: Array<[string, string]> = [
['/responses', 'post'],
['/images/generations', 'post'],
['/images/edits', 'post'],
- ['/search', 'post'],
- ['/mcp', 'post'],
];
/** NaN's real catalogue (src/data/modelos.json + the API reference). */
diff --git a/src/lib/openapiToText.ts b/src/lib/openapiToText.ts
index bc4f531..f2d2dc7 100644
--- a/src/lib/openapiToText.ts
+++ b/src/lib/openapiToText.ts
@@ -151,9 +151,9 @@ function propertyTable(schema: SchemaLike | undefined): string {
*
* They are written for Scalar, which paints them inside the endpoint panel, so
* they start at `##`. Dumped as-is into a flat document they would outrank the
- * `###` of the very endpoint they belong to: the `## Use it as an agent tool`
- * of /search read as a sibling section of "Search" instead of part of it. They
- * sink until they sit below, capped at `######`.
+ * `###` of the very endpoint they belong to: a `## Usage` inside an
+ * endpoint would read as a sibling section of its tag instead of part of it.
+ * They sink until they sit below, capped at `######`.
*
* Only headings at the start of a line and outside a code fence count: inside
* a ``` a `#` is usually a shell comment.
diff --git a/src/middleware.ts b/src/middleware.ts
index 42c3738..8dd6f08 100644
--- a/src/middleware.ts
+++ b/src/middleware.ts
@@ -89,6 +89,28 @@ function hackatonRedirect(context: Parameters[0]): Response |
return context.redirect(`${prefix}/events/${LEGACY_HACKATON_SLUG}${rest}${url.search}`, 301);
}
+/**
+ * Páginas de docs retiradas, con su destino.
+ *
+ * `/docs/mcp` documentaba el servidor MCP de NaN, cuya única herramienta era la
+ * búsqueda web; se retiraron los dos (el endpoint `/mcp` y `POST /v1/search`).
+ * Los enlaces siguen circulando, así que se mandan con 301 a la página desde la
+ * que se llegaba a ella, en vez de dejar un 404. Se conserva el prefijo `/es`.
+ */
+const REMOVED_DOCS: Readonly> = {
+ mcp: 'agent-setup',
+};
+
+function removedDocsRedirect(context: Parameters[0]): Response | null {
+ const { url } = context;
+ const m = /^(\/es)?\/docs\/([^/]+)\/?$/.exec(url.pathname);
+ if (!m) return null;
+ const [, prefix = '', slug] = m;
+ const target = REMOVED_DOCS[slug];
+ if (!target) return null;
+ return context.redirect(`${prefix}/docs/${target}${url.search}`, 301);
+}
+
/**
* OJO con prerenderizar: el middleware NO corre en las rutas prerenderizadas
* (Astro lo ejecuta en tiempo de build para esas), así que activar
@@ -97,7 +119,7 @@ function hackatonRedirect(context: Parameters[0]): Response |
* Hoy no hay ninguna prerenderizada, y por eso esto vale para todo el sitio.
*/
export const onRequest: MiddlewareHandler = async (context, next) => {
- const response = langRedirect(context) ?? hackatonRedirect(context) ?? (await next());
+ const response = langRedirect(context) ?? hackatonRedirect(context) ?? removedDocsRedirect(context) ?? (await next());
for (const [name, value] of Object.entries(SECURITY_HEADERS)) {
response.headers.set(name, value);
diff --git a/src/tests/lib/docsGlm53.test.ts b/src/tests/lib/docsGlm53.test.ts
index 66d819d..761d65d 100644
--- a/src/tests/lib/docsGlm53.test.ts
+++ b/src/tests/lib/docsGlm53.test.ts
@@ -133,8 +133,8 @@ describe('docs/api — glm5.3 is callable', () => {
});
test('the errors table documents the statuses the limits return', () => {
- // The "## Errors" table, not the per-endpoint ones (web search has its own
- // 429 rows and would answer first).
+ // The "## Errors" table, not the per-endpoint ones (an endpoint with its
+ // own 429 rows would answer first).
const section = api.slice(api.indexOf('## Errors'));
expect(section.length).toBeGreaterThan(0);
const rows = section.split('\n').filter((l) => l.startsWith('|'));
diff --git a/src/tests/lib/middleware.test.ts b/src/tests/lib/middleware.test.ts
index 448e036..d86bc5f 100644
--- a/src/tests/lib/middleware.test.ts
+++ b/src/tests/lib/middleware.test.ts
@@ -109,6 +109,25 @@ describe('middleware — rutas antiguas del hackatón', () => {
});
});
+describe('middleware — páginas de docs retiradas', () => {
+ test('/docs/mcp lleva a /docs/agent-setup con 301', async () => {
+ const r = await run('https://nan.builders/docs/mcp');
+ expect(r.status).toBe(301);
+ expect(r.location).toBe('/docs/agent-setup');
+ expect(r.passedThrough).toBe(false);
+ });
+
+ test('conserva el prefijo de idioma', async () => {
+ expect((await run('https://nan.builders/es/docs/mcp')).location).toBe('/es/docs/agent-setup');
+ });
+
+ test('no toca las páginas que siguen vivas', async () => {
+ const r = await run('https://nan.builders/docs/agent-setup');
+ expect(r.location).toBeNull();
+ expect(r.passedThrough).toBe(true);
+ });
+});
+
describe('middleware — cabeceras de seguridad', () => {
test('las emite en las páginas', async () => {
const { res } = await run('https://nan.builders/community');