From af0fcb628b81096c9667b9467789f4e77a126773 Mon Sep 17 00:00:00 2001 From: barckcode Date: Sun, 27 Sep 2026 12:39:44 +0200 Subject: [PATCH] docs(api): single-source the /usage rate limit and unify cutover caveats MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-ups from the blocking reviews of the /v1/usage docs: - The usage budget (30 requests per minute per member) now lives once, in rateLimits.ts (USAGE_REQUESTS_PER_MINUTE, mirroring the cloud-api keyedLimiter value), and reaches info.description through a new {{USAGE_RATE_LIMIT}} placeholder resolved in resolveSpec — same mechanism as {{RATE_LIMITS}}; rendered text is byte-identical. The /usage endpoint description and its 429 stay static (the contract) and are pinned by test against the shared constant. - All four api_requests descriptions now use the exact same cutover sentence; the caveat test asserts the full sentence on every schema. Tests: 3 new single-source tests + strengthened caveat assertions; vitest 66 files / 1308 tests green; npm run build green. --- src/data/openapi.json | 4 +-- src/lib/apiDoc.ts | 23 +++++++++++--- src/lib/openapiSpec.test.ts | 61 +++++++++++++++++++++++++++++++++++-- src/lib/rateLimits.ts | 19 ++++++++++++ 4 files changed, 98 insertions(+), 9 deletions(-) diff --git a/src/data/openapi.json b/src/data/openapi.json index 0f15cb1..aa85c38 100644 --- a/src/data/openapi.json +++ b/src/data/openapi.json @@ -3,7 +3,7 @@ "info": { "title": "NaN API", "version": "1.0.0", - "description": "Open models on a shared EU inference cluster. Zero logs.\n\nThe NaN API is OpenAI-compatible: predictable, resource-oriented URLs, JSON request and response bodies, and standard HTTP verbs and status codes. Point any OpenAI SDK at our base URL and your existing code keeps working. Change the base URL and the API key, and that's it.\n\nOne schema across every model, so you only learn the API once. Change the `model` field to switch models; everything else stays the same.\n\n- Base URL: `https://api.nan.builders/v1`\n- OpenAPI spec: this document. Import it into Postman, Insomnia, or your own tooling.\n\nIf you use the [Helmcode](https://helmcode.com) enterprise service, the base URL is `https://api.helmcode.com/v1` instead. Every other endpoint is identical.\n\n## Authentication\n\nEvery request authenticates with an API key, sent as a Bearer token:\n\n```\nAuthorization: Bearer $NAN_API_KEY\n```\n\nYou must be a NaN community member. Generate your key from user settings, under \"API Keys\", on the [platform](https://cloud.nan.builders/). The key is personal and non-transferable. Keep it secret: never embed one in client-side code or commit it to source control. Requests must go over HTTPS; calls over plain HTTP fail.\n\n## Making requests\n\nThe API is OpenAI-compatible, so point an official OpenAI SDK at our base URL and change nothing else:\n\n```python\nfrom openai import OpenAI\n\nclient = OpenAI(\n api_key=\"$NAN_API_KEY\",\n base_url=\"https://api.nan.builders/v1\",\n)\n\nresp = client.chat.completions.create(\n model=\"deepseek-v4-flash\",\n messages=[{\"role\": \"user\", \"content\": \"Hello\"}],\n)\nprint(resp.choices[0].message.content)\n```\n\n## Streaming\n\nChat responses can stream token-by-token. Set `\"stream\": true` on `/chat/completions` and the response arrives as Server-Sent Events: each event is a `data:` line carrying a `chat.completion.chunk`, with the new text in `choices[0].delta.content`. A final `data: [DONE]` line ends the stream. Only `/chat/completions` streams incrementally; `/responses` currently emits a single terminal event.\n\n## Rate limits\n\n{{RATE_LIMITS}}\n\nImage endpoints run on their own budget, separate from the model endpoints: 20 requests per minute and 100 requests per month. The usage endpoint is metered separately too: 30 requests per minute per member. Exceed any limit and you get a `429`.\n\n## Errors\n\nNaN uses conventional HTTP status codes: `2xx` on success, `4xx` for a problem with the request (a missing parameter, an invalid key, an unavailable model) and `5xx` for a server-side error. Every error returns a JSON body in the OpenAI shape:\n\n```json\n{\n \"error\": {\n \"message\": \"The model 'foo' does not exist.\",\n \"type\": \"invalid_request_error\",\n \"param\": \"model\",\n \"code\": \"model_not_found\"\n }\n}\n```\n\n`message` is human-readable, `param` names the offending field when applicable, and `code` is a short machine-readable string you can branch on.\n\n| Status | Meaning | `code` |\n| --- | --- | --- |\n| `400` | Invalid or malformed parameter (`param` says which); or content blocked by the safety filter. | `invalid_request_error` · `content_policy_violation` |\n| `401` | Missing or invalid API key, or a key whose tier does not reach the requested model (`glm5.3`): \"This API key does not have access to the requested model\", `type: auth_error`. Measured 2026-09-12. | `invalid_api_key` |\n| `402` | The token allowance is spent on a model that carries one. Not retryable: the counter returns to zero when that model's quota period does, the calendar month for the models counted per month and your billing period for `glm5.3`. | `monthly_cap_reached` |\n| `403` | Your tier can't access this endpoint. Image generation requires inference membership. A model your tier cannot reach answers `401`, not this. | `tier_restricted` |\n| `404` | The requested model doesn't exist. | `model_not_found` |\n| `429` | Rate limit hit (`rpm_limit`, `max_parallel_requests`), the rolling 4h token budget of `glm5.3`, or a quota exhausted. | `rate_limit_exceeded` · `insufficient_quota` · `quota_exceeded` |\n| `500` | Something went wrong on our side (includes upstream model errors). | (none) |\n| `524` | Timeout, typical with large audio files on `/audio/transcriptions`. | (none) |\n\nRetry `429` and `5xx` responses with exponential backoff. Don't retry `400`, `401`, `403`, or `404` blindly: they'll fail the same way every time until you change the request. `402` cannot be fixed by repetition either: it clears when that model's quota period resets.\n\n## Model catalog\n\nEvery endpoint takes a `model` id. Capabilities vary by model:\n\n| Model | Use for | Capabilities |\n| --- | --- | --- |\n| `deepseek-v4-flash` | Chat, vision, reasoning | Streaming, tool calling, reasoning, image input, 1M-token context. 3B tokens/month per member |\n| `mimo-v2.5` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `mimo-v2.6-flash` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `qwen3.8-flash` | Chat, vision, agents | Streaming, tool calling, reasoning (on by default), vision, 262K-token context. 500M tokens/month per member |\n| `glm5.3-flash` | Chat, vision, agents | Streaming, tool calling, reasoning, vision, 1M-token context. 2B tokens/month per member |\n| `qwen3.6` | Chat, agents | Streaming, tool calling, vision, reasoning (opt-out, returns `reasoning_content`) |\n| `gemma4` | Chat, vision, agents | Streaming, tool calling, vision, reasoning (opt-in) |\n| `glm5.3` | Coding, long-horizon agents | Streaming, tool calling, reasoning trace, text-only input, 1M-token context. Premium tier only |\n| `qwen3-embedding` | Embeddings | 4096-dimension vectors |\n| `rerank` | RAG reranking | Qwen3-Reranker-8B, 100+ languages |\n| `kokoro` | Text-to-speech | Multiple voices and audio formats |\n| `whisper` | Speech-to-text | Transcription with word/segment timestamps |\n| `flux-2-klein` | Image generation | Text-to-image and image-to-image |\n| `qwen-image-2.1` | Image generation (text→image) | 512-1280 px, 1-4 per request, seed 0-2147483647. 100 images/month per member (shared pool with flux-2-klein) |\n\n`glm5.3` is served only to keys on the GLM 5.3 premium tier; every other model is available to any inference member. Call [List models](#tag/Models) for the exact set available to your key.\n\n## Versioning & compatibility\n\nThe API tracks the OpenAI API surface, so OpenAI SDKs and tools work against `https://api.nan.builders/v1` unchanged. This reference documents the stable public `/v1` endpoints, and we add capabilities without breaking existing fields.", + "description": "Open models on a shared EU inference cluster. Zero logs.\n\nThe NaN API is OpenAI-compatible: predictable, resource-oriented URLs, JSON request and response bodies, and standard HTTP verbs and status codes. Point any OpenAI SDK at our base URL and your existing code keeps working. Change the base URL and the API key, and that's it.\n\nOne schema across every model, so you only learn the API once. Change the `model` field to switch models; everything else stays the same.\n\n- Base URL: `https://api.nan.builders/v1`\n- OpenAPI spec: this document. Import it into Postman, Insomnia, or your own tooling.\n\nIf you use the [Helmcode](https://helmcode.com) enterprise service, the base URL is `https://api.helmcode.com/v1` instead. Every other endpoint is identical.\n\n## Authentication\n\nEvery request authenticates with an API key, sent as a Bearer token:\n\n```\nAuthorization: Bearer $NAN_API_KEY\n```\n\nYou must be a NaN community member. Generate your key from user settings, under \"API Keys\", on the [platform](https://cloud.nan.builders/). The key is personal and non-transferable. Keep it secret: never embed one in client-side code or commit it to source control. Requests must go over HTTPS; calls over plain HTTP fail.\n\n## Making requests\n\nThe API is OpenAI-compatible, so point an official OpenAI SDK at our base URL and change nothing else:\n\n```python\nfrom openai import OpenAI\n\nclient = OpenAI(\n api_key=\"$NAN_API_KEY\",\n base_url=\"https://api.nan.builders/v1\",\n)\n\nresp = client.chat.completions.create(\n model=\"deepseek-v4-flash\",\n messages=[{\"role\": \"user\", \"content\": \"Hello\"}],\n)\nprint(resp.choices[0].message.content)\n```\n\n## Streaming\n\nChat responses can stream token-by-token. Set `\"stream\": true` on `/chat/completions` and the response arrives as Server-Sent Events: each event is a `data:` line carrying a `chat.completion.chunk`, with the new text in `choices[0].delta.content`. A final `data: [DONE]` line ends the stream. Only `/chat/completions` streams incrementally; `/responses` currently emits a single terminal event.\n\n## Rate limits\n\n{{RATE_LIMITS}}\n\nImage endpoints run on their own budget, separate from the model endpoints: 20 requests per minute and 100 requests per month. The usage endpoint is metered separately too: {{USAGE_RATE_LIMIT}} requests per minute per member. Exceed any limit and you get a `429`.\n\n## Errors\n\nNaN uses conventional HTTP status codes: `2xx` on success, `4xx` for a problem with the request (a missing parameter, an invalid key, an unavailable model) and `5xx` for a server-side error. Every error returns a JSON body in the OpenAI shape:\n\n```json\n{\n \"error\": {\n \"message\": \"The model 'foo' does not exist.\",\n \"type\": \"invalid_request_error\",\n \"param\": \"model\",\n \"code\": \"model_not_found\"\n }\n}\n```\n\n`message` is human-readable, `param` names the offending field when applicable, and `code` is a short machine-readable string you can branch on.\n\n| Status | Meaning | `code` |\n| --- | --- | --- |\n| `400` | Invalid or malformed parameter (`param` says which); or content blocked by the safety filter. | `invalid_request_error` · `content_policy_violation` |\n| `401` | Missing or invalid API key, or a key whose tier does not reach the requested model (`glm5.3`): \"This API key does not have access to the requested model\", `type: auth_error`. Measured 2026-09-12. | `invalid_api_key` |\n| `402` | The token allowance is spent on a model that carries one. Not retryable: the counter returns to zero when that model's quota period does, the calendar month for the models counted per month and your billing period for `glm5.3`. | `monthly_cap_reached` |\n| `403` | Your tier can't access this endpoint. Image generation requires inference membership. A model your tier cannot reach answers `401`, not this. | `tier_restricted` |\n| `404` | The requested model doesn't exist. | `model_not_found` |\n| `429` | Rate limit hit (`rpm_limit`, `max_parallel_requests`), the rolling 4h token budget of `glm5.3`, or a quota exhausted. | `rate_limit_exceeded` · `insufficient_quota` · `quota_exceeded` |\n| `500` | Something went wrong on our side (includes upstream model errors). | (none) |\n| `524` | Timeout, typical with large audio files on `/audio/transcriptions`. | (none) |\n\nRetry `429` and `5xx` responses with exponential backoff. Don't retry `400`, `401`, `403`, or `404` blindly: they'll fail the same way every time until you change the request. `402` cannot be fixed by repetition either: it clears when that model's quota period resets.\n\n## Model catalog\n\nEvery endpoint takes a `model` id. Capabilities vary by model:\n\n| Model | Use for | Capabilities |\n| --- | --- | --- |\n| `deepseek-v4-flash` | Chat, vision, reasoning | Streaming, tool calling, reasoning, image input, 1M-token context. 3B tokens/month per member |\n| `mimo-v2.5` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `mimo-v2.6-flash` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `qwen3.8-flash` | Chat, vision, agents | Streaming, tool calling, reasoning (on by default), vision, 262K-token context. 500M tokens/month per member |\n| `glm5.3-flash` | Chat, vision, agents | Streaming, tool calling, reasoning, vision, 1M-token context. 2B tokens/month per member |\n| `qwen3.6` | Chat, agents | Streaming, tool calling, vision, reasoning (opt-out, returns `reasoning_content`) |\n| `gemma4` | Chat, vision, agents | Streaming, tool calling, vision, reasoning (opt-in) |\n| `glm5.3` | Coding, long-horizon agents | Streaming, tool calling, reasoning trace, text-only input, 1M-token context. Premium tier only |\n| `qwen3-embedding` | Embeddings | 4096-dimension vectors |\n| `rerank` | RAG reranking | Qwen3-Reranker-8B, 100+ languages |\n| `kokoro` | Text-to-speech | Multiple voices and audio formats |\n| `whisper` | Speech-to-text | Transcription with word/segment timestamps |\n| `flux-2-klein` | Image generation | Text-to-image and image-to-image |\n| `qwen-image-2.1` | Image generation (text→image) | 512-1280 px, 1-4 per request, seed 0-2147483647. 100 images/month per member (shared pool with flux-2-klein) |\n\n`glm5.3` is served only to keys on the GLM 5.3 premium tier; every other model is available to any inference member. Call [List models](#tag/Models) for the exact set available to your key.\n\n## Versioning & compatibility\n\nThe API tracks the OpenAI API surface, so OpenAI SDKs and tools work against `https://api.nan.builders/v1` unchanged. This reference documents the stable public `/v1` endpoints, and we add capabilities without breaking existing fields.", "contact": { "name": "NaN", "url": "https://nan.builders" @@ -2489,7 +2489,7 @@ }, "api_requests": { "type": "integer", - "description": "Total requests recorded for this member. Request counts start on 2026-09-02; older days report `0`." + "description": "Total requests recorded for this member. Request counts are only available from 2026-09-02 onward; older days report `0`." }, "cached_at": { "type": "string", diff --git a/src/lib/apiDoc.ts b/src/lib/apiDoc.ts index 4b6a528..38de6ff 100644 --- a/src/lib/apiDoc.ts +++ b/src/lib/apiDoc.ts @@ -1,6 +1,6 @@ import rawSpec from '../data/openapi.json'; import { openapiToText } from './openapiToText'; -import { rateLimitsToSpecMarkdown, type RateLimitsConfig } from './rateLimits'; +import { rateLimitsToSpecMarkdown, USAGE_REQUESTS_PER_MINUTE, type RateLimitsConfig } from './rateLimits'; /** * The API reference as a docs entry, generated from the spec. @@ -46,12 +46,25 @@ export const API_DOC_META = { */ const RATE_LIMITS_PLACEHOLDER = '{{RATE_LIMITS}}'; +/** + * The placeholder the spec carries where the /usage budget goes. + * + * Same story as {{RATE_LIMITS}}: the 30 lives in rateLimits.ts + * (USAGE_REQUESTS_PER_MINUTE, mirroring the backend's keyedLimiter value), so + * the overview prose cannot drift from the module every other surface reads. + * Unlike the rate-limit table this placeholder carries only the NUMBER — the + * sentence stays written in the spec, so the rendered text is byte-identical + * to what the spec said by hand. It is scoped to info.description: the /usage + * endpoint's own description and 429 response are part of the served contract + * and stay static (openapiSpec.test.ts pins their figures to the constant). + */ +const USAGE_RATE_LIMIT_PLACEHOLDER = '{{USAGE_RATE_LIMIT}}'; + /** The spec with its placeholders resolved, ready to serve or to render. */ export function resolveSpec(rateLimits: RateLimitsConfig): typeof rawSpec { - const description = rawSpec.info.description.replace( - RATE_LIMITS_PLACEHOLDER, - rateLimitsToSpecMarkdown(rateLimits), - ); + const description = rawSpec.info.description + .replace(RATE_LIMITS_PLACEHOLDER, rateLimitsToSpecMarkdown(rateLimits)) + .replace(USAGE_RATE_LIMIT_PLACEHOLDER, String(USAGE_REQUESTS_PER_MINUTE)); return { ...rawSpec, info: { ...rawSpec.info, description } }; } diff --git a/src/lib/openapiSpec.test.ts b/src/lib/openapiSpec.test.ts index 0711033..102ff37 100644 --- a/src/lib/openapiSpec.test.ts +++ b/src/lib/openapiSpec.test.ts @@ -5,7 +5,7 @@ import { dirname, resolve } from 'node:path'; import spec from '../data/openapi.json'; import modelos from '../data/modelos.json'; import { resolveSpec } from './apiDoc'; -import { DEFAULT_RATE_LIMITS, formatTokens, getRateLimitsConfig } from './rateLimits'; +import { DEFAULT_RATE_LIMITS, formatTokens, getRateLimitsConfig, USAGE_REQUESTS_PER_MINUTE } from './rateLimits'; /** * Tripwire over src/data/openapi.json, the spec Scalar renders at /docs/api @@ -217,6 +217,55 @@ describe('openapi.json: rate limits come from the single source of truth', () => expect(description).toContain('| Concurrent requests | per model — see the per-model limits below |'); }); + /** + * The /usage budget is a number the backend hardcodes, so it is a code + * constant in rateLimits.ts (like the tier ceilings) and reaches the + * overview prose through its own placeholder — not a third handwritten + * "30". The wording stays in the spec; only the number is shared. + */ + it('ships the usage budget as a placeholder rather than a number', () => { + expect(spec.info.description).toContain('{{USAGE_RATE_LIMIT}}'); + // The overview must not still carry the handwritten figure the + // placeholder replaces — one copy, not two. + expect(spec.info.description).not.toMatch( + /The usage endpoint is metered separately too: \d+ requests per minute per member/, + ); + }); + + /** + * The rendered overview is what the current spec published by hand, to the + * byte: the placeholder swap only shares the number, it rewrites nothing. + */ + it('renders the usage sentence from the shared constant, unchanged', () => { + const description = resolveSpec(DEFAULT_RATE_LIMITS).info.description; + expect(description).not.toContain('{{USAGE_RATE_LIMIT}}'); + expect(description).toContain( + 'Image endpoints run on their own budget, separate from the model endpoints: ' + + '20 requests per minute and 100 requests per month. ' + + `The usage endpoint is metered separately too: ${USAGE_REQUESTS_PER_MINUTE} requests per minute per member. ` + + 'Exceed any limit and you get a `429`.', + ); + }); + + /** + * The /usage endpoint's own contract (operation description and 429) stays + * static — no placeholder there — but its figures have to agree with the + * shared constant, or the overview and the endpoint disagree about the + * budget, which is the drift this single source exists to prevent. + */ + it('keeps the /usage endpoint contract static and consistent with the shared budget', () => { + const usage = (spec.paths as any)['/usage'].get; + expect(usage.description).toContain( + `Rate limit: ${USAGE_REQUESTS_PER_MINUTE} requests per minute per member, a budget separate from the model endpoints'.`, + ); + expect(usage.responses['429'].description).toContain( + `${USAGE_REQUESTS_PER_MINUTE} requests per minute per member (\`rate_limit_exceeded\`)`, + ); + // Static means static: no placeholder may leak into the served contract. + expect(usage.description).not.toContain('{{'); + expect(usage.responses['429'].description).not.toContain('{{'); + }); + it('publishes every model that carries a per-minute limit', () => { const description = resolveSpec(DEFAULT_RATE_LIMITS).info.description; for (const m of DEFAULT_RATE_LIMITS.tokensPerMinuteByModel) { @@ -515,9 +564,17 @@ describe('openapi.json: the /usage contract', () => { // Request counts only exist from the usage-hook cutover (2026-09-02); // older days report 0. Every api_requests description must say so, or // tokens-per-request math in third-party tools silently lies. + // + // The sentence is asserted in full, not just by date: the four schemas + // were written on different days and had already drifted into two + // wordings ("only available from ... onward" against "start on"), and a + // substring check on the date alone cannot see that. One sentence, four + // fields, byte for byte. + const CUTOVER = + 'Request counts are only available from 2026-09-02 onward; older days report `0`.'; for (const name of ['UsageRow', 'UsageTotals', 'UsageModelTotals', 'UsageAllTime']) { const description: string = schemas[name].properties.api_requests.description; - expect(description, `${name}.api_requests`).toContain('2026-09-02'); + expect(description, `${name}.api_requests`).toContain(CUTOVER); } }); }); diff --git a/src/lib/rateLimits.ts b/src/lib/rateLimits.ts index fb05ce5..dd80b22 100644 --- a/src/lib/rateLimits.ts +++ b/src/lib/rateLimits.ts @@ -88,6 +88,25 @@ export interface RateLimitsEnv { RATE_LIMIT_PARALLEL?: string; } +/** + * The /usage budget: 30 requests per minute, per member. + * + * Mirrors the backend exactly: cloud-api hardcodes `usageRequestsPerMinute = 30` + * (internal/handlers/usage_public.go) and feeds it to a keyedLimiter, so the + * bucket is per KEY across every /usage call a member makes — a member with + * five sk- keys gets 30 rpm in total, not 150. That is why the docs say "per + * member", not "per key". + * + * A code constant like tierMaxParallel, for the same reason: the backend + * hardcodes it, so there is no env var to follow, and an override here could + * only make the docs disagree with the enforcement. Consumed by apiDoc.ts's + * {{USAGE_RATE_LIMIT}} placeholder in the overview prose. The /usage + * endpoint's own description and 429 response carry the same figure as static + * spec text (openapiSpec.test.ts pins them to this constant), so a change to + * either side fails there until both move together. + */ +export const USAGE_REQUESTS_PER_MINUTE = 30; + /** * Per-model tables stay here rather than in env vars: they only change when a * model is added or removed, which is a code change anyway.