From eafe22c73ac7000c3e3bf6546c02da5a8306495d Mon Sep 17 00:00:00 2001 From: barckcode Date: Wed, 19 Aug 2026 14:35:26 +0200 Subject: [PATCH] docs: raise DeepSeek V4-Flash monthly quota from 500M to 2B Update the landing, model catalog, docs, and i18n FAQ to reflect the consolidated DeepSeek V4-Flash quota of 2B tokens/month per member (up from 500M). The model is now served via OpenRouter (slug deepseek/deepseek-v4-flash-0731); the quota is the only user-facing number that changed. --- i18n/en.json | 4 ++-- i18n/es.json | 4 ++-- src/content/docs-es/models.mdx | 4 ++-- src/content/docs/models.mdx | 4 ++-- src/data/modelos.json | 2 +- src/data/openapi.json | 2 +- 6 files changed, 10 insertions(+), 10 deletions(-) diff --git a/i18n/en.json b/i18n/en.json index ab9fc1c..664b113 100644 --- a/i18n/en.json +++ b/i18n/en.json @@ -295,7 +295,7 @@ "inferenceIncluded": "inference included", "memberIncludes": [ "Access to the shared cluster. Open models, no token caps.", - "DeepSeek V4-Flash (frontier 284B, 1M context, reasoning). 500M tokens/month.", + "DeepSeek V4-Flash (frontier 284B, 1M context, reasoning). 2B tokens/month.", "Open models chosen by the community", "Personal API key compatible with OpenAI", "Private Discord channels for members only", @@ -357,7 +357,7 @@ }, { "q": "Is it really unlimited tokens?", - "a": "On the cluster models, no counter. The frontier ones come with a token allowance per member so the plan holds up: 500M tokens per month on DeepSeek V4-Flash, 1.0B per month on MiMo v2.5, and 3,000M per billing period on GLM 5.2 for members on the premium tier. Those limits are published here, not in fine print. The bill is the fixed subscription price and it doesn’t move." + "a": "On the cluster models, no counter. The frontier ones come with a token allowance per member so the plan holds up: 2B tokens per month on DeepSeek V4-Flash, 1.0B per month on MiMo v2.5, and 3,000M per billing period on GLM 5.2 for members on the premium tier. Those limits are published here, not in fine print. The bill is the fixed subscription price and it doesn’t move." }, { "q": "How are the models chosen?", diff --git a/i18n/es.json b/i18n/es.json index 85dfadc..fa0b424 100644 --- a/i18n/es.json +++ b/i18n/es.json @@ -295,7 +295,7 @@ "inferenceIncluded": "inferencia incluida", "memberIncludes": [ "Acceso al cluster compartido. Modelos abiertos, sin límite de tokens.", - "DeepSeek V4-Flash (frontier 284B, contexto 1M, razonamiento). 500M tokens/mes.", + "DeepSeek V4-Flash (frontier 284B, contexto 1M, razonamiento). 2B tokens/mes.", "Modelos abiertos elegidos por la comunidad", "API key personal compatible con OpenAI", "Canales privados de Discord solo para miembros", @@ -357,7 +357,7 @@ }, { "q": "¿De verdad son tokens ilimitados?", - "a": "En los modelos del cluster, sin contador. Los frontier llevan cuota de tokens por miembro para que el plan se sostenga: 500M tokens al mes en DeepSeek V4-Flash, 1.0B al mes en MiMo v2.5 y 3.000M por periodo de facturación en GLM 5.2 para los miembros del tier premium. Esos límites están publicados aquí, no en letra pequeña. La factura es el precio fijo de la suscripción y no se mueve." + "a": "En los modelos del cluster, sin contador. Los frontier llevan cuota de tokens por miembro para que el plan se sostenga: 2B tokens al mes en DeepSeek V4-Flash, 1.0B al mes en MiMo v2.5 y 3.000M por periodo de facturación en GLM 5.2 para los miembros del tier premium. Esos límites están publicados aquí, no en letra pequeña. La factura es el precio fijo de la suscripción y no se mueve." }, { "q": "¿Cómo se eligen los modelos?", diff --git a/src/content/docs-es/models.mdx b/src/content/docs-es/models.mdx index 529f7ef..e1ebcaf 100644 --- a/src/content/docs-es/models.mdx +++ b/src/content/docs-es/models.mdx @@ -20,12 +20,12 @@ OpenAI y la misma `base URL`. tag="284B-21B" leftLabel="generación de texto y chat" rightLabel="capacidades" - description="Modelo MoE de 284B parámetros (21B activos). Contexto de 1M tokens. Tool calling y razonamiento. Cuota de 500M tokens al mes por miembro." + description="Modelo MoE de 284B parámetros (21B activos). Contexto de 1M tokens. Tool calling y razonamiento. Cuota de 2B tokens al mes por miembro." specs={[ { label: 'Tipo', value: 'MoE (284B total · 21B activos)' }, { label: 'Cuantización', value: 'FP8' }, { label: 'Contexto', value: '1M tokens' }, - { label: 'Cuota mensual', value: '500M tokens / miembro' }, + { label: 'Cuota mensual', value: '2B tokens / miembro' }, ]} items={[ 'Tool calling', diff --git a/src/content/docs/models.mdx b/src/content/docs/models.mdx index 5d1f1b3..7c4c792 100644 --- a/src/content/docs/models.mdx +++ b/src/content/docs/models.mdx @@ -20,12 +20,12 @@ with the same `base URL`. tag="284B-21B" leftLabel="text generation & chat" rightLabel="capabilities" - description="284B parameter MoE model (21B active). 1M token context. Tool calling and reasoning. 500M token monthly quota per member." + description="284B parameter MoE model (21B active). 1M token context. Tool calling and reasoning. 2B token monthly quota per member." specs={[ { label: 'Type', value: 'MoE (284B total · 21B active)' }, { label: 'Quantization', value: 'FP8' }, { label: 'Context', value: '1M tokens' }, - { label: 'Monthly quota', value: '500M tokens / member' }, + { label: 'Monthly quota', value: '2B tokens / member' }, ]} items={[ 'Tool calling', diff --git a/src/data/modelos.json b/src/data/modelos.json index b867251..36b6b58 100644 --- a/src/data/modelos.json +++ b/src/data/modelos.json @@ -15,7 +15,7 @@ "id": "deepseek-v4-flash", "by": "DeepSeek", "specs": "284B-21B MoE · FP8 · 1M context · tool calling · reasoning", - "cuota": "500M tokens/mes", + "cuota": "2B tokens/mes", "frontier": true, "mostUsed": true }, diff --git a/src/data/openapi.json b/src/data/openapi.json index af89494..06225ea 100644 --- a/src/data/openapi.json +++ b/src/data/openapi.json @@ -3,7 +3,7 @@ "info": { "title": "NaN API", "version": "1.0.0", - "description": "Open models on a shared EU inference cluster. Zero logs.\n\nThe NaN API is OpenAI-compatible: predictable, resource-oriented URLs, JSON request and response bodies, and standard HTTP verbs and status codes. Point any OpenAI SDK at our base URL and your existing code keeps working. Change the base URL and the API key, and that's it.\n\nOne schema across every model, so you only learn the API once. Change the `model` field to switch models; everything else stays the same.\n\n- Base URL: `https://api.nan.builders/v1`\n- OpenAPI spec: this document. Import it into Postman, Insomnia, or your own tooling.\n\nIf you use the [Helmcode](https://helmcode.com) enterprise service, the base URL is `https://api.helmcode.com/v1` instead. Every other endpoint is identical.\n\n## Authentication\n\nEvery request authenticates with an API key, sent as a Bearer token:\n\n```\nAuthorization: Bearer $NAN_API_KEY\n```\n\nYou must be a NaN community member. Generate your key from user settings, under \"API Keys\", on the [platform](https://cloud.nan.builders/). The key is personal and non-transferable. Keep it secret: never embed one in client-side code or commit it to source control. Requests must go over HTTPS; calls over plain HTTP fail.\n\n## Making requests\n\nThe API is OpenAI-compatible, so point an official OpenAI SDK at our base URL and change nothing else:\n\n```python\nfrom openai import OpenAI\n\nclient = OpenAI(\n api_key=\"$NAN_API_KEY\",\n base_url=\"https://api.nan.builders/v1\",\n)\n\nresp = client.chat.completions.create(\n model=\"qwen3.6\",\n messages=[{\"role\": \"user\", \"content\": \"Hello\"}],\n)\nprint(resp.choices[0].message.content)\n```\n\n## Streaming\n\nChat responses can stream token-by-token. Set `\"stream\": true` on `/chat/completions` and the response arrives as Server-Sent Events: each event is a `data:` line carrying a `chat.completion.chunk`, with the new text in `choices[0].delta.content`. A final `data: [DONE]` line ends the stream. Only `/chat/completions` streams incrementally; `/responses` currently emits a single terminal event.\n\n## Rate limits\n\n{{RATE_LIMITS}}\n\nWeb search runs on its own budget, separate from the model endpoints: 20 requests per minute, 3 concurrent, and 500 searches per day per key. Image endpoints have their own too: 20 requests per minute and 100 requests per month. Exceed any limit and you get a `429`.\n\n## Errors\n\nNaN uses conventional HTTP status codes: `2xx` on success, `4xx` for a problem with the request (a missing parameter, an invalid key, an unavailable model) and `5xx` for a server-side error. Every error returns a JSON body in the OpenAI shape:\n\n```json\n{\n \"error\": {\n \"message\": \"The model 'foo' does not exist.\",\n \"type\": \"invalid_request_error\",\n \"param\": \"model\",\n \"code\": \"model_not_found\"\n }\n}\n```\n\n`message` is human-readable, `param` names the offending field when applicable, and `code` is a short machine-readable string you can branch on.\n\n| Status | Meaning | `code` |\n| --- | --- | --- |\n| `400` | Invalid or malformed parameter (`param` says which); or content blocked by the safety filter. | `invalid_request_error` · `content_policy_violation` |\n| `401` | Missing or invalid API key. | `invalid_api_key` |\n| `402` | The token allowance for the billing period is spent on a model that carries one, such as `glm5.2`. Not retryable: the counter returns to zero when your billing period starts. | `monthly_cap_reached` |\n| `403` | Your tier can't access this endpoint or model. Image generation requires inference membership. | `tier_restricted` |\n| `404` | The requested model doesn't exist. | `model_not_found` |\n| `429` | Rate limit hit (`rpm_limit`, `max_parallel_requests`), the rolling 4h token budget of `glm5.2`, or a quota exhausted. | `rate_limit_exceeded` · `insufficient_quota` · `quota_exceeded` |\n| `500` | Something went wrong on our side (includes upstream model errors). | (none) |\n| `503` | Web search is temporarily unavailable; retry shortly. | `search_unavailable` |\n| `524` | Timeout, typical with large audio files on `/audio/transcriptions`. | (none) |\n\nRetry `429` and `5xx` responses with exponential backoff. Don't retry `400`, `401`, `403`, or `404` blindly: they'll fail the same way every time until you change the request. `402` cannot be fixed by repetition either: it clears when your billing period starts.\n\n## Model catalog\n\nEvery endpoint takes a `model` id. Capabilities vary by model:\n\n| Model | Use for | Capabilities |\n| --- | --- | --- |\n| `deepseek-v4-flash` | Chat, reasoning | Streaming, tool calling, reasoning, 1M-token context. 500M tokens/month per member |\n| `mimo-v2.5` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `qwen3.6` | Chat, agents | Streaming, tool calling, vision, reasoning (opt-out, returns `reasoning_content`) |\n| `gemma4` | Chat, vision, agents | Streaming, tool calling, vision, reasoning (opt-in) |\n| `glm5.2` | Coding, long-horizon agents | Streaming, tool calling, reasoning trace, 500K-token context. Premium tier only |\n| `qwen3-embedding` | Embeddings | 4096-dimension vectors |\n| `rerank` | RAG reranking | Qwen3-Reranker-8B, 100+ languages |\n| `kokoro` | Text-to-speech | Multiple voices and audio formats |\n| `whisper` | Speech-to-text | Transcription with word/segment timestamps |\n| `flux-2-klein` | Image generation | Text-to-image and image-to-image |\n\n`glm5.2` is served only to keys on the GLM 5.2 premium tier; every other model is available to any inference member. Call [List models](#tag/Models) for the exact set available to your key.\n\n## Versioning & compatibility\n\nThe API tracks the OpenAI API surface, so OpenAI SDKs and tools work against `https://api.nan.builders/v1` unchanged. This reference documents the stable public `/v1` endpoints, and we add capabilities without breaking existing fields.", + "description": "Open models on a shared EU inference cluster. Zero logs.\n\nThe NaN API is OpenAI-compatible: predictable, resource-oriented URLs, JSON request and response bodies, and standard HTTP verbs and status codes. Point any OpenAI SDK at our base URL and your existing code keeps working. Change the base URL and the API key, and that's it.\n\nOne schema across every model, so you only learn the API once. Change the `model` field to switch models; everything else stays the same.\n\n- Base URL: `https://api.nan.builders/v1`\n- OpenAPI spec: this document. Import it into Postman, Insomnia, or your own tooling.\n\nIf you use the [Helmcode](https://helmcode.com) enterprise service, the base URL is `https://api.helmcode.com/v1` instead. Every other endpoint is identical.\n\n## Authentication\n\nEvery request authenticates with an API key, sent as a Bearer token:\n\n```\nAuthorization: Bearer $NAN_API_KEY\n```\n\nYou must be a NaN community member. Generate your key from user settings, under \"API Keys\", on the [platform](https://cloud.nan.builders/). The key is personal and non-transferable. Keep it secret: never embed one in client-side code or commit it to source control. Requests must go over HTTPS; calls over plain HTTP fail.\n\n## Making requests\n\nThe API is OpenAI-compatible, so point an official OpenAI SDK at our base URL and change nothing else:\n\n```python\nfrom openai import OpenAI\n\nclient = OpenAI(\n api_key=\"$NAN_API_KEY\",\n base_url=\"https://api.nan.builders/v1\",\n)\n\nresp = client.chat.completions.create(\n model=\"qwen3.6\",\n messages=[{\"role\": \"user\", \"content\": \"Hello\"}],\n)\nprint(resp.choices[0].message.content)\n```\n\n## Streaming\n\nChat responses can stream token-by-token. Set `\"stream\": true` on `/chat/completions` and the response arrives as Server-Sent Events: each event is a `data:` line carrying a `chat.completion.chunk`, with the new text in `choices[0].delta.content`. A final `data: [DONE]` line ends the stream. Only `/chat/completions` streams incrementally; `/responses` currently emits a single terminal event.\n\n## Rate limits\n\n{{RATE_LIMITS}}\n\nWeb search runs on its own budget, separate from the model endpoints: 20 requests per minute, 3 concurrent, and 500 searches per day per key. Image endpoints have their own too: 20 requests per minute and 100 requests per month. Exceed any limit and you get a `429`.\n\n## Errors\n\nNaN uses conventional HTTP status codes: `2xx` on success, `4xx` for a problem with the request (a missing parameter, an invalid key, an unavailable model) and `5xx` for a server-side error. Every error returns a JSON body in the OpenAI shape:\n\n```json\n{\n \"error\": {\n \"message\": \"The model 'foo' does not exist.\",\n \"type\": \"invalid_request_error\",\n \"param\": \"model\",\n \"code\": \"model_not_found\"\n }\n}\n```\n\n`message` is human-readable, `param` names the offending field when applicable, and `code` is a short machine-readable string you can branch on.\n\n| Status | Meaning | `code` |\n| --- | --- | --- |\n| `400` | Invalid or malformed parameter (`param` says which); or content blocked by the safety filter. | `invalid_request_error` · `content_policy_violation` |\n| `401` | Missing or invalid API key. | `invalid_api_key` |\n| `402` | The token allowance for the billing period is spent on a model that carries one, such as `glm5.2`. Not retryable: the counter returns to zero when your billing period starts. | `monthly_cap_reached` |\n| `403` | Your tier can't access this endpoint or model. Image generation requires inference membership. | `tier_restricted` |\n| `404` | The requested model doesn't exist. | `model_not_found` |\n| `429` | Rate limit hit (`rpm_limit`, `max_parallel_requests`), the rolling 4h token budget of `glm5.2`, or a quota exhausted. | `rate_limit_exceeded` · `insufficient_quota` · `quota_exceeded` |\n| `500` | Something went wrong on our side (includes upstream model errors). | (none) |\n| `503` | Web search is temporarily unavailable; retry shortly. | `search_unavailable` |\n| `524` | Timeout, typical with large audio files on `/audio/transcriptions`. | (none) |\n\nRetry `429` and `5xx` responses with exponential backoff. Don't retry `400`, `401`, `403`, or `404` blindly: they'll fail the same way every time until you change the request. `402` cannot be fixed by repetition either: it clears when your billing period starts.\n\n## Model catalog\n\nEvery endpoint takes a `model` id. Capabilities vary by model:\n\n| Model | Use for | Capabilities |\n| --- | --- | --- |\n| `deepseek-v4-flash` | Chat, reasoning | Streaming, tool calling, reasoning, 1M-token context. 2B tokens/month per member |\n| `mimo-v2.5` | Chat, vision, audio | Streaming, tool calling, reasoning, image input, audio input, 1M-token context. 1.0B tokens/month per member |\n| `qwen3.6` | Chat, agents | Streaming, tool calling, vision, reasoning (opt-out, returns `reasoning_content`) |\n| `gemma4` | Chat, vision, agents | Streaming, tool calling, vision, reasoning (opt-in) |\n| `glm5.2` | Coding, long-horizon agents | Streaming, tool calling, reasoning trace, 500K-token context. Premium tier only |\n| `qwen3-embedding` | Embeddings | 4096-dimension vectors |\n| `rerank` | RAG reranking | Qwen3-Reranker-8B, 100+ languages |\n| `kokoro` | Text-to-speech | Multiple voices and audio formats |\n| `whisper` | Speech-to-text | Transcription with word/segment timestamps |\n| `flux-2-klein` | Image generation | Text-to-image and image-to-image |\n\n`glm5.2` is served only to keys on the GLM 5.2 premium tier; every other model is available to any inference member. Call [List models](#tag/Models) for the exact set available to your key.\n\n## Versioning & compatibility\n\nThe API tracks the OpenAI API surface, so OpenAI SDKs and tools work against `https://api.nan.builders/v1` unchanged. This reference documents the stable public `/v1` endpoints, and we add capabilities without breaking existing fields.", "contact": { "name": "NaN", "url": "https://nan.builders"