diff --git a/i18n/en.json b/i18n/en.json index 0114d8a..8874353 100644 --- a/i18n/en.json +++ b/i18n/en.json @@ -4,8 +4,8 @@ "pageTitle": "NaN Community", "title": "Join the community while you wait for an inference seat.", "subtitle": "Discord access, conversations with builders, and a reserved spot in line for GPU access when capacity frees up.", - "priceLabel": "$14.99 / month", - "priceTaxNote": "Taxes included · billed in euros (€14.99) in the EU", + "priceLabel": "14,99€ / month", + "priceTaxNote": "Taxes included · billed in euros from any region", "includesLabel": "What's included", "notIncludedLabel": "Not included", "notIncluded": { @@ -302,7 +302,7 @@ "Events, workshops and hackathons", "Voice in the quarterly model vote" ], - "memberCond": "Month to month, no commitment. Limited spots: when it fills up, waitlist.", + "memberCond": "Month to month, no commitment. Billed in euros from any region. Limited spots: when it fills up, waitlist.", "memberCta": "Join the waitlist", "communityIncludes": [ "Discord access", @@ -311,13 +311,15 @@ "Events, workshops and hackathons", "A reserved spot in line for inference" ], - "communityCond": "No commitment. Cancel anytime. Billed in euros (€14.99) in the EU.", + "communityCond": "No commitment. Cancel anytime. Billed in euros from any region.", "communityCta": "Join the community", "premiumIncludes": [ - "GLM 5.2 — frontier open model with reasoning", - "1,000M token monthly allowance", - "Everything in nan_member · eu", - "Access granted immediately after billing" + "GLM 5.2, frontier open model with reasoning", + "3,000M token allowance per billing period", + "400M tokens per rolling 4h window: the limit a coding agent reaches first", + "500K context · 5 concurrent requests", + "Everything in nan_member", + "Access switches on a few minutes after payment" ], "premiumCond": "GLM 5.2 premium (200€/mo) is offered to members as seats open. Join the waitlist and we will flag you for this tier.", "premiumCta": "Join the waitlist", @@ -355,7 +357,7 @@ }, { "q": "Is it really unlimited tokens?", - "a": "On the cluster models, no counter. The frontier ones come with a monthly allowance per member so the plan holds up: 500M tokens on DeepSeek V4-Flash and 1.0B on MiMo v2.5. Those limits are published here, not in fine print. The bill is the fixed subscription price and it doesn’t move." + "a": "On the cluster models, no counter. The frontier ones come with a token allowance per member so the plan holds up: 500M tokens per month on DeepSeek V4-Flash, 1.0B per month on MiMo v2.5, and 3,000M per billing period on GLM 5.2 for members on the premium tier. Those limits are published here, not in fine print. The bill is the fixed subscription price and it doesn’t move." }, { "q": "How are the models chosen?", @@ -379,7 +381,7 @@ }, { "q": "What payment methods do you accept?", - "a": "Credit or debit card. Prices include taxes (VAT in the EU, taxes in USA/Latam). No commitment: cancel anytime." + "a": "Credit or debit card. New memberships are billed in euros from any region, taxes included; subscriptions opened earlier in US dollars keep their original price. No commitment: cancel anytime." }, { "q": "How do I get in?", diff --git a/i18n/es.json b/i18n/es.json index f1ad60f..ebace28 100644 --- a/i18n/es.json +++ b/i18n/es.json @@ -4,8 +4,8 @@ "pageTitle": "Comunidad NaN", "title": "Únete a la comunidad mientras esperas tu plaza de inferencia.", "subtitle": "Acceso al Discord, charlas con builders y plaza reservada en la cola para acceso a las GPUs cuando se libere capacidad.", - "priceLabel": "$14.99 / mes", - "priceTaxNote": "Impuestos incluidos · en EU se factura en euros (14,99€)", + "priceLabel": "14,99€ / mes", + "priceTaxNote": "Impuestos incluidos · se factura en euros desde cualquier región", "includesLabel": "Qué incluye", "notIncludedLabel": "Qué NO incluye", "notIncluded": { @@ -302,7 +302,7 @@ "Eventos, workshops y hackathons", "Voto en la votación trimestral de modelos" ], - "memberCond": "Mes a mes, sin compromiso. Plazas limitadas: cuando se llena, waitlist.", + "memberCond": "Mes a mes, sin compromiso. Se factura en euros desde cualquier región. Plazas limitadas: cuando se llena, waitlist.", "memberCta": "Únete a la waitlist", "communityIncludes": [ "Acceso a Discord", @@ -311,13 +311,15 @@ "Eventos, workshops y hackathons", "Un sitio reservado en la cola de inferencia" ], - "communityCond": "Sin compromiso. Cancela cuando quieras. En EU se factura en euros (14,99€).", + "communityCond": "Sin compromiso. Cancela cuando quieras. Se factura en euros desde cualquier región.", "communityCta": "Únete a la comunidad", "premiumIncludes": [ - "GLM 5.2 — modelo abierto frontier con razonamiento", - "Asignación mensual de 1.000M de tokens", - "Todo lo incluido en nan_member · eu", - "Acceso activado justo después del cobro" + "GLM 5.2, modelo abierto frontier con razonamiento", + "Asignación de 3.000M de tokens por periodo de facturación", + "400M de tokens por ventana deslizante de 4h: el límite que un agente de código alcanza primero", + "Contexto de 500K · 5 peticiones en paralelo", + "Todo lo incluido en nan_member", + "El acceso se activa pocos minutos después del pago" ], "premiumCond": "GLM 5.2 premium (200€/mes) se ofrece a miembros según haya plazas. Únete a la waitlist y te marcaremos para este tier.", "premiumCta": "Únete a la waitlist", @@ -355,7 +357,7 @@ }, { "q": "¿De verdad son tokens ilimitados?", - "a": "En los modelos del cluster, sin contador. Los frontier llevan cuota mensual por miembro para que el plan se sostenga: 500M tokens en DeepSeek V4-Flash y 1.0B en MiMo v2.5. Esos límites están publicados aquí, no en letra pequeña. La factura es el precio fijo de la suscripción y no se mueve." + "a": "En los modelos del cluster, sin contador. Los frontier llevan cuota de tokens por miembro para que el plan se sostenga: 500M tokens al mes en DeepSeek V4-Flash, 1.0B al mes en MiMo v2.5 y 3.000M por periodo de facturación en GLM 5.2 para los miembros del tier premium. Esos límites están publicados aquí, no en letra pequeña. La factura es el precio fijo de la suscripción y no se mueve." }, { "q": "¿Cómo se eligen los modelos?", @@ -379,7 +381,7 @@ }, { "q": "¿Qué métodos de pago aceptáis?", - "a": "Tarjeta de crédito o débito. Los precios incluyen impuestos (IVA en la UE, impuestos en USA/Latam). Sin compromiso: cancela cuando quieras." + "a": "Tarjeta de crédito o débito. Las altas nuevas se facturan en euros desde cualquier región, impuestos incluidos; las suscripciones abiertas antes en dólares mantienen su precio original. Sin compromiso: cancela cuando quieras." }, { "q": "¿Cómo entro?", diff --git a/src/components/docs/ModelCard.astro b/src/components/docs/ModelCard.astro index 4fb83aa..f36a871 100644 --- a/src/components/docs/ModelCard.astro +++ b/src/components/docs/ModelCard.astro @@ -13,30 +13,15 @@ interface Props { description: string; specs: Spec[]; items: string[]; - soon?: boolean; } -const { id, name, tag, leftLabel, rightLabel, description, specs, items, soon } = Astro.props; +const { id, name, tag, leftLabel, rightLabel, description, specs, items } = Astro.props; const heading = tag ? `${name} - ${tag}` : name; --- -

- {heading} - { - soon && ( - - coming soon - - ) - } -

+

{heading}

-
+

{leftLabel} diff --git a/src/components/docs/RateLimits.astro b/src/components/docs/RateLimits.astro index 0695057..66d22ee 100644 --- a/src/components/docs/RateLimits.astro +++ b/src/components/docs/RateLimits.astro @@ -1,8 +1,14 @@ --- import { env } from 'cloudflare:workers'; -import { getRateLimitsConfig } from '../../lib/rateLimits'; +import { + formatTokens, + getRateLimitsConfig, + windowedModelBody, + windowedModelHeadline, +} from '../../lib/rateLimits'; -const { perKey, tokensPerMinuteByModel, requestsPerMinuteByModel } = getRateLimitsConfig(env); +const { perKey, tokensPerMinuteByModel, requestsPerMinuteByModel, windowedModels } = + getRateLimitsConfig(env); ---

@@ -21,6 +27,44 @@ const { perKey, tokensPerMinuteByModel, requestsPerMinuteByModel } = getRateLimi
+{ + /* The premium models are gated by a sliding window, not by a per-minute + rate. The window is the limit an intensive coding-agent session reaches + first, so it gets its own highlighted block above the per-minute tables + instead of a line of small print underneath them. */ +} +{ + windowedModels.map((m) => ( +
+

+ {m.model} · premium tier limits +

+
+
+
Rolling {m.windowHours}h window
+
{formatTokens(m.windowTokens)} tokens
+
+
+
Allowance / billing period
+
{formatTokens(m.periodCapTokens)} tokens
+
+
+
Context window
+
{formatTokens(m.contextTokens)} tokens
+
+
+
Concurrent requests
+
{m.maxParallel}
+
+
+

+ {windowedModelHeadline(m)}{' '} + {windowedModelBody(m)} +

+
+ )) +} +

tokens / min por modelo diff --git a/src/components/nan/home/Pricing.astro b/src/components/nan/home/Pricing.astro index a7cb5aa..46e142a 100644 --- a/src/components/nan/home/Pricing.astro +++ b/src/components/nan/home/Pricing.astro @@ -1,5 +1,15 @@ --- -// Datos verificados contra nan.builders: Member EU 70€ · Member USA/Latam $75 · Community $14.99. +// Prices verified against nan.builders: Member 70€ · GLM 5.2 premium 200€ · Community 14,99€. +// +// There is no per-region tier and no per-region currency any more. Every new +// signup is charged in EUR from any region, on BOTH funnels: cloud-api's +// inferencePriceForNewCustomer and communityPriceForNewCustomer each always +// return the EU price. So the `nan_member · usa / latam — $75` tier was removed +// AND community is published in euros here, for the same reason: a prospect +// from outside the EU read a $ amount and landed on a EUR Checkout, a different +// price in a different currency, at the payment step. Legacy USD subscriptions +// still exist and keep their price; that is explained in the payment-methods +// FAQ entry, not here. import { getLang, useT } from '../../../lib/i18n'; const lang = getLang(Astro.url); @@ -7,7 +17,7 @@ const tt = useT(lang).pricing; const home = lang === 'es' ? '/es' : '/'; const tiers = [ - // GLM 5.2 premium — price swap on the existing nan_member · eu + // GLM 5.2 premium — price swap on the existing nan_member // subscription (200€/mo, EUR only). Shown first as the flagship tier. { name: 'nan_member · glm 5.2 premium', @@ -22,7 +32,7 @@ const tiers = [ primary: true, }, { - name: 'nan_member · eu', + name: 'nan_member', amount: '70€', per: tt.perVat, badge: tt.inferenceIncluded, @@ -33,21 +43,9 @@ const tiers = [ href: `${home}#waitlist`, primary: true, }, - { - name: 'nan_member · usa / latam', - amount: '$75', - per: tt.perTaxes, - badge: tt.inferenceIncluded, - ref: true, - includes: tt.memberIncludes, - cond: tt.memberCond, - cta: tt.memberCta, - href: `${home}#waitlist`, - primary: true, - }, { name: 'nan_community', - amount: '$14.99', + amount: '14,99€', per: tt.perTaxes, badge: null, ref: false, diff --git a/src/content/docs/api.mdx b/src/content/docs/api.mdx index 47baa08..03dad00 100644 --- a/src/content/docs/api.mdx +++ b/src/content/docs/api.mdx @@ -49,7 +49,7 @@ curl https://api.nan.builders/v1/models \

GET /v1/models

-Returns the list of available models for your key. Published models: `deepseek-v4-flash`, `mimo-v2.5`, `qwen3.6`, `gemma4`, `qwen3-embedding`, `rerank`, `kokoro`, `whisper`, `flux-2-klein` (includes the `flux-2-klein` image model). +Returns the list of available models for your key. Published models: `deepseek-v4-flash`, `mimo-v2.5`, `qwen3.6`, `gemma4`, `qwen3-embedding`, `rerank`, `kokoro`, `whisper`, `flux-2-klein` (includes the `flux-2-klein` image model). `glm5.2` is served too, and is callable only with a key on the GLM 5.2 premium tier. ### Request @@ -86,7 +86,7 @@ curl https://api.nan.builders/v1/models \

POST /v1/chat/completions

-The main chat endpoint. OpenAI Chat Completions compatible. Compatible models: `deepseek-v4-flash`, `mimo-v2.5`, `qwen3.6`, and `gemma4`. `glm5.2` is coming soon and is not callable yet. +The main chat endpoint. OpenAI Chat Completions compatible. Compatible models: `deepseek-v4-flash`, `mimo-v2.5`, `qwen3.6`, `gemma4`, and `glm5.2`. `glm5.2` needs a key on the GLM 5.2 premium tier; the other models are available to every inference member. @@ -114,7 +118,7 @@ The main chat endpoint. OpenAI Chat Completions compatible. Compatible models: ` | Field | Type | Description | |---|---|---| -| `model` | string · required | `deepseek-v4-flash`, `mimo-v2.5`, `qwen3.6` or `gemma4`. | +| `model` | string · required | `deepseek-v4-flash`, `mimo-v2.5`, `qwen3.6`, `gemma4` or `glm5.2` (premium tier). | | `messages` | array · required | List of messages `{ role, content }`. `content` can be a string or an array of parts `[{type:"text",text}, {type:"image_url",image_url:{url}}]` for multimodal input. | | `max_tokens` | integer · optional | Maximum tokens to generate. | | `stream` | boolean · optional | Default `false`. If `true`, the response arrives as SSE. | @@ -325,7 +329,7 @@ print(data["name"], data["age"]) ### Reasoning -All four LLMs generate reasoning and return it in `choices[0].message.reasoning_content`. The control mechanism varies by model: +Every chat LLM generates reasoning and returns it in `choices[0].message.reasoning_content`. The control mechanism varies by model: | Model | Control | |---|---| @@ -333,6 +337,7 @@ All four LLMs generate reasoning and return it in `choices[0].message.reasoning_ | `gemma4` | `chat_template_kwargs.enable_thinking` · desactivado por defecto | | `deepseek-v4-flash` | `reasoning_effort`: `low` \| `medium` \| `high` · default `medium` | | `mimo-v2.5` | siempre activo · no configurable por API hoy | +| `glm5.2` | razonamiento activo · sin parámetro publicado para controlarlo | #### enable_thinking (qwen3.6, gemma4) @@ -405,7 +410,7 @@ print(response.choices[0].message.reasoning_content) print(response.choices[0].message.content) ``` -A más `effort`, más tokens dedicados al razonamiento y mejor calidad en problemas complejos — a cambio de latencia y consumo de tu cuota mensual. +A más `effort`, más tokens dedicados al razonamiento y mejor calidad en problemas complejos — a cambio de latencia y consumo de tu cuota de tokens. #### mimo-v2.5 @@ -1137,9 +1142,10 @@ Errors follow the standard OpenAI format: HTTP non-2xx status with a JSON body d |---|---| | 400 | Parámetro inválido — the body includes `param` with the field that failed (e.g. `prompt`, `n`, `size`, `stream`, `mask` o `image` en los endpoints de imágenes). El filtro de seguridad returns `content_policy_violation`. | | 401 | `Authorization` header invalid or missing (`invalid_api_key`). | +| 402 | Token allowance for the billing period exhausted on a model that carries one (`monthly_cap_reached`), such as `glm5.2`. Not retryable: the counter goes back to zero when your billing period starts. | | 403 | Your tier does not have access to the endpoint (`tier_restricted`). Image generation requires inference membership. | | 404 | Model does not exist (campo `model`, `model_not_found`). | -| 429 | Rate limit exceeded — `rpm_limit` o `max_parallel_requests` (`rate_limit_exceeded`), or monthly quota exhausted (`quota_exceeded` / `insufficient_quota`, como la de 100 requests de imágenes). | +| 429 | Rate limit exceeded — `rpm_limit` o `max_parallel_requests` (`rate_limit_exceeded`), the rolling 4h token budget of `glm5.2` (the body says how much frees up and when), or monthly quota exhausted (`quota_exceeded` / `insufficient_quota`, como la de 100 requests de imágenes). | | 500 | Internal error (includes upstream model errors). | | 524 | Timeout (typical with large audios on [/v1/audio/transcriptions](#audio-transcriptions)). | diff --git a/src/content/docs/models.mdx b/src/content/docs/models.mdx index 36bb7f8..ca42326 100644 --- a/src/content/docs/models.mdx +++ b/src/content/docs/models.mdx @@ -64,24 +64,26 @@ with the same `base URL`. id="glm-5-2" name="glm5.2" tag="753B MoE" - soon leftLabel="text generation & chat — agentic coding" rightLabel="capabilities" - description="Not available yet: glm5.2 is not published on the community API, so requests naming it return a model error. ~753B parameter MoE model, focused on coding and long-horizon agentic tasks. 256K token context. Tool calling and reasoning (emits a reasoning trace). Text only." + description="Premium tier: callable only with a key on the GLM 5.2 premium membership. ~753B parameter MoE model, focused on coding and long-horizon agentic tasks. 500K token context. Tool calling and reasoning (emits a reasoning trace). Text only. 3,000M token quota per member, and the counter goes back to zero when your billing period starts. Also capped at 400M tokens per rolling 4h window, which is the limit a heavy coding-agent run reaches first." specs={[ { label: 'Type', value: 'MoE (~753B total)' }, { label: 'Quantization', value: 'FP8' }, { label: 'Attention', value: 'Sparse attention' }, - { label: 'Context', value: '256K tokens' }, + { label: 'Context', value: '500K tokens' }, { label: 'Input modalities', value: 'text' }, { label: 'Output modalities', value: 'text' }, + { label: 'Allowance / billing period', value: '3,000M tokens / member' }, + { label: 'Rolling 4h window', value: '400M tokens' }, ]} items={[ 'Tool calling (function calling)', 'Reasoning mode (reasoning trace)', 'Coding and long-horizon agentic tasks', - '256K token context', + '500K token context', 'Streaming generation (SSE)', + 'Requires the GLM 5.2 premium tier', ]} /> diff --git a/src/lib/__fixtures__/ratelimits.expected.md b/src/lib/__fixtures__/ratelimits.expected.md index 46febe9..5559207 100644 --- a/src/lib/__fixtures__/ratelimits.expected.md +++ b/src/lib/__fixtures__/ratelimits.expected.md @@ -3,6 +3,15 @@ - Requests / min: 60 rpm - Paralelo máximo: 5 concurrentes +**glm5.2 · premium tier limits** + +- Rolling 4h window: 400M tokens +- Allowance / billing period: 3,000M tokens +- Context window: 500K tokens +- Concurrent requests: 5 + +400M tokens per rolling 4 hours is the limit a heavy coding-agent run reaches first, well before the allowance. Once you hit it, glm5.2 requests are rejected until the window slides forward: it is a rolling window, not a daily reset. The allowance counter goes back to zero when your billing period starts, and if you upgrade part-way into a period that first allowance is prorated to the share of the period you paid for. + **tokens / min por modelo** - deepseek-v4-flash: 1.5M tpm diff --git a/src/lib/mdxToText.test.ts b/src/lib/mdxToText.test.ts index df03f75..e4ac8b1 100644 --- a/src/lib/mdxToText.test.ts +++ b/src/lib/mdxToText.test.ts @@ -85,17 +85,36 @@ describe('mdxToText rate limits', () => { perKey: { requestsPerMinute: 120, maxParallel: 8 }, tokensPerMinuteByModel: [{ model: 'foo', label: '2M tpm' }], requestsPerMinuteByModel: [{ model: 'bar', label: '500 rpm' }], + windowedModels: [ + { + model: 'baz', + contextTokens: 128_000, + maxParallel: 2, + windowHours: 6, + windowTokens: 7_000_000, + periodCapTokens: 9_000_000, + }, + ], }); expect(out).toContain('- Requests / min: 120 rpm'); expect(out).toContain('- Paralelo máximo: 8 concurrentes'); expect(out).toContain('- foo: 2M tpm'); expect(out).toContain('- bar: 500 rpm'); + expect(out).toContain('**baz · premium tier limits**'); + expect(out).toContain('- Rolling 6h window: 7M tokens'); + expect(out).toContain('- Allowance / billing period: 9M tokens'); + expect(out).toContain('- Context window: 128K tokens'); + expect(out).toContain('- Concurrent requests: 2'); }); it('defaults to the same numbers renders', async () => { const out = await mdxToText(input); expect(out).toContain('- Requests / min: 60 rpm'); expect(out).toContain('- Paralelo máximo: 5 concurrentes'); + // glm5.2 is gated by the window, not by a per-minute rate, and the docs + // had no row for it at all while the model was already being served. + expect(out).toContain('- Rolling 4h window: 400M tokens'); + expect(out).toContain('- Allowance / billing period: 3,000M tokens'); }); it('omits the per-model blocks when they are empty', async () => { @@ -103,9 +122,11 @@ describe('mdxToText rate limits', () => { perKey: { requestsPerMinute: 60, maxParallel: 5 }, tokensPerMinuteByModel: [], requestsPerMinuteByModel: [], + windowedModels: [], }); expect(out).not.toContain('tokens / min por modelo'); expect(out).not.toContain('requests / min por modelo'); + expect(out).not.toContain('premium tier limits'); }); }); diff --git a/src/lib/mdxToText.ts b/src/lib/mdxToText.ts index 828d6b6..8dcc954 100644 --- a/src/lib/mdxToText.ts +++ b/src/lib/mdxToText.ts @@ -3,7 +3,12 @@ import remarkParse from 'remark-parse'; import remarkGfm from 'remark-gfm'; import remarkMdx from 'remark-mdx'; import remarkStringify from 'remark-stringify'; -import { DEFAULT_RATE_LIMITS, type RateLimitsConfig } from './rateLimits'; +import { + DEFAULT_RATE_LIMITS, + formatTokens, + windowedModelNote, + type RateLimitsConfig, +} from './rateLimits'; const KNOWN_COMPONENTS = new Set([ 'ModelCard', @@ -347,6 +352,19 @@ function rateLimitsToMd(config: RateLimitsConfig): string { `- Paralelo máximo: ${config.perKey.maxParallel} concurrentes`, '', ]; + for (const m of config.windowedModels) { + lines.push( + `**${m.model} · premium tier limits**`, + '', + `- Rolling ${m.windowHours}h window: ${formatTokens(m.windowTokens)} tokens`, + `- Allowance / billing period: ${formatTokens(m.periodCapTokens)} tokens`, + `- Context window: ${formatTokens(m.contextTokens)} tokens`, + `- Concurrent requests: ${m.maxParallel}`, + '', + windowedModelNote(m), + '', + ); + } if (config.tokensPerMinuteByModel.length) { lines.push('**tokens / min por modelo**', ''); for (const m of config.tokensPerMinuteByModel) lines.push(`- ${m.model}: ${m.label}`); diff --git a/src/lib/rateLimits.ts b/src/lib/rateLimits.ts index c454303..6e6c058 100644 --- a/src/lib/rateLimits.ts +++ b/src/lib/rateLimits.ts @@ -20,10 +20,26 @@ export interface ModelRate { label: string; } +/** + * Limits for a model that is not gated by a per-minute rate but by a sliding + * window plus an allowance per billing period. Those are the two numbers a + * member has to plan against, so they are published as first-class rows + * instead of a footnote. + */ +export interface WindowedModelLimits { + model: string; + contextTokens: number; + maxParallel: number; + windowHours: number; + windowTokens: number; + periodCapTokens: number; +} + export interface RateLimitsConfig { perKey: PerKeyRateLimits; tokensPerMinuteByModel: ModelRate[]; requestsPerMinuteByModel: ModelRate[]; + windowedModels: WindowedModelLimits[]; } export interface RateLimitsEnv { @@ -44,8 +60,62 @@ export const DEFAULT_RATE_LIMITS: RateLimitsConfig = { { model: 'gemma4', label: '1.5M tpm' }, ], requestsPerMinuteByModel: [{ model: 'rerank', label: '1000 rpm' }], + // glm5.2 (premium tier) is absent from the per-minute tables on purpose: its + // gate is the 4h sliding window plus the allowance per billing period. These + // mirror the backend policy (cloud-api modelRateLimits + the token cap for + // glm5.2) and the usage hook's window budget, which is the same set of + // numbers the member portal publishes. + windowedModels: [ + { + model: 'glm5.2', + contextTokens: 500_000, + maxParallel: 5, + windowHours: 4, + windowTokens: 400_000_000, + periodCapTokens: 3_000_000_000, + }, + ], }; +/** + * Formats a token count the way every published surface writes it: 500K, + * 400M, 3,000M. Lives here so the docs page and /api/docs cannot drift from + * each other, which is the whole reason this module exists. + */ +export function formatTokens(tokens: number): string { + if (tokens >= 1_000_000) return `${(tokens / 1_000_000).toLocaleString('en-US')}M`; + if (tokens >= 1_000) return `${(tokens / 1_000).toLocaleString('en-US')}K`; + return String(tokens); +} + +/** + * The window wording, written once. + * + * emphasizes the headline clause and /api/docs serves plain + * text, so the sentence is split in two halves instead of being retyped on + * each surface: the page and the Discord bot were already caught disagreeing + * about rpm, and this is the number a premium member plans against. + */ +export function windowedModelHeadline(m: WindowedModelLimits): string { + return `${formatTokens(m.windowTokens)} tokens per rolling ${m.windowHours} hours`; +} + +/** The rest of the sentence started by windowedModelHeadline(). */ +export function windowedModelBody(m: WindowedModelLimits): string { + return ( + `is the limit a heavy coding-agent run reaches first, well before the allowance. ` + + `Once you hit it, ${m.model} requests are rejected until the window slides forward: ` + + `it is a rolling window, not a daily reset. The allowance counter goes back to zero ` + + `when your billing period starts, and if you upgrade part-way into a period that ` + + `first allowance is prorated to the share of the period you paid for.` + ); +} + +/** Headline plus body, for the surfaces that publish it as one paragraph. */ +export function windowedModelNote(m: WindowedModelLimits): string { + return `${windowedModelHeadline(m)} ${windowedModelBody(m)}`; +} + function parsePositiveInt(raw: string | undefined, fallback: number, varName: string): number { if (raw === undefined || raw.trim() === '') return fallback; const n = Number(raw); diff --git a/src/pages/api/waitlist.ts b/src/pages/api/waitlist.ts index 71f4cb1..77c47db 100644 --- a/src/pages/api/waitlist.ts +++ b/src/pages/api/waitlist.ts @@ -51,7 +51,7 @@ export const POST: APIRoute = async ({ request }) => { // GLM 5.2 premium interest: signups from the premium pricing card // (?premium=1) carry the flag through to the member row so the admin // panel can distinguish and invite them for the premium tier. - const wantsPremium = raw && typeof raw === 'object' && (raw as Record).wantsPremium === true; + const wantsPremium: boolean = !!(raw && typeof raw === 'object' && (raw as Record).wantsPremium === true); // Honeypot trap: reply 200 OK without persisting so bots get no signal. if (honeypot) { diff --git a/src/tests/landing/pricing.test.ts b/src/tests/landing/pricing.test.ts new file mode 100644 index 0000000..136e555 --- /dev/null +++ b/src/tests/landing/pricing.test.ts @@ -0,0 +1,218 @@ +import { describe, expect, test } from 'vitest'; +import { readFileSync } from 'node:fs'; +import { fileURLToPath } from 'node:url'; +import { dirname, resolve } from 'node:path'; +import { t, tArr, tObj } from '../../lib/i18n'; +import enData from '../../../i18n/en.json' with { type: 'json' }; +import esData from '../../../i18n/es.json' with { type: 'json' }; + +/** + * The pricing section is the surface a prospect reads BEFORE paying, so its + * copy is the one that has to be true first. It was selling a product that no + * longer exists: + * + * - "1,000M token monthly allowance" while the cap is 3,000M. + * - "Access granted immediately after billing", the exact promise removed + * from the member portal for being false (the grant can stay pending, and + * the API has an explicit state to say so). + * - a `nan_member · usa / latam — $75` tier, while every new signup is + * charged 70€ in any region, so the price and the currency changed between + * the card and the Checkout. + * + * That last one shipped TWICE because the currency asserts covered `memberCond` + * only: community kept publishing `$14.99` after cloud-api's + * communityPriceForNewCustomer started charging 14,99€ from every region. Every + * paid tier of the section is asserted here now, amount and condition, so a + * currency that only moves on one funnel cannot pass again. + */ + +const here = dirname(fileURLToPath(import.meta.url)); +const source = readFileSync(resolve(here, '../../components/nan/home/Pricing.astro'), 'utf-8'); +const locales = ['en', 'es'] as const; + +/** The tier objects the section renders, as written in the frontmatter. */ +const tiers = source.slice(0, source.indexOf('---', 4)); + +describe('Pricing — no per-region tier', () => { + test('the $75 tier is gone', () => { + expect(tiers).not.toMatch(/amount:\s*'\$75'/); + expect(tiers).not.toMatch(/name:\s*'nan_member · usa \/ latam'/); + }); + + test('the member tier is a single one, priced in euros', () => { + expect(tiers).toMatch(/name:\s*'nan_member',/); + expect(tiers).toMatch(/amount:\s*'70€'/); + }); + + test('the premium tier still leads at 200€', () => { + const first = tiers.indexOf("name: 'nan_member · glm 5.2 premium'"); + const member = tiers.indexOf("name: 'nan_member',"); + expect(first).toBeGreaterThan(-1); + expect(first).toBeLessThan(member); + expect(tiers).toMatch(/amount:\s*'200€'/); + }); + + test('the community tier is priced in euros too, like the Checkout charges', () => { + expect(tiers).toMatch(/name:\s*'nan_community',/); + expect(tiers).toMatch(/amount:\s*'14,99€'/); + }); + + /** + * Nothing in the section may quote a dollar amount: all three Checkouts are + * created in EUR for every region. The legacy USD subscriptions are true and + * stay explained in the payment-methods FAQ, which is not this file. + */ + test('no tier quotes a price in dollars', () => { + expect(tiers).not.toMatch(/amount:\s*'\$/); + for (const locale of locales) { + for (const key of ['premiumCond', 'memberCond', 'communityCond']) { + expect(t(`nan.pricing.${key}`, locale), `${locale}.${key}`).not.toMatch(/\$|USD/); + } + } + }); +}); + +/** Every string of a dictionary, keyed by its dotted path. */ +function flatten(node: unknown, path: string[] = [], out = new Map()) { + if (typeof node === 'string') out.set(path.join('.'), node); + else if (Array.isArray(node)) node.forEach((v, i) => flatten(v, [...path, String(i)], out)); + else if (node && typeof node === 'object') { + for (const [k, v] of Object.entries(node)) flatten(v, [...path, k], out); + } + return out; +} + +const dictionaries = { en: flatten(enData), es: flatten(esData) }; + +/** + * The two fixes this file guards were both "a price card kept a currency the + * Checkout had stopped using", found one tier at a time by a reviewer reading + * the pages. So the sweep is on the whole dictionary, not on the keys that + * happened to be reported: any NEW surface that quotes a dollar amount fails + * here without anyone remembering to add an assert for it. + * + * The word "dollars" in prose stays legal, because the legacy USD subscriptions + * are real and the payment-methods FAQ has to keep saying so. What cannot + * appear is an AMOUNT in dollars: every Checkout is created in EUR now. + */ +describe('i18n — no price is quoted in a currency we do not charge', () => { + test.each(locales)('%s quotes no dollar amount anywhere', (locale) => { + const offenders = [...dictionaries[locale]] + .filter(([, value]) => /\$\s?\d/.test(value)) + .map(([key, value]) => `${key}: ${value}`); + expect(offenders).toEqual([]); + }); + + /** + * The top-level `community` dictionary has no consumer left: /community + * hardcodes its copy and links to the home `#pricing` section instead of + * printing an amount. These two keys still carried `$14.99` and the + * "(€14.99) in the EU" caveat, so they are fixed and asserted rather than + * left to ship stale the day that page grows a price card again. + */ + test.each(locales)('%s community price card is in euros, for every region', (locale) => { + expect(t('community.priceLabel', locale)).toMatch(/14,99€/); + expect(t('community.priceLabel', locale)).not.toMatch(/\$/); + const note = t('community.priceTaxNote', locale); + expect(note).toMatch(/euros/); + expect(note).toMatch(/any region|cualquier región/); + expect(note).not.toMatch(/in the EU|en EU/); + }); +}); + +interface FaqItem { + q: string; + a: string; +} + +/** Every FAQ answer of a locale, joined. tArr() drops objects, so tObj(). */ +function faqAnswers(locale: string): string { + const faq = tObj<{ items?: FaqItem[] }>('nan.faq', locale); + expect(faq.items?.length ?? 0).toBeGreaterThan(0); + return (faq.items ?? []).map((item) => item.a).join(' '); +} + +describe.each(locales)('Pricing copy — %s', (locale) => { + const premium = () => tArr('nan.pricing.premiumIncludes', locale); + + test('publishes the real allowance and not the old 1,000M', () => { + const copy = premium().join(' '); + expect(copy).toMatch(/3[.,]000M/); + expect(copy).not.toMatch(/1[.,]000M/); + }); + + test('does not promise the access grant is immediate', () => { + const copy = premium().join(' ').toLowerCase(); + expect(copy).not.toContain('immediately'); + expect(copy).not.toContain('justo después'); + // What it says instead, matching the portal's wording. + expect(copy).toMatch(/few minutes after payment|pocos minutos después del pago/); + }); + + test('publishes the 4h window before the purchase, not only after it', () => { + const copy = premium().join(' '); + expect(copy).toMatch(/400M/); + expect(copy).toMatch(/4h/); + }); + + test('publishes context and concurrency', () => { + const copy = premium().join(' '); + expect(copy).toMatch(/500K/); + expect(copy).toMatch(/5 (concurrent requests|peticiones en paralelo)/); + }); + + test('refers to the member tier by the name the section renders', () => { + const copy = premium().join(' '); + expect(copy).toContain('nan_member'); + expect(copy).not.toContain('nan_member · eu'); + }); + + test('no em-dashes in the premium bullets', () => { + for (const item of premium()) expect(item).not.toContain('—'); + }); + + /** + * Both paid funnels charge EUR from every region, so both conditions have to + * say so. Asserting `memberCond` alone is what let community keep the "in the + * EU" caveat, which told a prospect from outside the EU that the euro did not + * apply to them right before a EUR Checkout. + */ + test.each(['memberCond', 'communityCond'])('%s states the billing currency', (key) => { + const cond = t(`nan.pricing.${key}`, locale); + expect(cond).toMatch(/euros/); + expect(cond).toMatch(/any region|cualquier región/); + expect(cond).not.toMatch(/in the EU|en EU/); + }); + + test('the payment-methods answer no longer prices by region, and keeps legacy USD true', () => { + const answers = faqAnswers(locale); + expect(answers).not.toMatch(/USA\/Latam/); + expect(answers).toMatch(/euros/); + expect(answers).toMatch(/dollars|dólares/); + }); + + test('the allowance FAQ lists the premium quota alongside the other frontier models', () => { + expect(faqAnswers(locale)).toMatch(/3[.,]000M/); + }); + + /** + * glm5.2's 3,000M is the one allowance that is NOT a calendar month: it resets + * with the Stripe billing period, which is why the portal and the docs stopped + * saying "monthly". The answer used to lump it in with DeepSeek's and MiMo's + * genuinely monthly quotas under a single "monthly allowance". + */ + test('the allowance FAQ names the billing period for the premium quota', () => { + const faq = tObj<{ items?: FaqItem[] }>('nan.faq', locale); + const answer = (faq.items ?? []).map((item) => item.a).find((a) => /3[.,]000M/.test(a)); + expect(answer).toBeDefined(); + expect(answer).toMatch(/per billing period|por periodo de facturación/); + expect(answer).not.toMatch(/monthly allowance|cuota mensual/); + }); +}); + +describe('Pricing copy — both locales stay parallel', () => { + test('the premium bullet lists have the same length', () => { + const [en, es] = locales.map((l) => tArr('nan.pricing.premiumIncludes', l)); + expect(en.length).toBe(es.length); + }); +}); diff --git a/src/tests/lib/docsGlm52.test.ts b/src/tests/lib/docsGlm52.test.ts new file mode 100644 index 0000000..4b41ec7 --- /dev/null +++ b/src/tests/lib/docsGlm52.test.ts @@ -0,0 +1,137 @@ +import { beforeAll, describe, expect, test } from 'vitest'; +import { readFileSync } from 'node:fs'; +import { fileURLToPath } from 'node:url'; +import { dirname, resolve } from 'node:path'; +import { mdxToText } from '../../lib/mdxToText'; + +/** + * What docs/models#glm-5-2 and docs/api#rate-limits publish about glm5.2. + * + * Both pages described a model that does not exist: "not available yet", + * "requests naming it return a model error" and a 256K context, while the + * model was being served in production and a premium member was using it. The + * asserts here are on the text /api/docs actually serves (same extractor the + * Discord bot consumes), so the page and the API are covered at once. + */ + +const here = dirname(fileURLToPath(import.meta.url)); + +function docBody(slug: string): string { + const raw = readFileSync(resolve(here, `../../content/docs/${slug}.mdx`), 'utf-8'); + // getEntry().body hands mdxToText the content without frontmatter. + return raw.replace(/^---\n[\s\S]*?\n---\n/, ''); +} + +let models = ''; +let api = ''; + +beforeAll(async () => { + models = await mdxToText(docBody('models')); + api = await mdxToText(docBody('api')); +}); + +/** The glm5.2 card only, so a 256K from another model cannot pass for it. */ +function glmSection(text: string): string { + const start = text.indexOf('### glm5.2'); + expect(start).toBeGreaterThan(-1); + const rest = text.slice(start + 1); + const end = rest.indexOf('\n### '); + return end === -1 ? rest : rest.slice(0, end); +} + +describe('docs/models — the glm5.2 card', () => { + test('does not say the model is unavailable', () => { + const section = glmSection(models); + expect(section).not.toMatch(/coming soon/i); + expect(section).not.toMatch(/not available yet/i); + expect(section).not.toMatch(/not published on the community api/i); + expect(section).not.toMatch(/return a model error/i); + }); + + test('publishes the 500K context and not the old 256K', () => { + const section = glmSection(models); + expect(section).toContain('500K token'); + expect(section).toContain('Context: 500K tokens'); + expect(section).not.toContain('256K'); + }); + + /** + * The spec label named the wrong window: `Monthly quota` sat two lines under a + * description saying the counter goes back to zero when the BILLING PERIOD + * starts. glm5.2 is the one model whose cap is not the calendar month + * (usage_quota.go clears the month default and stamps the Stripe period), so + * it uses the label RateLimits.astro already publishes for it. + */ + test('documents the allowance with the window it actually resets on', () => { + const section = glmSection(models); + expect(section).toContain('Allowance / billing period: 3,000M tokens / member'); + expect(section).not.toMatch(/Monthly quota/); + expect(section).toMatch(/billing period starts/); + }); + + test('documents the rolling window that a coding agent reaches first', () => { + const section = glmSection(models); + expect(section).toContain('400M tokens'); + expect(section).toMatch(/4h window/); + }); + + test('says the tier is required to call it', () => { + expect(glmSection(models)).toMatch(/premium/i); + }); +}); + +describe('docs/api — glm5.2 is callable', () => { + test('is listed among the compatible chat models', () => { + const line = api.split('\n').find((l) => l.includes('Compatible models:')); + expect(line).toBeDefined(); + expect(line).toContain('`glm5.2`'); + }); + + test('no surface still says it is coming soon or not callable', () => { + expect(api).not.toMatch(/coming soon/i); + expect(api).not.toMatch(/not callable/i); + }); + + test('the model field accepts it', () => { + // The extractor pads the table columns, hence the loose left edge. + const row = api.split('\n').find((l) => /^\|\s*`model`\s*\|/.test(l)); + expect(row).toBeDefined(); + expect(row).toContain('glm5.2'); + }); + + test('the rate limits block carries the four limits', () => { + expect(api).toContain('Rolling 4h window: 400M tokens'); + expect(api).toContain('Allowance / billing period: 3,000M tokens'); + expect(api).toContain('Context window: 500K tokens'); + expect(api).toContain('Concurrent requests: 5'); + }); + + test('the 4h window is spelled out, not left as small print', () => { + expect(api).toContain('400M tokens per rolling 4 hours'); + expect(api).toMatch(/rolling window, not a daily reset/); + }); + + test('the errors table documents the statuses the limits return', () => { + // The "## Errors" table, not the per-endpoint ones (web search has its own + // 429 rows and would answer first). + const section = api.slice(api.indexOf('## Errors')); + expect(section.length).toBeGreaterThan(0); + const rows = section.split('\n').filter((l) => l.startsWith('|')); + const status = (code: string) => rows.find((l) => new RegExp(`^\\|\\s*${code}\\s*\\|`).test(l)); + expect(status('402')).toMatch(/Token allowance for the billing period exhausted/); + expect(status('402')).toContain('monthly_cap_reached'); + expect(status('429')).toMatch(/rolling 4h token budget of `glm5\.2`/); + }); +}); + +describe('no published surface reveals how glm5.2 is sourced', () => { + test('neither page names an upstream provider or calls the model resold', () => { + for (const [name, text] of [ + ['models', () => models], + ['api', () => api], + ] as const) { + expect(text(), name).not.toMatch(/openrouter|z\.ai|zhipu|bigmodel/i); + expect(text(), name).not.toMatch(/resold|reseller|upstream provider|third.party provider/i); + } + }); +}); diff --git a/src/tests/lib/rateLimits.test.ts b/src/tests/lib/rateLimits.test.ts new file mode 100644 index 0000000..b7f38b7 --- /dev/null +++ b/src/tests/lib/rateLimits.test.ts @@ -0,0 +1,96 @@ +import { describe, expect, test } from 'vitest'; +import { + DEFAULT_RATE_LIMITS, + formatTokens, + getRateLimitsConfig, + windowedModelBody, + windowedModelHeadline, + windowedModelNote, +} from '../../lib/rateLimits'; + +/** + * The published numbers for glm5.2, and the reason this file exists. + * + * docs/api#rate-limits used to have no row at all for glm5.2 while the model + * was already being served, so the only limits a premium member could read + * were the ones that do not apply to them. These asserts pin the four numbers + * to what the platform actually enforces: + * + * context 500,000 cloud-api usage_quota.go modelRateLimits + * concurrency 5 idem, and the ratelimit hook + * 400M per rolling 4h ratelimit hook ROLLING_WINDOW_S / rolling budget + * 3,000M per period cloud-api usage_quota.go monthlyTokenCaps + * + * If any of them moves upstream, this test has to be updated in the same + * change: the failure is the point. + */ +const GLM = { + contextTokens: 500_000, + maxParallel: 5, + windowHours: 4, + windowTokens: 400_000_000, + periodCapTokens: 3_000_000_000, +}; + +describe('rateLimits — glm5.2 windowed limits', () => { + const glm = DEFAULT_RATE_LIMITS.windowedModels.find((m) => m.model === 'glm5.2'); + + test('glm5.2 is published', () => { + expect(glm).toBeDefined(); + }); + + test('publishes the four enforced limits', () => { + expect(glm).toMatchObject(GLM); + }); + + test('is absent from the per-minute tables, which do not gate it', () => { + const perMinute = [ + ...DEFAULT_RATE_LIMITS.tokensPerMinuteByModel, + ...DEFAULT_RATE_LIMITS.requestsPerMinuteByModel, + ]; + expect(perMinute.map((m) => m.model)).not.toContain('glm5.2'); + }); + + test('getRateLimitsConfig keeps the windowed models when env overrides the per-key values', () => { + const config = getRateLimitsConfig({ RATE_LIMIT_RPM: '120', RATE_LIMIT_PARALLEL: '9' }); + expect(config.perKey).toEqual({ requestsPerMinute: 120, maxParallel: 9 }); + expect(config.windowedModels).toEqual(DEFAULT_RATE_LIMITS.windowedModels); + }); +}); + +describe('formatTokens', () => { + test('writes the numbers the way every surface publishes them', () => { + expect(formatTokens(GLM.contextTokens)).toBe('500K'); + expect(formatTokens(GLM.windowTokens)).toBe('400M'); + expect(formatTokens(GLM.periodCapTokens)).toBe('3,000M'); + }); + + test('leaves small counts alone', () => { + expect(formatTokens(999)).toBe('999'); + }); +}); + +describe('the window wording', () => { + const glm = DEFAULT_RATE_LIMITS.windowedModels[0]; + + test('the headline carries the number and the window length', () => { + expect(windowedModelHeadline(glm)).toBe('400M tokens per rolling 4 hours'); + }); + + test('says it is reached first, is rolling, and when the allowance resets', () => { + const body = windowedModelBody(glm); + expect(body).toContain('reaches first'); + expect(body).toContain('rolling window, not a daily reset'); + expect(body).toContain('when your billing period starts'); + // The one case where the allowance is not the full 3,000M. + expect(body).toContain('prorated'); + }); + + test('the paragraph served by /api/docs is the same sentence the page shows', () => { + expect(windowedModelNote(glm)).toBe(`${windowedModelHeadline(glm)} ${windowedModelBody(glm)}`); + }); + + test('no em-dashes in member-facing copy', () => { + expect(windowedModelNote(glm)).not.toContain('—'); + }); +});