From 972366dc2e6e02a3a61ec0e766a164b049bc35c8 Mon Sep 17 00:00:00 2001 From: 1bcMax Date: Tue, 29 Sep 2026 20:57:46 +0800 Subject: [PATCH] fix(free): replace the delisted nemotron-3-nano-30b, and document the live free lineup blockrun delisted nvidia/nemotron-3-nano-30b on 2026-09-08 (NVIDIA deprovisioned it for the account). The gateway redirects the id to nano-omni. Code - router_adapter.FREE_TIERS: nano-30b was SIMPLE's primary and a fallback in MEDIUM, COMPLEX and REASONING. The adapter already dropped it at runtime, since it is not in the catalog, so SIMPLE had silently been opening on lightning. Every slot now goes to nemotron-3-nano-omni-30b-a3b-reasoning: the gateway's own redirect target, the same family and size, and the model those slots were actually served by. The 2026-08-31 reason for excluding it (it answered as nano-30b) is gone with nano-30b. - router_core: re-synced to router-core e38958b (BlockRunAI/router-core#3). eco SIMPLE fallback[0] goes nano-30b -> nano-omni; nano-30b is dropped from model_capabilities; nano-omni gets the supportsVision:false override. The decision fixture is a verbatim copy of upstream's regenerated snapshot, and the parity test passes on it. - New test: FREE_TIERS may only name ids from the $0 chat set /v1/models published on 2026-09-29, and never nano-30b. The existing membership check compared the table against a hand-kept list and could not catch this. Docs - README "Available free models", the `free` routing-profile row, the hero line, both code examples, the NVIDIA section and the FAQ count now name the six live free models instead of the 08-12 lineup (step-3.7-flash, mistral-nemotron, nemotron-nano v2, gpt-oss). gpt-oss-120b/20b were retired 2026-09-03, so their "direct calls still work" rows and privacy note are gone. Counts use the models.free marker. - client.py streaming docstring: example ids moved off the retired deepseek-v4-flash / llama-4-maverick. Co-Authored-By: Claude Opus 5.5 (1M context) --- README.md | 65 +++++++++---------- blockrun_llm/client.py | 6 +- blockrun_llm/router_adapter.py | 24 +++++-- blockrun_llm/router_core/__init__.py | 2 +- blockrun_llm/router_core/config.py | 9 ++- .../router_core/model_capabilities.py | 16 ++--- .../unit/router_core_decisions.snapshot.json | 2 +- tests/unit/test_router_adapter.py | 23 ++++++- tests/unit/test_router_core_snapshot.py | 2 +- tests/unit/test_routing_parity.py | 2 +- 10 files changed, 91 insertions(+), 60 deletions(-) diff --git a/README.md b/README.md index f3f7a0d..c55f5fe 100644 --- a/README.md +++ b/README.md @@ -8,7 +8,7 @@ > request settle itself over x402 — on **Solana or Base**. Every client takes > either credential in the same first argument. > -> 🆓 **Includes 8 fully-free NVIDIA-hosted models** — DeepSeek V4 Flash (1M context), Nemotron Nano Omni (vision), Qwen3 Next + Coder, Llama 4 Maverick, Mistral Small 4, plus `gpt-oss-120b/20b` (hidden from `/v1/models` but direct calls still work). Zero USDC, no rate-limit gimmicks. Use `routing_profile="free"` or call any `nvidia/*` model directly. +> 🆓 **Includes 6 free models** — Nemotron 3.5 Lightning (1M context), Nemotron 3 Ultra 550B, Nemotron 3 Nano Omni, Llama 3.2 11B Vision, Cohere North Mini Code and Poolside Laguna XS 2.1. Zero USDC, no rate-limit gimmicks. Use `routing_profile="free"` or call any of them directly. [![PyPI](https://img.shields.io/pypi/v/blockrun-llm.svg)](https://pypi.org/project/blockrun-llm/) [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](LICENSE) @@ -62,7 +62,7 @@ reports which rail you ended up on. ### Try It Free (No Balance Required) Want to kick the tires before topping up or funding a wallet? Route to -BlockRun's free NVIDIA tier — it settles $0 on both rails, so an unfunded wallet +BlockRun's free tier — it settles $0 on both rails, so an unfunded wallet or a $0 credit account is enough: ```python @@ -71,31 +71,30 @@ from blockrun_llm import LLMClient client = LLMClient() # a credential is still needed; a balance is not # Option 1: call a free model directly -response = client.chat("nvidia/step-3.7-flash", "Explain x402 in 1 sentence") +response = client.chat("nvidia/nemotron-3.5-lightning", "Explain x402 in 1 sentence") # Option 2: let the smart router pick the best free model per request result = client.smart_chat("What is 2+2?", routing_profile="free") -print(result.model) # e.g. 'nvidia/step-3.7-flash' (cheapest capable for SIMPLE tier) +print(result.model) # e.g. 'nvidia/nemotron-3-nano-omni-30b-a3b-reasoning' (free SIMPLE tier) print(result.response) # '4' ``` -**Available free models** (input + output both $0, all NVIDIA-hosted): +**Available free models** (input + output both $0; the live list is `GET /v1/models` filtered on $0 pricing): | Model ID | Context | Best For | |----------|---------|----------| -| `nvidia/step-3.7-flash` | 131K | Fast general-purpose chat + reasoning | -| `nvidia/mistral-nemotron` | 131K | Fast free Mistral (Mistral × NVIDIA) | -| `nvidia/nemotron-3-nano-omni-30b-a3b-reasoning` | 256K | Only vision-capable free model — text + images + video (≤2 min) + audio (≤1 hr) | -| `nvidia/nemotron-nano-9b-v2` | 131K | Compact fast chat | -| `nvidia/nemotron-nano-12b-v2-vl` | 131K | Compact vision | -| `nvidia/gpt-oss-120b` | 128K | OpenAI open-weight 120B — the free workhorse. Hidden from `/v1/models` (so SmartChat won't auto-pick it) but direct calls still work | -| `nvidia/gpt-oss-20b` | 128K | OpenAI open-weight 20B — 155 tok/s. Hidden from `/v1/models` but direct calls still work | +| `nvidia/nemotron-3.5-lightning` | 1M | The free default — thinking-mode reasoning, longest free context | +| `nvidia/nemotron-3-ultra-550b` | 1M | Largest free model (550B / 55B active MoE) — strongest, but slower | +| `nvidia/nemotron-3-nano-omni-30b-a3b-reasoning` | 256K | Fast general chat + reasoning (30B-A3B) | +| `nvidia/llama-3.2-11b-vision` | 128K | The free Meta Llama | +| `cohere/north-mini-code` | 256K | Coding, sub-second responses | +| `poolside/laguna-xs-2.1` | 131K | Coding, ~161 tok/s | -> Need V4-Pro-class reasoning? Use the paid `deepseek/deepseek-v4-pro` ($0.435/$0.87 — the 75% launch promo became the permanent list price after 2026-05-31) — `nvidia/deepseek-v4-pro` is hidden because NVIDIA's NIM deployment is hung; backend MODEL_REDIRECTS forwards calls to V4 Flash. +> Two of these (`nemotron-3-nano-omni` and `llama-3.2-11b-vision`) are catalogued as vision-capable, but image input did not hold up on real probes — both return HTTP 200 with a wrong answer. Send images to a paid vision model. -> **Privacy note for `gpt-oss-120b/20b`**: NVIDIA's free build.nvidia.com tier reserves the right to use prompts/outputs for service improvement. The models are hidden from `/v1/models` so SmartChat won't auto-route to them, but direct calls still work — use them only when prompts contain no sensitive data. +> Need V4-Pro-class reasoning? Use the paid `deepseek/deepseek-v4-pro` ($0.435/$0.87 — the 75% launch promo became the permanent list price after 2026-05-31) — `nvidia/deepseek-v4-pro` is hidden because NVIDIA's NIM deployment is hung; backend MODEL_REDIRECTS forwards calls to V4 Flash. -> **Retired**: NVIDIA has EOL'd (HTTP 410) most of its early free lineup — the free DeepSeek family (last: `nvidia/deepseek-v4-flash`, 2026-08-12), `llama-4-maverick`, the qwen3 SKUs, free Mistral small/large, and more. The gateway auto-redirects pinned callers to a healthy free model, so old model IDs still return 200. +> **Retired**: NVIDIA has EOL'd most of its early free lineup — the free DeepSeek family (last: `nvidia/deepseek-v4-flash`, 2026-08-12), `step-3.7-flash`, `mistral-nemotron`, `nemotron-nano-9b-v2` / `-12b-v2-vl` (2026-08-30), `gpt-oss-120b/20b` (2026-09-03), `nemotron-3-nano-30b` (2026-09-08), `llama-4-maverick`, the qwen3 SKUs, free Mistral small/large, and more. The gateway auto-redirects pinned callers to a live model, so old model IDs still return 200 — from a different model. ## Solana Support @@ -228,7 +227,7 @@ print(decision.reasoning) # human-readable explanation of the pick | Profile | Description | Best For | |---------|-------------|----------| -| `free` | NVIDIA free tier — smart-routes across the 6 $0 models (Step 3.7 Flash, Mistral Nemotron, Nemotron Nano Omni / 9B / 12B VL) | Zero-cost testing, dev, prod | +| `free` | Free tier — smart-routes across the $0 models (Nemotron 3 Nano Omni, Nemotron 3.5 Lightning, Llama 3.2 11B Vision, Cohere North Mini Code, Poolside Laguna XS 2.1). Nemotron 3 Ultra 550B is free too, by direct call. | Zero-cost testing, dev, prod | | `eco` | Cheapest capable model per tier | Cost-sensitive production | | `auto` | Best balance of cost/quality (default) | General use | | `premium` | Top-tier models (Anthropic, OpenAI, Moonshot) | Quality-critical tasks | @@ -366,8 +365,8 @@ print(link["url"]) # open https://pay.coinbase.com/... to buy USDC on Base # (b) Transfer existing Base USDC to your wallet address print(client.get_wallet_address()) # send USDC on Base to this 0x… address -# (c) Skip funding entirely — the free NVIDIA models cost $0 -client.chat("nvidia/step-3.7-flash", "Hello!") # routing_profile="free" also works +# (c) Skip funding entirely — the free models cost $0 +client.chat("nvidia/nemotron-3.5-lightning", "Hello!") # routing_profile="free" also works ``` `$5` of USDC covers thousands of paid requests. Check your balance any time: @@ -516,24 +515,22 @@ glm-5 and glm-5-turbo on 2026-06-06) — the whole family now bills per-token. ### NVIDIA (Free & Hosted) -Free tier refreshed 2026-08-12. NVIDIA has retired (HTTP 410 end-of-life) -the entire free DeepSeek family — `nvidia/deepseek-v4-flash` was the last to -go — along with `llama-4-maverick`, `qwen3-coder-480b`, the free Mistral -small/large SKUs, and others. Retired models stay callable by ID: the gateway -auto-redirects them to a healthy free model, so pinned callers still get a -200. `nvidia/gpt-oss-120b` and `nvidia/gpt-oss-20b` remain callable by direct -ID but are hidden from `/v1/models` over the NVIDIA free tier's -prompt-retention terms (so SmartChat won't auto-pick them). The live list is -`GET /v1/models` filtered on the free flag. +Free tier checked against `/v1/models` on 2026-09-29. NVIDIA has retired +(HTTP 410 end-of-life, or deprovisioned) most of its earlier free lineup — the +free DeepSeek family, `step-3.7-flash`, the `nemotron-nano` v2 SKUs, +`gpt-oss-120b/20b` (2026-09-03) and `nemotron-3-nano-30b` (2026-09-08) among +them. Retired models stay callable by ID: the gateway auto-redirects them to a +live model, so pinned callers still get a 200. The free tier also includes two +non-NVIDIA models, `cohere/north-mini-code` and `poolside/laguna-xs-2.1` (see +[Try It Free](#try-it-free-no-balance-required)). The live list is +`GET /v1/models` filtered on $0 pricing. | Model | Input Price | Output Price | Context | Notes | |-------|-------------|--------------|---------|-------| -| `nvidia/step-3.7-flash` | **FREE** | **FREE** | 131K | Fast general-purpose chat + reasoning | -| `nvidia/mistral-nemotron` | **FREE** | **FREE** | 131K | Fast free Mistral | -| `nvidia/nemotron-3-nano-omni-30b-a3b-reasoning` | **FREE** | **FREE** | 256K | First vision-capable free model — RGB images, mp4 video | -| `nvidia/nemotron-nano-9b-v2` | **FREE** | **FREE** | 131K | Compact fast chat | -| `nvidia/gpt-oss-120b` | **FREE** | **FREE** | 128K | OpenAI open-weight 120B — 123 tok/s. Hidden from `/v1/models`; direct calls work | -| `nvidia/gpt-oss-20b` | **FREE** | **FREE** | 128K | OpenAI open-weight 20B — 155 tok/s. Hidden from `/v1/models`; direct calls work | +| `nvidia/nemotron-3.5-lightning` | **FREE** | **FREE** | 1M | Free default — thinking-mode reasoning | +| `nvidia/nemotron-3-ultra-550b` | **FREE** | **FREE** | 1M | Largest free model (550B / 55B active MoE) | +| `nvidia/nemotron-3-nano-omni-30b-a3b-reasoning` | **FREE** | **FREE** | 256K | Fast general chat + reasoning | +| `nvidia/llama-3.2-11b-vision` | **FREE** | **FREE** | 128K | Meta Llama 3.2 11B | | `moonshot/kimi-k2.5` | $0.60/M | $3.00/M | 262K | Kimi K2.5 direct from Moonshot (replaces `nvidia/kimi-k2.5`) | | `moonshot/kimi-k2.6` | $0.95/M | $4.00/M | 256K | Moonshot flagship (vision + reasoning_content) | @@ -1856,7 +1853,7 @@ When you make an API call, the SDK automatically handles x402 payment. It signs Router Core is BlockRun's built-in routing engine — shared with the TypeScript SDK and the gateway, so the same request routes the same way everywhere. It scores your request across 15 dimensions, drops every model that can't actually handle it (context, output length, tools, vision), then picks the cheapest capable one and keeps the rest as a fallback chain. Routing happens locally in under 1ms and makes no extra model call. It can save up to 84% on LLM costs compared to using premium models for every request. ### How much does it cost? -Pay only for what you use. Prices start at **FREE** (11 NVIDIA-hosted models). Paid models start at $0.10/M tokens. There are no minimums, subscriptions, or monthly fees. $5 in USDC gets you thousands of requests. +Pay only for what you use. Prices start at **FREE** (6 free models). Paid models start at $0.10/M tokens. There are no minimums, subscriptions, or monthly fees. $5 in USDC gets you thousands of requests. ### Can I use it with Solana? Yes. Install with `pip install blockrun-llm[solana]` and use `SolanaLLMClient` instead of `LLMClient`. Same API, different payment chain. diff --git a/blockrun_llm/client.py b/blockrun_llm/client.py index e48a8e0..56cf4ea 100644 --- a/blockrun_llm/client.py +++ b/blockrun_llm/client.py @@ -1005,7 +1005,7 @@ def chat_completion_stream( request returns 402, the SDK signs an EIP-712 payment locally, then re-issues the request with ``stream=true`` and the ``PAYMENT-SIGNATURE`` header. Free models (e.g. - ``nvidia/deepseek-v4-flash``) skip the 402 and stream directly. + ``nvidia/nemotron-3.5-lightning``) skip the 402 and stream directly. Fallback semantics ------------------ @@ -1019,9 +1019,9 @@ def chat_completion_stream( Example:: for chunk in client.chat_completion_stream( - "nvidia/deepseek-v4-flash", + "nvidia/nemotron-3.5-lightning", [{"role": "user", "content": "Hello"}], - fallback_models=["nvidia/llama-4-maverick"], + fallback_models=["nvidia/nemotron-3-nano-omni-30b-a3b-reasoning"], ): delta = chunk.choices[0].delta if delta.content: diff --git a/blockrun_llm/router_adapter.py b/blockrun_llm/router_adapter.py index 6aaf7c7..373c2ff 100644 --- a/blockrun_llm/router_adapter.py +++ b/blockrun_llm/router_adapter.py @@ -73,7 +73,16 @@ class ResolvedRoutingDecision(RoutingDecision, total=False): #: ``step-3.7-flash``, ``nemotron-nano-9b-v2``, ``nemotron-nano-12b-v2-vl`` and #: ``mistral-nemotron`` (retired upstream 2026-08-30), plus ``nemotron-3-ultra-550b`` #: and ``nemotron-3-nano-omni-30b-a3b-reasoning`` — the latter two still list at -#: $0 in ``/v1/models`` but both answer as ``nemotron-3-nano-30b``. +#: $0 in ``/v1/models`` but both answered as ``nemotron-3-nano-30b``. +#: +#: 2026-09-29: ``nemotron-3-nano-30b`` itself is gone. NVIDIA deprovisioned it for +#: blockrun's account on 2026-09-08, it left ``/v1/models``, and the gateway now +#: redirects it to nano-omni. Every slot it held goes to nano-omni: that is the +#: model those slots were actually being served by, it is the same family and size +#: (30B-A3B), and with nano-30b gone the reason it was excluded above no longer +#: applies (a model-echo probe that day got nano-omni's own NIM deployment back). +#: The router adapter was already dropping nano-30b at runtime, because it is not +#: in the catalog, so SIMPLE had been opening on its first fallback. #: #: The table is no longer NVIDIA-only: ``cohere/north-mini-code`` and #: ``poolside/laguna-xs-2.1`` serve at $0 and carry the free coding load. @@ -81,9 +90,10 @@ class ResolvedRoutingDecision(RoutingDecision, total=False): #: the NVIDIA free tier's prompt-retention policy. FREE_TIERS: dict[str, TierConfig] = { "SIMPLE": { - # Fastest free model (~121 tok/s), and latency is the only axis that - # separates free rungs — they all cost $0. - "primary": "nvidia/nemotron-3-nano-30b", # 131K ctx + # Was nemotron-3-nano-30b, the fastest free model, until its 2026-09-08 + # delisting. Latency is the only axis that separates free rungs (they + # all cost $0); nano-omni answered in 1.4s on blockrun's direct probe. + "primary": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", # 256K ctx "fallback": [ "nvidia/nemotron-3.5-lightning", "nvidia/llama-3.2-11b-vision", @@ -93,7 +103,7 @@ class ResolvedRoutingDecision(RoutingDecision, total=False): "MEDIUM": { "primary": "nvidia/nemotron-3.5-lightning", # 1M ctx — free tier flagship "fallback": [ - "nvidia/nemotron-3-nano-30b", + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", "poolside/laguna-xs-2.1", "cohere/north-mini-code", ], @@ -104,14 +114,14 @@ class ResolvedRoutingDecision(RoutingDecision, total=False): "primary": "nvidia/nemotron-3.5-lightning", "fallback": [ "cohere/north-mini-code", # 256K ctx - "nvidia/nemotron-3-nano-30b", + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", "nvidia/llama-3.2-11b-vision", # free vision ], }, "REASONING": { "primary": "nvidia/nemotron-3.5-lightning", "fallback": [ - "nvidia/nemotron-3-nano-30b", + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", "cohere/north-mini-code", "poolside/laguna-xs-2.1", ], diff --git a/blockrun_llm/router_core/__init__.py b/blockrun_llm/router_core/__init__.py index e27de28..3c8d105 100644 --- a/blockrun_llm/router_core/__init__.py +++ b/blockrun_llm/router_core/__init__.py @@ -2,7 +2,7 @@ Router Core — deterministic, constraint-first model routing. Python port of `@blockrun/router-core `_ -(upstream commit ``5ee7c23``), the same routing engine the TypeScript SDK and +(upstream commit ``e38958b``), the same routing engine the TypeScript SDK and the BlockRun gateway use. The package is deliberately product-neutral: task classification, hard capability filtering, portfolio scoring, ordered fallbacks, and routing configuration. It contains no wallet, gateway client, diff --git a/blockrun_llm/router_core/config.py b/blockrun_llm/router_core/config.py index dc784fd..2068e0b 100644 --- a/blockrun_llm/router_core/config.py +++ b/blockrun_llm/router_core/config.py @@ -2,7 +2,7 @@ Default Routing Config Python port of ``@blockrun/router-core`` ``config.ts`` (upstream commit -``5ee7c23``, 2026-08-30 — the same pin the TypeScript SDK +``e38958b``, 2026-09-29 — the same pin the TypeScript SDK bundles). V3.5: every tier rung is a public catalog id. All routing parameters as a module constant. Hosts override by passing their @@ -1156,7 +1156,12 @@ "SIMPLE": { "primary": "nvidia/nemotron-3.5-lightning", # FREE — NVIDIA free tier flagship, 1M ctx "fallback": [ - "nvidia/nemotron-3-nano-30b", # FREE — fastest free model (~121 tok/s) + # Was nvidia/nemotron-3-nano-30b until blockrun delisted it on 2026-09-08 + # (NVIDIA deprovisioned it for the account — a structured per-account + # 404). nano-omni is blockrun's own redirect target for that id, so this + # follows the same rule as the head: the router and the gateway name the + # same model. + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", # FREE — nano-30b's successor, 256K ctx # The free head keeps rotting with NVIDIA's hosting (deepseek-v4-flash # 410 2026-08-12, seed-oss-36b 410 2026-08-03, gpt-oss-120b/20b 400 # 2026-08-21, and on 2026-08-30 FOUR of the five visible free models at diff --git a/blockrun_llm/router_core/model_capabilities.py b/blockrun_llm/router_core/model_capabilities.py index 9228ce0..6dcc6d2 100644 --- a/blockrun_llm/router_core/model_capabilities.py +++ b/blockrun_llm/router_core/model_capabilities.py @@ -188,18 +188,18 @@ "supports_tools": False, "supports_vision": True, }, - # supportsTools: not probed — fails closed - "nvidia/nemotron-3-nano-30b": { - "context_window": 131_072, - "max_output_tokens": 16_384, - "supports_tools": False, - "supports_vision": False, - }, + # override: The catalog tags this model "vision", but a correctly sized probe does not + # hold up (ClawRouter, 2026-08-31): a 64x64 solid-red PNG was named correctly 1 of 4 + # times on Base, and on Solana the image was silently dropped and a text model answered + # "white". An HTTP 200 with a confident wrong answer gives the caller nothing to branch + # on, so image turns must not be routed here. It is ecoTiers.SIMPLE.fallback[0] since + # nemotron-3-nano-30b was delisted (2026-09-08); remove once a probe of this size comes + # back right on both chains. "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": { "context_window": 256_000, "max_output_tokens": 16_384, "supports_tools": False, - "supports_vision": True, + "supports_vision": False, }, # supportsTools: not probed — fails closed "nvidia/nemotron-3-ultra-550b": { diff --git a/tests/unit/router_core_decisions.snapshot.json b/tests/unit/router_core_decisions.snapshot.json index 61fb442..04f259a 100644 --- a/tests/unit/router_core_decisions.snapshot.json +++ b/tests/unit/router_core_decisions.snapshot.json @@ -969,7 +969,7 @@ "agenticScore": 0, "candidates": [ "nvidia/nemotron-3.5-lightning", - "nvidia/nemotron-3-nano-30b", + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", "google/gemini-2.5-flash-lite", "zai/glm-5.3-flash", "openai/gpt-5.6-luna", diff --git a/tests/unit/test_router_adapter.py b/tests/unit/test_router_adapter.py index b7cb1da..915513c 100644 --- a/tests/unit/test_router_adapter.py +++ b/tests/unit/test_router_adapter.py @@ -21,11 +21,12 @@ from blockrun_llm.types import RoutingDecision # The free chat models that answer as themselves, not via a gateway redirect. -# Verified with a two-pass model-echo probe on 2026-08-31; keep in step with +# Verified with a two-pass model-echo probe on 2026-08-31 (nano-omni re-checked +# 2026-09-29, after nemotron-3-nano-30b was delisted); keep in step with # router_adapter.FREE_TIERS. FREE_MODELS = [ "nvidia/nemotron-3.5-lightning", - "nvidia/nemotron-3-nano-30b", + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", "nvidia/llama-3.2-11b-vision", "cohere/north-mini-code", "poolside/laguna-xs-2.1", @@ -175,6 +176,24 @@ def test_every_free_tier_entry_is_live_in_the_catalog(self): for model in [tier["primary"], *tier["fallback"]]: assert model in FREE_MODELS, model + def test_names_only_models_in_the_live_free_catalog(self): + # FREE_MODELS above is hand-kept, so checking the table against it alone + # passed while SIMPLE's primary was a delisted id. Pin the table to the + # $0 chat set /v1/models actually published (2026-09-29), and name the + # id that slipped through so it cannot come back. + live = { + "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", + "nvidia/nemotron-3.5-lightning", + "nvidia/llama-3.2-11b-vision", + "nvidia/nemotron-3-ultra-550b", + "cohere/north-mini-code", + "poolside/laguna-xs-2.1", + } + used = {m for tier in FREE_TIERS.values() for m in [tier["primary"], *tier["fallback"]]} + assert used, "FREE_TIERS is empty" + assert used <= live, sorted(used - live) + assert "nvidia/nemotron-3-nano-30b" not in used # delisted 2026-09-08 + def test_uses_the_rules_strategy_so_paid_evidence_models_cannot_leak_in(self): decision = route( "Fix the TypeScript payment retry bug, run tests, and update the patch.", diff --git a/tests/unit/test_router_core_snapshot.py b/tests/unit/test_router_core_snapshot.py index 922e293..e447e1a 100644 --- a/tests/unit/test_router_core_snapshot.py +++ b/tests/unit/test_router_core_snapshot.py @@ -2,7 +2,7 @@ Cross-language decision-snapshot parity. ``router_core_decisions.snapshot.json`` is a verbatim copy of upstream -``decisions.snapshot.json`` at commit ``5ee7c23`` — 88 complete decisions the +``decisions.snapshot.json`` at commit ``e38958b`` — 88 complete decisions the TypeScript engine produced for a frozen corpus (22 prompts x 4 profiles with rotating tool/vision/structured-output shapes, frozen pricing, frozen clock). This test recomputes every decision with the Python port and compares field diff --git a/tests/unit/test_routing_parity.py b/tests/unit/test_routing_parity.py index 9fbcdf5..440bf21 100644 --- a/tests/unit/test_routing_parity.py +++ b/tests/unit/test_routing_parity.py @@ -33,7 +33,7 @@ {"id": "deepseek/deepseek-v4-pro", "pricing": {"input": 0.435, "output": 0.87}}, {"id": "moonshot/kimi-k2.7", "pricing": {"input": 0.95, "output": 4}}, {"id": "nvidia/nemotron-3.5-lightning", "pricing": {"input": 0, "output": 0}}, - {"id": "nvidia/nemotron-3-nano-30b", "pricing": {"input": 0, "output": 0}}, + {"id": "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", "pricing": {"input": 0, "output": 0}}, {"id": "nvidia/llama-3.2-11b-vision", "pricing": {"input": 0, "output": 0}}, {"id": "cohere/north-mini-code", "pricing": {"input": 0, "output": 0}}, {"id": "poolside/laguna-xs-2.1", "pricing": {"input": 0, "output": 0}},