From 77e6fc93b741c4251d44e9bff30f576198ad1609 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 21 Sep 2026 21:37:30 +0000 Subject: [PATCH 1/2] chore: sync models with dashboard API MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - deepseek-v4.1-flash: $0.30/$1.20 -> $0.14/$0.57 (ZGPU_PRICING, test CATALOG) - qwen3-30b-a3b-fp8: $0.05/$0.30 -> $0.10/$0.45 (ZGPU_PRICING, test CATALOG) - llama-3.1-8b-instruct-fast: $0.02/$0.05 -> $0.15/$0.28 (ZGPU_PRICING, test CATALOG, and the sample-call assertion that hard-codes the rate) - add glm-5.3-flash, gpt-5.6-luna, gpt-4.1-mini, gpt-5.4-nano: pricing, chat --model (Responses), README + DOCUMENTATION rows, DOCUMENTATION §1 bullet - remove deepseek-v4-flash-0731: pricing, test CATALOG and model list, CHAT_MODELS, README row + example + routing sentence, DOCUMENTATION row, §1 bullet, routing paragraph and §5 exceptions, ADDING_COMMANDS list - version 3.10.0 -> 3.11.0 ZGPU_FALLBACK is unchanged: glm-5.2 still holds both the highest input and the highest output rate in the catalog. Source: https://api-dashboard.zerogpu.ai/api/models Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_014jFCNTvLBuLipdPRdTbg6u --- README.md | 10 +++++----- docs/ADDING_COMMANDS.md | 2 +- docs/DOCUMENTATION.md | 11 +++++++---- package-lock.json | 4 ++-- package.json | 2 +- src/commands/chat.ts | 9 ++++++--- src/lib/savings.ts | 11 +++++++---- tests/savings.test.ts | 14 ++++++++------ 8 files changed, 37 insertions(+), 26 deletions(-) diff --git a/README.md b/README.md index eca8232..3aa51e3 100644 --- a/README.md +++ b/README.md @@ -141,9 +141,6 @@ zerogpu chat "Explique la mise en cache en une phrase." -m qwen3-30b-a3b-fp8 # A 262K-token context, for whole repos and very long documents zerogpu chat "$(cat ARCHITECTURE.md)" -m glm-5.2 - -# Coding and agentic work, at a fraction of the flagship price -zerogpu chat "Port this helper to async/await." -m deepseek-v4-flash-0731 ``` | Option | Description | @@ -159,11 +156,14 @@ zerogpu chat "Port this helper to async/await." -m deepseek-v4-flash-0731 | `gpt-oss-120b` | 120B MoE, 131K context, reasoning + function calling. | | `llama-guard-4-12b` | 12B dense, 164K context, brand safety + text moderation. | | `deepseek-v4.1-flash` | Sparse MoE (8B active on input, 16B on output), 1M context, long-context agentic work + function calling. | +| `glm-5.3-flash` | Hybrid sparse + linear attention, 1M context, coding and long-horizon agentic work + function calling. | +| `gpt-5.6-luna` | Cost-optimized GPT-5.6, 272K context, coding, chat, function calling, reasoning, RAG, summarization + translation. | +| `gpt-4.1-mini` | Fast, cost-efficient GPT-4.1, 1M context, coding, chat, function calling, reasoning, RAG, summarization + translation. | +| `gpt-5.4-nano` | Most cost-efficient GPT-5.4, 400K context, coding, chat, function calling, reasoning, RAG, summarization + translation. | | `qwen3-30b-a3b-fp8` | 30B MoE, 100+ languages, reasoning + function calling. | | `glm-5.2` | 753B MoE, 262K context, reasoning + function calling. The platform's most capable model, and its priciest. | -| `deepseek-v4-flash-0731` | 284B MoE (13B active), 1M context, coding and agentic workflows. | -`qwen3-30b-a3b-fp8`, `glm-5.2`, and `deepseek-v4-flash-0731` are served by the Chat Completions API rather than the Responses API; the CLI routes them automatically. +`qwen3-30b-a3b-fp8` and `glm-5.2` are served by the Chat Completions API rather than the Responses API; the CLI routes them automatically. #### `chat_thinking` diff --git a/docs/ADDING_COMMANDS.md b/docs/ADDING_COMMANDS.md index 87cfcd4..e2ba432 100644 --- a/docs/ADDING_COMMANDS.md +++ b/docs/ADDING_COMMANDS.md @@ -9,7 +9,7 @@ This guide explains how to add a new CLI command to the ZeroGPU CLI. - `src/commands/` — one file per command, each exporting a `registerCommand(program)` function. - `src/cli.ts` — wires every command into the root program. - `src/lib/responses.ts` — shared `RESPONSES_ENDPOINT`, `ResponsesApiResponse`, and the `extractOutputText` / `extractReasoningText` helpers for `/v1/responses` calls. -- `src/lib/chatCompletions.ts` — the same for `/v1/chat/completions`, used by models the platform serves only there (currently `qwen3-30b-a3b-fp8`, `glm-5.2`, and `deepseek-v4-flash-0731`), plus `toResponsesUsage` to normalize token counts for savings tracking. +- `src/lib/chatCompletions.ts` — the same for `/v1/chat/completions`, used by models the platform serves only there (currently `qwen3-30b-a3b-fp8` and `glm-5.2`), plus `toResponsesUsage` to normalize token counts for savings tracking. - `src/lib/auth.ts` — `getApiKey()` for authenticated requests. - `src/lib/request.ts` — plumbing for the endpoint commands: `requireApiKey`, `resolveInput` (argument or stdin), `parseJsonObject`, `postJson`, and the print helpers. diff --git a/docs/DOCUMENTATION.md b/docs/DOCUMENTATION.md index f529cc8..fa94cd2 100644 --- a/docs/DOCUMENTATION.md +++ b/docs/DOCUMENTATION.md @@ -5,7 +5,7 @@ `zerogpu-cli` is the official command-line interface for [ZeroGPU](https://zerogpu.ai), a distributed / edge inference platform for small language models (SLMs) and nano language models. The CLI is a thin, OpenAI-compatible client around the ZeroGPU **Responses API** (`https://api.zerogpu.ai/v1/responses`) — and, for models served only there, the **Chat Completions API** (`https://api.zerogpu.ai/v1/chat/completions`) — that lets you call a curated set of edge-optimized models directly from your terminal for common NLP workloads: - Conversational chat (`LFM2.5-1.2B-Instruct`, `LFM2.5-1.2B-Thinking`) -- Reasoning and tool-use chat (`gpt-oss-120b`, `llama-guard-4-12b`, `deepseek-v4.1-flash`, `qwen3-30b-a3b-fp8`, `glm-5.2`, `deepseek-v4-flash-0731`) +- Reasoning and tool-use chat (`gpt-oss-120b`, `llama-guard-4-12b`, `deepseek-v4.1-flash`, `glm-5.3-flash`, `gpt-5.6-luna`, `gpt-4.1-mini`, `gpt-5.4-nano`, `qwen3-30b-a3b-fp8`, `glm-5.2`) - IAB content/audience classification (`zlm-v1-iab-classify-edge`, `zlm-v2-iab-classify-edge-enriched`) - Domain-level IAB classification (`zlm-v1-iab-domain-classifier`) - Zero-shot classification (`deberta-v3-small`) @@ -214,11 +214,14 @@ zerogpu chat [-i ] [-m ] [-r] | `gpt-oss-120b` | Responses | 120B MoE, 131K context, reasoning + function calling. | | `llama-guard-4-12b` | Responses | 12B dense, 163,840-token context, brand safety + text moderation. | | `deepseek-v4.1-flash` | Responses | Sparse MoE (8B active on input, 16B on output), 1,048,576-token context, long-context agentic work + function calling. | +| `glm-5.3-flash` | Responses | Hybrid sparse + linear attention, 1,048,576-token context, coding and long-horizon agentic work + function calling. | +| `gpt-5.6-luna` | Responses | Cost-optimized GPT-5.6, 272,000-token context, coding, chat, function calling, reasoning, RAG, summarization + translation. | +| `gpt-4.1-mini` | Responses | Fast, cost-efficient GPT-4.1, 1,047,576-token context, coding, chat, function calling, reasoning, RAG, summarization + translation. | +| `gpt-5.4-nano` | Responses | Most cost-efficient GPT-5.4, 400,000-token context, coding, chat, function calling, reasoning, RAG, summarization + translation. | | `qwen3-30b-a3b-fp8` | Chat Completions | 30B MoE, 100+ languages, reasoning + function calling. | | `glm-5.2` | Chat Completions | 753B MoE, 262,144-token context, reasoning + function calling. The most capable model on the platform, and the most expensive by an order of magnitude. | -| `deepseek-v4-flash-0731` | Chat Completions | 284B MoE (13B active), 1,048,576-token context, coding and agentic workflows. | -`qwen3-30b-a3b-fp8`, `glm-5.2`, and `deepseek-v4-flash-0731` have no Responses endpoint, so the CLI posts them to `/v1/chat/completions` instead, mapping `--instructions` to a `system` message and normalizing `prompt_tokens` / `completion_tokens` back to Responses token names for savings tracking. This routing is transparent — the command and its output are identical either way. +`qwen3-30b-a3b-fp8` and `glm-5.2` have no Responses endpoint, so the CLI posts them to `/v1/chat/completions` instead, mapping `--instructions` to a `system` message and normalizing `prompt_tokens` / `completion_tokens` back to Responses token names for savings tracking. This routing is transparent — the command and its output are identical either way. **Example** ```bash @@ -815,7 +818,7 @@ Content-Type: application/json x-api-key: ``` -The exceptions are `summarize`, `chat --model qwen3-30b-a3b-fp8`, `chat --model glm-5.2`, and `chat --model deepseek-v4-flash-0731`, whose models the ZeroGPU platform serves only through the OpenAI-compatible Chat Completions endpoint: +The exceptions are `summarize`, `chat --model qwen3-30b-a3b-fp8`, and `chat --model glm-5.2`, whose models the ZeroGPU platform serves only through the OpenAI-compatible Chat Completions endpoint: ``` POST https://api.zerogpu.ai/v1/chat/completions diff --git a/package-lock.json b/package-lock.json index 9b3c7d7..a1e933d 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "zerogpu-cli", - "version": "3.10.0", + "version": "3.11.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "zerogpu-cli", - "version": "3.10.0", + "version": "3.11.0", "license": "MIT", "dependencies": { "commander": "^12.1.0", diff --git a/package.json b/package.json index d40bda6..17ad9bb 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "zerogpu-cli", - "version": "3.10.0", + "version": "3.11.0", "description": "Command-line interface for ZeroGPU.", "type": "module", "bin": { diff --git a/src/commands/chat.ts b/src/commands/chat.ts index fb7db56..9d69e9a 100644 --- a/src/commands/chat.ts +++ b/src/commands/chat.ts @@ -17,8 +17,8 @@ import { recordAndMaybeNotify } from "../lib/savings.js"; const DEFAULT_MODEL = "LFM2.5-1.2B-Instruct"; // Text-generation models `--model` accepts, and the API each one speaks. -// qwen3-30b-a3b-fp8, glm-5.2, and deepseek-v4-flash-0731 are Chat Completions -// only — they have no Responses endpoint. +// qwen3-30b-a3b-fp8 and glm-5.2 are Chat Completions only — they have no +// Responses endpoint. // Source: https://docs.zerogpu.ai/docs/text-generation const CHAT_MODELS: Record = { "LFM2.5-1.2B-Instruct": "responses", @@ -26,9 +26,12 @@ const CHAT_MODELS: Record = { "gpt-oss-120b": "responses", "llama-guard-4-12b": "responses", "deepseek-v4.1-flash": "responses", + "glm-5.3-flash": "responses", + "gpt-5.6-luna": "responses", + "gpt-4.1-mini": "responses", + "gpt-5.4-nano": "responses", "qwen3-30b-a3b-fp8": "chat-completions", "glm-5.2": "chat-completions", - "deepseek-v4-flash-0731": "chat-completions", }; // Model ids are case-sensitive to the API but not to the person typing them. diff --git a/src/lib/savings.ts b/src/lib/savings.ts index e2de728..5a2456c 100644 --- a/src/lib/savings.ts +++ b/src/lib/savings.ts @@ -21,12 +21,15 @@ const DEFAULT_BASELINE = "claude-opus-4-8"; // this table to the published catalog — see tests/savings.test.ts. export const ZGPU_PRICING: Record = { "gpt-oss-120b": { in: 0.15, out: 0.6 }, - "qwen3-30b-a3b-fp8": { in: 0.05, out: 0.3 }, + "qwen3-30b-a3b-fp8": { in: 0.1, out: 0.45 }, "glm-5.2": { in: 1.1, out: 3.5 }, - "deepseek-v4.1-flash": { in: 0.3, out: 1.2 }, - "deepseek-v4-flash-0731": { in: 0.16, out: 0.38 }, + "deepseek-v4.1-flash": { in: 0.14, out: 0.57 }, + "glm-5.3-flash": { in: 0.1, out: 0.35 }, + "gpt-5.6-luna": { in: 0.2, out: 1.2 }, + "gpt-4.1-mini": { in: 0.4, out: 1.6 }, + "gpt-5.4-nano": { in: 0.2, out: 1.25 }, "llama-guard-4-12b": { in: 0.18, out: 0.18 }, - "llama-3.1-8b-instruct-fast": { in: 0.02, out: 0.05 }, + "llama-3.1-8b-instruct-fast": { in: 0.15, out: 0.28 }, "zlm-v1-iab-classify-edge": { in: 0.02, out: 0.05 }, "zlm-v2-iab-classify-edge-enriched": { in: 0.025, out: 0.15 }, "zlm-v1-iab-domain-classifier": { in: 0.02, out: 0.05 }, diff --git a/tests/savings.test.ts b/tests/savings.test.ts index 6fb0e69..e768a4a 100644 --- a/tests/savings.test.ts +++ b/tests/savings.test.ts @@ -52,7 +52,7 @@ describe("computeCallSavings", () => { ); expect(tokens).toBe(1020); const claude = (740 * 5 + 280 * 25) / 1e6; // 0.0107 - const zgpu = (740 * 0.02 + 280 * 0.05) / 1e6; // real ZeroGPU cost + const zgpu = (740 * 0.15 + 280 * 0.28) / 1e6; // real ZeroGPU cost expect(savingsUsd).toBeCloseTo(claude - zgpu, 8); }); @@ -93,7 +93,6 @@ describe("computeCallSavings", () => { "gpt-oss-120b", "qwen3-30b-a3b-fp8", "glm-5.2", - "deepseek-v4-flash-0731", "LFM2.5-1.2B-Instruct", ]) { expect(unknown).toBeLessThanOrEqual( @@ -112,12 +111,15 @@ describe("ZGPU_PRICING tracks the published model catalog", () => { // gpt-oss-120b sat at $0.03/$0.10 long after it was repriced to $0.15/$0.60. const CATALOG: Record = { "gpt-oss-120b": { in: 0.15, out: 0.6 }, - "qwen3-30b-a3b-fp8": { in: 0.05, out: 0.3 }, + "qwen3-30b-a3b-fp8": { in: 0.1, out: 0.45 }, "glm-5.2": { in: 1.1, out: 3.5 }, - "deepseek-v4.1-flash": { in: 0.3, out: 1.2 }, - "deepseek-v4-flash-0731": { in: 0.16, out: 0.38 }, + "deepseek-v4.1-flash": { in: 0.14, out: 0.57 }, + "glm-5.3-flash": { in: 0.1, out: 0.35 }, + "gpt-5.6-luna": { in: 0.2, out: 1.2 }, + "gpt-4.1-mini": { in: 0.4, out: 1.6 }, + "gpt-5.4-nano": { in: 0.2, out: 1.25 }, "llama-guard-4-12b": { in: 0.18, out: 0.18 }, - "llama-3.1-8b-instruct-fast": { in: 0.02, out: 0.05 }, + "llama-3.1-8b-instruct-fast": { in: 0.15, out: 0.28 }, "zlm-v2-iab-classify-edge-enriched": { in: 0.025, out: 0.15 }, "zlm-v1-iab-classify-edge": { in: 0.02, out: 0.05 }, "zlm-v1-iab-domain-classifier": { in: 0.02, out: 0.05 }, From 82e8e00adaea21493e1273e1bf5448ecf12086d1 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 22 Sep 2026 19:09:01 +0000 Subject: [PATCH 2/2] chore: price whisper-tiny and chatterbox-nano MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The dashboard API added two models after this branch was cut: - whisper-tiny (Speech-to-Text, 39M, $0/$0) - chatterbox-nano (Text-to-Speech, 110M, $0/$0) Neither is Text Generation, so per the model-sync rules they need a ZGPU_PRICING and test CATALOG entry only — no chat --model entry, no docs table row, and no new per-task command. Source: https://api-dashboard.zerogpu.ai/api/models Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_014jFCNTvLBuLipdPRdTbg6u --- src/lib/savings.ts | 3 +++ tests/savings.test.ts | 3 +++ 2 files changed, 6 insertions(+) diff --git a/src/lib/savings.ts b/src/lib/savings.ts index 5a2456c..c1fe88f 100644 --- a/src/lib/savings.ts +++ b/src/lib/savings.ts @@ -40,6 +40,9 @@ export const ZGPU_PRICING: Record = { "LFM2.5-1.2B-Thinking": { in: 0.02, out: 0.05 }, "LFM2.5-1.2B-Instruct": { in: 0.02, out: 0.05 }, "zlm-v1-moderation-edge": { in: 0.02, out: 0.05 }, + // Audio models: the catalog publishes $0 in and $0 out for both. + "whisper-tiny": { in: 0, out: 0 }, + "chatterbox-nano": { in: 0, out: 0 }, // Embedding models bill input tokens only; there are no output tokens to // charge, so `out: 0` is the real rate, not a placeholder. "all-minilm-l6-v2": { in: 0.004, out: 0 }, diff --git a/tests/savings.test.ts b/tests/savings.test.ts index e768a4a..e507153 100644 --- a/tests/savings.test.ts +++ b/tests/savings.test.ts @@ -130,6 +130,9 @@ describe("ZGPU_PRICING tracks the published model catalog", () => { "LFM2.5-1.2B-Thinking": { in: 0.02, out: 0.05 }, "LFM2.5-1.2B-Instruct": { in: 0.02, out: 0.05 }, "zlm-v1-moderation-edge": { in: 0.02, out: 0.05 }, + // Audio models: the catalog publishes $0 in and $0 out for both. + "whisper-tiny": { in: 0, out: 0 }, + "chatterbox-nano": { in: 0, out: 0 }, "all-minilm-l6-v2": { in: 0.004, out: 0 }, "bge-small-en-v1.5": { in: 0.004, out: 0 }, };